Upload files to 'scripts'
This commit is contained in:
@@ -0,0 +1,239 @@
|
||||
"""
|
||||
|
||||
AUTHOR:
|
||||
|
||||
Khushal P Soonderji
|
||||
|
||||
DATE:
|
||||
|
||||
CREATED: Wed, 25th Feb, 2026
|
||||
UPDATED: Wed, 25th Feb, 2026
|
||||
|
||||
OBJECTIVE:
|
||||
|
||||
To use fuzzy similarity scores to get the `User ID` of the `Reporting To` field for employees.
|
||||
|
||||
The problem is simple - we need to figure out hierarchy.
|
||||
In Cosec, we have `"User ID": "VIS07085"` with `"User Name": "Vibhor Iyer"`
|
||||
Under him, we have `"User ID": "VIS07024"` with `"User Name": "Punish Kumar"` and `"Direct Reporting": "Vibhor"`
|
||||
|
||||
For a computer "Vibor" and "Vibhor Iyer" are very distinct strings, so the next best thing to do is fuzzy
|
||||
matching. The danger is for cases where names overlap. For instance, an employee under "Dr. Nagaraj H" may get
|
||||
assigned to "Nagaraj Patel". This is a risk, BUT THERE IS NO COHERENCE IN THE INPUTS SO NO BETTER SOLUTION
|
||||
EXISTS AS OF THE TIME OF THE DEVELOPMENT.
|
||||
|
||||
REFERENCES:
|
||||
|
||||
N/A
|
||||
|
||||
DOWNLOADS:
|
||||
|
||||
N/A
|
||||
|
||||
"""
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** IMPORT ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# To make sibling directories accessible for imports:
|
||||
import sys
|
||||
sys.path.append(".")
|
||||
sys.path.append("..")
|
||||
|
||||
# For system-level activities:
|
||||
import os
|
||||
import copy
|
||||
from pathlib import Path
|
||||
|
||||
# To work with date and time:
|
||||
import time
|
||||
import datetime
|
||||
|
||||
# To work with tabulate data:
|
||||
import pandas as pd
|
||||
|
||||
# Cosec-related:
|
||||
from cosec_web.cosec_web import CosecWeb
|
||||
|
||||
# TCAOFF-related:
|
||||
from tcaoff.async_tcaoff import AsyncTheCAOffice
|
||||
|
||||
# My utils:
|
||||
from utils_v2.system import files
|
||||
from utils_v2.string import json
|
||||
from utils_v2.string import regex
|
||||
from utils_v2.string import fuzzy
|
||||
from utils_v2.date_time import date_time
|
||||
|
||||
# To work with datatypes:
|
||||
from typing import List, Dict, Any
|
||||
from collections import defaultdict
|
||||
|
||||
# To run a cron-like scheduler:
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
# For async activities:
|
||||
import asyncio
|
||||
|
||||
# For debugging:
|
||||
from icecream import IceCreamDebugger
|
||||
|
||||
# Common and Shared:
|
||||
from scripts import common
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** MACROS / ONE-TIME INIT ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# Debugging:
|
||||
printer = IceCreamDebugger(prefix = "Fuzzy Hierarchy | ", includeContext = True)
|
||||
err_printer = IceCreamDebugger(prefix = "[ERR] Fuzzy Hierarchy | ", includeContext = True)
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** VARIABLES ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# --- Nothing Yet
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** CLASSES ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
# --- Nothing Yet
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** FUNCTIONS ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
def get_hierarchy(
|
||||
cosec_muster_roll: str | Path,
|
||||
velankani_master_excel: str | Path,
|
||||
):
|
||||
|
||||
# Read the files:
|
||||
printer("Reading files.")
|
||||
print("\n\n")
|
||||
|
||||
# Read the Muster Roll from Cosec:
|
||||
cosec_muster_roll_df = CosecWeb.read_muster_roll_xls(str(cosec_muster_roll))
|
||||
cosec_muster_roll_df["Direct Reporting ID"] = None
|
||||
cosec_muster_roll_df["Level-1 ID"] = None
|
||||
cosec_muster_roll_df = cosec_muster_roll_df[[
|
||||
"User ID",
|
||||
"User Name",
|
||||
"Category Name",
|
||||
"Grade Name",
|
||||
"Branch Name",
|
||||
"Department Name",
|
||||
"Direct Reporting",
|
||||
"Direct Reporting ID",
|
||||
"Level-1",
|
||||
"Level-1 ID",
|
||||
]]
|
||||
printer("Cosec Muster Roll:")
|
||||
print(cosec_muster_roll_df[:25].to_string())
|
||||
print("\n\n")
|
||||
|
||||
# Read the Master file maintained by Velankani:
|
||||
velankani_master_df = pd.read_excel(velankani_master_excel)
|
||||
velankani_master_df = velankani_master_df.rename(
|
||||
columns = {
|
||||
"New Emp ID": "New Emp ID",
|
||||
"Emp ID": "Emp ID",
|
||||
"User Name": "User Name",
|
||||
"Personal Email": "Personal Email",
|
||||
"Moblie Number": "Mobile No.",
|
||||
"Birth Date": "DoB",
|
||||
"Gender": "Gender",
|
||||
"Joining Date": "Joining Date",
|
||||
"Shift Group": "Shift Group",
|
||||
"Shift ID": "Shift ID"
|
||||
}
|
||||
)
|
||||
printer("Velankani Master Excel:")
|
||||
print(velankani_master_df[:25].to_string())
|
||||
print("\n\n")
|
||||
|
||||
# Perform a left join to directly copy easily matchable data:
|
||||
# Merge selected columns from lookup_df into main_df
|
||||
printer("Performing left join.")
|
||||
cosec_muster_roll_df = cosec_muster_roll_df.merge(
|
||||
velankani_master_df[[
|
||||
"Emp ID", "Personal Email", "Mobile No.", "DoB", "Gender",
|
||||
"Joining Date", "Shift Group", "Shift ID"
|
||||
]],
|
||||
how = "left",
|
||||
left_on = "User ID",
|
||||
right_on = "Emp ID"
|
||||
)
|
||||
|
||||
# Now we start preparing the combined data.
|
||||
# Since the master file is just a reference, and we are focussing on automating Cosec,
|
||||
# we iterate over Cosec's records:
|
||||
printer("Performing fuzzy match.")
|
||||
time.sleep(1.0)
|
||||
print(f"| {'TARGET': <30} | {'BEST CHOICE': <30} | {'%': <4} |")
|
||||
print(f"| {'-' * 30} | {'-' * 30} | {'-' * 4} |")
|
||||
for i, target_row in cosec_muster_roll_df.iterrows():
|
||||
|
||||
# Loop through all the options in the lookup Master Excel and find the best match:
|
||||
direct_reporting_fuzzy_str = target_row["Direct Reporting"]
|
||||
best_match_score = 0.0
|
||||
best_match_entry = None
|
||||
for j, choice_row in cosec_muster_roll_df.iterrows():
|
||||
|
||||
# Make the key and check for the best match:
|
||||
score = fuzzy.get_match_score(
|
||||
target = direct_reporting_fuzzy_str,
|
||||
choice = choice_row["User Name"],
|
||||
partial = False
|
||||
)
|
||||
if score > best_match_score:
|
||||
best_match_score = score
|
||||
best_match_entry = choice_row.to_dict()
|
||||
|
||||
print(f"| {target_row['Direct Reporting']: <30} | {best_match_entry['User Name']: <30} | {best_match_score: <.2f} |")
|
||||
|
||||
|
||||
# *****************************************************************************************************************
|
||||
# ***** ****
|
||||
# *** MAIN PROGRAM ***
|
||||
# ***** ****
|
||||
# *****************************************************************************************************************
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
# Request the required files:
|
||||
print("Please paste the paths to the required files.")
|
||||
# muster_roll = input("COSEC MUSTER ROLL FILE: ")
|
||||
# master_excel = input("VEL. MASTER EXCEL FILE: ")
|
||||
muster_roll = r"D:\kps\PycharmProjects\cosec\cosec_web\sample_files\muster_roll.xls"
|
||||
master_excel = r"C:\Users\Khushal P Soonderji\Downloads\Master Data-25.02.2026.xlsx"
|
||||
|
||||
# Now we run the matching algo:
|
||||
get_hierarchy(
|
||||
cosec_muster_roll = muster_roll,
|
||||
velankani_master_excel = master_excel
|
||||
)
|
||||
Reference in New Issue
Block a user