""" AUTHOR: Khushal P Soonderji DATE: CREATED: Thu, 5th Feb, 2026 UPDATED: Thu, 5th Feb, 2026 OBJECTIVE: There will be some common actions across various scripts. This script holds those common actions. REFERENCES: N/A DOWNLOADS: N/A """ # ***************************************************************************************************************** # ***** **** # *** IMPORT *** # ***** **** # ***************************************************************************************************************** # To make sibling directories accessible for imports: import sys sys.path.append(".") sys.path.append("..") # For system-level activities: import os import copy # To work with date and time: import time import datetime # To work with tabulate data: import pandas as pd # Cosec-related: from cosec_web.cosec_web import CosecWeb # TCAOFF-related: from helpers.async_tcaoff import AsyncTheCAOffice # My utils: from utils_v2.system import files from utils_v2.string import json from utils_v2.string import regex from utils_v2.date_time import date_time # To work with datatypes: from typing import List, Dict, Any, Union from collections import defaultdict # To run a cron-like scheduler: from apscheduler.schedulers.asyncio import AsyncIOScheduler # For async activities: import asyncio # For debugging: from icecream import IceCreamDebugger # ***************************************************************************************************************** # ***** **** # *** MACROS / ONE-TIME INIT *** # ***** **** # ***************************************************************************************************************** # File paths: FILE_DIR = files.get_file_directory(include_filename = False) PROJ_DIR = files.get_parent_directory(FILE_DIR, depth = 1) CACHE_DIR = os.path.join(PROJ_DIR, "local", "cache") CREDS_DIR = os.path.join(PROJ_DIR, "creds") # --- COSEC_CREDS_FILE = os.path.join(CREDS_DIR, "cosec.json") TCAOFF_CREDS_FILE = os.path.join(CREDS_DIR, "tcaoff.json") MUSTER_ROLL_CACHE_FILE = os.path.join(CACHE_DIR, "muster_roll_cache.json") IN_OUT_SUMMARY_CACHE_FILE = os.path.join(CACHE_DIR, "in_out_summary_cache.json") PREV_DAY_IN_OUT_SUMMARY_CACHE_FILE = os.path.join(CACHE_DIR, "prev_day_in_out_summary_cache.json") CHROME_DRIVER_DIR = os.path.join(PROJ_DIR, "drivers", "chrome") USER_DATA_DIR = os.path.join(PROJ_DIR, "browser", "user_data") DOWNLOADS_DIR = os.path.join(PROJ_DIR, "downloads") # --- TEST_MODE_MUSTER_ROLL_FILE_PATH = os.path.join(PROJ_DIR, "cosec_web", "sample_files", "muster_roll.xls") TEST_MODE_IN_OUT_SUMMARY_FILE_PATH = os.path.join(PROJ_DIR, "cosec_web", "sample_files", "in_out_summary.xls") # Debugging: printer = IceCreamDebugger(prefix = "Common | ", includeContext = True) err_printer = IceCreamDebugger(prefix = "[ERR] Common | ", includeContext = True) # ***************************************************************************************************************** # ***** **** # *** VARIABLES *** # ***** **** # ***************************************************************************************************************** # --- Nothing Yet # ***************************************************************************************************************** # ***** **** # *** CLASSES *** # ***** **** # ***************************************************************************************************************** # --- Nothing Yet # ***************************************************************************************************************** # ***** **** # *** FUNCTIONS *** # ***** **** # ***************************************************************************************************************** def kill_chrome() -> None: """ To kill running Chrome processes so that they don't interfere with the one that will be spun up by automation code. :return: None """ # First kill the previous processes, # then wait if old processes were killed: kill_count = CosecWeb.kill_chrome_processes() if kill_count > 0: time.sleep(2.5) # --------------------------------------------------------------------------------------------------------------------- def get_in_out_summary( cosec_creds: dict, from_dt: date_time.datetime = None, to_dt: date_time.datetime = None, cache_file: str = IN_OUT_SUMMARY_CACHE_FILE, test_mode: bool = False ) -> bool: """ Get the latest In-Out-Summary from Matrix Cosec. It saves the data into a local cache file. :param cosec_creds: The credentials (and config) to operate Cosec Matrix. :param from_dt: The date from which to fetch In-Out-Summary. :param to_dt: The date till which to fetch In-Out-Summary. :param cache_file: The cache file to use to store the results. :param test_mode: If set to True, a past file will be used instead of getting new reports from Cosec. :return: True if the automated fetch was successful, else False. """ # Start by assuming failure: success = False # If test mode: if test_mode: report_path = TEST_MODE_IN_OUT_SUMMARY_FILE_PATH # When not working in test mode: else: # Force close other running Chrome processes: kill_chrome() # Create an instance of the automation object: cosec = CosecWeb( cosec_url = cosec_creds["creds"]["url"], username = cosec_creds["creds"]["username"], password = cosec_creds["creds"]["password"], driver_dir = CHROME_DRIVER_DIR, user_data_dir = USER_DATA_DIR, downloads_dir = DOWNLOADS_DIR, ) # Perform the login: cosec.login(initial_sleep = 2.5) # Get the in/out report: if to_dt is None: to_dt = date_time.get_current_ist_date_time() if from_dt is None: from_dt = to_dt - datetime.timedelta(days = 1) report_path = cosec.get_in_out_summary( initial_sleep = 1.0, from_date = from_dt, to_date = to_dt, group_ids = cosec_creds["inOutConfig"]["groupIds"], download_timeout = 60.0, timezone = cosec_creds["generalConfig"]["timezone"], ) # Log out to end the cycle: cosec.logout() # Close the browser window: cosec.quit() # Now process the report, # and save it to the JSON file: if report_path is not None: # Read the data: report_data = CosecWeb.read_in_out_summary_xls(report_path) # Assume that the punch time in the data is IST data, # then normalize it to UTC: def parse_dt(x): if pd.isnull(x): return None else: return date_time.to_timezone( datetime_object = date_time.as_if_timezone( datetime_object = date_time.parse_date_time(x), timezone = cosec_creds["generalConfig"]["timezone"] ), timezone = date_time.TIMEZONE_UTC ).timestamp() report_data["Punch Time"] = report_data["Punch Time"].apply(lambda x: parse_dt(x)) # Do the remaining cleanup and formatting: report_data = report_data.where(report_data.notna(), None) report_data = report_data.to_dict(orient = "records") report_data = { "ts": date_time.get_current_utc_date_time(as_string = False).timestamp(), "report": report_data } # Save the data to a JSON file: json.to_file( file = cache_file, python_data = report_data, no_space = True ) # Note down success: success = True # Done here: return success # --------------------------------------------------------------------------------------------------------------------- def compute_work_done( in_out_df: pd.DataFrame ) -> List[Dict[str, Union[str, int, float, None]]]: # Create the structure that will be given as the output: flattened_work_reports = [] # Create an internal dict with the structure: # work_reports["user"]["date"] = {...} work_reports = defaultdict(lambda: defaultdict(dict)) # Loop through the full DataFrame once, # and figure out the first in and last out times: for index, row in in_out_df.iterrows(): # Extract user and date: user_id = row["User ID"] punch_dt = date_time.parse_date_time(row["Punch Time"], timezone = date_time.TIMEZONE_IST) punch_date = punch_dt.strftime("%Y-%m-%d") punch_ts = punch_dt.timestamp() # Handle the first in time: if row["I/O Type"] == "In": if work_reports[user_id][punch_date].get("first_in") is None: work_reports[user_id][punch_date]["first_in"] = punch_ts # Handle the last out time: if row["I/O Type"] == "Out": if work_reports[user_id][punch_date].get("first_in") is not None: work_reports[user_id][punch_date]["last_out"] = punch_ts # Now use the first_in and last out times of each record to figure out the amount of work done: for user_id, user_reports in work_reports.items(): for punch_date, punch_info in user_reports.items(): # Some defaults: min_work_seconds = 10.0 * 60.0 * 60.0 work_seconds = 0.0 work_ot_seconds = 0.0 work_status = "A" # Extract, clean and compute punch timing: first_in = punch_info.get("first_in") last_out = punch_info.get("last_out") # When the user has a valid in-time, but no known out time, # we assume that he worked a full day: if first_in is not None and not last_out: last_out = first_in + min_work_seconds work_seconds = last_out - first_in work_ot_seconds = 0.0 # When the user has neither an in-time, nor an out-time, # we assume that he was absent the whole day: elif not first_in and not last_out: work_seconds = 0.0 work_ot_seconds = 0.0 # When we have both - an in-time and an out-time - we compute the work done. # Any work over 10 hours will be counted as overtime: elif first_in and last_out: work_seconds = last_out - first_in work_ot_seconds = max(0.0, work_seconds - min_work_seconds) # Finally, compute the work status. # 'A' --> Absent # 'H' --> Half Day # 'P' --> Present (Full Day) # 'OT' -> Overtime work_hours = work_seconds / (60 * 60) if work_hours > 10.0: work_status = "OT" elif 7.5 < work_hours <= 10.0: work_status = "P" elif 2.5 < work_hours <= 5.0: work_status = "H" else: work_status = "A" # Save the data: flattened_work_reports.append({ "user_id": user_id, "work_date": punch_date, "first_in": first_in, "last_out": last_out, "work_seconds": work_seconds, "work_ot_seconds": work_ot_seconds, "work_hours": work_hours, "work_status": work_status, }) # Done here: return flattened_work_reports # --------------------------------------------------------------------------------------------------------------------- async def sync_attendance_to_tcaoff( tcaoff_client: AsyncTheCAOffice, cosec_in_out_summary: dict, verbose: bool = False ) -> Dict[str, int]: # # Get the list of existing team members from TCAOFF: # # NOTE: `pseudonym` is the unique username of the user. # tcaoff_teams = await tcaoff_client.team_list() # # print("TCAOFF TEAMS:", json.to_string(tcaoff_teams)) # # # Get a mapping from Cosec id to TCAOFF record: # cosec_id_to_tcaoff_team = {} # for t in tcaoff_teams: # app_notes = json.from_string(t["json_notes"]).get("applicantNotes", {}) # if isinstance(app_notes, str): app_notes = json.from_string(app_notes) # if not app_notes: continue # cosec_notes = app_notes.get("cosec", {}) # if not cosec_notes: continue # print("Cosec Notes:", cosec_notes) # cosec_id_to_tcaoff_team[cosec_notes.get("User ID") or cosec_notes.get("UserID")] = t # # print("COSEC to TCAOFF TEAMS:", json.to_string(cosec_id_to_tcaoff_team)) # Convert the data to a DataFrame: in_out_df = pd.DataFrame(cosec_in_out_summary["report"]) # in_out_df["Punch Time"] = pd.to_datetime(in_out_df["Punch Time"], unit = "s", utc = True).dt.tz_convert("Asia/Kolkata") print(in_out_df) in_out_df.info() # Compute the work done: work_reports = compute_work_done(in_out_df) print("WORK REPORTS:", work_reports) print("WORK REPORTS:", json.to_string(work_reports)) # # Convert the dt column to actual dt objects and apply the timezone on them, # # then enlist the unique dates: # in_out_df["Punch Time"] = pd.to_datetime(in_out_df["Punch Time"], unit = "s", utc = True).dt.tz_convert("Asia/Kolkata") # unique_dates = sorted(in_out_df['Punch Time'].dt.date.unique()) # print(f"UNIQUE DATES ({len(unique_dates)}):", unique_dates) # # # Get the unique user ids: # unique_cosec_user_ids = in_out_df["User ID"].unique().tolist() # print(f"UNIQUE USER IDS ({len(unique_cosec_user_ids)}):", unique_cosec_user_ids) # # # We will create all async. tasks for firing attendance marking: # tasks = [] # # # For every user: # for user_count, cosec_user_id in enumerate(unique_cosec_user_ids): # # # Debugging: # now_time = date_time.get_current_date_time(as_string = True) # printer("Cosec User:", cosec_user_id, user_count, now_time) # # # Find the equivalent TCAOFF team member record: # tcaoff_team = cosec_id_to_tcaoff_team.get(cosec_user_id) # if tcaoff_team is None: # print(f"TCAOFF SYNC ERR: Cosesc User Id '{cosec_user_id}' not found in TCAOFF") # continue # # # For every date: # for work_dt in unique_dates: # # # Fetch only the successful events: # user_allowed_events = in_out_df[ # (in_out_df["User ID"] == cosec_user_id) & # (in_out_df["Event Status"] == "Allowed") & # (in_out_df["Punch Time"].dt.date == work_dt) # ] # # # Debugging: # weekday = work_dt.weekday() + 1 # ... 1 = Monday, 7 = Sunday # events = len(user_allowed_events) # if verbose: printer(cosec_user_id, work_dt, weekday, events) # # # Check if the summary is empty: # if user_allowed_events.empty: # if verbose: printer("Nothing to sync.", cosec_user_id, work_dt, weekday, events) # continue # # # We note down the first "In" time of the user # # and the last "Out" time of the user: # first_in = None # last_out = None # for idx, row in user_allowed_events.iterrows(): # # print(f"{row['I/O Type']: >4} at row: {idx: <5} | ts: {row['Punch Time']: <15} | dt: {date_time.parse_date_time(row['Punch Time'])}") # if row["I/O Type"] == "In" and first_in is None: first_in = row["Punch Time"].timestamp() # if row["I/O Type"] == "Out" and first_in is not None: last_out = row["Punch Time"].timestamp() # # # Figure out the worked time: # if first_in is None and last_out is None: # time_worked = { # "work_seconds": 0.0, # "work_date": work_dt # } # elif first_in is None or last_out is None: # time_worked = { # "work_seconds": 60 * 60 * 10.0, # ... 10 hours represented in seconds. # "work_date": work_dt # } # else: # time_worked = { # "work_seconds": last_out - first_in, # "work_date": work_dt # } # # # Figure out the number of hours worked: # hours_worked = time_worked["work_seconds"] / (60.0 * 60.0) # if hours_worked > 10.0: status = "OT" # elif 7.5 < hours_worked <= 10.0: status = "P" # elif 2.5 < hours_worked <= 5.0: status = "H" # else: status = "A" # # # Create this TCAOFF task: # task = tcaoff_client.attendance_mark( # user_id = tcaoff_team["user_id"], # status = status, # over_time = max(0.0, hours_worked - 10.0), # attendance_date = time_worked["work_date"].strftime("%Y-%m-%d"), # json_notes = { # "totHours": hours_worked, # "firstIn": first_in, # "lastOut": last_out, # } # ) # tasks.append(task) # # # break # # print("TASK COUNT:", len(tasks)) # # # Now we fire all the tasks: # now_time = date_time.get_current_date_time(as_string = True) # printer("Marking Attendance", len(tasks), now_time) # results = await asyncio.gather(*tasks) # now_time = date_time.get_current_date_time(as_string = True) # printer("Marked Attendance", len(tasks), now_time, results) # ***************************************************************************************************************** # ***** **** # *** MAIN PROGRAM *** # ***** **** # ***************************************************************************************************************** if __name__ == "__main__": pass