"""Reproducible vTiger-versus-Hub impact metrics for Project CT.""" from __future__ import annotations import hashlib import json from collections import defaultdict from datetime import date, datetime, timedelta from decimal import Decimal, InvalidOperation from typing import Any from app.core.database import execute_insert, execute_query, execute_query_single from app.services.vtiger_service import VTigerService DATE_FIELDS = ("worked_date", "date", "createdtime", "created_at", "modifiedtime") EMAIL_FIELDS = ("email1", "email", "email_address", "user_email") USER_REF_FIELDS = ("assigned_user_id", "creator", "created_by", "userid", "user_id") def _date(value: Any) -> date | None: if isinstance(value, datetime): return value.date() if isinstance(value, date): return value try: return datetime.fromisoformat(str(value or "").strip().replace("Z", "+00:00")).date() except (ValueError, TypeError): return None def _first(payload: dict, fields: tuple[str, ...]): return next((payload.get(field) for field in fields if payload.get(field) not in (None, "")), None) def _hours(payload: dict) -> Decimal: return VTigerService._extract_timelog_hours(payload) def _archive_snapshot(version_id: int) -> list[dict[str, Any]]: return execute_query( """SELECT DISTINCT ON (module,vtiger_id) module,vtiger_id,payload,is_deleted FROM vtiger_archive_records WHERE version_id <= %s ORDER BY module,vtiger_id,revision_no DESC""", (version_id,), ) or [] def _hub_users() -> tuple[dict[int, dict], dict[str, dict]]: rows = execute_query("SELECT user_id,email,full_name,username FROM users") or [] by_id = {int(row["user_id"]): dict(row) for row in rows} by_email = {str(row.get("email") or "").strip().lower(): dict(row) for row in rows if row.get("email")} return by_id, by_email def _employee_bucket(store: dict, key: str, name: str, email: str | None): if key not in store: store[key] = {"key": key, "name": name, "email": email, "hours": Decimal("0"), "time_entries": 0, "cases": 0, "orders": 0} return store[key] def _finalize(metrics: dict[str, Any], start: date, end: date) -> dict[str, Any]: days = (end - start).days + 1 workdays = max(sum(1 for offset in range(days) if (start + timedelta(days=offset)).weekday() < 5), 1) employees = [] for item in metrics["employees"].values(): normalized = dict(item) normalized["hours"] = round(float(item["hours"]), 2) normalized["productivity"] = { "hours_per_workday": round(normalized["hours"] / workdays, 2), "registrations_per_workday": round(normalized["time_entries"] / workdays, 2), "cases_per_workday": round(normalized["cases"] / workdays, 2), "orders_per_workday": round(normalized["orders"] / workdays, 2), "deliveries_per_workday": round((normalized["time_entries"] + normalized["cases"] + normalized["orders"]) / workdays, 2), } employees.append(normalized) employees.sort(key=lambda row: (row.get("name") or "").lower()) totals = {key: sum(float(row[key]) for row in employees) for key in ("hours", "time_entries", "cases", "orders")} totals["hours"] = round(totals["hours"], 2) totals["time_entries"] = int(totals["time_entries"]) totals["cases"] = int(totals["cases"]) totals["orders"] = int(totals["orders"]) totals["productivity"] = { "hours_per_workday": round(totals["hours"] / workdays, 2), "registrations_per_workday": round(totals["time_entries"] / workdays, 2), "cases_per_workday": round(totals["cases"] / workdays, 2), "orders_per_workday": round(totals["orders"] / workdays, 2), "deliveries_per_workday": round((totals["time_entries"] + totals["cases"] + totals["orders"]) / workdays, 2), "hours_per_case_or_order": round(totals["hours"] / max(totals["cases"] + totals["orders"], 1), 2), } return {"totals": totals, "employees": employees, "anomalies": metrics["anomalies"], "data_quality": metrics["data_quality"], "workdays": workdays} def _vtiger_metrics(records: list[dict], start: date, end: date, hub_email_users: dict[str, dict]) -> dict: active = [row for row in records if not row.get("is_deleted")] vt_users = {} for row in active: if row["module"] != "Users": continue payload = row["payload"] or {} email = str(_first(payload, EMAIL_FIELDS) or "").strip().lower() vt_users[row["vtiger_id"]] = { "email": email or None, "name": str(payload.get("first_name") or "") + " " + str(payload.get("last_name") or payload.get("user_name") or row["vtiger_id"]), } result = {"employees": {}, "anomalies": [], "data_quality": {"unmatched_employees": [], "missing_dates": 0}} module_metric = {"Timelog": "time_entries", "Cases": "cases", "SalesOrder": "orders"} for row in active: metric = module_metric.get(row["module"]) if not metric: continue payload = row["payload"] or {} occurred = _date(_first(payload, DATE_FIELDS)) if not occurred: result["data_quality"]["missing_dates"] += 1 continue if occurred < start or occurred > end: continue user_ref = str(_first(payload, USER_REF_FIELDS) or "") vt_user = vt_users.get(user_ref, {"email": None, "name": user_ref or "Ukendt vTiger-bruger"}) email = vt_user.get("email") matched = hub_email_users.get(email) if email else None key = f"hub:{matched['user_id']}" if matched else f"vtiger:{user_ref or 'unknown'}" name = str(matched.get("full_name") or matched.get("username")) if matched else str(vt_user.get("name") or key).strip() bucket = _employee_bucket(result["employees"], key, name, email) if not matched and key not in result["data_quality"]["unmatched_employees"]: result["data_quality"]["unmatched_employees"].append(key) if metric == "time_entries": hours = _hours(payload) if hours > 16: result["anomalies"].append({"source": "vtiger", "type": "over_16_hours", "id": row["vtiger_id"], "hours": float(hours)}) continue bucket["hours"] += hours bucket[metric] += 1 return result def _hub_metrics(start: date, end: date, hub_users: dict[int, dict]) -> dict: result = {"employees": {}, "anomalies": [], "data_quality": {"unmatched_employees": [], "missing_dates": 0}} times = execute_query( """SELECT id,medarbejder_id,worked_date,start_tid,created_at,aktiv_timer, COALESCE(faktisk_tid_min::numeric/60,approved_hours,original_hours,0) AS hours FROM tmodule_times WHERE vtiger_id IS NULL AND COALESCE(worked_date,start_tid::date,created_at::date) BETWEEN %s AND %s""", (start, end), ) or [] for row in times: user = hub_users.get(int(row["medarbejder_id"])) if row.get("medarbejder_id") else None key = f"hub:{user['user_id']}" if user else "hub:unknown" bucket = _employee_bucket(result["employees"], key, str((user or {}).get("full_name") or (user or {}).get("username") or "Ukendt Hub-bruger"), (user or {}).get("email")) if not user and key not in result["data_quality"]["unmatched_employees"]: result["data_quality"]["unmatched_employees"].append(key) hours = Decimal(str(row.get("hours") or 0)) if row.get("aktiv_timer") or hours > 16: result["anomalies"].append({"source": "hub", "type": "active_timer" if row.get("aktiv_timer") else "over_16_hours", "id": row["id"], "hours": float(hours)}) continue bucket["hours"] += hours bucket["time_entries"] += 1 cases = execute_query( """SELECT id,created_by_user_id,created_at FROM sag_sager WHERE deleted_at IS NULL AND created_at::date BETWEEN %s AND %s""", (start, end), ) or [] orders = execute_query( "SELECT id,created_by,order_date FROM tmodule_orders WHERE order_date BETWEEN %s AND %s", (start, end), ) or [] for rows, metric, user_field in ((cases, "cases", "created_by_user_id"), (orders, "orders", "created_by")): for row in rows: user = hub_users.get(int(row[user_field])) if row.get(user_field) else None key = f"hub:{user['user_id']}" if user else "hub:unknown" bucket = _employee_bucket(result["employees"], key, str((user or {}).get("full_name") or (user or {}).get("username") or "Ukendt Hub-bruger"), (user or {}).get("email")) if not user and key not in result["data_quality"]["unmatched_employees"]: result["data_quality"]["unmatched_employees"].append(key) bucket[metric] += 1 return result def build_impact_report(version_id: int, vtiger_from: date, vtiger_to: date, hub_from: date, hub_to: date) -> dict: vt_days = (vtiger_to - vtiger_from).days + 1 hub_days = (hub_to - hub_from).days + 1 if vt_days <= 0 or hub_days <= 0 or vt_days != hub_days: raise ValueError("vTiger- og Hub-perioderne skal være gyldige og lige lange") version = execute_query_single( "SELECT id,status,source_cutoff FROM vtiger_archive_versions WHERE id=%s", (version_id,), ) if not version or version.get("status") != "completed": raise ValueError("Vælg en fuldført arkivversion") by_id, by_email = _hub_users() vtiger = _finalize(_vtiger_metrics(_archive_snapshot(version_id), vtiger_from, vtiger_to, by_email), vtiger_from, vtiger_to) hub = _finalize(_hub_metrics(hub_from, hub_to, by_id), hub_from, hub_to) keys = ("hours", "time_entries", "cases", "orders") delta = {key: round(float(hub["totals"][key]) - float(vtiger["totals"][key]), 2) for key in keys} percent = { key: (round(delta[key] / float(vtiger["totals"][key]) * 100, 1) if float(vtiger["totals"][key]) else None) for key in keys } productivity_keys = ("hours_per_workday", "registrations_per_workday", "cases_per_workday", "orders_per_workday", "deliveries_per_workday", "hours_per_case_or_order") productivity_delta = {} for key in productivity_keys: old, new = float(vtiger["totals"]["productivity"][key]), float(hub["totals"]["productivity"][key]) productivity_delta[key] = {"delta": round(new - old, 2), "percent": round((new - old) / old * 100, 1) if old else None} evidence = [] for key, label in (("deliveries_per_workday", "samlet registreret output pr. arbejdsdag"), ("cases_per_workday", "sager pr. arbejdsdag"), ("orders_per_workday", "ordrer pr. arbejdsdag"), ("registrations_per_workday", "tidsregistreringer pr. arbejdsdag")): change = productivity_delta[key] if change["percent"] is not None: direction = "mere" if change["percent"] >= 0 else "mindre" evidence.append(f"Hub registrerer {abs(change['percent']):.1f}% {direction} {label}.") return { "archive_version": {"id": version_id, "source_cutoff": version.get("source_cutoff")}, "periods": {"vtiger": {"from": vtiger_from, "to": vtiger_to, "days": vt_days}, "hub": {"from": hub_from, "to": hub_to, "days": hub_days}}, "vtiger": vtiger, "hub": hub, "comparison": delta, "improvement": {"percent": percent, "productivity": productivity_delta, "evidence": evidence}, } def save_impact_report(result: dict, generated_by: int) -> int: canonical = json.dumps(result, ensure_ascii=False, sort_keys=True, separators=(",", ":"), default=str) periods = result["periods"] return int(execute_insert( """INSERT INTO hub_impact_reports (archive_version_id,vtiger_from,vtiger_to,hub_from,hub_to,result,result_sha256,generated_by) VALUES (%s,%s,%s,%s,%s,%s::jsonb,%s,%s) RETURNING id""", (result["archive_version"]["id"], periods["vtiger"]["from"], periods["vtiger"]["to"], periods["hub"]["from"], periods["hub"]["to"], canonical, hashlib.sha256(canonical.encode()).hexdigest(), generated_by), ))