from __future__ import annotations import json import re from collections import defaultdict from difflib import SequenceMatcher from typing import Any from sqlalchemy import select from app.modules.accounting.chart_models import AccountingChartGroup, AccountingChartLedger from app.modules.accounting.chart_service import ROLE_LABELS, effective_role from app.modules.accounting.stock_models import AccountingStockGroup, AccountingStockItem, AccountingStockUnit from app.modules.accounting.stock_service import normalize_unit def _s(value: Any) -> str: return str(value or "").strip() def _key(value: Any) -> str: return re.sub(r"[^a-z0-9]+", " ", _s(value).casefold()).strip() def _compact(value: Any) -> str: return re.sub(r"[^a-z0-9]+", "", _s(value).casefold()) def _payload(value: str) -> dict: try: data = json.loads(value or "{}") return data if isinstance(data, dict) else {} except Exception: return {} ROLE_EXPECTED_ROOTS = { "CUSTOMER": {"sundry debtors"}, "SUPPLIER": {"sundry creditors"}, "SALES": {"sales accounts"}, "PURCHASE": {"purchase accounts"}, "DIRECT_INCOME": {"direct incomes"}, "OTHER_INCOME": {"indirect incomes"}, "DIRECT_EXPENSE": {"direct expenses"}, "INDIRECT_EXPENSE": {"indirect expenses"}, "BANK": {"bank accounts"}, "CASH": {"cash in hand"}, "FIXED_ASSET": {"fixed assets"}, "CURRENT_ASSET": {"current assets"}, "CURRENT_LIABILITY": {"current liabilities"}, "LOAN": {"loans liability", "secured loans", "unsecured loans", "bank od a c"}, "CAPITAL": {"capital account"}, "RESERVE": {"reserves surplus"}, "INVENTORY": {"stock in hand"}, "INVESTMENT": {"investments"}, "GST_INPUT": {"duties taxes"}, "GST_OUTPUT": {"duties taxes"}, "TAX_OTHER": {"duties taxes"}, "INTER_BRANCH": {"branch divisions"}, "SUSPENSE": {"suspense a c"}, } def _issue(kind: str, severity: str, master_type: str, name: str, current: str, suggested: str, reason: str, *, master_id: int | None = None, related: str = "") -> dict: return { "kind": kind, "severity": severity, "master_type": master_type, "master_id": master_id, "name": name, "current": current, "suggested": suggested, "reason": reason, "related": related, } def _duplicate_pairs(rows, name_getter, threshold: float = 0.94): buckets: dict[str, list] = defaultdict(list) for row in rows: key = _compact(name_getter(row)) if key: buckets[key[:1]].append(row) pairs = [] seen = set() for bucket in buckets.values(): for i, left in enumerate(bucket): lk = _compact(name_getter(left)) if len(lk) < 4: continue for right in bucket[i + 1:]: rk = _compact(name_getter(right)) if len(rk) < 4: continue pair_key = tuple(sorted((int(left.id), int(right.id)))) if pair_key in seen: continue seen.add(pair_key) if lk == rk: score = 1.0 elif abs(len(lk) - len(rk)) > max(4, int(max(len(lk), len(rk)) * 0.20)): continue else: score = SequenceMatcher(None, lk, rk).ratio() if score >= threshold: pairs.append((left, right, score)) return pairs def build_master_integrity_report(db, *, tenant_id: int, client_id: int, tally_guid: str) -> dict: filters = ( lambda model: ( model.tenant_id == int(tenant_id), model.client_id == int(client_id), model.tally_guid == _s(tally_guid), ) ) groups = list(db.execute(select(AccountingChartGroup).where(*filters(AccountingChartGroup))).scalars().all()) ledgers = list(db.execute(select(AccountingChartLedger).where(*filters(AccountingChartLedger))).scalars().all()) stock_groups = list(db.execute(select(AccountingStockGroup).where(*filters(AccountingStockGroup))).scalars().all()) stock_items = list(db.execute(select(AccountingStockItem).where(*filters(AccountingStockItem))).scalars().all()) units = list(db.execute(select(AccountingStockUnit).where(*filters(AccountingStockUnit))).scalars().all()) issues: list[dict] = [] group_names = {_key(row.name) for row in groups} stock_group_names = {_key(row.name) for row in stock_groups} unit_names = {_key(row.name) for row in units} # Ledger parent integrity + accounting-role/root-group consistency. for row in ledgers: parent_key = _key(row.parent_group_name) root_key = _key(row.root_group_name) if row.parent_group_name and parent_key not in group_names: issues.append(_issue( "MISSING_LEDGER_PARENT", "high", "Ledger", row.name, row.parent_group_name, "Review / sync Tally group", "Ledger refers to a parent group that is not present in the synced Chart of Accounts.", master_id=row.id, )) role = effective_role(row) expected = ROLE_EXPECTED_ROOTS.get(role, set()) if expected and root_key and root_key not in expected: issues.append(_issue( "LEDGER_GROUP_MISMATCH", "high", "Ledger", row.name, row.root_group_name or row.parent_group_name, " / ".join(sorted(expected)), f"Effective role is {ROLE_LABELS.get(role, role)}, but the ledger is under a different Tally root group.", master_id=row.id, )) # Party master integrity. gstin = _compact(row.party_gstin).upper() if gstin and len(gstin) != 15: issues.append(_issue( "PARTY_GSTIN_LENGTH", "high", "Ledger", row.name, row.party_gstin, "15-character GSTIN", "Party GSTIN is present but is not 15 characters.", master_id=row.id, )) if role in {"CUSTOMER", "SUPPLIER"} and _s(row.gst_registration_type) and not gstin: reg = _key(row.gst_registration_type) if any(x in reg for x in ("regular", "composition", "consumer", "registered")) and "unregistered" not in reg: issues.append(_issue( "REGISTERED_PARTY_WITHOUT_GSTIN", "medium", "Ledger", row.name, row.gst_registration_type, "Verify GSTIN / registration type", "Party appears to carry a GST registration classification but no GSTIN is stored in the synced master.", master_id=row.id, )) # Tax ledger sanity. if role in {"GST_INPUT", "GST_OUTPUT", "TAX_OTHER"}: tax_text = " ".join((_key(row.tax_type), _key(row.gst_applicable), _key(row.name))) if "gst" not in tax_text and role in {"GST_INPUT", "GST_OUTPUT"}: issues.append(_issue( "GST_LEDGER_METADATA", "medium", "Ledger", row.name, row.tax_type or row.gst_applicable or "Blank", "Verify GST tax metadata", "Ledger is classified as GST input/output but synced Tally GST/tax metadata does not clearly identify GST.", master_id=row.id, )) # Exact/near duplicate ledgers. for left, right, score in _duplicate_pairs(ledgers, lambda x: x.name): issues.append(_issue( "POSSIBLE_DUPLICATE_LEDGER", "medium" if score < 0.995 else "high", "Ledger", left.name, left.parent_group_name, "Review duplicate / naming", f"Name is {score * 100:.0f}% similar to another ledger.", master_id=left.id, related=right.name, )) # Stock master integrity. for row in stock_items: if row.parent_group_name and _key(row.parent_group_name) not in stock_group_names: issues.append(_issue( "MISSING_STOCK_GROUP", "high", "Stock Item", row.name, row.parent_group_name, "Review / sync stock group", "Stock item refers to a parent Stock Group that is not present in the synced master snapshot.", master_id=row.id, )) if row.base_units and _key(row.base_units) not in unit_names: issues.append(_issue( "MISSING_STOCK_UNIT", "high", "Stock Item", row.name, row.base_units, "Create / correct Unit", "Stock item's base unit is not present in the synced Tally Unit master.", master_id=row.id, )) if row.additional_units and _key(row.additional_units) not in unit_names: issues.append(_issue( "MISSING_ADDITIONAL_UNIT", "medium", "Stock Item", row.name, row.additional_units, "Create / correct Unit", "Stock item's additional unit is not present in the synced Tally Unit master.", master_id=row.id, )) gst_app = _key(row.gst_applicable) if gst_app and gst_app not in {"not applicable", "none", "no"} and not _s(row.hsn_code): issues.append(_issue( "GST_STOCK_WITHOUT_HSN", "medium", "Stock Item", row.name, row.gst_applicable, "Review HSN/SAC", "GST is applicable to the stock item but no HSN/SAC code is present in the synced master.", master_id=row.id, )) for left, right, score in _duplicate_pairs(stock_items, lambda x: x.name): issues.append(_issue( "POSSIBLE_DUPLICATE_STOCK_ITEM", "medium" if score < 0.995 else "high", "Stock Item", left.name, left.parent_group_name, "Review duplicate / naming", f"Name is {score * 100:.0f}% similar to another stock item.", master_id=left.id, related=right.name, )) # Unit aliases which are operationally equivalent but separately maintained. normalized_units: dict[str, list] = defaultdict(list) for row in units: normalized_units[normalize_unit(row.name)].append(row) for norm, rows in normalized_units.items(): if norm and len(rows) > 1: names = sorted({_s(r.name) for r in rows}, key=str.casefold) if len(names) > 1: issues.append(_issue( "DUPLICATE_UNIT_ALIAS", "medium", "Unit", names[0], ", ".join(names), norm, "Multiple Tally units normalize to the same accounting unit and may cause stock-item mapping inconsistencies.", related=", ".join(names[1:]), )) severity_rank = {"high": 0, "medium": 1, "low": 2} issues.sort(key=lambda x: (severity_rank.get(x["severity"], 9), x["master_type"], x["name"].casefold(), x["kind"])) summary = { "groups": len(groups), "ledgers": len(ledgers), "stock_groups": len(stock_groups), "stock_items": len(stock_items), "units": len(units), "issues": len(issues), "high": sum(1 for x in issues if x["severity"] == "high"), "medium": sum(1 for x in issues if x["severity"] == "medium"), "low": sum(1 for x in issues if x["severity"] == "low"), } issue_types = sorted({x["kind"] for x in issues}) master_types = sorted({x["master_type"] for x in issues}) return { "summary": summary, "issues": issues, "issue_types": issue_types, "master_types": master_types, } def filter_integrity_issues(report: dict, *, q: str = "", severity: str = "", master_type: str = "", issue_type: str = "", page: int = 1, per_page: int = 50) -> dict: rows = list(report.get("issues") or []) qk = _key(q) if qk: rows = [ row for row in rows if qk in _key(" ".join([ row.get("name", ""), row.get("current", ""), row.get("suggested", ""), row.get("reason", ""), row.get("related", ""), row.get("kind", ""), ])) ] if severity: rows = [row for row in rows if row.get("severity") == severity] if master_type: rows = [row for row in rows if row.get("master_type") == master_type] if issue_type: rows = [row for row in rows if row.get("kind") == issue_type] per_page = max(10, min(int(per_page or 50), 200)) total = len(rows) pages = max(1, (total + per_page - 1) // per_page) page = max(1, min(int(page or 1), pages)) start = (page - 1) * per_page return { "rows": rows[start:start + per_page], "total": total, "page": page, "pages": pages, "per_page": per_page, }