Files
arrr-erp/app/modules/accounting/master_integrity_service.py
T
2026-08-26 22:12:29 +05:30

310 lines
13 KiB
Python

from __future__ import annotations
import json
import re
from collections import defaultdict
from difflib import SequenceMatcher
from typing import Any
from sqlalchemy import select
from app.modules.accounting.chart_models import AccountingChartGroup, AccountingChartLedger
from app.modules.accounting.chart_service import ROLE_LABELS, effective_role
from app.modules.accounting.stock_models import AccountingStockGroup, AccountingStockItem, AccountingStockUnit
from app.modules.accounting.stock_service import normalize_unit
def _s(value: Any) -> str:
return str(value or "").strip()
def _key(value: Any) -> str:
return re.sub(r"[^a-z0-9]+", " ", _s(value).casefold()).strip()
def _compact(value: Any) -> str:
return re.sub(r"[^a-z0-9]+", "", _s(value).casefold())
def _payload(value: str) -> dict:
try:
data = json.loads(value or "{}")
return data if isinstance(data, dict) else {}
except Exception:
return {}
ROLE_EXPECTED_ROOTS = {
"CUSTOMER": {"sundry debtors"},
"SUPPLIER": {"sundry creditors"},
"SALES": {"sales accounts"},
"PURCHASE": {"purchase accounts"},
"DIRECT_INCOME": {"direct incomes"},
"OTHER_INCOME": {"indirect incomes"},
"DIRECT_EXPENSE": {"direct expenses"},
"INDIRECT_EXPENSE": {"indirect expenses"},
"BANK": {"bank accounts"},
"CASH": {"cash in hand"},
"FIXED_ASSET": {"fixed assets"},
"CURRENT_ASSET": {"current assets"},
"CURRENT_LIABILITY": {"current liabilities"},
"LOAN": {"loans liability", "secured loans", "unsecured loans", "bank od a c"},
"CAPITAL": {"capital account"},
"RESERVE": {"reserves surplus"},
"INVENTORY": {"stock in hand"},
"INVESTMENT": {"investments"},
"GST_INPUT": {"duties taxes"},
"GST_OUTPUT": {"duties taxes"},
"TAX_OTHER": {"duties taxes"},
"INTER_BRANCH": {"branch divisions"},
"SUSPENSE": {"suspense a c"},
}
def _issue(kind: str, severity: str, master_type: str, name: str, current: str, suggested: str,
reason: str, *, master_id: int | None = None, related: str = "") -> dict:
return {
"kind": kind,
"severity": severity,
"master_type": master_type,
"master_id": master_id,
"name": name,
"current": current,
"suggested": suggested,
"reason": reason,
"related": related,
}
def _duplicate_pairs(rows, name_getter, threshold: float = 0.94):
buckets: dict[str, list] = defaultdict(list)
for row in rows:
key = _compact(name_getter(row))
if key:
buckets[key[:1]].append(row)
pairs = []
seen = set()
for bucket in buckets.values():
for i, left in enumerate(bucket):
lk = _compact(name_getter(left))
if len(lk) < 4:
continue
for right in bucket[i + 1:]:
rk = _compact(name_getter(right))
if len(rk) < 4:
continue
pair_key = tuple(sorted((int(left.id), int(right.id))))
if pair_key in seen:
continue
seen.add(pair_key)
if lk == rk:
score = 1.0
elif abs(len(lk) - len(rk)) > max(4, int(max(len(lk), len(rk)) * 0.20)):
continue
else:
score = SequenceMatcher(None, lk, rk).ratio()
if score >= threshold:
pairs.append((left, right, score))
return pairs
def build_master_integrity_report(db, *, tenant_id: int, client_id: int, tally_guid: str) -> dict:
filters = (
lambda model: (
model.tenant_id == int(tenant_id),
model.client_id == int(client_id),
model.tally_guid == _s(tally_guid),
)
)
groups = list(db.execute(select(AccountingChartGroup).where(*filters(AccountingChartGroup))).scalars().all())
ledgers = list(db.execute(select(AccountingChartLedger).where(*filters(AccountingChartLedger))).scalars().all())
stock_groups = list(db.execute(select(AccountingStockGroup).where(*filters(AccountingStockGroup))).scalars().all())
stock_items = list(db.execute(select(AccountingStockItem).where(*filters(AccountingStockItem))).scalars().all())
units = list(db.execute(select(AccountingStockUnit).where(*filters(AccountingStockUnit))).scalars().all())
issues: list[dict] = []
group_names = {_key(row.name) for row in groups}
stock_group_names = {_key(row.name) for row in stock_groups}
unit_names = {_key(row.name) for row in units}
# Ledger parent integrity + accounting-role/root-group consistency.
for row in ledgers:
parent_key = _key(row.parent_group_name)
root_key = _key(row.root_group_name)
if row.parent_group_name and parent_key not in group_names:
issues.append(_issue(
"MISSING_LEDGER_PARENT", "high", "Ledger", row.name,
row.parent_group_name, "Review / sync Tally group",
"Ledger refers to a parent group that is not present in the synced Chart of Accounts.",
master_id=row.id,
))
role = effective_role(row)
expected = ROLE_EXPECTED_ROOTS.get(role, set())
if expected and root_key and root_key not in expected:
issues.append(_issue(
"LEDGER_GROUP_MISMATCH", "high", "Ledger", row.name,
row.root_group_name or row.parent_group_name,
" / ".join(sorted(expected)),
f"Effective role is {ROLE_LABELS.get(role, role)}, but the ledger is under a different Tally root group.",
master_id=row.id,
))
# Party master integrity.
gstin = _compact(row.party_gstin).upper()
if gstin and len(gstin) != 15:
issues.append(_issue(
"PARTY_GSTIN_LENGTH", "high", "Ledger", row.name,
row.party_gstin, "15-character GSTIN",
"Party GSTIN is present but is not 15 characters.",
master_id=row.id,
))
if role in {"CUSTOMER", "SUPPLIER"} and _s(row.gst_registration_type) and not gstin:
reg = _key(row.gst_registration_type)
if any(x in reg for x in ("regular", "composition", "consumer", "registered")) and "unregistered" not in reg:
issues.append(_issue(
"REGISTERED_PARTY_WITHOUT_GSTIN", "medium", "Ledger", row.name,
row.gst_registration_type, "Verify GSTIN / registration type",
"Party appears to carry a GST registration classification but no GSTIN is stored in the synced master.",
master_id=row.id,
))
# Tax ledger sanity.
if role in {"GST_INPUT", "GST_OUTPUT", "TAX_OTHER"}:
tax_text = " ".join((_key(row.tax_type), _key(row.gst_applicable), _key(row.name)))
if "gst" not in tax_text and role in {"GST_INPUT", "GST_OUTPUT"}:
issues.append(_issue(
"GST_LEDGER_METADATA", "medium", "Ledger", row.name,
row.tax_type or row.gst_applicable or "Blank",
"Verify GST tax metadata",
"Ledger is classified as GST input/output but synced Tally GST/tax metadata does not clearly identify GST.",
master_id=row.id,
))
# Exact/near duplicate ledgers.
for left, right, score in _duplicate_pairs(ledgers, lambda x: x.name):
issues.append(_issue(
"POSSIBLE_DUPLICATE_LEDGER", "medium" if score < 0.995 else "high",
"Ledger", left.name, left.parent_group_name,
"Review duplicate / naming",
f"Name is {score * 100:.0f}% similar to another ledger.",
master_id=left.id, related=right.name,
))
# Stock master integrity.
for row in stock_items:
if row.parent_group_name and _key(row.parent_group_name) not in stock_group_names:
issues.append(_issue(
"MISSING_STOCK_GROUP", "high", "Stock Item", row.name,
row.parent_group_name, "Review / sync stock group",
"Stock item refers to a parent Stock Group that is not present in the synced master snapshot.",
master_id=row.id,
))
if row.base_units and _key(row.base_units) not in unit_names:
issues.append(_issue(
"MISSING_STOCK_UNIT", "high", "Stock Item", row.name,
row.base_units, "Create / correct Unit",
"Stock item's base unit is not present in the synced Tally Unit master.",
master_id=row.id,
))
if row.additional_units and _key(row.additional_units) not in unit_names:
issues.append(_issue(
"MISSING_ADDITIONAL_UNIT", "medium", "Stock Item", row.name,
row.additional_units, "Create / correct Unit",
"Stock item's additional unit is not present in the synced Tally Unit master.",
master_id=row.id,
))
gst_app = _key(row.gst_applicable)
if gst_app and gst_app not in {"not applicable", "none", "no"} and not _s(row.hsn_code):
issues.append(_issue(
"GST_STOCK_WITHOUT_HSN", "medium", "Stock Item", row.name,
row.gst_applicable, "Review HSN/SAC",
"GST is applicable to the stock item but no HSN/SAC code is present in the synced master.",
master_id=row.id,
))
for left, right, score in _duplicate_pairs(stock_items, lambda x: x.name):
issues.append(_issue(
"POSSIBLE_DUPLICATE_STOCK_ITEM", "medium" if score < 0.995 else "high",
"Stock Item", left.name, left.parent_group_name,
"Review duplicate / naming",
f"Name is {score * 100:.0f}% similar to another stock item.",
master_id=left.id, related=right.name,
))
# Unit aliases which are operationally equivalent but separately maintained.
normalized_units: dict[str, list] = defaultdict(list)
for row in units:
normalized_units[normalize_unit(row.name)].append(row)
for norm, rows in normalized_units.items():
if norm and len(rows) > 1:
names = sorted({_s(r.name) for r in rows}, key=str.casefold)
if len(names) > 1:
issues.append(_issue(
"DUPLICATE_UNIT_ALIAS", "medium", "Unit",
names[0], ", ".join(names), norm,
"Multiple Tally units normalize to the same accounting unit and may cause stock-item mapping inconsistencies.",
related=", ".join(names[1:]),
))
severity_rank = {"high": 0, "medium": 1, "low": 2}
issues.sort(key=lambda x: (severity_rank.get(x["severity"], 9), x["master_type"], x["name"].casefold(), x["kind"]))
summary = {
"groups": len(groups),
"ledgers": len(ledgers),
"stock_groups": len(stock_groups),
"stock_items": len(stock_items),
"units": len(units),
"issues": len(issues),
"high": sum(1 for x in issues if x["severity"] == "high"),
"medium": sum(1 for x in issues if x["severity"] == "medium"),
"low": sum(1 for x in issues if x["severity"] == "low"),
}
issue_types = sorted({x["kind"] for x in issues})
master_types = sorted({x["master_type"] for x in issues})
return {
"summary": summary,
"issues": issues,
"issue_types": issue_types,
"master_types": master_types,
}
def filter_integrity_issues(report: dict, *, q: str = "", severity: str = "", master_type: str = "",
issue_type: str = "", page: int = 1, per_page: int = 50) -> dict:
rows = list(report.get("issues") or [])
qk = _key(q)
if qk:
rows = [
row for row in rows
if qk in _key(" ".join([
row.get("name", ""), row.get("current", ""), row.get("suggested", ""),
row.get("reason", ""), row.get("related", ""), row.get("kind", ""),
]))
]
if severity:
rows = [row for row in rows if row.get("severity") == severity]
if master_type:
rows = [row for row in rows if row.get("master_type") == master_type]
if issue_type:
rows = [row for row in rows if row.get("kind") == issue_type]
per_page = max(10, min(int(per_page or 50), 200))
total = len(rows)
pages = max(1, (total + per_page - 1) // per_page)
page = max(1, min(int(page or 1), pages))
start = (page - 1) * per_page
return {
"rows": rows[start:start + per_page],
"total": total,
"page": page,
"pages": pages,
"per_page": per_page,
}