from __future__ import annotations import json import math import re from collections import defaultdict from datetime import datetime, timezone from sqlalchemy import select from app.modules.accounting.historical_learning_models import ( AccountingHistoricalLedgerEvidence, AccountingLedgerNatureMapping, ) from app.modules.accounting.ledger_learning_models import ( AccountingLedgerLearningEvent, AccountingLedgerLearningRule, ) from app.modules.accounting.taxonomy_models import AccountingNature from app.modules.clients.models import ClientBusinessProfile TOKEN_RE = re.compile(r"[A-Z0-9]+") STOP_WORDS = { "THE", "AND", "FOR", "PVT", "PRIVATE", "LIMITED", "LTD", "LLP", "INDIA", "INVOICE", "BILL", "TAX", "GST", "GSTIN", "SERVICES", "SERVICE", "TRADERS", "TRADING", "ENTERPRISES", "ENTERPRISE", "COMPANY", "CO", } KEYWORD_PRIORS = { "TELEPHONE": {"AIRTEL", "VODAFONE", "IDEA", "JIO", "MOBILE", "TELEPHONE", "TELECOM"}, "INTERNET": {"BROADBAND", "INTERNET", "FIBER", "FIBRE", "LEASEDLINE"}, "ELECTRICITY": {"ELECTRICITY", "POWER", "TANGEDCO", "EB"}, "INSURANCE": {"INSURANCE", "PREMIUM"}, "VEHICLE_MAINTENANCE": {"TYRE", "TYRES", "TIRE", "BRAKE", "CLUTCH", "VEHICLE", "CAR", "MOTOR", "SERVICECENTER"}, "COMPUTER_MAINTENANCE": {"LAPTOPREPAIR", "COMPUTERREPAIR", "PRINTERREPAIR", "AMCIT"}, "BUILDING_MAINTENANCE": {"PLUMBING", "PLUMBER", "PAINTING", "CIVILREPAIR", "BUILDINGREPAIR"}, "MACHINERY_MAINTENANCE": {"MACHINERYREPAIR", "MACHINEPART", "BEARING", "INDUSTRIALREPAIR"}, "PROFESSIONAL_CHARGES": {"CONSULTANCY", "CONSULTANT", "PROFESSIONAL"}, "LEGAL_FEES": {"ADVOCATE", "LEGAL", "LAWYER"}, "AUDIT_FEES": {"AUDIT", "AUDITOR"}, "TRAVELLING": {"FLIGHT", "AIRLINE", "TRAVEL", "RAILWAY", "TRAIN", "TAXI", "CAB"}, "HOTEL_ACCOMMODATION": {"HOTEL", "RESORT", "ACCOMMODATION"}, "FREIGHT_CARRIAGE": {"FREIGHT", "TRANSPORT", "CARRIAGE", "LOGISTICS"}, "PRINTING_STATIONERY": {"STATIONERY", "PRINTING", "PAPER", "TONER", "CARTRIDGE"}, "SOFTWARE_SUBSCRIPTION": {"SOFTWARE", "SUBSCRIPTION", "SAAS", "LICENSE", "LICENCE"}, "CLOUD_HOSTING": {"HOSTING", "CLOUD", "DOMAIN", "AWS", "AZURE"}, "ADVERTISEMENT_MARKETING": {"ADVERTISEMENT", "ADVERTISING", "MARKETING", "PROMOTION"}, "COMMISSION_BROKERAGE": {"COMMISSION", "BROKERAGE"}, "COMPUTERS": {"LAPTOP", "DESKTOP", "COMPUTER", "SERVER"}, "FURNITURE_FIXTURES": {"FURNITURE", "CHAIR", "TABLE", "WORKSTATION"}, "MOTOR_VEHICLES": {"MOTORCAR", "MOTORVEHICLE", "NEWCAR", "NEWVEHICLE"}, "PLANT_MACHINERY": {"MACHINERY", "MACHINE", "EQUIPMENTPLANT"}, } def _utcnow(): return datetime.now(timezone.utc) def normalize_text(value: str | None) -> str: return " ".join(TOKEN_RE.findall(str(value or "").upper())).strip() def normalize_gstin(value: str | None) -> str: return re.sub(r"[^A-Z0-9]", "", str(value or "").upper())[:15] def normalize_hsn(value: str | None) -> str: return re.sub(r"\D", "", str(value or ""))[:8] def description_signature(value: str | None) -> str: tokens = [t for t in TOKEN_RE.findall(str(value or "").upper()) if len(t) >= 3 and t not in STOP_WORDS] return " ".join(sorted(dict.fromkeys(tokens))[:10]) def _confidence(confirmations: int, rejections: int) -> int: total = max(0, int(confirmations)) + max(0, int(rejections)) if total <= 0: return 0 # Bayesian smoothing prevents one confirmation from becoming 100% certainty. ratio = (confirmations + 2.0) / (total + 4.0) volume = min(1.0, math.log1p(total) / math.log(11)) return max(1, min(99, round((ratio * 75) + (volume * 24)))) def business_profile(db, client_id: int): return db.execute(select(ClientBusinessProfile).where(ClientBusinessProfile.client_id == client_id)).scalar_one_or_none() def active_natures(db, tenant_id: int): return list(db.execute( select(AccountingNature).where( AccountingNature.tenant_id == tenant_id, AccountingNature.is_active.is_(True), AccountingNature.is_posting_nature.is_(True), ).order_by(AccountingNature.sort_order, AccountingNature.name) ).scalars().all()) def _nature_maps(db, tenant_id: int): rows = active_natures(db, tenant_id) return rows, {r.id: r for r in rows}, {r.code: r for r in rows} def learned_rules(db, tenant_id: int, client_id: int, tally_guid: str = ""): stmt = select(AccountingLedgerLearningRule).where( AccountingLedgerLearningRule.tenant_id == tenant_id, AccountingLedgerLearningRule.client_id == client_id, AccountingLedgerLearningRule.is_active.is_(True), ) if tally_guid: stmt = stmt.where( (AccountingLedgerLearningRule.tally_guid == tally_guid) | (AccountingLedgerLearningRule.tally_guid == "") ) return list(db.execute(stmt).scalars().all()) def recent_events(db, tenant_id: int, client_id: int, limit: int = 30): return list(db.execute( select(AccountingLedgerLearningEvent).where( AccountingLedgerLearningEvent.tenant_id == tenant_id, AccountingLedgerLearningEvent.client_id == client_id, ).order_by(AccountingLedgerLearningEvent.id.desc()).limit(limit) ).scalars().all()) def available_tally_guids(db, tenant_id: int, client_id: int): vals = set() for row in db.execute(select( AccountingLedgerNatureMapping.tally_guid, AccountingLedgerNatureMapping.company_name, ).where( AccountingLedgerNatureMapping.tenant_id == tenant_id, AccountingLedgerNatureMapping.client_id == client_id, )).all(): if row[0]: vals.add((row[0], row[1] or "")) for row in db.execute(select( AccountingHistoricalLedgerEvidence.tally_guid, AccountingHistoricalLedgerEvidence.company_name, ).where( AccountingHistoricalLedgerEvidence.tenant_id == tenant_id, AccountingHistoricalLedgerEvidence.client_id == client_id, )).all(): if row[0]: vals.add((row[0], row[1] or "")) return sorted(vals, key=lambda x: ((x[1] or "").casefold(), x[0])) def _rule_contexts(*, supplier_name: str, supplier_gstin: str, hsn_code: str, description: str): contexts = [] gstin = normalize_gstin(supplier_gstin) supplier = normalize_text(supplier_name) hsn = normalize_hsn(hsn_code) desc = description_signature(description) if gstin: contexts.append(("supplier_gstin", gstin, supplier_gstin)) if supplier: contexts.append(("supplier_name", supplier, supplier_name)) if hsn: contexts.append(("hsn", hsn, hsn_code)) if desc: contexts.append(("description_signature", desc, description[:500])) if supplier and hsn: contexts.append(("supplier_hsn", f"{supplier}|{hsn}", f"{supplier_name} / HSN {hsn}")) if gstin and hsn: contexts.append(("gstin_hsn", f"{gstin}|{hsn}", f"{gstin} / HSN {hsn}")) return contexts def _add_score(scores, nature_id, points, reason): if not nature_id or points <= 0: return row = scores.setdefault(nature_id, {"score": 0.0, "reasons": []}) row["score"] += float(points) row["reasons"].append(reason) def rank_suggestions( db, *, tenant_id: int, client_id: int, tally_guid: str = "", supplier_name: str = "", supplier_gstin: str = "", hsn_code: str = "", description: str = "", amount: float | None = None, ): natures, nature_by_id, nature_by_code = _nature_maps(db, tenant_id) if not natures: return [] scores = {} contexts = _rule_contexts( supplier_name=supplier_name, supplier_gstin=supplier_gstin, hsn_code=hsn_code, description=description, ) context_lookup = {(t, k) for t, k, _ in contexts} # 1. User-confirmed client-specific learning. rule_weights = { "gstin_hsn": 72, "supplier_gstin": 68, "supplier_hsn": 62, "supplier_name": 52, "hsn": 34, "description_signature": 28, } for rule in learned_rules(db, tenant_id, client_id, tally_guid): if (rule.rule_type, rule.rule_key) not in context_lookup: continue strength = max(0.10, rule.confidence_percent / 100) points = rule_weights.get(rule.rule_type, 20) * strength _add_score( scores, rule.nature_id, points, f"Confirmed {rule.rule_type.replace('_', ' ')} rule: " f"{rule.confirmation_count} confirmation(s), {rule.rejection_count} correction(s), " f"{rule.confidence_percent}% learned confidence." ) # 2. Phase 5 historical supplier treatment. supplier_norm = normalize_text(supplier_name) if supplier_norm: evidence = list(db.execute(select(AccountingHistoricalLedgerEvidence).where( AccountingHistoricalLedgerEvidence.tenant_id == tenant_id, AccountingHistoricalLedgerEvidence.client_id == client_id, )).scalars().all()) mappings = list(db.execute(select(AccountingLedgerNatureMapping).where( AccountingLedgerNatureMapping.tenant_id == tenant_id, AccountingLedgerNatureMapping.client_id == client_id, )).scalars().all()) mapping_by_ledger = {(m.tally_guid, normalize_text(m.ledger_name)): m for m in mappings} buckets = defaultdict(int) total = 0 for row in evidence: if tally_guid and row.tally_guid != tally_guid: continue if normalize_text(row.party_ledger_name) != supplier_norm: continue m = mapping_by_ledger.get((row.tally_guid, normalize_text(row.counter_ledger_name))) if not m: continue count = max(0, int(row.voucher_count or 0)) buckets[m.nature_id] += count total += count if total: for nature_id, count in buckets.items(): share = count / total _add_score( scores, nature_id, 46 * share, f"Historical Tally treatment: {count} of {total} mapped purchase voucher(s) " f"for this supplier used this accounting nature." ) # 3. Deterministic text priors. These are intentionally weaker than confirmed history. combined = normalize_text(" ".join([supplier_name, description])) compact = combined.replace(" ", "") token_set = set(combined.split()) for code, words in KEYWORD_PRIORS.items(): nature = nature_by_code.get(code) if not nature: continue hits = [w for w in words if w in token_set or w in compact] if hits: _add_score(scores, nature.id, min(24, 12 + 4 * len(hits)), f"Description/supplier keyword match: {', '.join(sorted(hits)[:4])}.") # 4. Business profile context. Never overrides client-confirmed mappings. profile = business_profile(db, client_id) if profile: model = str(profile.business_model or "") activity = normalize_text(profile.primary_business_activity) products = normalize_text(profile.main_products) services = normalize_text(profile.main_services) if model in {"trading", "manufacturing_and_trading"}: n = nature_by_code.get("TRADING_PURCHASE") if n: _add_score(scores, n.id, 10 if model == "trading" else 6, f"Client business model is {model.replace('_', ' ')}.") if model in {"manufacturing", "manufacturing_and_trading"}: n = nature_by_code.get("RAW_MATERIAL_PURCHASE") if n: _add_score(scores, n.id, 9, f"Client business model is {model.replace('_', ' ')}.") if model in {"construction", "contracting"}: for code in ("RAW_MATERIAL_PURCHASE", "CAPITAL_WIP"): n = nature_by_code.get(code) if n: _add_score(scores, n.id, 5, f"Client business model is {model}.") if profile.vehicle_intensive: n = nature_by_code.get("VEHICLE_MAINTENANCE") if n: _add_score(scores, n.id, 5, "Client business profile is marked vehicle-intensive.") if profile.capital_intensive: for code in ("PLANT_MACHINERY", "CAPITAL_WIP"): n = nature_by_code.get(code) if n: _add_score(scores, n.id, 3, "Client business profile is marked capital-intensive.") if combined and activity and any(tok in activity for tok in combined.split() if len(tok) >= 5): # This supports context but deliberately does not pick a new nature by itself. pass # Find candidate client-specific Tally ledgers for each nature. mappings = list(db.execute(select(AccountingLedgerNatureMapping).where( AccountingLedgerNatureMapping.tenant_id == tenant_id, AccountingLedgerNatureMapping.client_id == client_id, )).scalars().all()) ledgers_by_nature = defaultdict(list) for m in mappings: if tally_guid and m.tally_guid != tally_guid: continue ledgers_by_nature[m.nature_id].append(m) max_raw = max([v["score"] for v in scores.values()], default=0.0) result = [] for nature_id, info in scores.items(): nature = nature_by_id.get(nature_id) if not nature: continue raw = info["score"] # Confidence is capped for suggestion-only Phase 6. Human confirmation remains required. confidence = min(99, max(1, round((raw / max(70.0, max_raw)) * 96))) if raw else 0 candidates = sorted( ledgers_by_nature.get(nature_id, []), key=lambda m: (-int(m.confidence_percent or 0), (m.ledger_name or "").casefold()) ) result.append({ "nature": nature, "raw_score": round(raw, 2), "confidence": confidence, "reasons": info["reasons"], "ledger_candidates": candidates[:5], "suggested_ledger": candidates[0].ledger_name if candidates else "", }) # If evidence is weak, REVIEW_REQUIRED must be visible rather than pretending certainty. result.sort(key=lambda x: (-x["raw_score"], x["nature"].sort_order, x["nature"].name)) if not result or (result and result[0]["confidence"] < 45): review = nature_by_code.get("REVIEW_REQUIRED") if review and not any(r["nature"].id == review.id for r in result): result.append({ "nature": review, "raw_score": 1.0, "confidence": 100 if not result else max(55, 100 - result[0]["confidence"]), "reasons": ["Available evidence is not strong enough for a reliable automatic classification."], "ledger_candidates": [], "suggested_ledger": "", }) return result[:8] def _find_rule(db, *, tenant_id: int, client_id: int, tally_guid: str, rule_type: str, rule_key: str, nature_id: int, ledger_name: str): return db.execute(select(AccountingLedgerLearningRule).where( AccountingLedgerLearningRule.tenant_id == tenant_id, AccountingLedgerLearningRule.client_id == client_id, AccountingLedgerLearningRule.tally_guid == (tally_guid or ""), AccountingLedgerLearningRule.rule_type == rule_type, AccountingLedgerLearningRule.rule_key == rule_key, AccountingLedgerLearningRule.nature_id == nature_id, AccountingLedgerLearningRule.ledger_name == (ledger_name or ""), )).scalar_one_or_none() def _touch_rule( db, *, tenant_id: int, client_id: int, tally_guid: str, rule_type: str, rule_key: str, display_value: str, nature_id: int, ledger_name: str, confirmed: bool, user_id: int, ): row = _find_rule( db, tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid, rule_type=rule_type, rule_key=rule_key, nature_id=nature_id, ledger_name=ledger_name, ) if not row: row = AccountingLedgerLearningRule( tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid or "", rule_type=rule_type, rule_key=rule_key, display_value=display_value or rule_key, nature_id=nature_id, ledger_name=ledger_name or "", ) if confirmed: row.confirmation_count = int(row.confirmation_count or 0) + 1 row.last_confirmed_by_user_id = user_id row.last_confirmed_at_utc = _utcnow() else: row.rejection_count = int(row.rejection_count or 0) + 1 row.confidence_percent = _confidence(row.confirmation_count, row.rejection_count) row.is_active = True row.updated_at_utc = _utcnow() db.add(row) return row def record_review( db, *, tenant_id: int, client_id: int, tally_guid: str, supplier_name: str, supplier_gstin: str, hsn_code: str, description: str, amount: float | None, suggested_nature_id: int | None, suggested_ledger_name: str, suggested_confidence: int, final_nature_id: int, final_ledger_name: str, user_id: int, explanation: list[str] | None = None, ): active_rows = active_natures(db, tenant_id) active_by_id = {n.id: n for n in active_rows} if final_nature_id not in active_by_id: raise ValueError("Select an active accounting nature.") final_ledger = (final_ledger_name or "").strip() if final_ledger: ledger_stmt = select(AccountingLedgerNatureMapping).where( AccountingLedgerNatureMapping.tenant_id == tenant_id, AccountingLedgerNatureMapping.client_id == client_id, AccountingLedgerNatureMapping.ledger_name == final_ledger, AccountingLedgerNatureMapping.nature_id == final_nature_id, ) if tally_guid: ledger_stmt = ledger_stmt.where(AccountingLedgerNatureMapping.tally_guid == tally_guid) if db.execute(ledger_stmt).scalar_one_or_none() is None: raise ValueError("The selected Tally ledger is not mapped to the selected accounting nature for this client/company.") accepted = bool( suggested_nature_id and int(suggested_nature_id) == int(final_nature_id) and normalize_text(suggested_ledger_name) == normalize_text(final_ledger_name) ) profile = business_profile(db, client_id) profile_snapshot = {} if profile: profile_snapshot = { "primary_industry": profile.primary_industry, "primary_business_activity": profile.primary_business_activity, "business_model": profile.business_model, "main_products": profile.main_products, "main_services": profile.main_services, "inventory_maintained": profile.inventory_maintained, "project_job_based": profile.project_job_based, "capital_intensive": profile.capital_intensive, "vehicle_intensive": profile.vehicle_intensive, "profile_status": profile.profile_status, } event = AccountingLedgerLearningEvent( tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid or "", source_type="manual_review", supplier_name=(supplier_name or "").strip(), supplier_gstin=normalize_gstin(supplier_gstin), hsn_code=normalize_hsn(hsn_code), description_text=(description or "").strip() or None, amount=amount, suggested_nature_id=suggested_nature_id, suggested_ledger_name=(suggested_ledger_name or "").strip(), suggested_confidence=max(0, min(100, int(suggested_confidence or 0))), final_nature_id=final_nature_id, final_ledger_name=final_ledger, accepted_suggestion=accepted, explanation_json=json.dumps(explanation or [], ensure_ascii=False), business_profile_snapshot_json=json.dumps(profile_snapshot, ensure_ascii=False, default=str), reviewed_by_user_id=user_id, ) db.add(event) contexts = _rule_contexts( supplier_name=supplier_name, supplier_gstin=supplier_gstin, hsn_code=hsn_code, description=description, ) # REVIEW_REQUIRED is an audit outcome, not a training label. It is stored in # the event history but never strengthened into a future automatic rule. final_is_review = active_by_id[final_nature_id].code == "REVIEW_REQUIRED" # A correction weakens the exact contextual rule that produced the wrong choice. if suggested_nature_id and int(suggested_nature_id) != int(final_nature_id): for rule_type, rule_key, display_value in contexts: _touch_rule( db, tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid, rule_type=rule_type, rule_key=rule_key, display_value=display_value, nature_id=int(suggested_nature_id), ledger_name=(suggested_ledger_name or "").strip(), confirmed=False, user_id=user_id, ) # Final human choice is the training label, unless the reviewer intentionally # chose the non-classification REVIEW_REQUIRED bucket. if not final_is_review: for rule_type, rule_key, display_value in contexts: _touch_rule( db, tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid, rule_type=rule_type, rule_key=rule_key, display_value=display_value, nature_id=final_nature_id, ledger_name=final_ledger, confirmed=True, user_id=user_id, ) db.commit() db.refresh(event) return event def learning_summary(db, tenant_id: int, client_id: int): rules = list(db.execute(select(AccountingLedgerLearningRule).where( AccountingLedgerLearningRule.tenant_id == tenant_id, AccountingLedgerLearningRule.client_id == client_id, AccountingLedgerLearningRule.is_active.is_(True), )).order_by( AccountingLedgerLearningRule.confidence_percent.desc(), AccountingLedgerLearningRule.confirmation_count.desc(), AccountingLedgerLearningRule.id.desc(), ).scalars().all()) natures = {n.id: n for n in active_natures(db, tenant_id)} return [{ "rule": r, "nature": natures.get(r.nature_id), "net_confirmations": max(0, int(r.confirmation_count or 0) - int(r.rejection_count or 0)), } for r in rules[:100] if natures.get(r.nature_id)]