Add Phase 6 self improving ledger selection engine

This commit is contained in:
A R R R Associates
2026-08-22 13:30:38 +05:30
parent ba0d638dfb
commit 7e1b9c24ef
7 changed files with 1096 additions and 0 deletions
@@ -0,0 +1,532 @@
from __future__ import annotations
import json
import math
import re
from collections import defaultdict
from datetime import datetime, timezone
from sqlalchemy import select
from app.modules.accounting.historical_learning_models import (
AccountingHistoricalLedgerEvidence,
AccountingLedgerNatureMapping,
)
from app.modules.accounting.ledger_learning_models import (
AccountingLedgerLearningEvent,
AccountingLedgerLearningRule,
)
from app.modules.accounting.taxonomy_models import AccountingNature
from app.modules.clients.models import ClientBusinessProfile
TOKEN_RE = re.compile(r"[A-Z0-9]+")
STOP_WORDS = {
"THE", "AND", "FOR", "PVT", "PRIVATE", "LIMITED", "LTD", "LLP", "INDIA",
"INVOICE", "BILL", "TAX", "GST", "GSTIN", "SERVICES", "SERVICE", "TRADERS",
"TRADING", "ENTERPRISES", "ENTERPRISE", "COMPANY", "CO",
}
KEYWORD_PRIORS = {
"TELEPHONE": {"AIRTEL", "VODAFONE", "IDEA", "JIO", "MOBILE", "TELEPHONE", "TELECOM"},
"INTERNET": {"BROADBAND", "INTERNET", "FIBER", "FIBRE", "LEASEDLINE"},
"ELECTRICITY": {"ELECTRICITY", "POWER", "TANGEDCO", "EB"},
"INSURANCE": {"INSURANCE", "PREMIUM"},
"VEHICLE_MAINTENANCE": {"TYRE", "TYRES", "TIRE", "BRAKE", "CLUTCH", "VEHICLE", "CAR", "MOTOR", "SERVICECENTER"},
"COMPUTER_MAINTENANCE": {"LAPTOPREPAIR", "COMPUTERREPAIR", "PRINTERREPAIR", "AMCIT"},
"BUILDING_MAINTENANCE": {"PLUMBING", "PLUMBER", "PAINTING", "CIVILREPAIR", "BUILDINGREPAIR"},
"MACHINERY_MAINTENANCE": {"MACHINERYREPAIR", "MACHINEPART", "BEARING", "INDUSTRIALREPAIR"},
"PROFESSIONAL_CHARGES": {"CONSULTANCY", "CONSULTANT", "PROFESSIONAL"},
"LEGAL_FEES": {"ADVOCATE", "LEGAL", "LAWYER"},
"AUDIT_FEES": {"AUDIT", "AUDITOR"},
"TRAVELLING": {"FLIGHT", "AIRLINE", "TRAVEL", "RAILWAY", "TRAIN", "TAXI", "CAB"},
"HOTEL_ACCOMMODATION": {"HOTEL", "RESORT", "ACCOMMODATION"},
"FREIGHT_CARRIAGE": {"FREIGHT", "TRANSPORT", "CARRIAGE", "LOGISTICS"},
"PRINTING_STATIONERY": {"STATIONERY", "PRINTING", "PAPER", "TONER", "CARTRIDGE"},
"SOFTWARE_SUBSCRIPTION": {"SOFTWARE", "SUBSCRIPTION", "SAAS", "LICENSE", "LICENCE"},
"CLOUD_HOSTING": {"HOSTING", "CLOUD", "DOMAIN", "AWS", "AZURE"},
"ADVERTISEMENT_MARKETING": {"ADVERTISEMENT", "ADVERTISING", "MARKETING", "PROMOTION"},
"COMMISSION_BROKERAGE": {"COMMISSION", "BROKERAGE"},
"COMPUTERS": {"LAPTOP", "DESKTOP", "COMPUTER", "SERVER"},
"FURNITURE_FIXTURES": {"FURNITURE", "CHAIR", "TABLE", "WORKSTATION"},
"MOTOR_VEHICLES": {"MOTORCAR", "MOTORVEHICLE", "NEWCAR", "NEWVEHICLE"},
"PLANT_MACHINERY": {"MACHINERY", "MACHINE", "EQUIPMENTPLANT"},
}
def _utcnow():
return datetime.now(timezone.utc)
def normalize_text(value: str | None) -> str:
return " ".join(TOKEN_RE.findall(str(value or "").upper())).strip()
def normalize_gstin(value: str | None) -> str:
return re.sub(r"[^A-Z0-9]", "", str(value or "").upper())[:15]
def normalize_hsn(value: str | None) -> str:
return re.sub(r"\D", "", str(value or ""))[:8]
def description_signature(value: str | None) -> str:
tokens = [t for t in TOKEN_RE.findall(str(value or "").upper()) if len(t) >= 3 and t not in STOP_WORDS]
return " ".join(sorted(dict.fromkeys(tokens))[:10])
def _confidence(confirmations: int, rejections: int) -> int:
total = max(0, int(confirmations)) + max(0, int(rejections))
if total <= 0:
return 0
# Bayesian smoothing prevents one confirmation from becoming 100% certainty.
ratio = (confirmations + 2.0) / (total + 4.0)
volume = min(1.0, math.log1p(total) / math.log(11))
return max(1, min(99, round((ratio * 75) + (volume * 24))))
def business_profile(db, client_id: int):
return db.execute(select(ClientBusinessProfile).where(ClientBusinessProfile.client_id == client_id)).scalar_one_or_none()
def active_natures(db, tenant_id: int):
return list(db.execute(
select(AccountingNature).where(
AccountingNature.tenant_id == tenant_id,
AccountingNature.is_active.is_(True),
AccountingNature.is_posting_nature.is_(True),
).order_by(AccountingNature.sort_order, AccountingNature.name)
).scalars().all())
def _nature_maps(db, tenant_id: int):
rows = active_natures(db, tenant_id)
return rows, {r.id: r for r in rows}, {r.code: r for r in rows}
def learned_rules(db, tenant_id: int, client_id: int, tally_guid: str = ""):
stmt = select(AccountingLedgerLearningRule).where(
AccountingLedgerLearningRule.tenant_id == tenant_id,
AccountingLedgerLearningRule.client_id == client_id,
AccountingLedgerLearningRule.is_active.is_(True),
)
if tally_guid:
stmt = stmt.where(
(AccountingLedgerLearningRule.tally_guid == tally_guid) |
(AccountingLedgerLearningRule.tally_guid == "")
)
return list(db.execute(stmt).scalars().all())
def recent_events(db, tenant_id: int, client_id: int, limit: int = 30):
return list(db.execute(
select(AccountingLedgerLearningEvent).where(
AccountingLedgerLearningEvent.tenant_id == tenant_id,
AccountingLedgerLearningEvent.client_id == client_id,
).order_by(AccountingLedgerLearningEvent.id.desc()).limit(limit)
).scalars().all())
def available_tally_guids(db, tenant_id: int, client_id: int):
vals = set()
for row in db.execute(select(
AccountingLedgerNatureMapping.tally_guid,
AccountingLedgerNatureMapping.company_name,
).where(
AccountingLedgerNatureMapping.tenant_id == tenant_id,
AccountingLedgerNatureMapping.client_id == client_id,
)).all():
if row[0]:
vals.add((row[0], row[1] or ""))
for row in db.execute(select(
AccountingHistoricalLedgerEvidence.tally_guid,
AccountingHistoricalLedgerEvidence.company_name,
).where(
AccountingHistoricalLedgerEvidence.tenant_id == tenant_id,
AccountingHistoricalLedgerEvidence.client_id == client_id,
)).all():
if row[0]:
vals.add((row[0], row[1] or ""))
return sorted(vals, key=lambda x: ((x[1] or "").casefold(), x[0]))
def _rule_contexts(*, supplier_name: str, supplier_gstin: str, hsn_code: str, description: str):
contexts = []
gstin = normalize_gstin(supplier_gstin)
supplier = normalize_text(supplier_name)
hsn = normalize_hsn(hsn_code)
desc = description_signature(description)
if gstin:
contexts.append(("supplier_gstin", gstin, supplier_gstin))
if supplier:
contexts.append(("supplier_name", supplier, supplier_name))
if hsn:
contexts.append(("hsn", hsn, hsn_code))
if desc:
contexts.append(("description_signature", desc, description[:500]))
if supplier and hsn:
contexts.append(("supplier_hsn", f"{supplier}|{hsn}", f"{supplier_name} / HSN {hsn}"))
if gstin and hsn:
contexts.append(("gstin_hsn", f"{gstin}|{hsn}", f"{gstin} / HSN {hsn}"))
return contexts
def _add_score(scores, nature_id, points, reason):
if not nature_id or points <= 0:
return
row = scores.setdefault(nature_id, {"score": 0.0, "reasons": []})
row["score"] += float(points)
row["reasons"].append(reason)
def rank_suggestions(
db, *,
tenant_id: int,
client_id: int,
tally_guid: str = "",
supplier_name: str = "",
supplier_gstin: str = "",
hsn_code: str = "",
description: str = "",
amount: float | None = None,
):
natures, nature_by_id, nature_by_code = _nature_maps(db, tenant_id)
if not natures:
return []
scores = {}
contexts = _rule_contexts(
supplier_name=supplier_name,
supplier_gstin=supplier_gstin,
hsn_code=hsn_code,
description=description,
)
context_lookup = {(t, k) for t, k, _ in contexts}
# 1. User-confirmed client-specific learning.
rule_weights = {
"gstin_hsn": 72,
"supplier_gstin": 68,
"supplier_hsn": 62,
"supplier_name": 52,
"hsn": 34,
"description_signature": 28,
}
for rule in learned_rules(db, tenant_id, client_id, tally_guid):
if (rule.rule_type, rule.rule_key) not in context_lookup:
continue
strength = max(0.10, rule.confidence_percent / 100)
points = rule_weights.get(rule.rule_type, 20) * strength
_add_score(
scores, rule.nature_id, points,
f"Confirmed {rule.rule_type.replace('_', ' ')} rule: "
f"{rule.confirmation_count} confirmation(s), {rule.rejection_count} correction(s), "
f"{rule.confidence_percent}% learned confidence."
)
# 2. Phase 5 historical supplier treatment.
supplier_norm = normalize_text(supplier_name)
if supplier_norm:
evidence = list(db.execute(select(AccountingHistoricalLedgerEvidence).where(
AccountingHistoricalLedgerEvidence.tenant_id == tenant_id,
AccountingHistoricalLedgerEvidence.client_id == client_id,
)).scalars().all())
mappings = list(db.execute(select(AccountingLedgerNatureMapping).where(
AccountingLedgerNatureMapping.tenant_id == tenant_id,
AccountingLedgerNatureMapping.client_id == client_id,
)).scalars().all())
mapping_by_ledger = {(m.tally_guid, normalize_text(m.ledger_name)): m for m in mappings}
buckets = defaultdict(int)
total = 0
for row in evidence:
if tally_guid and row.tally_guid != tally_guid:
continue
if normalize_text(row.party_ledger_name) != supplier_norm:
continue
m = mapping_by_ledger.get((row.tally_guid, normalize_text(row.counter_ledger_name)))
if not m:
continue
count = max(0, int(row.voucher_count or 0))
buckets[m.nature_id] += count
total += count
if total:
for nature_id, count in buckets.items():
share = count / total
_add_score(
scores, nature_id, 46 * share,
f"Historical Tally treatment: {count} of {total} mapped purchase voucher(s) "
f"for this supplier used this accounting nature."
)
# 3. Deterministic text priors. These are intentionally weaker than confirmed history.
combined = normalize_text(" ".join([supplier_name, description]))
compact = combined.replace(" ", "")
token_set = set(combined.split())
for code, words in KEYWORD_PRIORS.items():
nature = nature_by_code.get(code)
if not nature:
continue
hits = [w for w in words if w in token_set or w in compact]
if hits:
_add_score(scores, nature.id, min(24, 12 + 4 * len(hits)), f"Description/supplier keyword match: {', '.join(sorted(hits)[:4])}.")
# 4. Business profile context. Never overrides client-confirmed mappings.
profile = business_profile(db, client_id)
if profile:
model = str(profile.business_model or "")
activity = normalize_text(profile.primary_business_activity)
products = normalize_text(profile.main_products)
services = normalize_text(profile.main_services)
if model in {"trading", "manufacturing_and_trading"}:
n = nature_by_code.get("TRADING_PURCHASE")
if n:
_add_score(scores, n.id, 10 if model == "trading" else 6, f"Client business model is {model.replace('_', ' ')}.")
if model in {"manufacturing", "manufacturing_and_trading"}:
n = nature_by_code.get("RAW_MATERIAL_PURCHASE")
if n:
_add_score(scores, n.id, 9, f"Client business model is {model.replace('_', ' ')}.")
if model in {"construction", "contracting"}:
for code in ("RAW_MATERIAL_PURCHASE", "CAPITAL_WIP"):
n = nature_by_code.get(code)
if n:
_add_score(scores, n.id, 5, f"Client business model is {model}.")
if profile.vehicle_intensive:
n = nature_by_code.get("VEHICLE_MAINTENANCE")
if n:
_add_score(scores, n.id, 5, "Client business profile is marked vehicle-intensive.")
if profile.capital_intensive:
for code in ("PLANT_MACHINERY", "CAPITAL_WIP"):
n = nature_by_code.get(code)
if n:
_add_score(scores, n.id, 3, "Client business profile is marked capital-intensive.")
if combined and activity and any(tok in activity for tok in combined.split() if len(tok) >= 5):
# This supports context but deliberately does not pick a new nature by itself.
pass
# Find candidate client-specific Tally ledgers for each nature.
mappings = list(db.execute(select(AccountingLedgerNatureMapping).where(
AccountingLedgerNatureMapping.tenant_id == tenant_id,
AccountingLedgerNatureMapping.client_id == client_id,
)).scalars().all())
ledgers_by_nature = defaultdict(list)
for m in mappings:
if tally_guid and m.tally_guid != tally_guid:
continue
ledgers_by_nature[m.nature_id].append(m)
max_raw = max([v["score"] for v in scores.values()], default=0.0)
result = []
for nature_id, info in scores.items():
nature = nature_by_id.get(nature_id)
if not nature:
continue
raw = info["score"]
# Confidence is capped for suggestion-only Phase 6. Human confirmation remains required.
confidence = min(99, max(1, round((raw / max(70.0, max_raw)) * 96))) if raw else 0
candidates = sorted(
ledgers_by_nature.get(nature_id, []),
key=lambda m: (-int(m.confidence_percent or 0), (m.ledger_name or "").casefold())
)
result.append({
"nature": nature,
"raw_score": round(raw, 2),
"confidence": confidence,
"reasons": info["reasons"],
"ledger_candidates": candidates[:5],
"suggested_ledger": candidates[0].ledger_name if candidates else "",
})
# If evidence is weak, REVIEW_REQUIRED must be visible rather than pretending certainty.
result.sort(key=lambda x: (-x["raw_score"], x["nature"].sort_order, x["nature"].name))
if not result or (result and result[0]["confidence"] < 45):
review = nature_by_code.get("REVIEW_REQUIRED")
if review and not any(r["nature"].id == review.id for r in result):
result.append({
"nature": review,
"raw_score": 1.0,
"confidence": 100 if not result else max(55, 100 - result[0]["confidence"]),
"reasons": ["Available evidence is not strong enough for a reliable automatic classification."],
"ledger_candidates": [],
"suggested_ledger": "",
})
return result[:8]
def _find_rule(db, *, tenant_id: int, client_id: int, tally_guid: str, rule_type: str, rule_key: str, nature_id: int, ledger_name: str):
return db.execute(select(AccountingLedgerLearningRule).where(
AccountingLedgerLearningRule.tenant_id == tenant_id,
AccountingLedgerLearningRule.client_id == client_id,
AccountingLedgerLearningRule.tally_guid == (tally_guid or ""),
AccountingLedgerLearningRule.rule_type == rule_type,
AccountingLedgerLearningRule.rule_key == rule_key,
AccountingLedgerLearningRule.nature_id == nature_id,
AccountingLedgerLearningRule.ledger_name == (ledger_name or ""),
)).scalar_one_or_none()
def _touch_rule(
db, *, tenant_id: int, client_id: int, tally_guid: str,
rule_type: str, rule_key: str, display_value: str,
nature_id: int, ledger_name: str, confirmed: bool, user_id: int,
):
row = _find_rule(
db, tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid,
rule_type=rule_type, rule_key=rule_key, nature_id=nature_id, ledger_name=ledger_name,
)
if not row:
row = AccountingLedgerLearningRule(
tenant_id=tenant_id,
client_id=client_id,
tally_guid=tally_guid or "",
rule_type=rule_type,
rule_key=rule_key,
display_value=display_value or rule_key,
nature_id=nature_id,
ledger_name=ledger_name or "",
)
if confirmed:
row.confirmation_count = int(row.confirmation_count or 0) + 1
row.last_confirmed_by_user_id = user_id
row.last_confirmed_at_utc = _utcnow()
else:
row.rejection_count = int(row.rejection_count or 0) + 1
row.confidence_percent = _confidence(row.confirmation_count, row.rejection_count)
row.is_active = True
row.updated_at_utc = _utcnow()
db.add(row)
return row
def record_review(
db, *,
tenant_id: int,
client_id: int,
tally_guid: str,
supplier_name: str,
supplier_gstin: str,
hsn_code: str,
description: str,
amount: float | None,
suggested_nature_id: int | None,
suggested_ledger_name: str,
suggested_confidence: int,
final_nature_id: int,
final_ledger_name: str,
user_id: int,
explanation: list[str] | None = None,
):
active_rows = active_natures(db, tenant_id)
active_by_id = {n.id: n for n in active_rows}
if final_nature_id not in active_by_id:
raise ValueError("Select an active accounting nature.")
final_ledger = (final_ledger_name or "").strip()
if final_ledger:
ledger_stmt = select(AccountingLedgerNatureMapping).where(
AccountingLedgerNatureMapping.tenant_id == tenant_id,
AccountingLedgerNatureMapping.client_id == client_id,
AccountingLedgerNatureMapping.ledger_name == final_ledger,
AccountingLedgerNatureMapping.nature_id == final_nature_id,
)
if tally_guid:
ledger_stmt = ledger_stmt.where(AccountingLedgerNatureMapping.tally_guid == tally_guid)
if db.execute(ledger_stmt).scalar_one_or_none() is None:
raise ValueError("The selected Tally ledger is not mapped to the selected accounting nature for this client/company.")
accepted = bool(
suggested_nature_id
and int(suggested_nature_id) == int(final_nature_id)
and normalize_text(suggested_ledger_name) == normalize_text(final_ledger_name)
)
profile = business_profile(db, client_id)
profile_snapshot = {}
if profile:
profile_snapshot = {
"primary_industry": profile.primary_industry,
"primary_business_activity": profile.primary_business_activity,
"business_model": profile.business_model,
"main_products": profile.main_products,
"main_services": profile.main_services,
"inventory_maintained": profile.inventory_maintained,
"project_job_based": profile.project_job_based,
"capital_intensive": profile.capital_intensive,
"vehicle_intensive": profile.vehicle_intensive,
"profile_status": profile.profile_status,
}
event = AccountingLedgerLearningEvent(
tenant_id=tenant_id,
client_id=client_id,
tally_guid=tally_guid or "",
source_type="manual_review",
supplier_name=(supplier_name or "").strip(),
supplier_gstin=normalize_gstin(supplier_gstin),
hsn_code=normalize_hsn(hsn_code),
description_text=(description or "").strip() or None,
amount=amount,
suggested_nature_id=suggested_nature_id,
suggested_ledger_name=(suggested_ledger_name or "").strip(),
suggested_confidence=max(0, min(100, int(suggested_confidence or 0))),
final_nature_id=final_nature_id,
final_ledger_name=final_ledger,
accepted_suggestion=accepted,
explanation_json=json.dumps(explanation or [], ensure_ascii=False),
business_profile_snapshot_json=json.dumps(profile_snapshot, ensure_ascii=False, default=str),
reviewed_by_user_id=user_id,
)
db.add(event)
contexts = _rule_contexts(
supplier_name=supplier_name,
supplier_gstin=supplier_gstin,
hsn_code=hsn_code,
description=description,
)
# REVIEW_REQUIRED is an audit outcome, not a training label. It is stored in
# the event history but never strengthened into a future automatic rule.
final_is_review = active_by_id[final_nature_id].code == "REVIEW_REQUIRED"
# A correction weakens the exact contextual rule that produced the wrong choice.
if suggested_nature_id and int(suggested_nature_id) != int(final_nature_id):
for rule_type, rule_key, display_value in contexts:
_touch_rule(
db, tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid,
rule_type=rule_type, rule_key=rule_key, display_value=display_value,
nature_id=int(suggested_nature_id), ledger_name=(suggested_ledger_name or "").strip(),
confirmed=False, user_id=user_id,
)
# Final human choice is the training label, unless the reviewer intentionally
# chose the non-classification REVIEW_REQUIRED bucket.
if not final_is_review:
for rule_type, rule_key, display_value in contexts:
_touch_rule(
db, tenant_id=tenant_id, client_id=client_id, tally_guid=tally_guid,
rule_type=rule_type, rule_key=rule_key, display_value=display_value,
nature_id=final_nature_id, ledger_name=final_ledger,
confirmed=True, user_id=user_id,
)
db.commit()
db.refresh(event)
return event
def learning_summary(db, tenant_id: int, client_id: int):
rules = list(db.execute(select(AccountingLedgerLearningRule).where(
AccountingLedgerLearningRule.tenant_id == tenant_id,
AccountingLedgerLearningRule.client_id == client_id,
AccountingLedgerLearningRule.is_active.is_(True),
)).order_by(
AccountingLedgerLearningRule.confidence_percent.desc(),
AccountingLedgerLearningRule.confirmation_count.desc(),
AccountingLedgerLearningRule.id.desc(),
).scalars().all())
natures = {n.id: n for n in active_natures(db, tenant_id)}
return [{
"rule": r,
"nature": natures.get(r.nature_id),
"net_confirmations": max(0, int(r.confirmation_count or 0) - int(r.rejection_count or 0)),
} for r in rules[:100] if natures.get(r.nature_id)]