Add Phase 12 server AI semantic accounting layer

This commit is contained in:
A R R R Associates
2026-08-22 21:36:27 +05:30
parent 016299cf0c
commit ef2b0e1f83
14 changed files with 1023 additions and 5 deletions
+460
View File
@@ -0,0 +1,460 @@
from __future__ import annotations
import hashlib
import json
import os
from datetime import datetime, timezone
from typing import Any
from sqlalchemy import func, select
from app.modules.accounting.ai_models import AccountingAISemanticDecision
from app.modules.accounting.ai_provider import AccountingAIProviderError, get_accounting_semantic_provider
from app.modules.accounting.gstr2b_models import AccountingGSTR2BPurchase
from app.modules.accounting.taxonomy_models import AccountingNature
from app.modules.clients.models import Client, ClientBusinessProfile
PROMPT_VERSION = "phase12-v1"
DEFAULT_AI_GATE = 90
def _utcnow():
return datetime.now(timezone.utc)
def _s(value) -> str:
return str(value or "").strip()
def _i(value, default=0) -> int:
try:
return int(round(float(value)))
except Exception:
return int(default)
def ai_gate() -> int:
return max(50, min(99, _i(os.getenv("ACCOUNTING_AI_GATE_CONFIDENCE", DEFAULT_AI_GATE), DEFAULT_AI_GATE)))
def provider_status() -> dict[str, Any]:
try:
provider = get_accounting_semantic_provider()
return {
"provider": provider.provider_name,
"model": provider.model_name(),
"configured": bool(provider.is_configured()),
"gate": ai_gate(),
}
except Exception as exc:
return {"provider": "", "model": "", "configured": False, "gate": ai_gate(), "error": str(exc)}
def _profile_context(db, client_id: int) -> dict[str, Any]:
client = db.get(Client, int(client_id))
profile = db.execute(
select(ClientBusinessProfile).where(ClientBusinessProfile.client_id == int(client_id))
).scalar_one_or_none()
data = {
"client_name": _s(getattr(client, "client_name", "")),
"trade_name": _s(getattr(client, "trade_name", "")),
"primary_industry": "",
"primary_business_activity": "",
"secondary_business_activities": "",
"business_model": "",
"main_products": "",
"main_services": "",
"inventory_maintained": None,
"project_job_based": None,
"capital_intensive": None,
"vehicle_intensive": None,
"profile_status": "",
"profile_confidence": 0,
}
if profile:
for key in (
"primary_industry", "primary_business_activity", "secondary_business_activities",
"business_model", "main_products", "main_services", "inventory_maintained",
"project_job_based", "capital_intensive", "vehicle_intensive", "profile_status",
):
data[key] = getattr(profile, key, None)
data["profile_confidence"] = int(getattr(profile, "confidence_score", 0) or 0)
return data
def _taxonomy(db, tenant_id: int) -> list[AccountingNature]:
return list(db.execute(
select(AccountingNature).where(
AccountingNature.tenant_id == int(tenant_id),
AccountingNature.is_active.is_(True),
AccountingNature.is_posting_nature.is_(True),
).order_by(AccountingNature.sort_order, AccountingNature.name)
).scalars().all())
def _taxonomy_payload(rows: list[AccountingNature]) -> list[dict[str, Any]]:
return [{
"code": row.code,
"name": row.name,
"group": row.classification_group,
"capital_revenue": row.capital_revenue,
"description": _s(row.description)[:500],
} for row in rows]
def _schema(codes: list[str]) -> dict[str, Any]:
return {
"type": "object",
"properties": {
"nature_code": {"type": "string", "enum": codes},
"confidence": {"type": "integer", "minimum": 0, "maximum": 100},
"capital_revenue": {"type": "string", "enum": ["capital", "revenue", "uncertain"]},
"business_personal": {"type": "string", "enum": ["business", "personal", "uncertain"]},
"reason": {"type": "string", "maxLength": 500},
"alternatives": {
"type": "array",
"maxItems": 3,
"items": {
"type": "object",
"properties": {
"nature_code": {"type": "string", "enum": codes},
"confidence": {"type": "integer", "minimum": 0, "maximum": 100},
},
"required": ["nature_code", "confidence"],
"additionalProperties": False,
},
},
"review_flags": {
"type": "array",
"maxItems": 5,
"items": {"type": "string", "maxLength": 120},
},
},
"required": [
"nature_code", "confidence", "capital_revenue", "business_personal",
"reason", "alternatives", "review_flags",
],
"additionalProperties": False,
}
SYSTEM_TEXT = """You classify accounting transactions for an Indian accounting/audit ERP.
You are a semantic fallback layer, not the primary rules engine.
Choose exactly one accounting nature from ALLOWED_TAXONOMY. Never invent a nature or ledger.
Use the client business profile, supplier/party information, HSN/description, source type,
amount and deterministic evidence. Distinguish revenue expense from capital acquisition
conservatively. Flag uncertainty, possible personal expenditure, mixed-purpose transactions,
or insufficient evidence. Return only the required structured object. Do not create ledgers,
do not decide GST eligibility, and do not authorize posting to Tally."""
def _combined_confidence(*, ai_conf: int, deterministic_conf: int, profile_conf: int, agreement: bool) -> int:
ai_conf = max(0, min(100, int(ai_conf)))
deterministic_conf = max(0, min(100, int(deterministic_conf)))
profile_conf = max(0, min(100, int(profile_conf)))
value = (0.70 * ai_conf) + (0.20 * profile_conf) + (0.10 * deterministic_conf)
if agreement and deterministic_conf > 0:
value += 5
return max(0, min(99, int(round(value))))
def _source_fingerprint(source_type: str, source_record_id: int, payload: dict[str, Any]) -> str:
raw = f"{source_type}|{source_record_id}|{json.dumps(payload, sort_keys=True, ensure_ascii=False)}"
return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest()
def latest_decision(db, *, tenant_id: int, source_type: str, source_record_id: int):
return db.execute(
select(AccountingAISemanticDecision).where(
AccountingAISemanticDecision.tenant_id == int(tenant_id),
AccountingAISemanticDecision.source_type == source_type,
AccountingAISemanticDecision.source_record_id == int(source_record_id),
).order_by(AccountingAISemanticDecision.id.desc()).limit(1)
).scalar_one_or_none()
def _run(
db, *,
tenant_id: int,
client_id: int,
source_type: str,
source_record_id: int,
deterministic_nature_id: int | None,
deterministic_confidence: int,
transaction: dict[str, Any],
created_by_user_id: int,
):
gate = ai_gate()
if int(deterministic_confidence or 0) >= gate:
raise ValueError(
f"AI is not required because deterministic confidence is already "
f"{int(deterministic_confidence or 0)}% (AI gate {gate}%)."
)
natures = _taxonomy(db, tenant_id)
if not natures:
raise ValueError("Accounting taxonomy is empty. Configure Phase 4 taxonomy first.")
profile = _profile_context(db, client_id)
nature_by_code = {row.code: row for row in natures}
nature_by_id = {row.id: row for row in natures}
deterministic = nature_by_id.get(int(deterministic_nature_id)) if deterministic_nature_id else None
payload = {
"source_type": source_type,
"client_business_profile": profile,
"transaction": transaction,
"deterministic_evidence": {
"nature_code": deterministic.code if deterministic else "",
"nature_name": deterministic.name if deterministic else "",
"confidence": int(deterministic_confidence or 0),
},
"allowed_taxonomy": _taxonomy_payload(natures),
}
provider = get_accounting_semantic_provider()
decision = AccountingAISemanticDecision(
tenant_id=tenant_id,
client_id=client_id,
source_type=source_type,
source_record_id=source_record_id,
source_fingerprint=_source_fingerprint(source_type, source_record_id, payload),
provider_name=provider.provider_name,
model_name=provider.model_name(),
prompt_version=PROMPT_VERSION,
deterministic_nature_id=deterministic_nature_id,
deterministic_confidence=int(deterministic_confidence or 0),
context_json=json.dumps(payload, ensure_ascii=False),
status="running",
created_by_user_id=created_by_user_id,
)
db.add(decision)
db.commit()
db.refresh(decision)
try:
result = provider.classify(
system_text=SYSTEM_TEXT,
user_payload=payload,
schema=_schema(list(nature_by_code)),
)
data = result.data
nature = nature_by_code.get(_s(data.get("nature_code")))
if not nature:
raise AccountingAIProviderError("AI returned a nature outside the active ERP taxonomy.")
ai_conf = max(0, min(100, _i(data.get("confidence"))))
agreement = bool(deterministic and deterministic.id == nature.id)
combined = _combined_confidence(
ai_conf=ai_conf,
deterministic_conf=int(deterministic_confidence or 0),
profile_conf=int(profile.get("profile_confidence") or 0),
agreement=agreement,
)
decision.provider_name = result.provider
decision.model_name = result.model
decision.ai_nature_id = nature.id
decision.ai_nature_code = nature.code
decision.ai_confidence = ai_conf
decision.combined_confidence = combined
decision.capital_revenue = _s(data.get("capital_revenue"))
decision.business_personal = _s(data.get("business_personal")) or "business"
decision.concise_reason = _s(data.get("reason"))[:1000]
decision.alternatives_json = json.dumps(data.get("alternatives") or [], ensure_ascii=False)
decision.review_flags_json = json.dumps(data.get("review_flags") or [], ensure_ascii=False)
decision.raw_response_json = json.dumps(result.raw, ensure_ascii=False)[:20000]
decision.input_tokens = int(result.input_tokens or 0)
decision.output_tokens = int(result.output_tokens or 0)
decision.latency_ms = int(result.latency_ms or 0)
decision.status = "completed"
db.add(decision)
db.commit()
db.refresh(decision)
return decision, nature
except Exception as exc:
decision.status = "failed"
decision.error_message = str(exc)[:4000]
db.add(decision)
db.commit()
raise
def ai_assist_purchase(db, *, row: AccountingGSTR2BPurchase, user_id: int):
if row.review_status == "reviewed":
raise ValueError("This purchase is already reviewed.")
transaction = {
"supplier_name": row.supplier_name,
"supplier_gstin": row.supplier_gstin,
"invoice_number": row.invoice_number,
"invoice_date": row.invoice_date,
"document_type": row.document_type,
"invoice_type": row.invoice_type,
"hsn_code": row.hsn_code,
"description": row.description_text or "",
"taxable_value": row.taxable_value,
"invoice_value": row.invoice_value,
"place_of_supply": row.place_of_supply,
"reverse_charge": row.reverse_charge,
"itc_availability": row.itc_availability,
}
decision, nature = _run(
db,
tenant_id=row.tenant_id,
client_id=row.client_id,
source_type="gstr2b",
source_record_id=row.id,
deterministic_nature_id=row.suggested_nature_id,
deterministic_confidence=row.suggested_confidence,
transaction=transaction,
created_by_user_id=user_id,
)
# AI may strengthen/change the accounting nature, but it never invents a ledger.
# Preserve any existing mapped ledger only when the nature did not change.
if row.suggested_nature_id != nature.id:
row.suggested_ledger_name = ""
row.suggested_nature_id = nature.id
row.suggested_confidence = decision.combined_confidence
existing = []
try:
existing = json.loads(row.suggestion_explanation_json or "[]")
except Exception:
pass
existing.append(
f"AI semantic fallback ({decision.model_name}): {decision.concise_reason} "
f"Combined confidence {decision.combined_confidence}%."
)
row.suggestion_explanation_json = json.dumps(existing[-8:], ensure_ascii=False)
row.review_status = "suggested" if decision.combined_confidence >= 60 else "review_required"
decision.applied_to_source = True
db.add(row)
db.add(decision)
db.commit()
return decision
def ai_assist_bank(db, *, tx, user_id: int):
if tx.review_status == "reviewed":
raise ValueError("This bank transaction is already reviewed.")
if _s(getattr(tx, "contra_pair_id", "")):
raise ValueError("Matched inter-bank contra does not require AI classification.")
transaction = {
"bank_name": tx.bank_name,
"account_number": tx.account_number,
"transaction_date": tx.transaction_date,
"direction": tx.direction,
"debit": tx.debit,
"credit": tx.credit,
"amount": tx.amount,
"narration": tx.narration,
"reference_no": tx.reference_no,
"transfer_reference": tx.transfer_reference,
"detected_party": tx.auto_party,
"analyzer_category": tx.analyzer_category,
"analyzer_nature": tx.analyzer_nature,
}
decision, nature = _run(
db,
tenant_id=tx.tenant_id,
client_id=tx.client_id,
source_type="bank",
source_record_id=tx.id,
deterministic_nature_id=tx.suggested_nature_id,
deterministic_confidence=tx.suggested_confidence,
transaction=transaction,
created_by_user_id=user_id,
)
if tx.suggested_nature_id != nature.id:
tx.suggested_ledger_name = ""
tx.suggested_nature_id = nature.id
tx.suggested_confidence = decision.combined_confidence
reasons = []
try:
reasons = json.loads(tx.suggestion_reason_json or "[]")
except Exception:
pass
reasons.append(
f"AI semantic fallback ({decision.model_name}): {decision.concise_reason} "
f"Combined confidence {decision.combined_confidence}%."
)
tx.suggestion_reason_json = json.dumps(reasons[-8:], ensure_ascii=False)
decision.applied_to_source = True
db.add(tx)
db.add(decision)
db.commit()
return decision
def mark_review_outcome(
db, *,
tenant_id: int,
source_type: str,
source_record_id: int,
final_nature_id: int | None,
final_ledger_name: str,
user_id: int,
):
decision = latest_decision(
db,
tenant_id=tenant_id,
source_type=source_type,
source_record_id=source_record_id,
)
if not decision or decision.status != "completed":
return None
decision.final_nature_id = final_nature_id
decision.final_ledger_name = _s(final_ledger_name)
decision.suggestion_accepted = bool(
final_nature_id and decision.ai_nature_id and int(final_nature_id) == int(decision.ai_nature_id)
)
decision.reviewed_by_user_id = user_id
decision.reviewed_at_utc = _utcnow()
db.add(decision)
db.commit()
return decision
def usage_summary(db, *, tenant_id: int, client_id: int | None = None):
where = [AccountingAISemanticDecision.tenant_id == int(tenant_id)]
if client_id:
where.append(AccountingAISemanticDecision.client_id == int(client_id))
total = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(*where)) or 0)
completed = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(
*where, AccountingAISemanticDecision.status == "completed"
)) or 0)
accepted = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(
*where, AccountingAISemanticDecision.suggestion_accepted.is_(True)
)) or 0)
corrected = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(
*where, AccountingAISemanticDecision.suggestion_accepted.is_(False)
)) or 0)
input_tokens = int(db.scalar(select(func.coalesce(func.sum(AccountingAISemanticDecision.input_tokens), 0)).where(*where)) or 0)
output_tokens = int(db.scalar(select(func.coalesce(func.sum(AccountingAISemanticDecision.output_tokens), 0)).where(*where)) or 0)
return {
"total": total,
"completed": completed,
"accepted": accepted,
"corrected": corrected,
"reviewed": accepted + corrected,
"acceptance_rate": round((accepted * 100 / (accepted + corrected)), 1) if (accepted + corrected) else 0,
"input_tokens": input_tokens,
"output_tokens": output_tokens,
}
def recent_decisions(db, *, tenant_id: int, client_id: int | None = None, limit: int = 100):
stmt = select(AccountingAISemanticDecision).where(
AccountingAISemanticDecision.tenant_id == int(tenant_id)
)
if client_id:
stmt = stmt.where(AccountingAISemanticDecision.client_id == int(client_id))
return list(db.execute(
stmt.order_by(AccountingAISemanticDecision.id.desc()).limit(max(1, min(500, int(limit))))
).scalars().all())