from __future__ import annotations import hashlib import json import os from datetime import datetime, timezone from typing import Any from sqlalchemy import func, select from app.modules.accounting.ai_models import AccountingAISemanticDecision from app.modules.accounting.ai_provider import AccountingAIProviderError, get_accounting_semantic_provider from app.modules.accounting.gstr2b_models import AccountingGSTR2BPurchase from app.modules.accounting.taxonomy_models import AccountingNature from app.modules.clients.models import Client, ClientBusinessProfile PROMPT_VERSION = "phase12-v1" DEFAULT_AI_GATE = 90 def _utcnow(): return datetime.now(timezone.utc) def _s(value) -> str: return str(value or "").strip() def _i(value, default=0) -> int: try: return int(round(float(value))) except Exception: return int(default) def ai_gate() -> int: return max(50, min(99, _i(os.getenv("ACCOUNTING_AI_GATE_CONFIDENCE", DEFAULT_AI_GATE), DEFAULT_AI_GATE))) def provider_status() -> dict[str, Any]: try: provider = get_accounting_semantic_provider() return { "provider": provider.provider_name, "model": provider.model_name(), "configured": bool(provider.is_configured()), "gate": ai_gate(), } except Exception as exc: return {"provider": "", "model": "", "configured": False, "gate": ai_gate(), "error": str(exc)} def _profile_context(db, client_id: int) -> dict[str, Any]: client = db.get(Client, int(client_id)) profile = db.execute( select(ClientBusinessProfile).where(ClientBusinessProfile.client_id == int(client_id)) ).scalar_one_or_none() data = { "client_name": _s(getattr(client, "client_name", "")), "trade_name": _s(getattr(client, "trade_name", "")), "primary_industry": "", "primary_business_activity": "", "secondary_business_activities": "", "business_model": "", "main_products": "", "main_services": "", "inventory_maintained": None, "project_job_based": None, "capital_intensive": None, "vehicle_intensive": None, "profile_status": "", "profile_confidence": 0, } if profile: for key in ( "primary_industry", "primary_business_activity", "secondary_business_activities", "business_model", "main_products", "main_services", "inventory_maintained", "project_job_based", "capital_intensive", "vehicle_intensive", "profile_status", ): data[key] = getattr(profile, key, None) data["profile_confidence"] = int(getattr(profile, "confidence_score", 0) or 0) return data def _taxonomy(db, tenant_id: int) -> list[AccountingNature]: return list(db.execute( select(AccountingNature).where( AccountingNature.tenant_id == int(tenant_id), AccountingNature.is_active.is_(True), AccountingNature.is_posting_nature.is_(True), ).order_by(AccountingNature.sort_order, AccountingNature.name) ).scalars().all()) def _taxonomy_payload(rows: list[AccountingNature]) -> list[dict[str, Any]]: return [{ "code": row.code, "name": row.name, "group": row.classification_group, "capital_revenue": row.capital_revenue, "description": _s(row.description)[:500], } for row in rows] def _schema(codes: list[str]) -> dict[str, Any]: return { "type": "object", "properties": { "nature_code": {"type": "string", "enum": codes}, "confidence": {"type": "integer", "minimum": 0, "maximum": 100}, "capital_revenue": {"type": "string", "enum": ["capital", "revenue", "uncertain"]}, "business_personal": {"type": "string", "enum": ["business", "personal", "uncertain"]}, "reason": {"type": "string", "maxLength": 500}, "alternatives": { "type": "array", "maxItems": 3, "items": { "type": "object", "properties": { "nature_code": {"type": "string", "enum": codes}, "confidence": {"type": "integer", "minimum": 0, "maximum": 100}, }, "required": ["nature_code", "confidence"], "additionalProperties": False, }, }, "review_flags": { "type": "array", "maxItems": 5, "items": {"type": "string", "maxLength": 120}, }, }, "required": [ "nature_code", "confidence", "capital_revenue", "business_personal", "reason", "alternatives", "review_flags", ], "additionalProperties": False, } SYSTEM_TEXT = """You classify accounting transactions for an Indian accounting/audit ERP. You are a semantic fallback layer, not the primary rules engine. Choose exactly one accounting nature from ALLOWED_TAXONOMY. Never invent a nature or ledger. Use the client business profile, supplier/party information, HSN/description, source type, amount and deterministic evidence. Distinguish revenue expense from capital acquisition conservatively. Flag uncertainty, possible personal expenditure, mixed-purpose transactions, or insufficient evidence. Return only the required structured object. Do not create ledgers, do not decide GST eligibility, and do not authorize posting to Tally.""" def _combined_confidence(*, ai_conf: int, deterministic_conf: int, profile_conf: int, agreement: bool) -> int: ai_conf = max(0, min(100, int(ai_conf))) deterministic_conf = max(0, min(100, int(deterministic_conf))) profile_conf = max(0, min(100, int(profile_conf))) value = (0.70 * ai_conf) + (0.20 * profile_conf) + (0.10 * deterministic_conf) if agreement and deterministic_conf > 0: value += 5 return max(0, min(99, int(round(value)))) def _source_fingerprint(source_type: str, source_record_id: int, payload: dict[str, Any]) -> str: raw = f"{source_type}|{source_record_id}|{json.dumps(payload, sort_keys=True, ensure_ascii=False)}" return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest() def latest_decision(db, *, tenant_id: int, source_type: str, source_record_id: int): return db.execute( select(AccountingAISemanticDecision).where( AccountingAISemanticDecision.tenant_id == int(tenant_id), AccountingAISemanticDecision.source_type == source_type, AccountingAISemanticDecision.source_record_id == int(source_record_id), ).order_by(AccountingAISemanticDecision.id.desc()).limit(1) ).scalar_one_or_none() def _run( db, *, tenant_id: int, client_id: int, source_type: str, source_record_id: int, deterministic_nature_id: int | None, deterministic_confidence: int, transaction: dict[str, Any], created_by_user_id: int, ): gate = ai_gate() if int(deterministic_confidence or 0) >= gate: raise ValueError( f"AI is not required because deterministic confidence is already " f"{int(deterministic_confidence or 0)}% (AI gate {gate}%)." ) natures = _taxonomy(db, tenant_id) if not natures: raise ValueError("Accounting taxonomy is empty. Configure Phase 4 taxonomy first.") profile = _profile_context(db, client_id) nature_by_code = {row.code: row for row in natures} nature_by_id = {row.id: row for row in natures} deterministic = nature_by_id.get(int(deterministic_nature_id)) if deterministic_nature_id else None payload = { "source_type": source_type, "client_business_profile": profile, "transaction": transaction, "deterministic_evidence": { "nature_code": deterministic.code if deterministic else "", "nature_name": deterministic.name if deterministic else "", "confidence": int(deterministic_confidence or 0), }, "allowed_taxonomy": _taxonomy_payload(natures), } provider = get_accounting_semantic_provider() decision = AccountingAISemanticDecision( tenant_id=tenant_id, client_id=client_id, source_type=source_type, source_record_id=source_record_id, source_fingerprint=_source_fingerprint(source_type, source_record_id, payload), provider_name=provider.provider_name, model_name=provider.model_name(), prompt_version=PROMPT_VERSION, deterministic_nature_id=deterministic_nature_id, deterministic_confidence=int(deterministic_confidence or 0), context_json=json.dumps(payload, ensure_ascii=False), status="running", created_by_user_id=created_by_user_id, ) db.add(decision) db.commit() db.refresh(decision) try: result = provider.classify( system_text=SYSTEM_TEXT, user_payload=payload, schema=_schema(list(nature_by_code)), ) data = result.data nature = nature_by_code.get(_s(data.get("nature_code"))) if not nature: raise AccountingAIProviderError("AI returned a nature outside the active ERP taxonomy.") ai_conf = max(0, min(100, _i(data.get("confidence")))) agreement = bool(deterministic and deterministic.id == nature.id) combined = _combined_confidence( ai_conf=ai_conf, deterministic_conf=int(deterministic_confidence or 0), profile_conf=int(profile.get("profile_confidence") or 0), agreement=agreement, ) decision.provider_name = result.provider decision.model_name = result.model decision.ai_nature_id = nature.id decision.ai_nature_code = nature.code decision.ai_confidence = ai_conf decision.combined_confidence = combined decision.capital_revenue = _s(data.get("capital_revenue")) decision.business_personal = _s(data.get("business_personal")) or "business" decision.concise_reason = _s(data.get("reason"))[:1000] decision.alternatives_json = json.dumps(data.get("alternatives") or [], ensure_ascii=False) decision.review_flags_json = json.dumps(data.get("review_flags") or [], ensure_ascii=False) decision.raw_response_json = json.dumps(result.raw, ensure_ascii=False)[:20000] decision.input_tokens = int(result.input_tokens or 0) decision.output_tokens = int(result.output_tokens or 0) decision.latency_ms = int(result.latency_ms or 0) decision.status = "completed" db.add(decision) db.commit() db.refresh(decision) return decision, nature except Exception as exc: decision.status = "failed" decision.error_message = str(exc)[:4000] db.add(decision) db.commit() raise def ai_assist_purchase(db, *, row: AccountingGSTR2BPurchase, user_id: int): if row.review_status == "reviewed": raise ValueError("This purchase is already reviewed.") transaction = { "supplier_name": row.supplier_name, "supplier_gstin": row.supplier_gstin, "invoice_number": row.invoice_number, "invoice_date": row.invoice_date, "document_type": row.document_type, "invoice_type": row.invoice_type, "hsn_code": row.hsn_code, "description": row.description_text or "", "taxable_value": row.taxable_value, "invoice_value": row.invoice_value, "place_of_supply": row.place_of_supply, "reverse_charge": row.reverse_charge, "itc_availability": row.itc_availability, } decision, nature = _run( db, tenant_id=row.tenant_id, client_id=row.client_id, source_type="gstr2b", source_record_id=row.id, deterministic_nature_id=row.suggested_nature_id, deterministic_confidence=row.suggested_confidence, transaction=transaction, created_by_user_id=user_id, ) # AI may strengthen/change the accounting nature, but it never invents a ledger. # Preserve any existing mapped ledger only when the nature did not change. if row.suggested_nature_id != nature.id: row.suggested_ledger_name = "" row.suggested_nature_id = nature.id row.suggested_confidence = decision.combined_confidence existing = [] try: existing = json.loads(row.suggestion_explanation_json or "[]") except Exception: pass existing.append( f"AI semantic fallback ({decision.model_name}): {decision.concise_reason} " f"Combined confidence {decision.combined_confidence}%." ) row.suggestion_explanation_json = json.dumps(existing[-8:], ensure_ascii=False) row.review_status = "suggested" if decision.combined_confidence >= 60 else "review_required" decision.applied_to_source = True db.add(row) db.add(decision) db.commit() return decision def ai_assist_bank(db, *, tx, user_id: int): if tx.review_status == "reviewed": raise ValueError("This bank transaction is already reviewed.") if _s(getattr(tx, "contra_pair_id", "")): raise ValueError("Matched inter-bank contra does not require AI classification.") transaction = { "bank_name": tx.bank_name, "account_number": tx.account_number, "transaction_date": tx.transaction_date, "direction": tx.direction, "debit": tx.debit, "credit": tx.credit, "amount": tx.amount, "narration": tx.narration, "reference_no": tx.reference_no, "transfer_reference": tx.transfer_reference, "detected_party": tx.auto_party, "analyzer_category": tx.analyzer_category, "analyzer_nature": tx.analyzer_nature, } decision, nature = _run( db, tenant_id=tx.tenant_id, client_id=tx.client_id, source_type="bank", source_record_id=tx.id, deterministic_nature_id=tx.suggested_nature_id, deterministic_confidence=tx.suggested_confidence, transaction=transaction, created_by_user_id=user_id, ) if tx.suggested_nature_id != nature.id: tx.suggested_ledger_name = "" tx.suggested_nature_id = nature.id tx.suggested_confidence = decision.combined_confidence reasons = [] try: reasons = json.loads(tx.suggestion_reason_json or "[]") except Exception: pass reasons.append( f"AI semantic fallback ({decision.model_name}): {decision.concise_reason} " f"Combined confidence {decision.combined_confidence}%." ) tx.suggestion_reason_json = json.dumps(reasons[-8:], ensure_ascii=False) decision.applied_to_source = True db.add(tx) db.add(decision) db.commit() return decision def mark_review_outcome( db, *, tenant_id: int, source_type: str, source_record_id: int, final_nature_id: int | None, final_ledger_name: str, user_id: int, ): decision = latest_decision( db, tenant_id=tenant_id, source_type=source_type, source_record_id=source_record_id, ) if not decision or decision.status != "completed": return None decision.final_nature_id = final_nature_id decision.final_ledger_name = _s(final_ledger_name) decision.suggestion_accepted = bool( final_nature_id and decision.ai_nature_id and int(final_nature_id) == int(decision.ai_nature_id) ) decision.reviewed_by_user_id = user_id decision.reviewed_at_utc = _utcnow() db.add(decision) db.commit() return decision def usage_summary(db, *, tenant_id: int, client_id: int | None = None): where = [AccountingAISemanticDecision.tenant_id == int(tenant_id)] if client_id: where.append(AccountingAISemanticDecision.client_id == int(client_id)) total = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(*where)) or 0) completed = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where( *where, AccountingAISemanticDecision.status == "completed" )) or 0) accepted = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where( *where, AccountingAISemanticDecision.suggestion_accepted.is_(True) )) or 0) corrected = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where( *where, AccountingAISemanticDecision.suggestion_accepted.is_(False) )) or 0) input_tokens = int(db.scalar(select(func.coalesce(func.sum(AccountingAISemanticDecision.input_tokens), 0)).where(*where)) or 0) output_tokens = int(db.scalar(select(func.coalesce(func.sum(AccountingAISemanticDecision.output_tokens), 0)).where(*where)) or 0) return { "total": total, "completed": completed, "accepted": accepted, "corrected": corrected, "reviewed": accepted + corrected, "acceptance_rate": round((accepted * 100 / (accepted + corrected)), 1) if (accepted + corrected) else 0, "input_tokens": input_tokens, "output_tokens": output_tokens, } def recent_decisions(db, *, tenant_id: int, client_id: int | None = None, limit: int = 100): stmt = select(AccountingAISemanticDecision).where( AccountingAISemanticDecision.tenant_id == int(tenant_id) ) if client_id: stmt = stmt.where(AccountingAISemanticDecision.client_id == int(client_id)) return list(db.execute( stmt.order_by(AccountingAISemanticDecision.id.desc()).limit(max(1, min(500, int(limit)))) ).scalars().all())