461 lines
17 KiB
Python
461 lines
17 KiB
Python
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import os
|
|
from datetime import datetime, timezone
|
|
from typing import Any
|
|
|
|
from sqlalchemy import func, select
|
|
|
|
from app.modules.accounting.ai_models import AccountingAISemanticDecision
|
|
from app.modules.accounting.ai_provider import AccountingAIProviderError, get_accounting_semantic_provider
|
|
from app.modules.accounting.gstr2b_models import AccountingGSTR2BPurchase
|
|
from app.modules.accounting.taxonomy_models import AccountingNature
|
|
from app.modules.clients.models import Client, ClientBusinessProfile
|
|
|
|
|
|
PROMPT_VERSION = "phase12-v1"
|
|
DEFAULT_AI_GATE = 90
|
|
|
|
|
|
def _utcnow():
|
|
return datetime.now(timezone.utc)
|
|
|
|
|
|
def _s(value) -> str:
|
|
return str(value or "").strip()
|
|
|
|
|
|
def _i(value, default=0) -> int:
|
|
try:
|
|
return int(round(float(value)))
|
|
except Exception:
|
|
return int(default)
|
|
|
|
|
|
def ai_gate() -> int:
|
|
return max(50, min(99, _i(os.getenv("ACCOUNTING_AI_GATE_CONFIDENCE", DEFAULT_AI_GATE), DEFAULT_AI_GATE)))
|
|
|
|
|
|
def provider_status() -> dict[str, Any]:
|
|
try:
|
|
provider = get_accounting_semantic_provider()
|
|
return {
|
|
"provider": provider.provider_name,
|
|
"model": provider.model_name(),
|
|
"configured": bool(provider.is_configured()),
|
|
"gate": ai_gate(),
|
|
}
|
|
except Exception as exc:
|
|
return {"provider": "", "model": "", "configured": False, "gate": ai_gate(), "error": str(exc)}
|
|
|
|
|
|
def _profile_context(db, client_id: int) -> dict[str, Any]:
|
|
client = db.get(Client, int(client_id))
|
|
profile = db.execute(
|
|
select(ClientBusinessProfile).where(ClientBusinessProfile.client_id == int(client_id))
|
|
).scalar_one_or_none()
|
|
|
|
data = {
|
|
"client_name": _s(getattr(client, "client_name", "")),
|
|
"trade_name": _s(getattr(client, "trade_name", "")),
|
|
"primary_industry": "",
|
|
"primary_business_activity": "",
|
|
"secondary_business_activities": "",
|
|
"business_model": "",
|
|
"main_products": "",
|
|
"main_services": "",
|
|
"inventory_maintained": None,
|
|
"project_job_based": None,
|
|
"capital_intensive": None,
|
|
"vehicle_intensive": None,
|
|
"profile_status": "",
|
|
"profile_confidence": 0,
|
|
}
|
|
if profile:
|
|
for key in (
|
|
"primary_industry", "primary_business_activity", "secondary_business_activities",
|
|
"business_model", "main_products", "main_services", "inventory_maintained",
|
|
"project_job_based", "capital_intensive", "vehicle_intensive", "profile_status",
|
|
):
|
|
data[key] = getattr(profile, key, None)
|
|
data["profile_confidence"] = int(getattr(profile, "confidence_score", 0) or 0)
|
|
return data
|
|
|
|
|
|
def _taxonomy(db, tenant_id: int) -> list[AccountingNature]:
|
|
return list(db.execute(
|
|
select(AccountingNature).where(
|
|
AccountingNature.tenant_id == int(tenant_id),
|
|
AccountingNature.is_active.is_(True),
|
|
AccountingNature.is_posting_nature.is_(True),
|
|
).order_by(AccountingNature.sort_order, AccountingNature.name)
|
|
).scalars().all())
|
|
|
|
|
|
def _taxonomy_payload(rows: list[AccountingNature]) -> list[dict[str, Any]]:
|
|
return [{
|
|
"code": row.code,
|
|
"name": row.name,
|
|
"group": row.classification_group,
|
|
"capital_revenue": row.capital_revenue,
|
|
"description": _s(row.description)[:500],
|
|
} for row in rows]
|
|
|
|
|
|
def _schema(codes: list[str]) -> dict[str, Any]:
|
|
return {
|
|
"type": "object",
|
|
"properties": {
|
|
"nature_code": {"type": "string", "enum": codes},
|
|
"confidence": {"type": "integer", "minimum": 0, "maximum": 100},
|
|
"capital_revenue": {"type": "string", "enum": ["capital", "revenue", "uncertain"]},
|
|
"business_personal": {"type": "string", "enum": ["business", "personal", "uncertain"]},
|
|
"reason": {"type": "string", "maxLength": 500},
|
|
"alternatives": {
|
|
"type": "array",
|
|
"maxItems": 3,
|
|
"items": {
|
|
"type": "object",
|
|
"properties": {
|
|
"nature_code": {"type": "string", "enum": codes},
|
|
"confidence": {"type": "integer", "minimum": 0, "maximum": 100},
|
|
},
|
|
"required": ["nature_code", "confidence"],
|
|
"additionalProperties": False,
|
|
},
|
|
},
|
|
"review_flags": {
|
|
"type": "array",
|
|
"maxItems": 5,
|
|
"items": {"type": "string", "maxLength": 120},
|
|
},
|
|
},
|
|
"required": [
|
|
"nature_code", "confidence", "capital_revenue", "business_personal",
|
|
"reason", "alternatives", "review_flags",
|
|
],
|
|
"additionalProperties": False,
|
|
}
|
|
|
|
|
|
SYSTEM_TEXT = """You classify accounting transactions for an Indian accounting/audit ERP.
|
|
You are a semantic fallback layer, not the primary rules engine.
|
|
Choose exactly one accounting nature from ALLOWED_TAXONOMY. Never invent a nature or ledger.
|
|
Use the client business profile, supplier/party information, HSN/description, source type,
|
|
amount and deterministic evidence. Distinguish revenue expense from capital acquisition
|
|
conservatively. Flag uncertainty, possible personal expenditure, mixed-purpose transactions,
|
|
or insufficient evidence. Return only the required structured object. Do not create ledgers,
|
|
do not decide GST eligibility, and do not authorize posting to Tally."""
|
|
|
|
|
|
def _combined_confidence(*, ai_conf: int, deterministic_conf: int, profile_conf: int, agreement: bool) -> int:
|
|
ai_conf = max(0, min(100, int(ai_conf)))
|
|
deterministic_conf = max(0, min(100, int(deterministic_conf)))
|
|
profile_conf = max(0, min(100, int(profile_conf)))
|
|
value = (0.70 * ai_conf) + (0.20 * profile_conf) + (0.10 * deterministic_conf)
|
|
if agreement and deterministic_conf > 0:
|
|
value += 5
|
|
return max(0, min(99, int(round(value))))
|
|
|
|
|
|
def _source_fingerprint(source_type: str, source_record_id: int, payload: dict[str, Any]) -> str:
|
|
raw = f"{source_type}|{source_record_id}|{json.dumps(payload, sort_keys=True, ensure_ascii=False)}"
|
|
return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest()
|
|
|
|
|
|
def latest_decision(db, *, tenant_id: int, source_type: str, source_record_id: int):
|
|
return db.execute(
|
|
select(AccountingAISemanticDecision).where(
|
|
AccountingAISemanticDecision.tenant_id == int(tenant_id),
|
|
AccountingAISemanticDecision.source_type == source_type,
|
|
AccountingAISemanticDecision.source_record_id == int(source_record_id),
|
|
).order_by(AccountingAISemanticDecision.id.desc()).limit(1)
|
|
).scalar_one_or_none()
|
|
|
|
|
|
def _run(
|
|
db, *,
|
|
tenant_id: int,
|
|
client_id: int,
|
|
source_type: str,
|
|
source_record_id: int,
|
|
deterministic_nature_id: int | None,
|
|
deterministic_confidence: int,
|
|
transaction: dict[str, Any],
|
|
created_by_user_id: int,
|
|
):
|
|
gate = ai_gate()
|
|
if int(deterministic_confidence or 0) >= gate:
|
|
raise ValueError(
|
|
f"AI is not required because deterministic confidence is already "
|
|
f"{int(deterministic_confidence or 0)}% (AI gate {gate}%)."
|
|
)
|
|
|
|
natures = _taxonomy(db, tenant_id)
|
|
if not natures:
|
|
raise ValueError("Accounting taxonomy is empty. Configure Phase 4 taxonomy first.")
|
|
|
|
profile = _profile_context(db, client_id)
|
|
nature_by_code = {row.code: row for row in natures}
|
|
nature_by_id = {row.id: row for row in natures}
|
|
|
|
deterministic = nature_by_id.get(int(deterministic_nature_id)) if deterministic_nature_id else None
|
|
payload = {
|
|
"source_type": source_type,
|
|
"client_business_profile": profile,
|
|
"transaction": transaction,
|
|
"deterministic_evidence": {
|
|
"nature_code": deterministic.code if deterministic else "",
|
|
"nature_name": deterministic.name if deterministic else "",
|
|
"confidence": int(deterministic_confidence or 0),
|
|
},
|
|
"allowed_taxonomy": _taxonomy_payload(natures),
|
|
}
|
|
|
|
provider = get_accounting_semantic_provider()
|
|
decision = AccountingAISemanticDecision(
|
|
tenant_id=tenant_id,
|
|
client_id=client_id,
|
|
source_type=source_type,
|
|
source_record_id=source_record_id,
|
|
source_fingerprint=_source_fingerprint(source_type, source_record_id, payload),
|
|
provider_name=provider.provider_name,
|
|
model_name=provider.model_name(),
|
|
prompt_version=PROMPT_VERSION,
|
|
deterministic_nature_id=deterministic_nature_id,
|
|
deterministic_confidence=int(deterministic_confidence or 0),
|
|
context_json=json.dumps(payload, ensure_ascii=False),
|
|
status="running",
|
|
created_by_user_id=created_by_user_id,
|
|
)
|
|
db.add(decision)
|
|
db.commit()
|
|
db.refresh(decision)
|
|
|
|
try:
|
|
result = provider.classify(
|
|
system_text=SYSTEM_TEXT,
|
|
user_payload=payload,
|
|
schema=_schema(list(nature_by_code)),
|
|
)
|
|
data = result.data
|
|
nature = nature_by_code.get(_s(data.get("nature_code")))
|
|
if not nature:
|
|
raise AccountingAIProviderError("AI returned a nature outside the active ERP taxonomy.")
|
|
|
|
ai_conf = max(0, min(100, _i(data.get("confidence"))))
|
|
agreement = bool(deterministic and deterministic.id == nature.id)
|
|
combined = _combined_confidence(
|
|
ai_conf=ai_conf,
|
|
deterministic_conf=int(deterministic_confidence or 0),
|
|
profile_conf=int(profile.get("profile_confidence") or 0),
|
|
agreement=agreement,
|
|
)
|
|
|
|
decision.provider_name = result.provider
|
|
decision.model_name = result.model
|
|
decision.ai_nature_id = nature.id
|
|
decision.ai_nature_code = nature.code
|
|
decision.ai_confidence = ai_conf
|
|
decision.combined_confidence = combined
|
|
decision.capital_revenue = _s(data.get("capital_revenue"))
|
|
decision.business_personal = _s(data.get("business_personal")) or "business"
|
|
decision.concise_reason = _s(data.get("reason"))[:1000]
|
|
decision.alternatives_json = json.dumps(data.get("alternatives") or [], ensure_ascii=False)
|
|
decision.review_flags_json = json.dumps(data.get("review_flags") or [], ensure_ascii=False)
|
|
decision.raw_response_json = json.dumps(result.raw, ensure_ascii=False)[:20000]
|
|
decision.input_tokens = int(result.input_tokens or 0)
|
|
decision.output_tokens = int(result.output_tokens or 0)
|
|
decision.latency_ms = int(result.latency_ms or 0)
|
|
decision.status = "completed"
|
|
db.add(decision)
|
|
db.commit()
|
|
db.refresh(decision)
|
|
return decision, nature
|
|
except Exception as exc:
|
|
decision.status = "failed"
|
|
decision.error_message = str(exc)[:4000]
|
|
db.add(decision)
|
|
db.commit()
|
|
raise
|
|
|
|
|
|
def ai_assist_purchase(db, *, row: AccountingGSTR2BPurchase, user_id: int):
|
|
if row.review_status == "reviewed":
|
|
raise ValueError("This purchase is already reviewed.")
|
|
transaction = {
|
|
"supplier_name": row.supplier_name,
|
|
"supplier_gstin": row.supplier_gstin,
|
|
"invoice_number": row.invoice_number,
|
|
"invoice_date": row.invoice_date,
|
|
"document_type": row.document_type,
|
|
"invoice_type": row.invoice_type,
|
|
"hsn_code": row.hsn_code,
|
|
"description": row.description_text or "",
|
|
"taxable_value": row.taxable_value,
|
|
"invoice_value": row.invoice_value,
|
|
"place_of_supply": row.place_of_supply,
|
|
"reverse_charge": row.reverse_charge,
|
|
"itc_availability": row.itc_availability,
|
|
}
|
|
decision, nature = _run(
|
|
db,
|
|
tenant_id=row.tenant_id,
|
|
client_id=row.client_id,
|
|
source_type="gstr2b",
|
|
source_record_id=row.id,
|
|
deterministic_nature_id=row.suggested_nature_id,
|
|
deterministic_confidence=row.suggested_confidence,
|
|
transaction=transaction,
|
|
created_by_user_id=user_id,
|
|
)
|
|
|
|
# AI may strengthen/change the accounting nature, but it never invents a ledger.
|
|
# Preserve any existing mapped ledger only when the nature did not change.
|
|
if row.suggested_nature_id != nature.id:
|
|
row.suggested_ledger_name = ""
|
|
row.suggested_nature_id = nature.id
|
|
row.suggested_confidence = decision.combined_confidence
|
|
existing = []
|
|
try:
|
|
existing = json.loads(row.suggestion_explanation_json or "[]")
|
|
except Exception:
|
|
pass
|
|
existing.append(
|
|
f"AI semantic fallback ({decision.model_name}): {decision.concise_reason} "
|
|
f"Combined confidence {decision.combined_confidence}%."
|
|
)
|
|
row.suggestion_explanation_json = json.dumps(existing[-8:], ensure_ascii=False)
|
|
row.review_status = "suggested" if decision.combined_confidence >= 60 else "review_required"
|
|
decision.applied_to_source = True
|
|
db.add(row)
|
|
db.add(decision)
|
|
db.commit()
|
|
return decision
|
|
|
|
|
|
def ai_assist_bank(db, *, tx, user_id: int):
|
|
if tx.review_status == "reviewed":
|
|
raise ValueError("This bank transaction is already reviewed.")
|
|
if _s(getattr(tx, "contra_pair_id", "")):
|
|
raise ValueError("Matched inter-bank contra does not require AI classification.")
|
|
|
|
transaction = {
|
|
"bank_name": tx.bank_name,
|
|
"account_number": tx.account_number,
|
|
"transaction_date": tx.transaction_date,
|
|
"direction": tx.direction,
|
|
"debit": tx.debit,
|
|
"credit": tx.credit,
|
|
"amount": tx.amount,
|
|
"narration": tx.narration,
|
|
"reference_no": tx.reference_no,
|
|
"transfer_reference": tx.transfer_reference,
|
|
"detected_party": tx.auto_party,
|
|
"analyzer_category": tx.analyzer_category,
|
|
"analyzer_nature": tx.analyzer_nature,
|
|
}
|
|
decision, nature = _run(
|
|
db,
|
|
tenant_id=tx.tenant_id,
|
|
client_id=tx.client_id,
|
|
source_type="bank",
|
|
source_record_id=tx.id,
|
|
deterministic_nature_id=tx.suggested_nature_id,
|
|
deterministic_confidence=tx.suggested_confidence,
|
|
transaction=transaction,
|
|
created_by_user_id=user_id,
|
|
)
|
|
|
|
if tx.suggested_nature_id != nature.id:
|
|
tx.suggested_ledger_name = ""
|
|
tx.suggested_nature_id = nature.id
|
|
tx.suggested_confidence = decision.combined_confidence
|
|
reasons = []
|
|
try:
|
|
reasons = json.loads(tx.suggestion_reason_json or "[]")
|
|
except Exception:
|
|
pass
|
|
reasons.append(
|
|
f"AI semantic fallback ({decision.model_name}): {decision.concise_reason} "
|
|
f"Combined confidence {decision.combined_confidence}%."
|
|
)
|
|
tx.suggestion_reason_json = json.dumps(reasons[-8:], ensure_ascii=False)
|
|
decision.applied_to_source = True
|
|
db.add(tx)
|
|
db.add(decision)
|
|
db.commit()
|
|
return decision
|
|
|
|
|
|
def mark_review_outcome(
|
|
db, *,
|
|
tenant_id: int,
|
|
source_type: str,
|
|
source_record_id: int,
|
|
final_nature_id: int | None,
|
|
final_ledger_name: str,
|
|
user_id: int,
|
|
):
|
|
decision = latest_decision(
|
|
db,
|
|
tenant_id=tenant_id,
|
|
source_type=source_type,
|
|
source_record_id=source_record_id,
|
|
)
|
|
if not decision or decision.status != "completed":
|
|
return None
|
|
decision.final_nature_id = final_nature_id
|
|
decision.final_ledger_name = _s(final_ledger_name)
|
|
decision.suggestion_accepted = bool(
|
|
final_nature_id and decision.ai_nature_id and int(final_nature_id) == int(decision.ai_nature_id)
|
|
)
|
|
decision.reviewed_by_user_id = user_id
|
|
decision.reviewed_at_utc = _utcnow()
|
|
db.add(decision)
|
|
db.commit()
|
|
return decision
|
|
|
|
|
|
def usage_summary(db, *, tenant_id: int, client_id: int | None = None):
|
|
where = [AccountingAISemanticDecision.tenant_id == int(tenant_id)]
|
|
if client_id:
|
|
where.append(AccountingAISemanticDecision.client_id == int(client_id))
|
|
|
|
total = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(*where)) or 0)
|
|
completed = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(
|
|
*where, AccountingAISemanticDecision.status == "completed"
|
|
)) or 0)
|
|
accepted = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(
|
|
*where, AccountingAISemanticDecision.suggestion_accepted.is_(True)
|
|
)) or 0)
|
|
corrected = int(db.scalar(select(func.count(AccountingAISemanticDecision.id)).where(
|
|
*where, AccountingAISemanticDecision.suggestion_accepted.is_(False)
|
|
)) or 0)
|
|
input_tokens = int(db.scalar(select(func.coalesce(func.sum(AccountingAISemanticDecision.input_tokens), 0)).where(*where)) or 0)
|
|
output_tokens = int(db.scalar(select(func.coalesce(func.sum(AccountingAISemanticDecision.output_tokens), 0)).where(*where)) or 0)
|
|
|
|
return {
|
|
"total": total,
|
|
"completed": completed,
|
|
"accepted": accepted,
|
|
"corrected": corrected,
|
|
"reviewed": accepted + corrected,
|
|
"acceptance_rate": round((accepted * 100 / (accepted + corrected)), 1) if (accepted + corrected) else 0,
|
|
"input_tokens": input_tokens,
|
|
"output_tokens": output_tokens,
|
|
}
|
|
|
|
|
|
def recent_decisions(db, *, tenant_id: int, client_id: int | None = None, limit: int = 100):
|
|
stmt = select(AccountingAISemanticDecision).where(
|
|
AccountingAISemanticDecision.tenant_id == int(tenant_id)
|
|
)
|
|
if client_id:
|
|
stmt = stmt.where(AccountingAISemanticDecision.client_id == int(client_id))
|
|
return list(db.execute(
|
|
stmt.order_by(AccountingAISemanticDecision.id.desc()).limit(max(1, min(500, int(limit))))
|
|
).scalars().all())
|