from __future__ import annotations import json import re from difflib import SequenceMatcher from pathlib import Path from types import SimpleNamespace from sqlalchemy import or_, select from app.modules.clients.models import Client, ClientBranch, ClientBusinessUnit from app.modules.consultants.models import ClientConsultantLink from app.modules.consultants.service import get_consultant_by_user from app.modules.documents.services import ( build_document_scope, list_visible_engagements, save_uploaded_revision, user_can_upload_to_engagement, ) from app.modules.registrations.models import ClientRegistration from app.modules.services.models import ClientServiceSubscription _LEGAL_NOISE = { "M", "S", "MS", "MR", "MRS", "MISS", "PRIVATE", "PVT", "LIMITED", "LTD", "LLP", "PROPRIETOR", "PROPRIETORSHIP", "PROP", "THE", "INDIA", } def _text(value) -> str: return str(value or "").strip() def _normalise_name(value) -> str: raw = re.sub(r"[^A-Z0-9]+", " ", _text(value).upper()).strip() tokens = [token for token in raw.split() if token and token not in _LEGAL_NOISE] return " ".join(tokens) def _name_match(extracted: str, alias: str) -> tuple[bool, int, str]: left = _normalise_name(extracted) right = _normalise_name(alias) if not left or not right: return False, 0, "" if left == right: return True, 100, "exact normalized account-holder match" # Conservative containment is allowed only for reasonably descriptive names. if min(len(left), len(right)) >= 8 and (left in right or right in left): score = int(round(100 * min(len(left), len(right)) / max(len(left), len(right)))) if score >= 82: return True, max(90, score), "legal/trade-name containment match" ratio = int(round(SequenceMatcher(None, left, right).ratio() * 100)) if ratio >= 92: return True, ratio, "high-confidence legal/trade-name similarity match" return False, ratio, "" def client_owner_aliases(db, *, tenant_id: int, client_id: int) -> list[str]: client = db.get(Client, int(client_id)) if not client or client.tenant_id != int(tenant_id) or client.is_archived: raise ValueError("Selected client was not found.") values = [client.client_name, client.trade_name] for unit in db.execute( select(ClientBusinessUnit).where( ClientBusinessUnit.tenant_id == tenant_id, ClientBusinessUnit.client_id == client.id, ClientBusinessUnit.is_active.is_(True), ) ).scalars().all(): values.extend([unit.business_name, unit.trade_name]) for branch in db.execute( select(ClientBranch).where( ClientBranch.tenant_id == tenant_id, ClientBranch.client_id == client.id, ClientBranch.is_active.is_(True), ) ).scalars().all(): values.append(branch.branch_name) for registration in db.execute( select(ClientRegistration).where( ClientRegistration.tenant_id == tenant_id, ClientRegistration.client_id == client.id, ClientRegistration.status.in_(["active", "valid", "registered"]), ) ).scalars().all(): values.extend([registration.legal_name, registration.trade_name]) result = [] seen = set() for value in values: value = _text(value) key = _normalise_name(value) if value and key and key not in seen: seen.add(key) result.append(value) return result def _consultant_context(db, user, tenant_id: int): consultant = get_consultant_by_user( db, tenant_id=tenant_id, user_id=int(user.id), ) return consultant def analyzer_client_context(db, *, request, user, roles: list[str]) -> dict: """Return role-scoped clients and archive-capable engagements. Partner/Manager/Staff reuse the existing Document module visibility rules. Consultant reuses the existing ClientConsultantLink permissions. """ tenant_id = int( request.session.get("active_tenant_id") or getattr(user, "tenant_id", 0) or 0 ) role_set = set(roles or []) clients: dict[int, Client] = {} engagements: dict[int, ClientServiceSubscription] = {} if "Consultant" in role_set: consultant = _consultant_context(db, user, tenant_id) if not consultant: return {"clients": [], "engagements": [], "consultant": None} links = list(db.execute( select(ClientConsultantLink).where( ClientConsultantLink.tenant_id == tenant_id, ClientConsultantLink.consultant_id == consultant.id, ClientConsultantLink.is_active.is_(True), ClientConsultantLink.can_view_client.is_(True), ) ).scalars().all()) client_ids = {int(link.client_id) for link in links} upload_ids = { int(link.client_id) for link in links if bool(link.can_upload_documents) } if client_ids: for client in db.execute( select(Client).where( Client.tenant_id == tenant_id, Client.id.in_(client_ids), Client.is_archived.is_(False), Client.is_active.is_(True), ).order_by(Client.client_name.asc()) ).scalars().all(): clients[int(client.id)] = client if upload_ids: for engagement in db.execute( select(ClientServiceSubscription).where( ClientServiceSubscription.tenant_id == tenant_id, ClientServiceSubscription.client_id.in_(upload_ids), ClientServiceSubscription.is_active.is_(True), ).order_by( ClientServiceSubscription.financial_year.desc(), ClientServiceSubscription.id.desc(), ) ).scalars().all(): engagements[int(engagement.id)] = engagement return { "clients": list(clients.values()), "engagements": list(engagements.values()), "consultant": consultant, } # Firm Partner / Manager / Staff use the existing engagement-document scope. scope = build_document_scope(request, db, user) visible = list_visible_engagements(db, user, scope, limit=1000) for engagement in visible: client = getattr(engagement, "client", None) or db.get(Client, engagement.client_id) if client and client.is_active and not client.is_archived: clients[int(client.id)] = client if user_can_upload_to_engagement(db, user, engagement, scope): engagements[int(engagement.id)] = engagement return { "clients": sorted(clients.values(), key=lambda c: (c.client_name or "").casefold()), "engagements": list(engagements.values()), "consultant": None, } def validate_selected_client_and_engagement( db, *, request, user, roles: list[str], client_id: int | None, engagement_id: int | None, ): if not client_id: if engagement_id: raise ValueError("Select a client before selecting an engagement.") return None, None context = analyzer_client_context(db, request=request, user=user, roles=roles) clients = {int(row.id): row for row in context["clients"]} engagements = {int(row.id): row for row in context["engagements"]} client = clients.get(int(client_id)) if not client: raise ValueError("Selected client is not available in your Bank Analyzer scope.") if not engagement_id: raise ValueError( "Select an engagement document destination when a client is selected." ) engagement = engagements.get(int(engagement_id)) if not engagement or int(engagement.client_id) != int(client.id): raise ValueError( "Selected engagement is not available for document upload for this client." ) return client, engagement def validate_statement_ownership( db, *, tenant_id: int, client_id: int | None, metas: list, owner_override: str = "", ownership_confirmation: bool = False, ) -> dict: """Validate that all statements belong to one ERP client/legal/trade-name family. With an ERP client selected the check is strict. Every statement must expose an owner name or use a manually confirmed owner override, and that name must match one of the ERP legal/trade aliases. """ extracted = [] for position, meta in enumerate(metas, start=1): raw = _text( getattr(meta, "extracted_customer_name", None) or getattr(meta, "customer_name", None) ) extracted.append({ "statement_id": getattr(meta, "statement_id", f"STMT-{position:03d}"), "source_file": _text(getattr(meta, "source_file", "")), "owner_name": raw, }) if not client_id: # Preserve standalone behavior. We record the owner names for review but # do not change current storage or reject mixed standalone analyses. return { "status": "standalone", "client_id": None, "aliases": [], "statements": extracted, "message": "Standalone analysis; no ERP client ownership binding requested.", } aliases = client_owner_aliases( db, tenant_id=int(tenant_id), client_id=int(client_id), ) if not aliases: raise ValueError( "Selected client does not have a usable legal/trade name for bank ownership validation." ) override = _text(owner_override) if override: matched_override = any(_name_match(override, alias)[0] for alias in aliases) if not matched_override: raise ValueError( f"Account-holder override '{override}' does not match the selected client's " "legal/trade names." ) results = [] failures = [] for statement in extracted: owner = statement["owner_name"] or override if not owner: failures.append( f"{statement['statement_id']}: account-holder name could not be extracted. " "Use Account holder override only after verifying the statement manually." ) results.append({**statement, "matched": False, "score": 0, "matched_alias": ""}) continue best = (False, 0, "", "") for alias in aliases: matched, score, reason = _name_match(owner, alias) if score > best[1]: best = (matched, score, alias, reason) matched, score, alias, reason = best results.append({ **statement, "effective_owner_name": owner, "matched": bool(matched), "score": int(score), "matched_alias": alias, "reason": reason, }) if not matched: failures.append( f"{statement['statement_id']}: '{owner}' does not match the selected client's " "legal/trade names." ) if failures: raise ValueError( "Bank account ownership validation failed. " + " ".join(failures) ) if len(metas) > 1 and not ownership_confirmation: raise ValueError( "Confirm that all uploaded bank statements belong to the selected client " "or its approved trade/business names." ) return { "status": "confirmed", "client_id": int(client_id), "aliases": aliases, "statements": results, "message": f"Ownership confirmed for {len(results)} statement(s).", } def _upload_proxy(path: Path, content_type: str): handle = path.open("rb") return SimpleNamespace( filename=path.name, content_type=content_type, file=handle, ) def archive_analysis_to_engagement( db, *, job, engagement: ClientServiceSubscription | None, metas: list, source_paths: list[Path], output_path: Path, user, ) -> dict: if not job.client_id or not job.engagement_id or engagement is None: return {"status": "not_requested", "documents": []} archived = [] reused = [] try: reused = json.loads(getattr(job, "stored_source_versions_json", None) or "[]") except Exception: reused = [] reused_by_copied_name = { _text(row.get("copied_filename")): row for row in reused if isinstance(row, dict) and _text(row.get("copied_filename")) } meta_by_source = { Path(_text(getattr(meta, "source_file", ""))).name: meta for meta in metas } for source in source_paths: reused_row = reused_by_copied_name.get(source.name) if reused_row: archived.append( { "document_id": int(reused_row.get("document_id") or 0), "version_id": int(reused_row.get("version_id") or 0), "filename": reused_row.get("original_filename") or source.name, "type": "BANK_STATEMENT", "reused_existing_document": True, } ) continue upload = _upload_proxy(source, "application/pdf") try: meta = meta_by_source.get(source.name) bank = _text(getattr(meta, "bank_name", "")) if meta else "" account = _text(getattr(meta, "account_number", "")) if meta else "" title_bits = ["Bank Statement"] if bank: title_bits.append(bank) if account: title_bits.append(account[-6:]) document = save_uploaded_revision( db, engagement=engagement, upload_file=upload, title=" - ".join(title_bits), document_type="BANK_STATEMENT", description=( f"Bank Statement Analyzer source. Job {job.id}. " f"Ownership status: {job.ownership_status}." ), remarks="Automatically archived from Bank Statement Analyzer.", user=user, udin_required=False, ) archived.append({ "document_id": int(document.id), "filename": source.name, "type": "BANK_STATEMENT", }) finally: upload.file.close() workbook_upload = _upload_proxy( output_path, "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", ) try: document = save_uploaded_revision( db, engagement=engagement, upload_file=workbook_upload, title="Bank Statement Analyzer - Analysis Workbook", document_type="WORKING_PAPER", description=( f"Generated Bank Statement Analyzer workbook for job {job.id}; " f"includes multi-bank contra review where applicable." ), remarks="Automatically archived from Bank Statement Analyzer.", user=user, udin_required=False, ) archived.append({ "document_id": int(document.id), "filename": output_path.name, "type": "WORKING_PAPER", }) finally: workbook_upload.file.close() return { "status": "archived", "client_id": int(job.client_id), "engagement_id": int(job.engagement_id), "documents": archived, }