from __future__ import annotations from pathlib import Path import re import pandas as pd from .parsers import parse_pdf from .parsers.common import infer_mode HIGH_VALUE_THRESHOLD = 50000.0 LOW_BALANCE_THRESHOLD = 1000.0 def clean_key(value): value = re.sub(r"\s+", " ", str(value or "").upper()).strip() return re.sub(r"\b\d{8,}\b", "", value) def _text(value) -> str: return re.sub(r"\s+", " ", str(value or "")).strip() def _amount(row) -> float: return float((row.get("debit") or 0) + (row.get("credit") or 0)) def _direction(row) -> str: return "Debit" if pd.notna(row.get("debit")) and float(row.get("debit") or 0) > 0 else "Credit" def _extract_counterparty(narration: str, direction: str) -> str: raw = _text(narration) upper = raw.upper() if not raw: return "Unidentified" if "CASH DEP" in upper or "CASH DEPOSIT" in upper: return "Cash Deposit" if "CASH WITHDRAW" in upper or "ATM WDL" in upper or "ATM/CASH" in upper and direction == "Debit": return "Cash Withdrawal" if "BANK CHARGE" in upper or "CHARGES" in upper or "COMMISSION" in upper: return "Bank Charges" if "INTEREST" in upper: return "Bank Interest" if "/UPI/" in upper or upper.startswith("UPI/"): parts = [p.strip() for p in re.split(r"/", raw) if p.strip()] ignored = {"UPI", "DR", "CR", "NA", "PAYMENT", "PHONEPE", "PAYTM"} for part in parts[1:]: cleaned = re.sub(r"^X+\d+$", "", part, flags=re.I).strip() if cleaned and cleaned.upper() not in ignored and not re.fullmatch(r"\d{6,}", cleaned): return cleaned[:120] return "UPI / Mobile" for marker in ("NEFT/", "RTGS/", "IMPS/", "TRANSFER TO", "TRANSFER FROM"): if marker in upper: parts = [p.strip() for p in re.split(r"/", raw) if p.strip()] for part in parts[1:]: if re.fullmatch(r"[A-Z0-9]{10,}", part.upper()): continue if re.fullmatch(r"\d+(?:\.\d+)?", part): continue if part.upper() in {"NEFT", "RTGS", "IMPS", "P2A", "P2P", "BRANCH : ATM SERVICE BRANCH"}: continue return part[:120] if "CTS-CHQ" in upper or "CLEARING" in upper: return "Cheque Clearing" return raw[:120] def _classify_row(row) -> tuple[str, str]: narration = _text(row.get("narration")) upper = narration.upper() direction = row.get("direction") or _direction(row) mode = str(row.get("mode") or "Other") if "CASH DEP" in upper or "CASH DEPOSIT" in upper: return "Cash Deposit / Cash Sales", "Cash Deposit" if "CASH WITHDRAW" in upper or "ATM WDL" in upper or ("ATM/CASH" in upper and direction == "Debit"): return "Cash Withdrawal", "Cash Withdrawal" if any(token in upper for token in ("BANK CHARGE", "SERVICE CHARGE", "COMMISSION CHARGE", "IMPS COMMISSION", "SMS CHARGE")): return "Bank Charges", "Bank Charges" if "INTEREST" in upper: return ("Interest Paid" if direction == "Debit" else "Interest Received"), "Bank Interest" if any(token in upper for token in ("GST PAYMENT", "GST PMT", "CPIN", "GSTIN")) and direction == "Debit": return "GST Payment", _extract_counterparty(narration, direction) if any(token in upper for token in ("INCOME TAX", "ITNS", "TDS", "OLTAS", "NSDL TAX")) and direction == "Debit": return "Tax / TDS Payment", _extract_counterparty(narration, direction) if any(token in upper for token in ("SALARY", "PAYROLL")) and direction == "Debit": return "Salary Payment", _extract_counterparty(narration, direction) if "RENT" in upper and direction == "Debit": return "Rent Expense", _extract_counterparty(narration, direction) if any(token in upper for token in ("ELECTRICITY", "TANGEDCO", "EB BILL")) and direction == "Debit": return "Electricity Expense", _extract_counterparty(narration, direction) if any(token in upper for token in ("PHONEPE", "PAYTM", "RAZORPAY", "ONE 97", "PAYMENT AGGREGATOR", "ESCROW")) and direction == "Credit": return "Payment Aggregator Settlement", _extract_counterparty(narration, direction) if "REFUND" in upper or "REVERSAL" in upper: return "Refund / Reversal", _extract_counterparty(narration, direction) if any(token in upper for token in ("SELF", "OWN ACCOUNT", "SELF TRANSFER")) or "TRANSFER FROM" in upper and "MOBILE TRANSFER" in upper: return "Self Transfer / Contra", _extract_counterparty(narration, direction) if mode == "UPI" or "/UPI/" in upper or upper.startswith("UPI/"): return ("UPI Payment / Expense" if direction == "Debit" else "UPI Customer Receipt"), _extract_counterparty(narration, direction) if mode == "NEFT" or "NEFT" in upper: return ("NEFT Payment" if direction == "Debit" else "NEFT Receipt"), _extract_counterparty(narration, direction) if mode == "RTGS" or "RTGS" in upper: return ("RTGS Payment" if direction == "Debit" else "RTGS Receipt"), _extract_counterparty(narration, direction) if mode == "IMPS" or "IMPS" in upper: return ("IMPS Payment" if direction == "Debit" else "IMPS Receipt"), _extract_counterparty(narration, direction) if "CTS-CHQ" in upper or "CLEARING" in upper or mode == "Cheque": return ("Cheque Payment" if direction == "Debit" else "Cheque Deposit"), "Cheque Clearing" if direction == "Debit": return "Other Bank Payment / Review", _extract_counterparty(narration, direction) return "Other Bank Receipt / Review", _extract_counterparty(narration, direction) def _review_note(row) -> str: notes: list[str] = [] amount = float(row.get("amount") or 0) balance = row.get("balance") category = str(row.get("category") or "") counterparty = str(row.get("counterparty") or "") if amount >= HIGH_VALUE_THRESHOLD: notes.append(f"High value transaction >= {HIGH_VALUE_THRESHOLD:,.0f}") if category == "Cash Deposit / Cash Sales": notes.append("Cash deposit - verify cash book/source") elif category == "Cash Withdrawal": notes.append("Cash withdrawal - verify cash book/use") if category.endswith("/ Review") or counterparty == "Unidentified": notes.append("Narration requires manual classification") if amount >= 10000 and amount % 10000 == 0: notes.append("Round/high amount - verify nature") if pd.notna(balance) and float(balance) < LOW_BALANCE_THRESHOLD: notes.append("Low balance after transaction") if bool(row.get("exact_duplicate")): notes.append("Exact duplicate / overlapping statement row") elif bool(row.get("possible_duplicate")): notes.append("Possible duplicate - review before exclusion") return "; ".join(dict.fromkeys(notes)) def enrich(df, classification_enabled: bool = True): if df.empty: return df x = df.copy() x["mode"] = x["narration"].map(infer_mode) x["amount"] = x["debit"].fillna(0) + x["credit"].fillna(0) x["direction"] = x.apply(_direction, axis=1) x["narration_key"] = x["narration"].map(clean_key) x["exact_key"] = x.apply( lambda r: f"{r.transaction_date}|{r.value_date}|{r.debit}|{r.credit}|{r.balance}|{clean_key(r.narration)}", axis=1, ) x["exact_duplicate"] = x.duplicated("exact_key", keep=False) x["possible_key"] = x.apply( lambda r: f"{r.transaction_date}|{r.direction}|{r.amount:.2f}|{r.narration_key}", axis=1 ) x["possible_duplicate"] = x.duplicated("possible_key", keep=False) & ~x["exact_duplicate"] if classification_enabled: classified = x.apply(_classify_row, axis=1, result_type="expand") x["category"] = classified[0] x["counterparty"] = classified[1] x["review_note"] = x.apply(_review_note, axis=1) else: x["category"] = "Classification disabled" x["counterparty"] = x["narration"].map(lambda value: _text(value)[:120] or "Unidentified") x["review_note"] = x.apply(_review_note, axis=1) return x def analyze_files(paths, customer_override="", account_override="", bank_hint="auto", classification_enabled=True): metas = [] frames = [] for path in paths: meta, df = parse_pdf(path, bank_hint=bank_hint) if customer_override: meta.customer_name = customer_override df["customer_name"] = customer_override if account_override: meta.account_number = account_override df["account_number"] = account_override metas.append(meta) frames.append(df) all_df = enrich(pd.concat(frames, ignore_index=True) if frames else pd.DataFrame(), classification_enabled) unique_df = all_df.drop_duplicates("exact_key", keep="first").copy() if not all_df.empty else all_df.copy() return metas, all_df, unique_df def reconcile(metas, all_df): rows = [] for meta in metas: data = all_df[all_df.source_file.eq(meta.source_file)] if not all_df.empty else pd.DataFrame() extracted_debit = float(data.debit.sum()) if not data.empty else 0 extracted_credit = float(data.credit.sum()) if not data.empty else 0 last_balance = float(data.balance.dropna().iloc[-1]) if not data.empty and data.balance.notna().any() else None debit_diff = None if meta.total_debit is None else round(extracted_debit - meta.total_debit, 2) credit_diff = None if meta.total_credit is None else round(extracted_credit - meta.total_credit, 2) closing_diff = None if meta.closing_balance is None or last_balance is None else round(last_balance - meta.closing_balance, 2) rows.append( { **meta.to_dict(), "extracted_transactions": len(data), "extracted_debit": extracted_debit, "debit_difference": debit_diff, "extracted_credit": extracted_credit, "credit_difference": credit_diff, "extracted_closing_balance": last_balance, "closing_difference": closing_diff, "status": "Reconciled" if all(value in (None, 0, 0.0) for value in (debit_diff, credit_diff, closing_diff)) else "Review", } ) return pd.DataFrame(rows) def monthly_summary(df): if df.empty: return pd.DataFrame() x = df.copy() x["month"] = x.transaction_date.dt.to_period("M").astype(str) summary = x.groupby("month", dropna=False).agg( transaction_count=("amount", "size"), opening_balance=("balance", "first"), total_debit=("debit", "sum"), total_credit=("credit", "sum"), closing_balance=("balance", "last"), ).reset_index() summary["net_movement"] = summary.total_credit - summary.total_debit return summary def mode_summary(df): if df.empty: return pd.DataFrame() return df.groupby(["mode", "direction"], dropna=False).agg(transaction_count=("amount", "size"), amount=("amount", "sum")).reset_index() def category_summary(df): if df.empty: return pd.DataFrame() result = df.groupby("category", dropna=False).agg( transaction_count=("amount", "size"), debit=("debit", "sum"), credit=("credit", "sum") ).reset_index() result["net_credit_debit"] = result.credit - result.debit return result.sort_values(["transaction_count", "category"], ascending=[False, True]) def party_summary(df): if df.empty: return pd.DataFrame() result = df.groupby(["counterparty", "category"], dropna=False).agg( transaction_count=("amount", "size"), debit=("debit", "sum"), credit=("credit", "sum") ).reset_index() result["net_credit_debit"] = result.credit - result.debit return result.sort_values(["transaction_count", "counterparty"], ascending=[False, True]) def duplicate_summary(all_df): return pd.DataFrame( [ {"check": "All extracted rows", "count": len(all_df)}, {"check": "Exact duplicate rows", "count": int(all_df.exact_duplicate.sum()) if not all_df.empty else 0}, {"check": "Possible duplicate rows", "count": int(all_df.possible_duplicate.sum()) if not all_df.empty else 0}, {"check": "Unique rows after exact deduplication", "count": int(all_df.exact_key.nunique()) if not all_df.empty else 0}, ] ) def draft_financials(df, metas): if df.empty: return pd.DataFrame(columns=["particulars", "amount", "treatment_notes", "finalisation_requirement"]) categories = df.groupby("category", dropna=False).agg(debit=("debit", "sum"), credit=("credit", "sum")) receipt_categories = ["Cash Deposit / Cash Sales", "UPI Customer Receipt", "Payment Aggregator Settlement", "NEFT Receipt", "RTGS Receipt", "IMPS Receipt", "Cheque Deposit"] payment_categories = ["UPI Payment / Expense", "NEFT Payment", "RTGS Payment", "IMPS Payment", "Cheque Payment"] business_receipts = float(sum(categories.loc[c, "credit"] for c in receipt_categories if c in categories.index)) other_receipts = float(sum(categories.loc[c, "credit"] for c in categories.index if c not in receipt_categories and categories.loc[c, "credit"] > 0)) classified_payments = float(sum(categories.loc[c, "debit"] for c in payment_categories if c in categories.index)) other_payments = float(sum(categories.loc[c, "debit"] for c in categories.index if c not in payment_categories and categories.loc[c, "debit"] > 0)) opening = next((meta.opening_balance for meta in metas if meta.opening_balance is not None), None) closing = next((meta.closing_balance for meta in reversed(metas) if meta.closing_balance is not None), None) return pd.DataFrame( [ ["Potential business receipts / collections", business_receipts, "Narration-based grouping of customer, cash and settlement credits", "Verify with GST/sales register and cash book"], ["Other / unclassified receipts", other_receipts, "May include capital, loans, contra and other receipts", "Classify with books and supporting records"], ["Classified bank payments", classified_payments, "Narration-based payment classification", "Verify expense, purchase and drawings treatment"], ["Other / unclassified bank payments", other_payments, "Requires review", "Classify before final accounts"], ["Opening bank balance", opening, "Per first uploaded statement", "Verify statement continuity"], ["Closing bank balance", closing, "Per last uploaded statement", "Agree with bank reconciliation"], ], columns=["particulars", "amount", "treatment_notes", "finalisation_requirement"], ) def _workbook_columns(df): preferred = [ "transaction_date", "value_date", "narration", "counterparty", "category", "mode", "direction", "debit", "credit", "balance", "reference_no", "bank_name", "customer_name", "account_number", "source_file", "source_page", "parser_name", "exact_duplicate", "possible_duplicate", "review_note", ] return [column for column in preferred if column in df.columns] def export_excel(output, metas, all_df, unique_df, financial_year="", selected_bank="auto", classification_enabled=True): recon = reconcile(metas, all_df) customer = next((meta.customer_name for meta in metas if meta.customer_name), "") account = next((meta.account_number for meta in metas if meta.account_number), "") banks = ", ".join(sorted({meta.bank_name for meta in metas})) exact_count = int(all_df.exact_duplicate.sum()) if not all_df.empty else 0 possible_count = int(all_df.possible_duplicate.sum()) if not all_df.empty else 0 opening = next((meta.opening_balance for meta in metas if meta.opening_balance is not None), None) closing = next((meta.closing_balance for meta in reversed(metas) if meta.closing_balance is not None), None) dashboard = pd.DataFrame( [ ["Customer / Account Holder", customer, "Extracted from statement or user override"], ["Account Number", account, "Extracted from statement or user override"], ["Bank(s)", banks, "Detected/selected parser result"], ["Financial Year", financial_year or "Not specified", "Optional user input"], ["Statements Uploaded", len(metas), "PDF files processed"], ["Rows Extracted", len(all_df), "Before exact duplicate removal"], ["Exact Duplicate Rows", exact_count, "Includes overlapping statement rows"], ["Possible Duplicate Rows", possible_count, "Requires manual review"], ["Unique Transactions", len(unique_df), "Used for summaries"], ["Opening Balance", opening, "From first available statement summary"], ["Total Debit (Unique)", float(unique_df.debit.sum()) if not unique_df.empty else 0, "After exact duplicate removal"], ["Total Credit (Unique)", float(unique_df.credit.sum()) if not unique_df.empty else 0, "After exact duplicate removal"], ["Closing Balance", closing, "From last available statement summary"], ["Classification", "Enabled" if classification_enabled else "Disabled", "Generic narration-based classification"], ], columns=["Metric", "Value", "Remarks"], ) all_export = all_df[_workbook_columns(all_df)].copy() if not all_df.empty else all_df.copy() unique_export = unique_df[_workbook_columns(unique_df)].copy() if not unique_df.empty else unique_df.copy() review_items = unique_export[unique_export["review_note"].fillna("").ne("")].copy() if not unique_export.empty else unique_export.copy() notes = pd.DataFrame( { "Assumption / Method": [ "Extraction method", "Debit/Credit identification", "Exact duplicate detection", "Possible duplicate detection", "Bank selection", "Classification", "Important limitation", "Files analysed", ], "Details": [ "Transactions are extracted using the selected bank parser or automatic parser detection.", "Debit and credit fields are taken from the bank-specific parser and reconciled to statement summaries where available.", "Exact duplicates use transaction date, value date, debit, credit, balance and normalized narration; the first occurrence is retained.", "Possible duplicates use the same date, direction, amount and normalized narration and are not removed automatically.", f"Selected option: {selected_bank or 'auto'}. Manual selection still validates that the PDF matches the selected bank format.", "Categories, counterparties and review notes are generic narration-based indicators and must be verified against books and source records.", "This workbook is an analysis aid, not a substitute for ledger, GST, inventory, receivable/payable, cash-book and supporting records.", "; ".join(meta.source_file for meta in metas), ], } ) with pd.ExcelWriter(output, engine="xlsxwriter", datetime_format="dd-mmm-yyyy") as writer: dashboard.to_excel(writer, sheet_name="Dashboard", index=False) recon.to_excel(writer, sheet_name="Statement Reconciliation", index=False) all_export.to_excel(writer, sheet_name="All Extracted Rows", index=False) unique_export.to_excel(writer, sheet_name="Unique Transactions", index=False) all_export[all_df.exact_duplicate.to_numpy()].to_excel(writer, sheet_name="Exact Duplicates", index=False) all_export[all_df.possible_duplicate.to_numpy()].to_excel(writer, sheet_name="Possible Duplicates", index=False) monthly_summary(unique_df).to_excel(writer, sheet_name="Monthly Summary", index=False) mode_summary(unique_df).to_excel(writer, sheet_name="Mode Summary", index=False) category_summary(unique_df).to_excel(writer, sheet_name="Category Summary", index=False) party_summary(unique_df).to_excel(writer, sheet_name="Party Summary", index=False) review_items.to_excel(writer, sheet_name="Review Items", index=False) draft_financials(unique_df, metas).to_excel(writer, sheet_name="Draft Financials", index=False) duplicate_summary(all_df).to_excel(writer, sheet_name="Duplicate Summary", index=False) notes.to_excel(writer, sheet_name="Assumptions", index=False) workbook = writer.book header = workbook.add_format({"bold": True, "bg_color": "#1F4E78", "font_color": "white", "border": 1}) money = workbook.add_format({"num_format": "#,##0.00"}) date_format = workbook.add_format({"num_format": "dd-mmm-yyyy"}) for name, worksheet in writer.sheets.items(): worksheet.freeze_panes(1, 0) worksheet.set_row(0, 22, header) max_col = max(0, worksheet.dim_colmax) if worksheet.dim_rowmax >= 0: worksheet.autofilter(0, 0, worksheet.dim_rowmax, max_col) worksheet.set_column(0, max_col, 18) writer.sheets["Dashboard"].set_column("A:A", 34) writer.sheets["Dashboard"].set_column("B:B", 28) writer.sheets["Dashboard"].set_column("C:C", 44) for sheet in ("All Extracted Rows", "Unique Transactions", "Exact Duplicates", "Possible Duplicates", "Review Items"): ws = writer.sheets[sheet] ws.set_column("A:B", 13, date_format) ws.set_column("C:C", 58) ws.set_column("D:E", 30) for sheet in ("Category Summary", "Party Summary", "Monthly Summary", "Mode Summary", "Draft Financials"): writer.sheets[sheet].set_column(1, max(1, writer.sheets[sheet].dim_colmax), 18, money) return output