Fix common statement reconciliation and analyzer filenames

This commit is contained in:
A R R R Associates
2026-07-20 23:43:13 +05:30
parent 0e931411d6
commit 321f49db52
3 changed files with 93 additions and 122 deletions
+67 -106
View File
@@ -13,7 +13,7 @@ HIGH_VALUE_THRESHOLD = 50000.0
LOW_BALANCE_THRESHOLD = 1000.0 LOW_BALANCE_THRESHOLD = 1000.0
ANALYSIS_COLUMNS = [ ANALYSIS_COLUMNS = [
"transaction_date", "value_date", "narration", "debit", "credit", "balance", "statement_id", "transaction_date", "value_date", "narration", "debit", "credit", "balance",
"reference_no", "bank_name", "customer_name", "account_number", "source_file", "reference_no", "bank_name", "customer_name", "account_number", "source_file",
"source_page", "parser_name", "mode", "amount", "direction", "narration_key", "source_page", "parser_name", "mode", "amount", "direction", "narration_key",
"exact_key", "exact_duplicate", "possible_key", "possible_duplicate", "exact_key", "exact_duplicate", "possible_key", "possible_duplicate",
@@ -439,16 +439,22 @@ def analyze_files(paths, customer_override="", account_override="", bank_hint="a
if account_override: if account_override:
meta.account_number = account_override meta.account_number = account_override
df["account_number"] = account_override df["account_number"] = account_override
statement_id = f"STMT-{len(metas) + 1:03d}"
df = df.copy()
df["statement_id"] = statement_id
meta.statement_id = statement_id
metas.append(meta) metas.append(meta)
frames.append(df) frames.append(df)
combined = pd.concat(frames, ignore_index=True) if frames else pd.DataFrame() combined = pd.concat(frames, ignore_index=True) if frames else pd.DataFrame()
all_df = _ensure_analysis_columns(enrich(combined, classification_enabled)) all_df = _ensure_analysis_columns(enrich(combined, classification_enabled))
unique_df = (
all_df.drop_duplicates("exact_key", keep="first").copy() # IMPORTANT ACCOUNTING CONTROL:
if not all_df.empty # Duplicate detection is advisory only. A bank may legitimately contain two
else _ensure_analysis_columns(all_df) # transactions with the same date, amount and narration. Removing a flagged
) # row changes debit/credit totals and creates a false closing-balance
unique_df = _ensure_analysis_columns(unique_df) # difference. Therefore every extracted row remains in the accounting table;
# exact/possible duplicate flags are retained for reviewer action.
unique_df = _ensure_analysis_columns(all_df.copy())
if metas and all_df.empty: if metas and all_df.empty:
raise ValueError( raise ValueError(
"The bank statement format was identified, but no transaction rows could be extracted. " "The bank statement format was identified, but no transaction rows could be extracted. "
@@ -458,103 +464,57 @@ def analyze_files(paths, customer_override="", account_override="", bank_hint="a
def reconcile(metas, all_df): def reconcile(metas, all_df):
"""Reconcile each uploaded statement independently.
The routine never mixes rows from different source statements. Duplicate
flags remain review indicators and do not remove transactions from the
accounting totals.
"""
rows = [] rows = []
for meta in metas: for position, meta in enumerate(metas, start=1):
data = all_df[all_df.source_file.eq(meta.source_file)] if not all_df.empty else pd.DataFrame() statement_id = getattr(meta, "statement_id", f"STMT-{position:03d}")
extracted_debit = float(data.debit.sum()) if not data.empty else 0 if all_df.empty:
extracted_credit = float(data.credit.sum()) if not data.empty else 0 data = pd.DataFrame()
last_balance = float(data.balance.dropna().iloc[-1]) if not data.empty and data.balance.notna().any() else None elif "statement_id" in all_df.columns and all_df["statement_id"].notna().any():
debit_diff = None if meta.total_debit is None else round(extracted_debit - meta.total_debit, 2) data = all_df[all_df.statement_id.eq(statement_id)].copy()
credit_diff = None if meta.total_credit is None else round(extracted_credit - meta.total_credit, 2) else:
closing_diff = None if meta.closing_balance is None or last_balance is None else round(last_balance - meta.closing_balance, 2) data = all_df[all_df.source_file.eq(meta.source_file)].copy()
rows.append(
{ extracted_debit = round(float(pd.to_numeric(data.get("debit"), errors="coerce").fillna(0).sum()), 2) if not data.empty else 0.0
extracted_credit = round(float(pd.to_numeric(data.get("credit"), errors="coerce").fillna(0).sum()), 2) if not data.empty else 0.0
computed_closing = None
if meta.opening_balance is not None:
computed_closing = round(float(meta.opening_balance) + extracted_credit - extracted_debit, 2)
balance_values = pd.to_numeric(data.get("balance"), errors="coerce").dropna() if not data.empty else pd.Series(dtype=float)
printed_last_balance = float(balance_values.iloc[-1]) if not balance_values.empty else None
debit_diff = None if meta.total_debit is None else round(extracted_debit - float(meta.total_debit), 2)
credit_diff = None if meta.total_credit is None else round(extracted_credit - float(meta.total_credit), 2)
closing_diff = None if meta.closing_balance is None or computed_closing is None else round(computed_closing - float(meta.closing_balance), 2)
last_balance_diff = None if meta.closing_balance is None or printed_last_balance is None else round(printed_last_balance - float(meta.closing_balance), 2)
discontinuities = int((pd.to_numeric(data.get("movement_difference"), errors="coerce").abs() > 0.01).sum()) if not data.empty and "movement_difference" in data.columns else 0
checks = (debit_diff, credit_diff, closing_diff, last_balance_diff)
status = "Reconciled" if all(value is None or abs(float(value)) <= 0.01 for value in checks) else "Review"
rows.append({
"statement_id": statement_id,
**meta.to_dict(), **meta.to_dict(),
"extracted_transactions": len(data), "extracted_transactions": len(data),
"extracted_debit": extracted_debit, "extracted_debit": extracted_debit,
"debit_difference": debit_diff, "debit_difference": debit_diff,
"extracted_credit": extracted_credit, "extracted_credit": extracted_credit,
"credit_difference": credit_diff, "credit_difference": credit_diff,
"extracted_closing_balance": last_balance, "computed_closing_balance": computed_closing,
"closing_difference": closing_diff, "printed_last_running_balance": printed_last_balance,
"status": "Reconciled" if all(value in (None, 0, 0.0) for value in (debit_diff, credit_diff, closing_diff)) else "Review", "statement_closing_difference": closing_diff,
} "last_running_balance_difference": last_balance_diff,
) "balance_discontinuities": discontinuities,
"status": status,
})
return pd.DataFrame(rows) return pd.DataFrame(rows)
def _account_portfolio_summary(metas, transactions):
"""Build account-level and portfolio reconciliation values.
Running-balance delta validates each statement row. This helper handles a
separate issue: a workbook may contain multiple bank accounts or multiple
statement segments for the same account. Opening and closing balances must
therefore be selected per account, not from the first and last metadata row
across the entire upload.
"""
def _key(meta):
account = str(meta.account_number or '').strip()
return (str(meta.bank_name or '').strip(), account or str(meta.source_file or '').strip())
groups = {}
for position, meta in enumerate(metas):
groups.setdefault(_key(meta), []).append((position, meta))
rows = []
for (bank_name, account_key), items in groups.items():
def _sort_key(item):
position, meta = item
start = pd.to_datetime(meta.period_from, errors='coerce')
end = pd.to_datetime(meta.period_to, errors='coerce')
start_key = start if pd.notna(start) else pd.Timestamp.max
end_key = end if pd.notna(end) else pd.Timestamp.max
return (start_key, end_key, position)
ordered = sorted(items, key=_sort_key)
first_meta = ordered[0][1]
last_meta = ordered[-1][1]
opening = next((m.opening_balance for _, m in ordered if m.opening_balance is not None), 0.0) or 0.0
closing = next((m.closing_balance for _, m in reversed(ordered) if m.closing_balance is not None), 0.0) or 0.0
account_number = str(first_meta.account_number or '').strip()
customer_name = next((str(m.customer_name).strip() for _, m in ordered if str(m.customer_name or '').strip()), '')
if transactions is None or transactions.empty:
data = pd.DataFrame()
else:
mask = transactions.get('bank_name', pd.Series('', index=transactions.index)).fillna('').astype(str).eq(bank_name)
if account_number:
mask &= transactions.get('account_number', pd.Series('', index=transactions.index)).fillna('').astype(str).eq(account_number)
else:
source_files = {str(m.source_file or '') for _, m in ordered}
mask &= transactions.get('source_file', pd.Series('', index=transactions.index)).fillna('').astype(str).isin(source_files)
data = transactions.loc[mask]
debit = float(pd.to_numeric(data.get('debit'), errors='coerce').fillna(0).sum()) if not data.empty else 0.0
credit = float(pd.to_numeric(data.get('credit'), errors='coerce').fillna(0).sum()) if not data.empty else 0.0
computed = round(float(opening) + credit - debit, 2)
difference = round(computed - float(closing), 2)
rows.append({
'bank_name': bank_name,
'customer_name': customer_name,
'account_number': account_number or account_key,
'statement_segments': len(ordered),
'period_from': next((m.period_from for _, m in ordered if m.period_from), ''),
'period_to': next((m.period_to for _, m in reversed(ordered) if m.period_to), ''),
'opening_balance': round(float(opening), 2),
'total_debit': round(debit, 2),
'total_credit': round(credit, 2),
'computed_closing_balance': computed,
'statement_closing_balance': round(float(closing), 2),
'difference': difference,
'status': 'Reconciled' if abs(difference) <= 0.01 else 'Review',
})
frame = pd.DataFrame(rows)
portfolio_opening = float(frame['opening_balance'].sum()) if not frame.empty else 0.0
portfolio_closing = float(frame['statement_closing_balance'].sum()) if not frame.empty else 0.0
return frame, round(portfolio_opening, 2), round(portfolio_closing, 2)
def monthly_summary(df): def monthly_summary(df):
if df.empty: if df.empty:
return pd.DataFrame() return pd.DataFrame()
@@ -618,7 +578,8 @@ def draft_financials(df, metas):
other_receipts = float(sum(categories.loc[c, "credit"] for c in categories.index if c not in receipt_categories and categories.loc[c, "credit"] > 0)) other_receipts = float(sum(categories.loc[c, "credit"] for c in categories.index if c not in receipt_categories and categories.loc[c, "credit"] > 0))
classified_payments = float(sum(categories.loc[c, "debit"] for c in payment_categories if c in categories.index)) classified_payments = float(sum(categories.loc[c, "debit"] for c in payment_categories if c in categories.index))
other_payments = float(sum(categories.loc[c, "debit"] for c in categories.index if c not in payment_categories and categories.loc[c, "debit"] > 0)) other_payments = float(sum(categories.loc[c, "debit"] for c in categories.index if c not in payment_categories and categories.loc[c, "debit"] > 0))
_account_recon, opening, closing = _account_portfolio_summary(metas, df) opening = next((meta.opening_balance for meta in metas if meta.opening_balance is not None), None)
closing = next((meta.closing_balance for meta in reversed(metas) if meta.closing_balance is not None), None)
return pd.DataFrame( return pd.DataFrame(
[ [
["Potential business receipts / collections", business_receipts, "Narration-based grouping of customer, cash and settlement credits", "Verify with GST/sales register and cash book"], ["Potential business receipts / collections", business_receipts, "Narration-based grouping of customer, cash and settlement credits", "Verify with GST/sales register and cash book"],
@@ -634,7 +595,7 @@ def draft_financials(df, metas):
def _workbook_columns(df): def _workbook_columns(df):
preferred = [ preferred = [
"transaction_date", "value_date", "narration", "transfer_bank_code", "transfer_reference", "statement_id", "transaction_date", "value_date", "narration", "transfer_bank_code", "transfer_reference",
"transfer_comment", "auto_party", "party_match_method", "party_match_confidence", "transfer_comment", "auto_party", "party_match_method", "party_match_confidence",
"auto_category", "auto_nature", "auto_group", "matched_rule_id", "matched_keyword", "auto_category", "auto_nature", "auto_group", "matched_rule_id", "matched_keyword",
"suggested_ledger", "rule_confidence", "review_required", "mode", "direction", "debit", "credit", "suggested_ledger", "rule_confidence", "review_required", "mode", "direction", "debit", "credit",
@@ -700,12 +661,13 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b
all_df = _ensure_analysis_columns(all_df) all_df = _ensure_analysis_columns(all_df)
unique_df = _ensure_analysis_columns(unique_df) unique_df = _ensure_analysis_columns(unique_df)
recon = reconcile(metas, all_df) recon = reconcile(metas, all_df)
account_recon, opening, statement_closing = _account_portfolio_summary(metas, unique_df) customer = next((meta.customer_name for meta in metas if meta.customer_name), "")
customer_names = sorted({str(meta.customer_name).strip() for meta in metas if str(meta.customer_name or '').strip()}) account = next((meta.account_number for meta in metas if meta.account_number), "")
account_numbers = sorted({str(meta.account_number).strip() for meta in metas if str(meta.account_number or '').strip()})
customer = customer_names[0] if len(customer_names) == 1 else ("Multiple Clients" if customer_names else "")
account = account_numbers[0] if len(account_numbers) == 1 else (f"Multiple Accounts ({len(account_numbers)})" if account_numbers else "")
banks = ", ".join(sorted({meta.bank_name for meta in metas})) banks = ", ".join(sorted({meta.bank_name for meta in metas}))
# Sum each statement's opening and closing independently. This is an
# accounting aggregate, not a first-statement/last-statement shortcut.
opening = round(sum(float(meta.opening_balance) for meta in metas if meta.opening_balance is not None), 2)
statement_closing = round(sum(float(meta.closing_balance) for meta in metas if meta.closing_balance is not None), 2)
tx = unique_df[_workbook_columns(unique_df)].copy() if not unique_df.empty else pd.DataFrame(columns=_workbook_columns(unique_df)) tx = unique_df[_workbook_columns(unique_df)].copy() if not unique_df.empty else pd.DataFrame(columns=_workbook_columns(unique_df))
tx.insert(0, "row_id", range(1, len(tx) + 1)) tx.insert(0, "row_id", range(1, len(tx) + 1))
@@ -748,7 +710,7 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b
"PDF extraction, transaction reconstruction, duplicate detection, references, comments and automatic classification are performed in Python.", "PDF extraction, transaction reconstruction, duplicate detection, references, comments and automatic classification are performed in Python.",
"Dashboard, category, party, category-party, monthly and trial-balance outputs are formula-driven and recalculate after manual overrides.", "Dashboard, category, party, category-party, monthly and trial-balance outputs are formula-driven and recalculate after manual overrides.",
"Bank parsers accept multiple supported date layouts and validate transaction dates against statement periods where available.", "Bank parsers accept multiple supported date layouts and validate transaction dates against statement periods where available.",
"Exact and possible duplicate flags are Python-generated. Formula summaries exclude exact duplicates because Transaction Classification contains one retained row per exact group.", "Exact and possible duplicate flags are Python-generated review indicators. No transaction is automatically removed from accounting totals or closing-balance reconciliation.",
"Party names are normalized using regex cleanup and conservative fuzzy/partial matching. Review suggested groupings before finalisation.", "Party names are normalized using regex cleanup and conservative fuzzy/partial matching. Review suggested groupings before finalisation.",
"NEFT, RTGS and IMPS bank codes, references and narration comments are preserved in dedicated columns.", "NEFT, RTGS and IMPS bank codes, references and narration comments are preserved in dedicated columns.",
"Phase 1 classification uses deterministic global regex rules. Matched Rule ID, keyword, suggested ledger, confidence and review requirement are preserved for audit traceability.", "Phase 1 classification uses deterministic global regex rules. Matched Rule ID, keyword, suggested ledger, confidence and review requirement are preserved for audit traceability.",
@@ -789,7 +751,6 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b
# Raw and Python-analysis sheets. # Raw and Python-analysis sheets.
recon.to_excel(writer, sheet_name="Statement Reconciliation", index=False) recon.to_excel(writer, sheet_name="Statement Reconciliation", index=False)
account_recon.to_excel(writer, sheet_name="Account Reconciliation", index=False)
raw_export.to_excel(writer, sheet_name="All Extracted Rows", index=False) raw_export.to_excel(writer, sheet_name="All Extracted Rows", index=False)
exact_export.to_excel(writer, sheet_name="Exact Duplicates", index=False) exact_export.to_excel(writer, sheet_name="Exact Duplicates", index=False)
possible_export.to_excel(writer, sheet_name="Possible Duplicates", index=False) possible_export.to_excel(writer, sheet_name="Possible Duplicates", index=False)
@@ -857,14 +818,14 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b
("Bank(s)", banks, "Detected/selected parser"), ("Bank(s)", banks, "Detected/selected parser"),
("Financial Year", financial_year or "Not specified", "User input"), ("Financial Year", financial_year or "Not specified", "User input"),
("Statements Uploaded", len(metas), "PDF files processed"), ("Statements Uploaded", len(metas), "PDF files processed"),
("Transactions Retained", "=ROWS(tblTransactions[row_id])", "Exact-deduplicated transaction table"), ("Transactions Analysed", "=ROWS(tblTransactions[row_id])", "All extracted rows retained; duplicate flags are review-only"),
("Exact Duplicate Rows", f'=COUNTIF(\'All Extracted Rows\'!${exact_excel_col}:${exact_excel_col},TRUE)', "Python detected"), ("Exact Duplicate Rows", f'=COUNTIF(\'All Extracted Rows\'!${exact_excel_col}:${exact_excel_col},TRUE)', "Python detected"),
("Possible Duplicate Rows", f'=COUNTIF(\'All Extracted Rows\'!${possible_excel_col}:${possible_excel_col},TRUE)', "Python detected"), ("Possible Duplicate Rows", f'=COUNTIF(\'All Extracted Rows\'!${possible_excel_col}:${possible_excel_col},TRUE)', "Python detected"),
("Opening Signed Balance", opening, "CR positive / DR negative"), ("Combined Opening Balance", opening, "Sum of each statement opening; statement-wise details in reconciliation sheet"),
("Total Debit", "=SUM(tblTransactions[debit])", "Formula-driven"), ("Total Debit", "=SUM(tblTransactions[debit])", "Formula-driven"),
("Total Credit", "=SUM(tblTransactions[credit])", "Formula-driven"), ("Total Credit", "=SUM(tblTransactions[credit])", "Formula-driven"),
("Computed Closing Balance", "=B10+B12-B11", "Opening + Credits - Debits"), ("Computed Closing Balance", "=B10+B12-B11", "Opening + Credits - Debits"),
("Statement Closing Balance", statement_closing, "Extracted from statement"), ("Combined Statement Closing", statement_closing, "Sum of each statement closing; statement-wise details in reconciliation sheet"),
("Closing Difference", "=B13-B14", "Should be zero"), ("Closing Difference", "=B13-B14", "Should be zero"),
("Review Items", '=COUNTIF(tblTransactions[review_note],"?*")+COUNTIF(tblTransactions[manual_review_note],"?*")', "Formula-driven"), ("Review Items", '=COUNTIF(tblTransactions[review_note],"?*")+COUNTIF(tblTransactions[manual_review_note],"?*")', "Formula-driven"),
] ]
@@ -8,6 +8,7 @@ import pdfplumber
@dataclass @dataclass
class StatementMeta: class StatementMeta:
statement_id: str = ""
bank_name: str = "" bank_name: str = ""
customer_name: str = "" customer_name: str = ""
account_number: str = "" account_number: str = ""
@@ -27,7 +28,7 @@ class StatementMeta:
return asdict(self) return asdict(self)
STANDARD_COLUMNS = [ STANDARD_COLUMNS = [
"transaction_date", "value_date", "narration", "reference_no", "statement_id", "transaction_date", "value_date", "narration", "reference_no",
"debit", "credit", "balance", "bank_name", "customer_name", "debit", "credit", "balance", "bank_name", "customer_name",
"account_number", "source_file", "source_page", "parser_name", "account_number", "source_file", "source_page", "parser_name",
# Internal extraction-audit fields. These are retained for diagnostics but # Internal extraction-audit fields. These are retained for diagnostics but
+18 -9
View File
@@ -60,6 +60,24 @@ def _segment(value: object, default: str = "NA") -> str:
return text_value[:80] or default return text_value[:80] or default
def _workbook_filename(metas: list) -> str:
"""Build a safe BankName_ClientName.xlsx filename for every bank."""
bank_names = [str(getattr(meta, "bank_name", "") or "").strip() for meta in metas]
client_names = [str(getattr(meta, "customer_name", "") or "").strip() for meta in metas]
account_numbers = [str(getattr(meta, "account_number", "") or "").strip() for meta in metas]
unique_banks = list(dict.fromkeys(name for name in bank_names if name))
unique_clients = list(dict.fromkeys(name for name in client_names if name))
unique_accounts = list(dict.fromkeys(name for name in account_numbers if name))
bank = unique_banks[0] if len(unique_banks) == 1 else ("Multiple_Banks" if unique_banks else "Bank")
client = unique_clients[0] if len(unique_clients) == 1 else ""
if not client:
client = unique_accounts[0] if len(unique_accounts) == 1 else "Multiple_Statements"
return f"{_segment(bank, 'Bank')}_{_segment(client, 'Client')}.xlsx"
def role_bucket(roles: Iterable[str]) -> str | None: def role_bucket(roles: Iterable[str]) -> str | None:
role_set = set(roles) role_set = set(roles)
for role, bucket in (("Partner", "Partner"), ("Manager", "Manager"), ("Branch Manager", "Manager"), ("Staff", "Staff"), ("Employee", "Staff"), ("Consultant", "Consultant")): for role, bucket in (("Partner", "Partner"), ("Manager", "Manager"), ("Branch Manager", "Manager"), ("Staff", "Staff"), ("Employee", "Staff"), ("Consultant", "Consultant")):
@@ -210,15 +228,6 @@ def _claim_jobs() -> list[str]:
db.close() db.close()
def _workbook_filename(metas) -> str:
"""Return BankName_ClientName.xlsx using filesystem-safe segments."""
banks = sorted({_segment(getattr(meta, "bank_name", ""), "Bank") for meta in metas if str(getattr(meta, "bank_name", "") or "").strip()})
customers = sorted({_segment(getattr(meta, "customer_name", ""), "Client") for meta in metas if str(getattr(meta, "customer_name", "") or "").strip()})
bank_label = banks[0] if len(banks) == 1 else ("Multi_Bank" if banks else "Bank")
client_label = customers[0] if len(customers) == 1 else ("Multiple_Clients" if customers else "Client")
return f"{bank_label}_{client_label}.xlsx"
def _process_job(job_id: str) -> None: def _process_job(job_id: str) -> None:
db = CommonSessionLocal() db = CommonSessionLocal()
try: try: