diff --git a/app/modules/bank_statement_analyzer/analyzer.py b/app/modules/bank_statement_analyzer/analyzer.py index e1af724..15275c9 100644 --- a/app/modules/bank_statement_analyzer/analyzer.py +++ b/app/modules/bank_statement_analyzer/analyzer.py @@ -683,8 +683,6 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b exact_export.to_excel(writer, sheet_name="Exact Duplicates", index=False) possible_export.to_excel(writer, sheet_name="Possible Duplicates", index=False) duplicate_summary(all_df).to_excel(writer, sheet_name="Duplicate Summary", index=False) - rules_export.to_excel(writer, sheet_name="Classification Rules", index=False) - notes.to_excel(writer, sheet_name="Assumptions", index=False) # Masters first so validation ranges exist. max_master = max(len(categories), len(parties), len(natures), len(groups), 1) @@ -888,16 +886,7 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b for name in ("Category Summary", "Party Summary", "Category Party Summary", "Trial Balance"): writer.sheets[name].set_column("A:B", 30) writer.sheets["Trial Balance"].set_column("C:E", 18, money) - writer.sheets["Assumptions"].set_column("A:A", 28) - writer.sheets["Assumptions"].set_column("B:B", 90) writer.sheets["Masters"].set_column("A:D", 34) - ws_rules = writer.sheets.get("Classification Rules") - if ws_rules: - ws_rules.freeze_panes(1, 0) - ws_rules.set_column("A:B", 18) - ws_rules.set_column("C:C", 52) - ws_rules.set_column("D:F", 32) - ws_rules.set_column("G:I", 18) - + # Formula/dropdown support only; not exposed as a visible report sheet. writer.sheets["Masters"].hide() return output diff --git a/app/modules/bank_statement_analyzer/parsers/axis.py b/app/modules/bank_statement_analyzer/parsers/axis.py index f2256a5..d03c1b0 100644 --- a/app/modules/bank_statement_analyzer/parsers/axis.py +++ b/app/modules/bank_statement_analyzer/parsers/axis.py @@ -2,7 +2,7 @@ from __future__ import annotations import re, pandas as pd from pathlib import Path from .base import * -from .common import find +from .common import DATE_TOKEN_PATTERN, date_iso, find, find_date_tokens class AxisParser(BaseParser): bank_name='Axis Bank'; parser_name='AxisParser' @@ -14,10 +14,10 @@ class AxisParser(BaseParser): m=re.search(r'Smart Statement Report\s*\n\s*([^\n]+)',text,re.I); meta.customer_name=norm(m.group(1)) if m else '' meta.account_number=find(r'Statement of Account No\s*-\s*([^\s]*)',text) meta.ifsc=find(r'IFSC:\s*([A-Z0-9]+)',text) - m=re.search(r'for period\s*\((\d{2}/\d{2}/\d{4})\s+to\s+(\d{2}/\d{2}/\d{4})\)',text,re.I) - if m: meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d') + m=re.search(rf'for period\s*\(({DATE_TOKEN_PATTERN})\s+to\s+({DATE_TOKEN_PATTERN})\)', text, re.I) + if m: meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2)) meta.opening_balance=amount(find(r'Opening Balance:\s*INR\s*([\d,]+\.\d{2})',text)) - pat=re.compile(r'^\s*(\d+)\s+(\d{2}/\d{2}/\d{4})\s+(\d{2}/\d{2}/\d{4})\s+(.*)$') + pat=re.compile(rf'^\s*(\d+)\s+({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$', re.I) rows=[]; cur=None; page=1 for line in text.splitlines(): if '\f' in line: page+=line.count('\f') diff --git a/app/modules/bank_statement_analyzer/parsers/base.py b/app/modules/bank_statement_analyzer/parsers/base.py index 6574f8a..56d86f6 100644 --- a/app/modules/bank_statement_analyzer/parsers/base.py +++ b/app/modules/bank_statement_analyzer/parsers/base.py @@ -69,8 +69,13 @@ def finalize(df: pd.DataFrame, meta: StatementMeta) -> pd.DataFrame: return pd.DataFrame(columns=STANDARD_COLUMNS) for c in ['debit','credit','balance']: df[c]=pd.to_numeric(df.get(c),errors='coerce') + from .common import parse_flexible_date for c in ['transaction_date','value_date']: - df[c]=pd.to_datetime(df.get(c),errors='coerce',dayfirst=True) + values = df.get(c) + if values is None: + df[c] = pd.NaT + else: + df[c] = values.map(parse_flexible_date) df['narration']=df.get('narration','').fillna('').map(norm) df['reference_no']=df.get('reference_no','').fillna('').map(norm) df['bank_name']=meta.bank_name diff --git a/app/modules/bank_statement_analyzer/parsers/common.py b/app/modules/bank_statement_analyzer/parsers/common.py index 162c35b..6211bd4 100644 --- a/app/modules/bank_statement_analyzer/parsers/common.py +++ b/app/modules/bank_statement_analyzer/parsers/common.py @@ -1,20 +1,157 @@ from __future__ import annotations + import re -from .base import amount, norm +from datetime import datetime +from typing import Iterable -def find(pattern,text,group=1,flags=re.I|re.M): - m=re.search(pattern,text,flags) - return norm(m.group(group)) if m else '' +import pandas as pd -def date_iso(s): - import pandas as pd - x=pd.to_datetime(s,errors='coerce',dayfirst=True) - return '' if pd.isna(x) else x.strftime('%Y-%m-%d') +from .base import norm + + +# Shared date support for all bank parsers. Indian bank statements and the +# other supported banks can be rendered differently by pdftotext/pdfplumber, +# so parsers must not depend on one literal date layout. +SUPPORTED_DATE_FORMATS: tuple[str, ...] = ( + "%d %b %Y", + "%d %B %Y", + "%b %d %Y", + "%B %d %Y", + "%d %b, %Y", + "%d %B, %Y", + "%b %d, %Y", + "%B %d, %Y", + "%d-%b-%Y", + "%d-%B-%Y", + "%d/%b/%Y", + "%d/%B/%Y", + "%d-%m-%Y", + "%d/%m/%Y", + "%d.%m.%Y", + "%Y-%m-%d", + "%Y/%m/%d", + "%d %b %y", + "%d %B %y", + "%b %d %y", + "%B %d %y", + "%d-%b-%y", + "%d-%B-%y", + "%d/%b/%y", + "%d/%B/%y", + "%d-%m-%y", + "%d/%m/%y", + "%d.%m.%y", +) + +# A permissive token used only to locate a candidate date at the start of a +# transaction line. parse_flexible_date() performs the actual validation. +DATE_TOKEN_PATTERN = ( + r"(?:" + r"\d{1,2}\s+[A-Za-z]{3,9},?\s+\d{2,4}" + r"|[A-Za-z]{3,9}\s+\d{1,2},?\s+\d{2,4}" + r"|\d{1,2}[-/.]\d{1,2}[-/.]\d{2,4}" + r"|\d{4}[-/.]\d{1,2}[-/.]\d{1,2}" + r"|\d{1,2}[-/.][A-Za-z]{3,9}[-/.]\d{2,4}" + r")" +) +DATE_TOKEN_RE = re.compile(DATE_TOKEN_PATTERN, re.I) +LEADING_DATE_RE = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})(?:\s+|$)(.*)$", re.I) + + +def find(pattern, text, group=1, flags=re.I | re.M): + m = re.search(pattern, text, flags) + return norm(m.group(group)) if m else "" + + +def _normalise_date_text(value: str) -> str: + value = norm(value) + value = value.replace("–", "-").replace("—", "-") + value = re.sub(r"\s*,\s*", ", ", value) + value = re.sub(r"\s+", " ", value) + return value.strip(" ,") + + +def parse_flexible_date(value, *, default_year: int | None = None) -> pd.Timestamp | pd.NaT: + """Parse an Indian bank date without silently swapping day and month. + + Explicit formats are attempted before pandas' parser. Numeric dates are + always interpreted day-first because all supported statements are Indian + banking statements. A default year may be supplied for legacy DD/MM rows. + """ + if value is None or (isinstance(value, float) and pd.isna(value)): + return pd.NaT + if isinstance(value, (pd.Timestamp, datetime)): + return pd.Timestamp(value) + + text = _normalise_date_text(str(value)) + if not text: + return pd.NaT + + if default_year and re.fullmatch(r"\d{1,2}[-/.]\d{1,2}", text): + text = f"{text}/{default_year}" + + for fmt in SUPPORTED_DATE_FORMATS: + try: + return pd.Timestamp(datetime.strptime(text, fmt)) + except ValueError: + continue + + # Final guarded fallback for extraction noise. dayfirst=True is explicit + # and yearfirst is enabled only when the candidate begins with four digits. + parsed = pd.to_datetime( + text, + errors="coerce", + dayfirst=not bool(re.match(r"^\d{4}[-/.]", text)), + yearfirst=bool(re.match(r"^\d{4}[-/.]", text)), + ) + return pd.NaT if pd.isna(parsed) else pd.Timestamp(parsed) + + +def date_iso(value, *, default_year: int | None = None) -> str: + parsed = parse_flexible_date(value, default_year=default_year) + return "" if pd.isna(parsed) else parsed.strftime("%Y-%m-%d") + + +def leading_dates(line: str, *, maximum: int = 2) -> tuple[list[str], str]: + """Return up to ``maximum`` validated dates from the start of a line.""" + remaining = line + dates: list[str] = [] + for _ in range(maximum): + match = LEADING_DATE_RE.match(remaining) + if not match: + break + candidate = match.group(1) + if pd.isna(parse_flexible_date(candidate)): + break + dates.append(candidate) + remaining = match.group(2) + return dates, remaining + + +def find_date_tokens(text: str) -> list[str]: + return [m.group(0) for m in DATE_TOKEN_RE.finditer(text or "") if not pd.isna(parse_flexible_date(m.group(0)))] + + +def split_pages(text): + return text.split("\f") -def split_pages(text): return text.split('\f') def infer_mode(n): - u=(n or '').upper() - for k,v in [('UPI','UPI'),('NEFT','NEFT'),('IMPS','IMPS'),('RTGS','RTGS'),('CASH DEPOSIT','Cash Deposit'),('CASH WITHDRAWAL','Cash Withdrawal'),('ATM','ATM'),('CHEQUE','Cheque'),('CHQ','Cheque'),('POS','POS'),('EDC','Card Settlement'),('ACH','ACH')]: - if k in u:return v - return 'Other' + u = (n or "").upper() + for k, v in ( + ("UPI", "UPI"), + ("NEFT", "NEFT"), + ("IMPS", "IMPS"), + ("RTGS", "RTGS"), + ("CASH DEPOSIT", "Cash Deposit"), + ("CASH WITHDRAWAL", "Cash Withdrawal"), + ("ATM", "ATM"), + ("CHEQUE", "Cheque"), + ("CHQ", "Cheque"), + ("POS", "POS"), + ("EDC", "Card Settlement"), + ("ACH", "ACH"), + ): + if k in u: + return v + return "Other" diff --git a/app/modules/bank_statement_analyzer/parsers/hdfc.py b/app/modules/bank_statement_analyzer/parsers/hdfc.py index 19375f6..0a29ce9 100644 --- a/app/modules/bank_statement_analyzer/parsers/hdfc.py +++ b/app/modules/bank_statement_analyzer/parsers/hdfc.py @@ -2,7 +2,7 @@ from __future__ import annotations import re, pandas as pd from pathlib import Path from .base import * -from .common import find +from .common import DATE_TOKEN_PATTERN, DATE_TOKEN_RE, date_iso, find class HDFCParser(BaseParser): bank_name='HDFC Bank'; parser_name='HDFCParser' @@ -14,15 +14,15 @@ class HDFCParser(BaseParser): meta.account_number=find(r'Account No\s*:?\s*([0-9X*]+)',text) meta.customer_id=find(r'Cust ID\s*:?\s*([0-9X*]+)',text) meta.ifsc=find(r'(?:RTGS/\s*NEFT IFSC|IFSC)\s*:?\s*([A-Z0-9]+)',text) - m=re.search(r'From\s*:\s*(\d{2}/\d{2}/\d{4})\s+To\s*:\s*(\d{2}/\d{2}/\d{4})',text,re.I) + m=re.search(rf'From\s*:\s*({DATE_TOKEN_PATTERN})\s+To\s*:\s*({DATE_TOKEN_PATTERN})', text, re.I) if m: - meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d') + meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2)) # first non-empty line before address usually contains customer name; allow manual override in UI m=re.search(r'\n\s{2,}([^\n:]{3,60})\s*\n.*?Address\s*:',text,re.S|re.I) if m: meta.customer_name=norm(m.group(1).splitlines()[-1]) lines=text.splitlines(); rows=[]; cur=None; page=1 wd_pos,dep_pos,bal_pos=130,165,190 - date_re=re.compile(r'^\s*(\d{2}/\d{2}/\d{2})\s+(.*)$') + date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I) for line in lines: if '\f' in line: page += line.count('\f') if 'Withdrawal Amt.' in line and 'Deposit Amt.' in line: @@ -37,7 +37,7 @@ class HDFCParser(BaseParser): for r in rows: first=r['raw_lines'][0]; body=' '.join(x.strip() for x in r['raw_lines']) # identify value date and ref on first line - dts=list(re.finditer(r'\d{2}/\d{2}/\d{2}',first)) + dts=list(DATE_TOKEN_RE.finditer(first)) if len(dts)>1: r['value_date']=dts[-1].group() else: r['value_date']=r['transaction_date'] nums=list(re.finditer(r'(? float | None: - """Return CR as positive and DR as negative.""" parsed = amount(value) if parsed is None: return None return -parsed if (suffix or "").upper() == "DR" else parsed -def _parse_modern_date(value: str) -> str: - """Normalize both '01 Apr 2025' and 'Apr 01 2025' to ISO date.""" - return pd.to_datetime(value, dayfirst=True, errors="raise").strftime("%Y-%m-%d") +def _nearest_column(position: int, debit_pos: int, credit_pos: int, balance_pos: int) -> str: + distances = { + "debit": abs(position - debit_pos), + "credit": abs(position - credit_pos), + "balance": abs(position - balance_pos), + } + return min(distances, key=distances.get) class IndianBankModernParser(BaseParser): @@ -28,18 +31,8 @@ class IndianBankModernParser(BaseParser): @classmethod def detect(cls, text): - # pdfplumber's layout mode can insert multiple spaces inside headings - # (for example, 'ACCOUNT STATEMENT'). Normalize whitespace before - # matching so the same bank PDF is detected whether pdftotext is - # installed in the runtime image or the pdfplumber fallback is used. u = norm(text).upper() - return ( - 0.98 - if "ACCOUNT STATEMENT" in u - and "TRANSACTION DETAILS" in u - and "TOTAL CREDITS" in u - else 0 - ) + return 0.98 if "ACCOUNT STATEMENT" in u and "TRANSACTION DETAILS" in u and "TOTAL CREDITS" in u else 0 def parse(self, path, text=None): text = text or extract_text(path) @@ -50,202 +43,158 @@ class IndianBankModernParser(BaseParser): confidence="High", ) - # Some Indian Bank statements leave these values blank. Do not let a - # value from the adjacent ACCOUNT SUMMARY column become the customer - # name merely because PDF text extraction merges the two columns. - customer = find(r"Account Holder Name[ \t]*([^\n]*)", text) - customer = norm(customer) - if customer and not re.search( - r"^(Opening Balance|Account Type|Account Number|Customer)", - customer, - re.I, - ): + customer = norm(find(r"Account Holder Name[ \t]*([^\n]*)", text)) + if customer and not re.search(r"^(Opening Balance|Account Type|Account Number|Customer)", customer, re.I): meta.customer_name = customer + meta.account_number = find(r"Account Number[ \t]*([0-9X*]{4,})", text) + meta.ifsc = find(r"IFSC[ \t]*([A-Z0-9]{8,})", text) - meta.account_number = find( - r"Account Number[ \t]*([0-9X*]{4,})", - text, - ) + period = re.search(rf"For period:\s*({DATE_TOKEN_PATTERN})\s*(?:-|to)\s*({DATE_TOKEN_PATTERN})", text, re.I) + if period: + meta.period_from = date_iso(period.group(1)) + meta.period_to = date_iso(period.group(2)) - period_match = re.search( - r"For period:\s*(\d{2}\s+[A-Za-z]{3}\s+\d{4})\s*-\s*" - r"(\d{2}\s+[A-Za-z]{3}\s+\d{4})", - text, - re.I, - ) - if period_match: - meta.period_from = _parse_modern_date(period_match.group(1)) - meta.period_to = _parse_modern_date(period_match.group(2)) + opening = re.search(r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", text, re.I) + if opening: + meta.opening_balance = _signed_amount(opening.group(1), opening.group(2)) + meta.total_credit = amount(find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text)) + meta.total_debit = amount(find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text)) + closing = re.search(r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", text, re.I) + if closing: + meta.closing_balance = _signed_amount(closing.group(1), closing.group(2)) - opening_match = re.search( - r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", - text, - re.I, - ) - if opening_match: - meta.opening_balance = _signed_amount( - opening_match.group(1), - opening_match.group(2), - ) + date_re = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})\s+(.*)$", re.I) + amount_re = re.compile(r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", re.I) - meta.total_credit = amount( - find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text) - ) - meta.total_debit = amount( - find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text) - ) - - closing_match = re.search( - r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", - text, - re.I, - ) - if closing_match: - meta.closing_balance = _signed_amount( - closing_match.group(1), - closing_match.group(2), - ) - - # Indian Bank's current PDF layout is seen in both date orders, - # depending on the text extraction engine: - # 01 Apr 2025 ... - # Apr 01 2025 ... - # Accept both without changing the parser selected for older samples. - date_re = re.compile( - r"^\s*((?:\d{2}\s+[A-Za-z]{3}|[A-Za-z]{3}\s+\d{2})\s+\d{4})\s+(.*)$" - ) - - rows = [] - current = None + rows: list[dict] = [] + current: dict | None = None page = 1 - for line in text.splitlines(): - if "\f" in line: - page += line.count("\f") + # Defaults match the wide first-page layout. They are replaced from + # every page header, which also supports narrower subsequent pages. + debit_pos, credit_pos, balance_pos = 56, 80, 94 + + for raw_line in text.splitlines(): + if "\f" in raw_line: + page += raw_line.count("\f") + line = raw_line.rstrip("\n") + upper = line.upper() + + if "TRANSACTION DETAILS" in upper and "DEBITS" in upper and "CREDITS" in upper and "BALANCE" in upper: + debit_pos = line.upper().find("DEBITS") + credit_pos = line.upper().find("CREDITS") + balance_pos = line.upper().rfind("BALANCE") + continue match = date_re.match(line) - if match: + if match and not pd.isna(parse_flexible_date(match.group(1))): if current: rows.append(current) - transaction_date = _parse_modern_date(match.group(1)) current = { - "transaction_date": transaction_date, - "value_date": transaction_date, - "lines": [match.group(2)], + "transaction_date": date_iso(match.group(1)), + "value_date": date_iso(match.group(1)), + "first_line": line, + "detail_start": match.start(2), + "continuation": [], "source_page": page, + "debit_pos": debit_pos, + "credit_pos": credit_pos, + "balance_pos": balance_pos, } - elif ( - current - and line.strip() - and not re.match( - r"^(Date\s+Transaction|ACCOUNT STATEMENT|Page)", - line.strip(), - re.I, - ) - ): - current["lines"].append(line) + continue + + if not current or not line.strip(): + continue + if re.match(r"^(\s*Date\s+Transaction|\s*ACCOUNT STATEMENT|\s*Indian Bank\s*\||\s*Ending Balance|\s*Total\s+INR)", line, re.I): + continue + current["continuation"].append(line.strip()) if current: rows.append(current) - out = [] + output: list[dict] = [] previous_balance = meta.opening_balance for row in rows: - first_line = row["lines"][0] - all_text = norm(" ".join(row["lines"])) + first_line = row["first_line"] + tokens = list(amount_re.finditer(first_line)) + if not tokens: + continue - # Balance is always the last INR amount and may be CR or DR. - balance_match = re.search( - r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)\s*$", - first_line, - re.I, - ) - if balance_match: - balance = _signed_amount( - balance_match.group(1), - balance_match.group(2), - ) - prefix = first_line[: balance_match.start()] - else: - balance = None - prefix = first_line + debit = credit = balance = None + balance_suffix = None + token_columns: list[tuple[str, re.Match]] = [] + for token in tokens: + column = _nearest_column(token.start(), row["debit_pos"], row["credit_pos"], row["balance_pos"]) + token_columns.append((column, token)) - amount_matches = list( - re.finditer(r"INR\s*([\d,]+\.\d{2})", prefix, re.I) - ) - transaction_amount = ( - amount(amount_matches[-1].group(1)) if amount_matches else None - ) + # The running balance is the rightmost amount and is also checked + # against the header-derived balance column. + balance_token = tokens[-1] + balance_suffix = balance_token.group(2) + balance = _signed_amount(balance_token.group(1), balance_suffix) - debit = None - credit = None + for column, token in token_columns[:-1]: + value = amount(token.group(1)) + if column == "debit": + debit = value + elif column == "credit": + credit = value - if ( - transaction_amount is not None - and previous_balance is not None - and balance is not None - ): - if abs((previous_balance - transaction_amount) - balance) < 0.05: - debit = transaction_amount - elif abs((previous_balance + transaction_amount) - balance) < 0.05: - credit = transaction_amount + # When text extraction collapses spacing, infer the transaction side + # from the running balance rather than guessing by token position. + if debit is None and credit is None and len(tokens) >= 2: + txn = amount(tokens[-2].group(1)) + if txn is not None and previous_balance is not None and balance is not None: + if abs((previous_balance - txn) - balance) < 0.05: + debit = txn + elif abs((previous_balance + txn) - balance) < 0.05: + credit = txn - # Fallback for the first row or when running-balance inference is - # unavailable. In the rendered Indian Bank table, a dash occupies - # the empty debit/credit column. Determine the side from text - # preceding the transaction amount. - if transaction_amount is not None and debit is None and credit is None: - before_amount = prefix[: amount_matches[-1].start()] - after_amount = prefix[amount_matches[-1].end() :] - - # "- INR 1,000.00" means debit is empty, therefore credit. - if re.search(r"-\s*$", before_amount): - credit = transaction_amount - # "INR 1,000.00 -" means credit is empty, therefore debit. - elif re.match(r"^\s*-", after_amount): - debit = transaction_amount - else: - # Position fallback retained for extraction engines that - # preserve table spacing but omit the dash placeholder. - if amount_matches[-1].start() < 52: - debit = transaction_amount - else: - credit = transaction_amount - - narration = re.split( - r"\s+INR\s*[\d,]+\.\d{2}", - all_text, - 1, - flags=re.I, - )[0] - - reference_no = "" - reference_match = re.search( - r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", - all_text, - re.I, - ) - if reference_match: - reference_no = reference_match.group(0) + # If the balance has no explicit DR/CR suffix, preserve continuity. + if balance is not None and previous_balance is not None: + expected = previous_balance + (credit or 0) - (debit or 0) + if abs(expected - balance) >= 0.05 and abs(expected + balance) < 0.05: + balance = -balance if debit is None and credit is None: continue - out.append( + first_detail_end = min((m.start() for m in tokens), default=len(first_line)) + first_detail = first_line[row["detail_start"]:first_detail_end].strip() + narration_parts = [first_detail] + row["continuation"] + narration = norm(" ".join(part for part in narration_parts if part)) + + reference_no = "" + reference = re.search(r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", narration, re.I) + if reference: + reference_no = reference.group(0) + + output.append( { - **row, + "transaction_date": row["transaction_date"], + "value_date": row["value_date"], "narration": narration, "reference_no": reference_no, "debit": debit, "credit": credit, "balance": balance, + "source_page": row["source_page"], } ) - if balance is not None: previous_balance = balance - return meta, finalize(pd.DataFrame(out), meta) + frame = finalize(pd.DataFrame(output), meta) + if not frame.empty: + # Validate and repair metadata from the parsed rows only when the + # statement summary did not supply it. + first = frame.iloc[0] + calculated_opening = (first["balance"] or 0) + (first["debit"] or 0) - (first["credit"] or 0) + if meta.opening_balance is None: + meta.opening_balance = round(calculated_opening, 2) + if meta.closing_balance is None: + meta.closing_balance = float(frame["balance"].dropna().iloc[-1]) + return meta, frame class IndianBankLegacyParser(BaseParser): @@ -255,13 +204,7 @@ class IndianBankLegacyParser(BaseParser): @classmethod def detect(cls, text): u = text.upper() - return ( - 0.97 - if "STATEMENT OF ACCOUNT FROM" in u - and "REMITTER" in u - and "CHEQUE NO" in u - else 0 - ) + return 0.97 if "STATEMENT OF ACCOUNT FROM" in u and "REMITTER" in u and "CHEQUE NO" in u else 0 def parse(self, path, text=None): text = text or extract_text(path) @@ -272,119 +215,74 @@ class IndianBankLegacyParser(BaseParser): confidence="Medium", ) meta.account_number = find(r"for Account Number\s*\.?\s*([0-9X*]+)", text) - m = re.search( - r"STATEMENT OF ACCOUNT from\s*(\d{2}/\d{2}/\d{4})\s*to\s*" - r"(\d{2}/\d{2}/\d{4})", - text, - re.I, - ) - if m: - meta.period_from = pd.to_datetime( - m.group(1), dayfirst=True - ).strftime("%Y-%m-%d") - meta.period_to = pd.to_datetime( - m.group(2), dayfirst=True - ).strftime("%Y-%m-%d") + period = re.search(rf"STATEMENT OF ACCOUNT from\s*({DATE_TOKEN_PATTERN})\s*to\s*({DATE_TOKEN_PATTERN})", text, re.I) + if period: + meta.period_from = date_iso(period.group(1)) + meta.period_to = date_iso(period.group(2)) - date_re = re.compile( - r"^\s*(\d{2}/\d{2})(?:/\d{4})?\s+" - r"(\d{2}/\d{2})(?:/\d{4})?\s+(.*)$" - ) + # Legacy files may omit the year on each transaction. Preserve the + # previous behaviour while accepting every supported full-date format. + default_year = int(meta.period_from[:4]) if meta.period_from else None + full_date_re = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$", re.I) + short_date_re = re.compile(r"^\s*(\d{1,2}[-/.]\d{1,2})\s+(\d{1,2}[-/.]\d{1,2})\s+(.*)$") rows = [] current = None page = 1 for line in text.splitlines(): if "\f" in line: page += line.count("\f") - match = date_re.match(line) + match = full_date_re.match(line) or short_date_re.match(line) if match: if current: rows.append(current) - year = meta.period_from[:4] if meta.period_from else "2024" - transaction_date = match.group(1).replace(" ", "") + "/" + year - value_date = match.group(2).replace(" ", "") + "/" + year + transaction_date = date_iso(match.group(1), default_year=default_year) + value_date = date_iso(match.group(2), default_year=default_year) current = { "transaction_date": transaction_date, "value_date": value_date, "lines": [match.group(3)], "source_page": page, } - elif ( - current - and line.strip() - and not re.match( - r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)", - line.strip(), - re.I, - ) - ): + elif current and line.strip() and not re.match(r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)", line.strip(), re.I): current["lines"].append(line) if current: rows.append(current) - out = [] + output = [] previous_balance = None for row in rows: - first_line = row["lines"][0] + first = row["lines"][0] all_text = norm(" ".join(row["lines"])) - balance_match = re.search( - r"([\d,]+\.\d{2})(CR|DR)\s*$", - first_line, - re.I, - ) + balance_match = re.search(r"([\d,]+\.\d{2})(CR|DR)\s*$", first, re.I) if not balance_match: continue - balance = _signed_amount( - balance_match.group(1), - balance_match.group(2), - ) - prefix = first_line[: balance_match.start()] - numbers = list( - re.finditer(r"(? 70 else None - debit = transaction_amount if numbers[-1].start() <= 70 else None - - reference_no = "" - reference_match = re.search( - r"(?:UPI|NEFT|IMPS|RTGS)[/A-Z0-9-]{6,}", - all_text, - re.I, - ) - if reference_match: - reference_no = reference_match.group(0) - - out.append( + if numbers[-1].start() > 70: + credit = transaction_amount + else: + debit = transaction_amount + if debit is None and credit is None: + continue + reference = re.search(r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", all_text, re.I) + output.append( { **row, - "narration": all_text, - "reference_no": reference_no, + "narration": norm(prefix[: numbers[-1].start()] + " " + " ".join(row["lines"][1:])), + "reference_no": reference.group(0) if reference else "", "debit": debit, "credit": credit, "balance": balance, } ) previous_balance = balance - - if out and meta.opening_balance is None: - first = out[0] - transaction_amount = (first.get("debit") or 0) - ( - first.get("credit") or 0 - ) - meta.opening_balance = round( - (first["balance"] or 0) + transaction_amount, - 2, - ) - - return meta, finalize(pd.DataFrame(out), meta) + return meta, finalize(pd.DataFrame(output), meta) diff --git a/app/modules/bank_statement_analyzer/parsers/indusind.py b/app/modules/bank_statement_analyzer/parsers/indusind.py index 7e5333f..abb7ae0 100644 --- a/app/modules/bank_statement_analyzer/parsers/indusind.py +++ b/app/modules/bank_statement_analyzer/parsers/indusind.py @@ -2,7 +2,7 @@ from __future__ import annotations import re, pandas as pd from pathlib import Path from .base import * -from .common import find +from .common import DATE_TOKEN_PATTERN, date_iso, find class IndusIndParser(BaseParser): bank_name='IndusInd Bank'; parser_name='IndusIndParser' @@ -13,13 +13,13 @@ class IndusIndParser(BaseParser): meta.account_number=find(r'Account Number\s*:\s*([0-9X*]+)',text) meta.customer_id=find(r'Cust\.Reln\.No\s*:\s*([0-9X*]+)',text) meta.ifsc=find(r'IFSC Code\s*:\s*([A-Z0-9]+)',text) - m=re.search(r'Period\s*:\s*(\d{2}[/-][A-Za-z0-9]{2,3}[/-]\d{2,4})\s*(?:to|TO|-)\s*(\d{2}[/-][A-Za-z0-9]{2,3}[/-]\d{2,4})',text,re.I) + m=re.search(rf'Period\s*:\s*({DATE_TOKEN_PATTERN})\s*(?:to|-)\s*({DATE_TOKEN_PATTERN})', text, re.I) if m: - meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d') + meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2)) meta.total_debit=amount(find(r'Total Withdrawal Amount\s*:\s*([\d,]+\.\d{2})',text)) meta.total_credit=amount(find(r'Total Deposit Amount\s*:\s*([\d,]+\.\d{2})',text)) # rows start with date and often end with amount Dr/Cr and balance Cr - date_re=re.compile(r'^\s*(\d{2}[-/]?[A-Za-z]{3}[-/]?\d{2,4}|\d{2}-\d{2}-\d{4})\s+(.*)$') + date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I) rows=[]; cur=None; page=1 for line in text.splitlines(): if '\f' in line: page+=line.count('\f') diff --git a/app/modules/bank_statement_analyzer/parsers/kotak.py b/app/modules/bank_statement_analyzer/parsers/kotak.py index 69704e7..4fe7230 100644 --- a/app/modules/bank_statement_analyzer/parsers/kotak.py +++ b/app/modules/bank_statement_analyzer/parsers/kotak.py @@ -2,7 +2,7 @@ from __future__ import annotations import re, pandas as pd from pathlib import Path from .base import * -from .common import find +from .common import DATE_TOKEN_PATTERN, date_iso, find class KotakParser(BaseParser): bank_name='Kotak Mahindra Bank'; parser_name='KotakParser' @@ -13,10 +13,10 @@ class KotakParser(BaseParser): text=text or extract_text(path); meta=StatementMeta(bank_name=self.bank_name,source_file=Path(path).name,parser_name=self.parser_name,confidence='High') meta.account_number=find(r'Account\s*#\s*Variant\s*KOTAK\s*\n.*?([0-9X*]{6,})',text,flags=re.I|re.S) meta.ifsc=find(r'IFSC\s+([A-Z0-9]+)',text) - m=re.search(r'(\d{2}\s+[A-Za-z]{3},\s*\d{4})\s*-\s*(\d{2}\s+[A-Za-z]{3},\s*\d{4})',text) + m=re.search(rf'({DATE_TOKEN_PATTERN})\s*-\s*({DATE_TOKEN_PATTERN})', text, re.I) if m: - meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d') - date_re=re.compile(r'^\s*(\d{2}\s+[A-Za-z]{3},\s*\d{4})\s+(.*)$') + meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2)) + date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I) rows=[]; cur=None; page=1 for line in text.splitlines(): if '\f' in line: page+=line.count('\f') diff --git a/app/modules/bank_statement_analyzer/parsers/sbi.py b/app/modules/bank_statement_analyzer/parsers/sbi.py index 058b423..9ee1c0b 100644 --- a/app/modules/bank_statement_analyzer/parsers/sbi.py +++ b/app/modules/bank_statement_analyzer/parsers/sbi.py @@ -2,7 +2,7 @@ from __future__ import annotations import re, pandas as pd from pathlib import Path from .base import * -from .common import find +from .common import DATE_TOKEN_PATTERN, date_iso, find class SBIModernParser(BaseParser): bank_name='State Bank of India'; parser_name='SBIModernParser' @@ -14,12 +14,12 @@ class SBIModernParser(BaseParser): meta.customer_name=find(r'Account Name\s*:?\s*([^\n]+)',text) meta.account_number=find(r'Account Number\s*:?\s*([0-9X*]+)',text) meta.ifsc=find(r'IFS Code\s*:?\s*([A-Z0-9]+)',text) - m=re.search(r'Account Statement from\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+to\s+(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})',text,re.I) + m=re.search(rf'Account Statement from\s*({DATE_TOKEN_PATTERN})\s+to\s+({DATE_TOKEN_PATTERN})', text, re.I) if m: - meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d') + meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2)) meta.opening_balance=amount(find(r'Balance as on[^\n]*\n\s*([\d,]+\.\d{2})',text)) # Format A: date details ref debit credit balance - date_re=re.compile(r'^\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+(.*)$') + date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I) rows=[]; cur=None; page=1 for line in text.splitlines(): if '\f' in line: page += line.count('\f') @@ -66,11 +66,11 @@ class SBIOtherParser(BaseParser): meta.customer_name=find(r'Account Name\s*:\s*([^\n]+)',text) meta.account_number=find(r'Account Number\s*:?\s*([0-9X*]+)',text) meta.ifsc=find(r'IFS Code\s*:?\s*([A-Z0-9]+)',text) - m=re.search(r'Account Statement from\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+to\s+(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})',text,re.I) + m=re.search(rf'Account Statement from\s*({DATE_TOKEN_PATTERN})\s+to\s+({DATE_TOKEN_PATTERN})', text, re.I) if m: - meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d') + meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2)) m=re.search(r'Balance as on\s+[^\n]+\n',text,re.I) - date_re=re.compile(r'^\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+(.*)$') + date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$', re.I) rows=[]; cur=None; page=1 for line in text.splitlines(): if '\f' in line: page += line.count('\f')