From b1b088d4f2a7c3c72931f0b97283dffb62a10e47 Mon Sep 17 00:00:00 2001 From: A R R R Associates Date: Mon, 3 Aug 2026 15:53:22 +0530 Subject: [PATCH] Add ICICI bank statement parser --- .../bank_statement_analyzer/parsers/icici.py | 239 ++++++++++++++++++ .../parsers/registry.py | 13 +- 2 files changed, 247 insertions(+), 5 deletions(-) create mode 100644 app/modules/bank_statement_analyzer/parsers/icici.py diff --git a/app/modules/bank_statement_analyzer/parsers/icici.py b/app/modules/bank_statement_analyzer/parsers/icici.py new file mode 100644 index 0000000..f4f2895 --- /dev/null +++ b/app/modules/bank_statement_analyzer/parsers/icici.py @@ -0,0 +1,239 @@ +from __future__ import annotations + +import re +from pathlib import Path + +import pandas as pd +import pdfplumber + +from .base import BaseParser, StatementMeta, amount, extract_text, finalize, norm +from .common import date_iso, find, infer_mode + + +_DATE_RE = re.compile(r"^\d{2}-\d{2}-\d{4}$") +_MONEY_RE = re.compile(r"^-?(?:\d{1,3}(?:,\d{2,3})+|\d+)\.\d{2}$") + + +class ICICIParser(BaseParser): + """Parser for ICICI Bank retail/Privilege PDF account statements. + + The statement uses a fixed visual table with DATE, MODE, PARTICULARS, + DEPOSITS, WITHDRAWALS and BALANCE columns. Transactions may span several + visual lines, so parsing is coordinate based rather than dependent on + pdftotext line wrapping. + """ + + bank_name = "ICICI Bank" + parser_name = "ICICIParser" + + @classmethod + def detect(cls, text: str) -> float: + upper = (text or "").upper() + score = 0.0 + if "ICICIBANK" in upper or "ICICI BANK" in upper: + score += 0.55 + if "STATEMENT OF TRANSACTIONS IN SAVINGS ACCOUNT" in upper: + score += 0.20 + if all(token in upper for token in ("DEPOSITS", "WITHDRAWALS", "BALANCE")): + score += 0.20 + if "ACCOUNT RELATED OTHER INFORMATION" in upper: + score += 0.05 + return min(score, 0.99) + + @staticmethod + def _table_geometry(page, words: list[dict]) -> tuple[float, list[float]] | None: + required = {"DATE", "PARTICULARS", "DEPOSITS", "WITHDRAWALS", "BALANCE"} + for date_word in [word for word in words if word["text"].strip().upper() == "DATE"]: + row = [word for word in words if abs(word["top"] - date_word["top"]) <= 2] + labels = {word["text"].strip().upper(): word for word in row} + if not required.issubset(labels): + continue + header_top = date_word["top"] + horizontal = [] + for line in page.lines: + top = page.height - line["y0"] + if abs(line["y0"] - line["y1"]) <= 1 and abs(top - header_top) <= 8: + horizontal.append(line) + endpoints = sorted({round(value, 2) for line in horizontal for value in (line["x0"], line["x1"])}) + if len(endpoints) >= 7: + return header_top, endpoints[:7] + # Safe fallback using the known visual order of the header labels. + return header_top, [ + max(0.0, labels["DATE"]["x0"] - 2), + (labels["DATE"]["x1"] + labels.get("MODE**", labels["PARTICULARS"])["x0"]) / 2, + (labels.get("MODE**", labels["PARTICULARS"])["x1"] + labels["PARTICULARS"]["x0"]) / 2, + (labels["PARTICULARS"]["x1"] + labels["DEPOSITS"]["x0"]) / 2, + (labels["DEPOSITS"]["x1"] + labels["WITHDRAWALS"]["x0"]) / 2, + (labels["WITHDRAWALS"]["x1"] + labels["BALANCE"]["x0"]) / 2, + min(page.width, labels["BALANCE"]["x1"] + 5), + ] + return None + + @staticmethod + def _cell_text(words: list[dict], left: float, right: float) -> str: + selected = [ + word for word in words + if left <= (word["x0"] + word["x1"]) / 2 < right + ] + selected.sort(key=lambda word: (round(word["top"], 1), word["x0"])) + return norm(" ".join(word["text"] for word in selected)) + + @staticmethod + def _cell_amount(words: list[dict], left: float, right: float) -> float | None: + selected = [ + word for word in words + if left <= (word["x0"] + word["x1"]) / 2 < right + and _MONEY_RE.fullmatch(word["text"].strip()) + ] + if not selected: + return None + selected.sort(key=lambda word: (word["top"], word["x0"])) + return amount(selected[-1]["text"]) + + @staticmethod + def _reference_number(narration: str) -> str: + patterns = ( + r"(?:MMT/IMPS/|UPI/[^/]+/[^/]+/[^/]+/[^/]+/)(\d{10,18})", + r"\b((?:ICI|HDF|AXI|SBI|IBL|PPPL|YJP)[A-Za-z0-9]{10,})\b", + r"\b(\d{12,18})\b", + ) + for pattern in patterns: + match = re.search(pattern, narration, re.I) + if match: + return match.group(1) + return "" + + def _parse_page(self, page, page_number: int) -> tuple[list[dict], float | None]: + words = page.extract_words(use_text_flow=False, keep_blank_chars=False) or [] + geometry = self._table_geometry(page, words) + if not geometry: + return [], None + header_top, boundaries = geometry + left, date_right, mode_right, particulars_right, deposits_right, withdrawals_right, right = boundaries + table_words = [word for word in words if word["top"] > header_top + 4] + date_words = [ + word for word in table_words + if _DATE_RE.fullmatch(word["text"].strip()) + and left <= (word["x0"] + word["x1"]) / 2 < date_right + ] + date_words.sort(key=lambda word: word["top"]) + if not date_words: + return [], None + + horizontal_groups: dict[float, list[dict]] = {} + for line in page.lines: + if abs(line["y0"] - line["y1"]) > 1: + continue + top = round(page.height - line["y0"], 2) + if top < header_top - 5: + continue + horizontal_groups.setdefault(top, []).append(line) + horizontal_tops = sorted( + top for top, lines in horizontal_groups.items() + if min(line["x0"] for line in lines) <= left + 2 + and max(line["x1"] for line in lines) >= right - 2 + ) + + rows: list[dict] = [] + opening_balance: float | None = None + for date_word in date_words: + center = (date_word["top"] + date_word["bottom"]) / 2 + upper_candidates = [top for top in horizontal_tops if top <= center] + lower_candidates = [top for top in horizontal_tops if top > center] + band_top = max(upper_candidates) + 0.1 if upper_candidates else header_top + 4 + band_bottom = min(lower_candidates) - 0.1 if lower_candidates else min(page.height - 15, center + 30) + band_words = [ + word for word in table_words + if band_top <= (word["top"] + word["bottom"]) / 2 <= band_bottom + and left <= (word["x0"] + word["x1"]) / 2 <= right + ] + + mode = self._cell_text(band_words, date_right, mode_right) + narration = self._cell_text(band_words, mode_right, particulars_right) + deposit = self._cell_amount(band_words, particulars_right, deposits_right) + withdrawal = self._cell_amount(band_words, deposits_right, withdrawals_right) + balance_value = self._cell_amount(band_words, withdrawals_right, right + 1) + combined_narration = norm(f"{mode} {narration}") + transaction_date = date_word["text"].strip() + + if re.fullmatch(r"(?:B/F|BROUGHT FORWARD)", narration.strip(), re.I): + opening_balance = balance_value + continue + if deposit is None and withdrawal is None: + continue + + rows.append({ + "transaction_date": transaction_date, + "value_date": transaction_date, + "narration": combined_narration, + "reference_no": self._reference_number(combined_narration), + "debit": withdrawal, + "credit": deposit, + "balance": balance_value, + "source_page": page_number, + "mode": mode or infer_mode(combined_narration), + }) + + return rows, opening_balance + + def parse(self, path, text=None): + pdf_path = Path(path) + text = text or extract_text(pdf_path) + meta = StatementMeta( + bank_name=self.bank_name, + source_file=pdf_path.name, + parser_name=self.parser_name, + confidence="High", + ) + + name_match = re.search(r"(?m)^\s*((?:MS|MR|MRS)\.?\s*[^\n]+)$", text, re.I) + meta.customer_name = norm(re.split(r"\s{2,}", name_match.group(1).strip())[0]) if name_match else "" + meta.customer_id = find(r"Cust ID\s*:\s*([0-9]+)", text) + meta.account_number = find(r"Savings Account Number\s*:\s*([0-9]+)", text) + if not meta.account_number: + meta.account_number = find(r"Savings\s+A/c\s+([0-9]+)", text) + meta.ifsc = find(r"IFSC CODE\s+NAME OF NOMINEE.*?Savings\s+[0-9]+\s+[0-9]+\s+([A-Z0-9]+)", text, flags=re.I | re.S) + + period = re.search( + r"for the period\s+([A-Za-z]+\s+\d{1,2},\s*\d{4})\s*-\s*([A-Za-z]+\s+\d{1,2},\s*\d{4})", + text, + re.I, + ) + if period: + meta.period_from = date_iso(period.group(1)) + meta.period_to = date_iso(period.group(2)) + + all_rows: list[dict] = [] + with pdfplumber.open(str(pdf_path)) as pdf: + for page_number, page in enumerate(pdf.pages, start=1): + page_rows, page_opening = self._parse_page(page, page_number) + if meta.opening_balance is None and page_opening is not None: + meta.opening_balance = page_opening + all_rows.extend(page_rows) + + data = pd.DataFrame(all_rows) + if not data.empty: + meta.closing_balance = float(data.iloc[-1]["balance"]) if pd.notna(data.iloc[-1]["balance"]) else None + meta.total_debit = round(float(pd.to_numeric(data["debit"], errors="coerce").fillna(0).sum()), 2) + meta.total_credit = round(float(pd.to_numeric(data["credit"], errors="coerce").fillna(0).sum()), 2) + + total_matches = re.findall( + r"\bTOTAL\s+([\d,]+\.\d{2})\s+([\d,]+\.\d{2})\s+([\d,]+\.\d{2})", + text, + re.I, + ) + total_match = total_matches[-1] if total_matches else None + if total_match: + printed_credit = amount(total_match[0]) + printed_debit = amount(total_match[1]) + printed_closing = amount(total_match[2]) + # Preserve printed totals where they agree; otherwise the common + # reconciliation sheet will expose the exact row-level difference. + if printed_credit is not None: + meta.total_credit = printed_credit + if printed_debit is not None: + meta.total_debit = printed_debit + if printed_closing is not None: + meta.closing_balance = printed_closing + + return meta, finalize(data, meta) diff --git a/app/modules/bank_statement_analyzer/parsers/registry.py b/app/modules/bank_statement_analyzer/parsers/registry.py index edfe084..77b43ed 100644 --- a/app/modules/bank_statement_analyzer/parsers/registry.py +++ b/app/modules/bank_statement_analyzer/parsers/registry.py @@ -3,31 +3,33 @@ from __future__ import annotations from .idfc import IDFCFirstParser from .axis import AxisParser from .hdfc import HDFCParser +from .icici import ICICIParser from .hsbc import HSBCParser from .indian_bank import IndianBankModernParser, IndianBankLegacyParser from .indusind import IndusIndParser from .kotak import KotakParser -from .sbi import SBIYONORelationshipParser, SBIStandardParser, SBICompactParser, SBIModernParser, SBIOtherParser +from .sbi import SBIModernParser, SBIOtherParser from .base import extract_text PARSERS = [ IDFCFirstParser, AxisParser, HDFCParser, + ICICIParser, HSBCParser, IndianBankModernParser, IndianBankLegacyParser, IndusIndParser, KotakParser, - SBIYONORelationshipParser, - SBIStandardParser, - SBICompactParser, + SBIOtherParser, + SBIModernParser, ] BANK_OPTIONS = [ ("auto", "Auto Detect"), ("axis", "Axis Bank"), ("hdfc", "HDFC Bank"), + ("icici", "ICICI Bank"), ("hsbc", "HSBC Bank"), ("idfc", "IDFC FIRST Bank"), ("indian_bank", "Indian Bank"), @@ -39,12 +41,13 @@ BANK_OPTIONS = [ BANK_PARSERS = { "axis": [AxisParser], "hdfc": [HDFCParser], + "icici": [ICICIParser], "hsbc": [HSBCParser], "idfc": [IDFCFirstParser], "indian_bank": [IndianBankModernParser, IndianBankLegacyParser], "indusind": [IndusIndParser], "kotak": [KotakParser], - "sbi": [SBIYONORelationshipParser, SBIStandardParser, SBICompactParser], + "sbi": [SBIOtherParser, SBIModernParser], }