from __future__ import annotations import re from pathlib import Path import pandas as pd import pdfplumber from .base import BaseParser, StatementMeta, amount, extract_text, finalize, norm from .common import date_iso, find, infer_mode _DATE_RE = re.compile(r"^(?:\d{2}-\d{2}-\d{4}|\d{2}-[A-Za-z]{3}-\d{4})$") _MONEY_RE = re.compile(r"^-?(?:\d{1,3}(?:,\d{2,3})+|\d+)\.\d{2}$") def _clean_date_cell(value: str) -> str: text = norm(value) return re.sub(r"\s*([/-])\s*", r"\1", text) class ICICIParser(BaseParser): """Parser for ICICI Bank retail/Privilege PDF account statements. The statement uses a fixed visual table with DATE, MODE, PARTICULARS, DEPOSITS, WITHDRAWALS and BALANCE columns. Transactions may span several visual lines, so parsing is coordinate based rather than dependent on pdftotext line wrapping. """ bank_name = "ICICI Bank" parser_name = "ICICIParser" @classmethod def detect(cls, text: str) -> float: upper = (text or "").upper() score = 0.0 if "ICICIBANK" in upper or "ICICI BANK" in upper: score += 0.55 if "STATEMENT OF TRANSACTIONS IN SAVINGS ACCOUNT" in upper: score += 0.20 if "ACCOUNT STATEMENT" in upper and "TRANSACTION ID" in upper: score += 0.20 if "AVAILABLE BALANCE" in upper and "WITHDRAWAL" in upper and "DEPOSIT" in upper: score += 0.15 if all(token in upper for token in ("DEPOSITS", "WITHDRAWALS", "BALANCE")): score += 0.20 if "ACCOUNT RELATED OTHER INFORMATION" in upper: score += 0.05 return min(score, 0.99) @staticmethod def _modern_header_mapping(row: list) -> dict[str, int] | None: cells = [norm(cell).upper() for cell in (row or [])] aliases = { "serial": ("S.NO", "S NO", "SERIAL NO"), "transaction_id": ("TRANSACTION ID", "TXN ID"), "transaction_date": ("TRANSACTION DATE", "TXN DATE"), "cheque_no": ("CHEQUE NO", "CHQ NO"), "description": ("DESCRIPTION", "PARTICULARS"), "debit": ("WITHDRAWAL (DR)", "WITHDRAWAL", "DEBIT"), "credit": ("DEPOSIT (CR)", "DEPOSIT", "CREDIT"), "balance": ("AVAILABLE BALANCE", "BALANCE"), } mapping: dict[str, int] = {} for field, names in aliases.items(): for index, cell in enumerate(cells): if any(name == cell or name in cell for name in names): mapping[field] = index break required = {"transaction_date", "description", "debit", "credit", "balance"} return mapping if required.issubset(mapping) else None def _parse_modern_tables(self, pdf) -> list[dict]: rows: list[dict] = [] active_mapping: dict[str, int] | None = None for page_number, page in enumerate(pdf.pages, start=1): for table in page.extract_tables() or []: if not table: continue start_index = 0 header_mapping = self._modern_header_mapping(table[0]) if header_mapping: active_mapping = header_mapping start_index = 1 if active_mapping is None: continue mapping = active_mapping for raw in table[start_index:]: cells = list(raw or []) max_index = max(mapping.values()) if len(cells) <= max_index: cells.extend([None] * (max_index + 1 - len(cells))) transaction_date = _clean_date_cell(cells[mapping["transaction_date"]]) if not _DATE_RE.fullmatch(transaction_date): continue narration = norm(cells[mapping["description"]]) transaction_id = norm(cells[mapping.get("transaction_id", -1)]) if "transaction_id" in mapping else "" cheque_no = norm(cells[mapping.get("cheque_no", -1)]) if "cheque_no" in mapping else "" debit = amount(cells[mapping["debit"]]) credit = amount(cells[mapping["credit"]]) balance_value = amount(cells[mapping["balance"]]) if balance_value is None or (debit is None and credit is None): continue reference = cheque_no if cheque_no and cheque_no != "-" else transaction_id rows.append({ "transaction_date": date_iso(transaction_date), "value_date": date_iso(transaction_date), "narration": narration, "reference_no": reference, "debit": debit, "credit": credit, "balance": balance_value, "source_page": page_number, "mode": infer_mode(narration), }) return rows @staticmethod def _table_geometry(page, words: list[dict]) -> tuple[float, list[float]] | None: required = {"DATE", "PARTICULARS", "DEPOSITS", "WITHDRAWALS", "BALANCE"} for date_word in [word for word in words if word["text"].strip().upper() == "DATE"]: row = [word for word in words if abs(word["top"] - date_word["top"]) <= 2] labels = {word["text"].strip().upper(): word for word in row} if not required.issubset(labels): continue header_top = date_word["top"] horizontal = [] for line in page.lines: top = page.height - line["y0"] if abs(line["y0"] - line["y1"]) <= 1 and abs(top - header_top) <= 8: horizontal.append(line) endpoints = sorted({round(value, 2) for line in horizontal for value in (line["x0"], line["x1"])}) if len(endpoints) >= 7: return header_top, endpoints[:7] # Safe fallback using the known visual order of the header labels. return header_top, [ max(0.0, labels["DATE"]["x0"] - 2), (labels["DATE"]["x1"] + labels.get("MODE**", labels["PARTICULARS"])["x0"]) / 2, (labels.get("MODE**", labels["PARTICULARS"])["x1"] + labels["PARTICULARS"]["x0"]) / 2, (labels["PARTICULARS"]["x1"] + labels["DEPOSITS"]["x0"]) / 2, (labels["DEPOSITS"]["x1"] + labels["WITHDRAWALS"]["x0"]) / 2, (labels["WITHDRAWALS"]["x1"] + labels["BALANCE"]["x0"]) / 2, min(page.width, labels["BALANCE"]["x1"] + 5), ] return None @staticmethod def _cell_text(words: list[dict], left: float, right: float) -> str: selected = [ word for word in words if left <= (word["x0"] + word["x1"]) / 2 < right ] selected.sort(key=lambda word: (round(word["top"], 1), word["x0"])) return norm(" ".join(word["text"] for word in selected)) @staticmethod def _cell_amount(words: list[dict], left: float, right: float) -> float | None: selected = [ word for word in words if left <= (word["x0"] + word["x1"]) / 2 < right and _MONEY_RE.fullmatch(word["text"].strip()) ] if not selected: return None selected.sort(key=lambda word: (word["top"], word["x0"])) return amount(selected[-1]["text"]) @staticmethod def _reference_number(narration: str) -> str: patterns = ( r"(?:MMT/IMPS/|UPI/[^/]+/[^/]+/[^/]+/[^/]+/)(\d{10,18})", r"\b((?:ICI|HDF|AXI|SBI|IBL|PPPL|YJP)[A-Za-z0-9]{10,})\b", r"\b(\d{12,18})\b", ) for pattern in patterns: match = re.search(pattern, narration, re.I) if match: return match.group(1) return "" def _parse_page(self, page, page_number: int) -> tuple[list[dict], float | None]: words = page.extract_words(use_text_flow=False, keep_blank_chars=False) or [] geometry = self._table_geometry(page, words) if not geometry: return [], None header_top, boundaries = geometry left, date_right, mode_right, particulars_right, deposits_right, withdrawals_right, right = boundaries table_words = [word for word in words if word["top"] > header_top + 4] date_words = [ word for word in table_words if _DATE_RE.fullmatch(word["text"].strip()) and left <= (word["x0"] + word["x1"]) / 2 < date_right ] date_words.sort(key=lambda word: word["top"]) if not date_words: return [], None horizontal_groups: dict[float, list[dict]] = {} for line in page.lines: if abs(line["y0"] - line["y1"]) > 1: continue top = round(page.height - line["y0"], 2) if top < header_top - 5: continue horizontal_groups.setdefault(top, []).append(line) horizontal_tops = sorted( top for top, lines in horizontal_groups.items() if min(line["x0"] for line in lines) <= left + 2 and max(line["x1"] for line in lines) >= right - 2 ) rows: list[dict] = [] opening_balance: float | None = None for date_word in date_words: center = (date_word["top"] + date_word["bottom"]) / 2 upper_candidates = [top for top in horizontal_tops if top <= center] lower_candidates = [top for top in horizontal_tops if top > center] band_top = max(upper_candidates) + 0.1 if upper_candidates else header_top + 4 band_bottom = min(lower_candidates) - 0.1 if lower_candidates else min(page.height - 15, center + 30) band_words = [ word for word in table_words if band_top <= (word["top"] + word["bottom"]) / 2 <= band_bottom and left <= (word["x0"] + word["x1"]) / 2 <= right ] mode = self._cell_text(band_words, date_right, mode_right) narration = self._cell_text(band_words, mode_right, particulars_right) deposit = self._cell_amount(band_words, particulars_right, deposits_right) withdrawal = self._cell_amount(band_words, deposits_right, withdrawals_right) balance_value = self._cell_amount(band_words, withdrawals_right, right + 1) combined_narration = norm(f"{mode} {narration}") transaction_date = date_word["text"].strip() if re.fullmatch(r"(?:B/F|BROUGHT FORWARD)", narration.strip(), re.I): opening_balance = balance_value continue if deposit is None and withdrawal is None: continue rows.append({ "transaction_date": transaction_date, "value_date": transaction_date, "narration": combined_narration, "reference_no": self._reference_number(combined_narration), "debit": withdrawal, "credit": deposit, "balance": balance_value, "source_page": page_number, "mode": mode or infer_mode(combined_narration), }) return rows, opening_balance def parse(self, path, text=None): pdf_path = Path(path) text = text or extract_text(pdf_path) meta = StatementMeta( bank_name=self.bank_name, source_file=pdf_path.name, parser_name=self.parser_name, confidence="High", ) name_match = re.search(r"(?m)^\s*((?:MS|MR|MRS)\.?\s*[^\n]+)$", text, re.I) meta.customer_name = norm(re.split(r"\s{2,}", name_match.group(1).strip())[0]) if name_match else "" meta.customer_id = find(r"(?:Cust ID|Customer ID)\s*:\s*([0-9]+)", text) meta.account_number = find(r"(?:Savings Account Number|Account number)\s*:\s*([0-9]+)", text, flags=re.I) if not meta.account_number: meta.account_number = find(r"Savings\s+A/c\s+([0-9]+)", text) if not meta.customer_name: meta.customer_name = norm(find(r"Account name\s*:\s*([^\n]+)", text, flags=re.I)) if not meta.ifsc: meta.ifsc = find(r"IFSC code\s*:\s*([A-Z0-9]+)", text, flags=re.I) meta.ifsc = find(r"IFSC CODE\s+NAME OF NOMINEE.*?Savings\s+[0-9]+\s+[0-9]+\s+([A-Z0-9]+)", text, flags=re.I | re.S) period = re.search( r"for the period\s+([A-Za-z]+\s+\d{1,2},\s*\d{4})\s*-\s*([A-Za-z]+\s+\d{1,2},\s*\d{4})", text, re.I, ) if period: meta.period_from = date_iso(period.group(1)) meta.period_to = date_iso(period.group(2)) all_rows: list[dict] = [] with pdfplumber.open(str(pdf_path)) as pdf: all_rows = self._parse_modern_tables(pdf) if not all_rows: for page_number, page in enumerate(pdf.pages, start=1): page_rows, page_opening = self._parse_page(page, page_number) if meta.opening_balance is None and page_opening is not None: meta.opening_balance = page_opening all_rows.extend(page_rows) data = pd.DataFrame(all_rows) if not data.empty: meta.closing_balance = float(data.iloc[-1]["balance"]) if pd.notna(data.iloc[-1]["balance"]) else None meta.total_debit = round(float(pd.to_numeric(data["debit"], errors="coerce").fillna(0).sum()), 2) meta.total_credit = round(float(pd.to_numeric(data["credit"], errors="coerce").fillna(0).sum()), 2) total_matches = re.findall( r"\bTOTAL\s+([\d,]+\.\d{2})\s+([\d,]+\.\d{2})\s+([\d,]+\.\d{2})", text, re.I, ) total_match = total_matches[-1] if total_matches else None if total_match: printed_credit = amount(total_match[0]) printed_debit = amount(total_match[1]) printed_closing = amount(total_match[2]) # Preserve printed totals where they agree; otherwise the common # reconciliation sheet will expose the exact row-level difference. if printed_credit is not None: meta.total_credit = printed_credit if printed_debit is not None: meta.total_debit = printed_debit if printed_closing is not None: meta.closing_balance = printed_closing return meta, finalize(data, meta)