Files
arrr-erp/app/modules/bank_statement_analyzer/parsers/icici.py
T
2026-08-03 15:53:22 +05:30

240 lines
10 KiB
Python

from __future__ import annotations
import re
from pathlib import Path
import pandas as pd
import pdfplumber
from .base import BaseParser, StatementMeta, amount, extract_text, finalize, norm
from .common import date_iso, find, infer_mode
_DATE_RE = re.compile(r"^\d{2}-\d{2}-\d{4}$")
_MONEY_RE = re.compile(r"^-?(?:\d{1,3}(?:,\d{2,3})+|\d+)\.\d{2}$")
class ICICIParser(BaseParser):
"""Parser for ICICI Bank retail/Privilege PDF account statements.
The statement uses a fixed visual table with DATE, MODE, PARTICULARS,
DEPOSITS, WITHDRAWALS and BALANCE columns. Transactions may span several
visual lines, so parsing is coordinate based rather than dependent on
pdftotext line wrapping.
"""
bank_name = "ICICI Bank"
parser_name = "ICICIParser"
@classmethod
def detect(cls, text: str) -> float:
upper = (text or "").upper()
score = 0.0
if "ICICIBANK" in upper or "ICICI BANK" in upper:
score += 0.55
if "STATEMENT OF TRANSACTIONS IN SAVINGS ACCOUNT" in upper:
score += 0.20
if all(token in upper for token in ("DEPOSITS", "WITHDRAWALS", "BALANCE")):
score += 0.20
if "ACCOUNT RELATED OTHER INFORMATION" in upper:
score += 0.05
return min(score, 0.99)
@staticmethod
def _table_geometry(page, words: list[dict]) -> tuple[float, list[float]] | None:
required = {"DATE", "PARTICULARS", "DEPOSITS", "WITHDRAWALS", "BALANCE"}
for date_word in [word for word in words if word["text"].strip().upper() == "DATE"]:
row = [word for word in words if abs(word["top"] - date_word["top"]) <= 2]
labels = {word["text"].strip().upper(): word for word in row}
if not required.issubset(labels):
continue
header_top = date_word["top"]
horizontal = []
for line in page.lines:
top = page.height - line["y0"]
if abs(line["y0"] - line["y1"]) <= 1 and abs(top - header_top) <= 8:
horizontal.append(line)
endpoints = sorted({round(value, 2) for line in horizontal for value in (line["x0"], line["x1"])})
if len(endpoints) >= 7:
return header_top, endpoints[:7]
# Safe fallback using the known visual order of the header labels.
return header_top, [
max(0.0, labels["DATE"]["x0"] - 2),
(labels["DATE"]["x1"] + labels.get("MODE**", labels["PARTICULARS"])["x0"]) / 2,
(labels.get("MODE**", labels["PARTICULARS"])["x1"] + labels["PARTICULARS"]["x0"]) / 2,
(labels["PARTICULARS"]["x1"] + labels["DEPOSITS"]["x0"]) / 2,
(labels["DEPOSITS"]["x1"] + labels["WITHDRAWALS"]["x0"]) / 2,
(labels["WITHDRAWALS"]["x1"] + labels["BALANCE"]["x0"]) / 2,
min(page.width, labels["BALANCE"]["x1"] + 5),
]
return None
@staticmethod
def _cell_text(words: list[dict], left: float, right: float) -> str:
selected = [
word for word in words
if left <= (word["x0"] + word["x1"]) / 2 < right
]
selected.sort(key=lambda word: (round(word["top"], 1), word["x0"]))
return norm(" ".join(word["text"] for word in selected))
@staticmethod
def _cell_amount(words: list[dict], left: float, right: float) -> float | None:
selected = [
word for word in words
if left <= (word["x0"] + word["x1"]) / 2 < right
and _MONEY_RE.fullmatch(word["text"].strip())
]
if not selected:
return None
selected.sort(key=lambda word: (word["top"], word["x0"]))
return amount(selected[-1]["text"])
@staticmethod
def _reference_number(narration: str) -> str:
patterns = (
r"(?:MMT/IMPS/|UPI/[^/]+/[^/]+/[^/]+/[^/]+/)(\d{10,18})",
r"\b((?:ICI|HDF|AXI|SBI|IBL|PPPL|YJP)[A-Za-z0-9]{10,})\b",
r"\b(\d{12,18})\b",
)
for pattern in patterns:
match = re.search(pattern, narration, re.I)
if match:
return match.group(1)
return ""
def _parse_page(self, page, page_number: int) -> tuple[list[dict], float | None]:
words = page.extract_words(use_text_flow=False, keep_blank_chars=False) or []
geometry = self._table_geometry(page, words)
if not geometry:
return [], None
header_top, boundaries = geometry
left, date_right, mode_right, particulars_right, deposits_right, withdrawals_right, right = boundaries
table_words = [word for word in words if word["top"] > header_top + 4]
date_words = [
word for word in table_words
if _DATE_RE.fullmatch(word["text"].strip())
and left <= (word["x0"] + word["x1"]) / 2 < date_right
]
date_words.sort(key=lambda word: word["top"])
if not date_words:
return [], None
horizontal_groups: dict[float, list[dict]] = {}
for line in page.lines:
if abs(line["y0"] - line["y1"]) > 1:
continue
top = round(page.height - line["y0"], 2)
if top < header_top - 5:
continue
horizontal_groups.setdefault(top, []).append(line)
horizontal_tops = sorted(
top for top, lines in horizontal_groups.items()
if min(line["x0"] for line in lines) <= left + 2
and max(line["x1"] for line in lines) >= right - 2
)
rows: list[dict] = []
opening_balance: float | None = None
for date_word in date_words:
center = (date_word["top"] + date_word["bottom"]) / 2
upper_candidates = [top for top in horizontal_tops if top <= center]
lower_candidates = [top for top in horizontal_tops if top > center]
band_top = max(upper_candidates) + 0.1 if upper_candidates else header_top + 4
band_bottom = min(lower_candidates) - 0.1 if lower_candidates else min(page.height - 15, center + 30)
band_words = [
word for word in table_words
if band_top <= (word["top"] + word["bottom"]) / 2 <= band_bottom
and left <= (word["x0"] + word["x1"]) / 2 <= right
]
mode = self._cell_text(band_words, date_right, mode_right)
narration = self._cell_text(band_words, mode_right, particulars_right)
deposit = self._cell_amount(band_words, particulars_right, deposits_right)
withdrawal = self._cell_amount(band_words, deposits_right, withdrawals_right)
balance_value = self._cell_amount(band_words, withdrawals_right, right + 1)
combined_narration = norm(f"{mode} {narration}")
transaction_date = date_word["text"].strip()
if re.fullmatch(r"(?:B/F|BROUGHT FORWARD)", narration.strip(), re.I):
opening_balance = balance_value
continue
if deposit is None and withdrawal is None:
continue
rows.append({
"transaction_date": transaction_date,
"value_date": transaction_date,
"narration": combined_narration,
"reference_no": self._reference_number(combined_narration),
"debit": withdrawal,
"credit": deposit,
"balance": balance_value,
"source_page": page_number,
"mode": mode or infer_mode(combined_narration),
})
return rows, opening_balance
def parse(self, path, text=None):
pdf_path = Path(path)
text = text or extract_text(pdf_path)
meta = StatementMeta(
bank_name=self.bank_name,
source_file=pdf_path.name,
parser_name=self.parser_name,
confidence="High",
)
name_match = re.search(r"(?m)^\s*((?:MS|MR|MRS)\.?\s*[^\n]+)$", text, re.I)
meta.customer_name = norm(re.split(r"\s{2,}", name_match.group(1).strip())[0]) if name_match else ""
meta.customer_id = find(r"Cust ID\s*:\s*([0-9]+)", text)
meta.account_number = find(r"Savings Account Number\s*:\s*([0-9]+)", text)
if not meta.account_number:
meta.account_number = find(r"Savings\s+A/c\s+([0-9]+)", text)
meta.ifsc = find(r"IFSC CODE\s+NAME OF NOMINEE.*?Savings\s+[0-9]+\s+[0-9]+\s+([A-Z0-9]+)", text, flags=re.I | re.S)
period = re.search(
r"for the period\s+([A-Za-z]+\s+\d{1,2},\s*\d{4})\s*-\s*([A-Za-z]+\s+\d{1,2},\s*\d{4})",
text,
re.I,
)
if period:
meta.period_from = date_iso(period.group(1))
meta.period_to = date_iso(period.group(2))
all_rows: list[dict] = []
with pdfplumber.open(str(pdf_path)) as pdf:
for page_number, page in enumerate(pdf.pages, start=1):
page_rows, page_opening = self._parse_page(page, page_number)
if meta.opening_balance is None and page_opening is not None:
meta.opening_balance = page_opening
all_rows.extend(page_rows)
data = pd.DataFrame(all_rows)
if not data.empty:
meta.closing_balance = float(data.iloc[-1]["balance"]) if pd.notna(data.iloc[-1]["balance"]) else None
meta.total_debit = round(float(pd.to_numeric(data["debit"], errors="coerce").fillna(0).sum()), 2)
meta.total_credit = round(float(pd.to_numeric(data["credit"], errors="coerce").fillna(0).sum()), 2)
total_matches = re.findall(
r"\bTOTAL\s+([\d,]+\.\d{2})\s+([\d,]+\.\d{2})\s+([\d,]+\.\d{2})",
text,
re.I,
)
total_match = total_matches[-1] if total_matches else None
if total_match:
printed_credit = amount(total_match[0])
printed_debit = amount(total_match[1])
printed_closing = amount(total_match[2])
# Preserve printed totals where they agree; otherwise the common
# reconciliation sheet will expose the exact row-level difference.
if printed_credit is not None:
meta.total_credit = printed_credit
if printed_debit is not None:
meta.total_debit = printed_debit
if printed_closing is not None:
meta.closing_balance = printed_closing
return meta, finalize(data, meta)