Fix modern ICICI template parsing and parser fallback
This commit is contained in:
@@ -10,10 +10,15 @@ from .base import BaseParser, StatementMeta, amount, extract_text, finalize, nor
|
||||
from .common import date_iso, find, infer_mode
|
||||
|
||||
|
||||
_DATE_RE = re.compile(r"^\d{2}-\d{2}-\d{4}$")
|
||||
_DATE_RE = re.compile(r"^(?:\d{2}-\d{2}-\d{4}|\d{2}-[A-Za-z]{3}-\d{4})$")
|
||||
_MONEY_RE = re.compile(r"^-?(?:\d{1,3}(?:,\d{2,3})+|\d+)\.\d{2}$")
|
||||
|
||||
|
||||
def _clean_date_cell(value: str) -> str:
|
||||
text = norm(value)
|
||||
return re.sub(r"\s*([/-])\s*", r"\1", text)
|
||||
|
||||
|
||||
class ICICIParser(BaseParser):
|
||||
"""Parser for ICICI Bank retail/Privilege PDF account statements.
|
||||
|
||||
@@ -34,12 +39,83 @@ class ICICIParser(BaseParser):
|
||||
score += 0.55
|
||||
if "STATEMENT OF TRANSACTIONS IN SAVINGS ACCOUNT" in upper:
|
||||
score += 0.20
|
||||
if "ACCOUNT STATEMENT" in upper and "TRANSACTION ID" in upper:
|
||||
score += 0.20
|
||||
if "AVAILABLE BALANCE" in upper and "WITHDRAWAL" in upper and "DEPOSIT" in upper:
|
||||
score += 0.15
|
||||
if all(token in upper for token in ("DEPOSITS", "WITHDRAWALS", "BALANCE")):
|
||||
score += 0.20
|
||||
if "ACCOUNT RELATED OTHER INFORMATION" in upper:
|
||||
score += 0.05
|
||||
return min(score, 0.99)
|
||||
|
||||
@staticmethod
|
||||
def _modern_header_mapping(row: list) -> dict[str, int] | None:
|
||||
cells = [norm(cell).upper() for cell in (row or [])]
|
||||
aliases = {
|
||||
"serial": ("S.NO", "S NO", "SERIAL NO"),
|
||||
"transaction_id": ("TRANSACTION ID", "TXN ID"),
|
||||
"transaction_date": ("TRANSACTION DATE", "TXN DATE"),
|
||||
"cheque_no": ("CHEQUE NO", "CHQ NO"),
|
||||
"description": ("DESCRIPTION", "PARTICULARS"),
|
||||
"debit": ("WITHDRAWAL (DR)", "WITHDRAWAL", "DEBIT"),
|
||||
"credit": ("DEPOSIT (CR)", "DEPOSIT", "CREDIT"),
|
||||
"balance": ("AVAILABLE BALANCE", "BALANCE"),
|
||||
}
|
||||
mapping: dict[str, int] = {}
|
||||
for field, names in aliases.items():
|
||||
for index, cell in enumerate(cells):
|
||||
if any(name == cell or name in cell for name in names):
|
||||
mapping[field] = index
|
||||
break
|
||||
required = {"transaction_date", "description", "debit", "credit", "balance"}
|
||||
return mapping if required.issubset(mapping) else None
|
||||
|
||||
def _parse_modern_tables(self, pdf) -> list[dict]:
|
||||
rows: list[dict] = []
|
||||
active_mapping: dict[str, int] | None = None
|
||||
for page_number, page in enumerate(pdf.pages, start=1):
|
||||
for table in page.extract_tables() or []:
|
||||
if not table:
|
||||
continue
|
||||
start_index = 0
|
||||
header_mapping = self._modern_header_mapping(table[0])
|
||||
if header_mapping:
|
||||
active_mapping = header_mapping
|
||||
start_index = 1
|
||||
if active_mapping is None:
|
||||
continue
|
||||
mapping = active_mapping
|
||||
for raw in table[start_index:]:
|
||||
cells = list(raw or [])
|
||||
max_index = max(mapping.values())
|
||||
if len(cells) <= max_index:
|
||||
cells.extend([None] * (max_index + 1 - len(cells)))
|
||||
transaction_date = _clean_date_cell(cells[mapping["transaction_date"]])
|
||||
if not _DATE_RE.fullmatch(transaction_date):
|
||||
continue
|
||||
narration = norm(cells[mapping["description"]])
|
||||
transaction_id = norm(cells[mapping.get("transaction_id", -1)]) if "transaction_id" in mapping else ""
|
||||
cheque_no = norm(cells[mapping.get("cheque_no", -1)]) if "cheque_no" in mapping else ""
|
||||
debit = amount(cells[mapping["debit"]])
|
||||
credit = amount(cells[mapping["credit"]])
|
||||
balance_value = amount(cells[mapping["balance"]])
|
||||
if balance_value is None or (debit is None and credit is None):
|
||||
continue
|
||||
reference = cheque_no if cheque_no and cheque_no != "-" else transaction_id
|
||||
rows.append({
|
||||
"transaction_date": date_iso(transaction_date),
|
||||
"value_date": date_iso(transaction_date),
|
||||
"narration": narration,
|
||||
"reference_no": reference,
|
||||
"debit": debit,
|
||||
"credit": credit,
|
||||
"balance": balance_value,
|
||||
"source_page": page_number,
|
||||
"mode": infer_mode(narration),
|
||||
})
|
||||
return rows
|
||||
|
||||
@staticmethod
|
||||
def _table_geometry(page, words: list[dict]) -> tuple[float, list[float]] | None:
|
||||
required = {"DATE", "PARTICULARS", "DEPOSITS", "WITHDRAWALS", "BALANCE"}
|
||||
@@ -188,10 +264,14 @@ class ICICIParser(BaseParser):
|
||||
|
||||
name_match = re.search(r"(?m)^\s*((?:MS|MR|MRS)\.?\s*[^\n]+)$", text, re.I)
|
||||
meta.customer_name = norm(re.split(r"\s{2,}", name_match.group(1).strip())[0]) if name_match else ""
|
||||
meta.customer_id = find(r"Cust ID\s*:\s*([0-9]+)", text)
|
||||
meta.account_number = find(r"Savings Account Number\s*:\s*([0-9]+)", text)
|
||||
meta.customer_id = find(r"(?:Cust ID|Customer ID)\s*:\s*([0-9]+)", text)
|
||||
meta.account_number = find(r"(?:Savings Account Number|Account number)\s*:\s*([0-9]+)", text, flags=re.I)
|
||||
if not meta.account_number:
|
||||
meta.account_number = find(r"Savings\s+A/c\s+([0-9]+)", text)
|
||||
if not meta.customer_name:
|
||||
meta.customer_name = norm(find(r"Account name\s*:\s*([^\n]+)", text, flags=re.I))
|
||||
if not meta.ifsc:
|
||||
meta.ifsc = find(r"IFSC code\s*:\s*([A-Z0-9]+)", text, flags=re.I)
|
||||
meta.ifsc = find(r"IFSC CODE\s+NAME OF NOMINEE.*?Savings\s+[0-9]+\s+[0-9]+\s+([A-Z0-9]+)", text, flags=re.I | re.S)
|
||||
|
||||
period = re.search(
|
||||
@@ -205,6 +285,8 @@ class ICICIParser(BaseParser):
|
||||
|
||||
all_rows: list[dict] = []
|
||||
with pdfplumber.open(str(pdf_path)) as pdf:
|
||||
all_rows = self._parse_modern_tables(pdf)
|
||||
if not all_rows:
|
||||
for page_number, page in enumerate(pdf.pages, start=1):
|
||||
page_rows, page_opening = self._parse_page(page, page_number)
|
||||
if meta.opening_balance is None and page_opening is not None:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
from .idfc import IDFCFirstParser
|
||||
from .axis import AxisParser
|
||||
from .hdfc import HDFCParser
|
||||
@@ -89,18 +93,65 @@ def _selected_parser(bank_key: str, text: str):
|
||||
return None, 0
|
||||
|
||||
|
||||
def _template_first(path, text: str, hint: str):
|
||||
"""Try the validated layout engine before bank-specific parsing.
|
||||
|
||||
Any template error is deliberately swallowed here so the existing bank
|
||||
parser path remains an unchanged and reliable fallback.
|
||||
"""
|
||||
def _usable_result(result) -> bool:
|
||||
if result is None:
|
||||
return False
|
||||
try:
|
||||
return parse_with_template(path, text=text, bank_hint=hint)
|
||||
_meta, frame = result
|
||||
return frame is not None and not frame.empty
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _attempt(parser, path, text: str, route: str):
|
||||
try:
|
||||
result = parser.parse(path, text)
|
||||
if _usable_result(result):
|
||||
meta, frame = result
|
||||
logger.info(
|
||||
"Bank analyzer route accepted: route=%s parser=%s rows=%s file=%s",
|
||||
route, getattr(meta, "parser_name", parser.__class__.__name__), len(frame), path,
|
||||
)
|
||||
return result
|
||||
logger.warning(
|
||||
"Bank analyzer route returned no rows: route=%s parser=%s file=%s",
|
||||
route, parser.__class__.__name__, path,
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.warning(
|
||||
"Bank analyzer route failed: route=%s parser=%s file=%s error=%s",
|
||||
route, parser.__class__.__name__, path, exc, exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _template_first(path, text: str, hint: str):
|
||||
try:
|
||||
result = parse_with_template(path, text=text, bank_hint=hint)
|
||||
if _usable_result(result):
|
||||
meta, frame = result
|
||||
logger.info(
|
||||
"Bank analyzer template accepted: parser=%s rows=%s file=%s",
|
||||
getattr(meta, "parser_name", "TemplateBasedStatementParser"), len(frame), path,
|
||||
)
|
||||
return result
|
||||
logger.warning("Bank analyzer template returned no rows: file=%s", path)
|
||||
except Exception as exc:
|
||||
logger.warning(
|
||||
"Bank analyzer template rejected: file=%s error=%s", path, exc, exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _ordered_bank_candidates(text: str):
|
||||
scored = sorted(
|
||||
((parser.detect(text), parser) for parser in PARSERS),
|
||||
key=lambda item: item[0],
|
||||
reverse=True,
|
||||
)
|
||||
return [parser for score, parser in scored if score > 0]
|
||||
|
||||
|
||||
def parse_pdf(path, bank_hint: str | None = None):
|
||||
hint = (bank_hint or "auto").strip().lower()
|
||||
text = extract_text(path)
|
||||
@@ -109,30 +160,42 @@ def parse_pdf(path, bank_hint: str | None = None):
|
||||
if template_result is not None:
|
||||
return template_result
|
||||
|
||||
# Preserve the existing HSBC OCR route for image-only statements and for
|
||||
# users/API clients that explicitly provide an HSBC override.
|
||||
if hint == "hsbc":
|
||||
return HSBCParser().parse(path, text)
|
||||
|
||||
if hint and hint != "auto":
|
||||
parser, score = _selected_parser(hint, text)
|
||||
if parser is None:
|
||||
candidates = BANK_PARSERS.get(hint, [])
|
||||
for parser_class in sorted(candidates, key=lambda cls: cls.detect(text), reverse=True):
|
||||
result = _attempt(parser_class(), path, text, f"selected:{hint}")
|
||||
if result is not None:
|
||||
return result
|
||||
label = dict(BANK_OPTIONS).get(hint, "the selected bank")
|
||||
raise ValueError(
|
||||
f"The uploaded statement could not be parsed by a validated layout template and does not match "
|
||||
f"the selected {label} format. Please verify the statement or use automatic detection."
|
||||
f"The uploaded statement could not be parsed by a validated layout template or the "
|
||||
f"selected {label} parser. Please verify that the PDF is readable and supported."
|
||||
)
|
||||
return parser.parse(path, text)
|
||||
|
||||
parser, score = detect_parser(text)
|
||||
if parser is None and not (text or "").strip():
|
||||
try:
|
||||
return HSBCParser().parse(path, text)
|
||||
except ValueError:
|
||||
pass
|
||||
if parser is None:
|
||||
attempted: set[type] = set()
|
||||
for parser_class in _ordered_bank_candidates(text):
|
||||
attempted.add(parser_class)
|
||||
result = _attempt(parser_class(), path, text, "auto-detected")
|
||||
if result is not None:
|
||||
return result
|
||||
|
||||
if not (text or "").strip():
|
||||
result = _attempt(HSBCParser(), path, text, "image-only-ocr")
|
||||
if result is not None:
|
||||
return result
|
||||
|
||||
# Last-resort compatibility pass: a detector may miss a new layout even
|
||||
# though the bank-specific parser can parse its visual table correctly.
|
||||
for parser_class in PARSERS:
|
||||
if parser_class in attempted:
|
||||
continue
|
||||
result = _attempt(parser_class(), path, text, "compatibility-fallback")
|
||||
if result is not None:
|
||||
return result
|
||||
|
||||
raise ValueError(
|
||||
"Unsupported statement format. The validated template engine and all available bank-specific parsers "
|
||||
"were unable to reconcile this statement."
|
||||
"The bank statement format was identified, but no transaction rows could be extracted. "
|
||||
"The template engine and all bank-specific parsers were attempted. "
|
||||
"Please verify that the PDF is text-readable and that this statement layout is supported."
|
||||
)
|
||||
return parser.parse(path, text)
|
||||
|
||||
|
||||
@@ -77,8 +77,15 @@ def _upper(value) -> str:
|
||||
return norm(value).upper().replace("\n", " ")
|
||||
|
||||
|
||||
def _is_date(value) -> bool:
|
||||
def _normalise_date_cell(value) -> str:
|
||||
"""Normalise PDF table line-wrap artefacts without changing the date value."""
|
||||
text = norm(value)
|
||||
text = re.sub(r"\s*([/-])\s*", r"\1", text)
|
||||
return text
|
||||
|
||||
|
||||
def _is_date(value) -> bool:
|
||||
text = _normalise_date_cell(value)
|
||||
return bool(_DATE_CELL_RE.fullmatch(text)) and not pd.isna(parse_flexible_date(text))
|
||||
|
||||
|
||||
@@ -102,7 +109,7 @@ def _find_header_mapping(row: list) -> dict[str, int]:
|
||||
|
||||
|
||||
def _looks_like_data_row(row: list) -> bool:
|
||||
return bool(row) and (_is_date(row[0]) or (len(row) > 1 and _is_date(row[1])))
|
||||
return bool(row) and any(_is_date(row[index]) for index in range(min(4, len(row))))
|
||||
|
||||
|
||||
def _infer_structural_template(table: list[list]) -> LayoutTemplate | None:
|
||||
@@ -110,6 +117,23 @@ def _infer_structural_template(table: list[list]) -> LayoutTemplate | None:
|
||||
if sample is None:
|
||||
return None
|
||||
width = len(sample)
|
||||
# Modern ICICI and similar indexed statements:
|
||||
# serial no, transaction id, transaction date, cheque no, description,
|
||||
# withdrawal, deposit, available balance.
|
||||
if width >= 8 and _is_date(sample[2]):
|
||||
return LayoutTemplate(
|
||||
"indexed_transaction_id_debit_credit_balance",
|
||||
{
|
||||
"transaction_date": 2,
|
||||
"value_date": 2,
|
||||
"reference_no": 1,
|
||||
"narration": 4,
|
||||
"debit": 5,
|
||||
"credit": 6,
|
||||
"balance": 7,
|
||||
},
|
||||
0.92,
|
||||
)
|
||||
# SBI Account Summary / SBI YONO family:
|
||||
# value date, post date, details, reference, debit, credit, balance.
|
||||
if width >= 7 and _is_date(sample[0]) and _is_date(sample[1]):
|
||||
@@ -354,11 +378,11 @@ class TemplateBasedStatementParser(BaseParser):
|
||||
max_index = max(mapping.values())
|
||||
if len(cells) <= max_index:
|
||||
cells += [None] * (max_index + 1 - len(cells))
|
||||
txn_text = norm(cells[mapping["transaction_date"]])
|
||||
txn_text = _normalise_date_cell(cells[mapping["transaction_date"]])
|
||||
if not _is_date(txn_text):
|
||||
continue
|
||||
value_index = mapping.get("value_date", mapping["transaction_date"])
|
||||
value_text = norm(cells[value_index]) or txn_text
|
||||
value_text = _normalise_date_cell(cells[value_index]) or txn_text
|
||||
narration = norm(cells[mapping["narration"]])
|
||||
reference = norm(cells[mapping["reference_no"]]) if "reference_no" in mapping else ""
|
||||
debit = amount(cells[mapping["debit"]])
|
||||
|
||||
Reference in New Issue
Block a user