Fix SBI multiline table parsing and restore template-first bank analysis

This commit is contained in:
A R R R Associates
2026-08-05 17:49:44 +05:30
parent 8bf409049b
commit 47cd5c2cf3
2 changed files with 165 additions and 49 deletions
@@ -1,5 +1,7 @@
from __future__ import annotations
import logging
from .idfc import IDFCFirstParser
from .axis import AxisParser
from .hdfc import HDFCParser
@@ -17,7 +19,9 @@ from .rbl_bank import RBLBankParser
from .andhra_pradesh_grameena_bank import AndhraPradeshGrameenaBankParser
from .bank_identity import apply_detected_bank_identity
from .base import extract_text
from .template_engine import parse_with_template
logger = logging.getLogger(__name__)
PARSERS = [
IDFCFirstParser,
@@ -88,54 +92,137 @@ def detect_parser(text):
return scored[0][1](), scored[0][0]
def _selected_parser(bank_key: str, text: str):
candidates = BANK_PARSERS.get(bank_key, [])
if not candidates:
return None, 0
def _usable_result(result) -> bool:
if result is None:
return False
try:
_meta, frame = result
return frame is not None and not frame.empty
except Exception:
return False
def _identified(result, text: str):
if not _usable_result(result):
return None
meta, frame = result
return apply_detected_bank_identity(meta, frame, text)
def _attempt(parser, path, text: str, route: str):
try:
result = parser.parse(path, text)
result = _identified(result, text)
if result is not None:
meta, frame = result
logger.info(
"Bank analyzer route accepted: route=%s parser=%s rows=%s file=%s",
route,
getattr(meta, "parser_name", parser.__class__.__name__),
len(frame),
path,
)
return result
logger.warning(
"Bank analyzer route returned no rows: route=%s parser=%s file=%s",
route,
parser.__class__.__name__,
path,
)
except Exception as exc:
logger.warning(
"Bank analyzer route failed: route=%s parser=%s file=%s error=%s",
route,
parser.__class__.__name__,
path,
exc,
exc_info=True,
)
return None
def _template_first(path, text: str, hint: str):
try:
result = parse_with_template(path, text=text, bank_hint=hint)
result = _identified(result, text)
if result is not None:
meta, frame = result
logger.info(
"Bank analyzer template accepted: parser=%s rows=%s file=%s",
getattr(meta, "parser_name", "TemplateBasedStatementParser"),
len(frame),
path,
)
return result
logger.warning("Bank analyzer template returned no rows: file=%s", path)
except Exception as exc:
logger.warning(
"Bank analyzer template rejected: file=%s error=%s",
path,
exc,
exc_info=True,
)
return None
def _ordered_bank_candidates(text: str):
scored = sorted(
((parser.detect(text), parser) for parser in candidates),
((parser.detect(text), parser) for parser in PARSERS),
key=lambda item: item[0],
reverse=True,
)
if scored and scored[0][0] > 0:
return scored[0][1](), scored[0][0]
return None, 0
def _parse_and_identify(parser, path, text: str):
meta, df = parser.parse(path, text)
return apply_detected_bank_identity(meta, df, text)
return [parser for score, parser in scored if score > 0]
def parse_pdf(path, bank_hint: str | None = None):
"""Parse a statement using a validated template first, then bank parsers.
Bank identity is resolved independently after extraction, so using a shared
layout never labels a statement as the bank whose parser happens to have a
similar table structure.
"""
hint = (bank_hint or "auto").strip().lower()
text = extract_text(path)
if hint == "hsbc":
meta, df = HSBCParser().parse(path, text)
return apply_detected_bank_identity(meta, df, text)
template_result = _template_first(path, text, hint)
if template_result is not None:
return template_result
if hint and hint != "auto":
parser, score = _selected_parser(hint, text)
if parser is None:
label = dict(BANK_OPTIONS).get(hint, "the selected bank")
raise ValueError(
f"The uploaded statement does not match the selected {label} format. "
"Please verify the selected bank or choose Auto Detect."
)
return _parse_and_identify(parser, path, text)
parser, score = detect_parser(text)
if parser is None and not (text or "").strip():
# Image-only statements cannot be identified by pdftotext. HSBCParser
# performs its own OCR identification and raises a precise mismatch error.
try:
meta, df = HSBCParser().parse(path, text)
return apply_detected_bank_identity(meta, df, text)
except ValueError:
pass
if parser is None:
candidates = BANK_PARSERS.get(hint, [])
for parser_class in sorted(candidates, key=lambda cls: cls.detect(text), reverse=True):
result = _attempt(parser_class(), path, text, f"selected:{hint}")
if result is not None:
return result
label = dict(BANK_OPTIONS).get(hint, "the selected bank")
raise ValueError(
"Unsupported statement format. Select the bank manually or add a bank-specific parser for this statement layout."
f"The uploaded statement could not be parsed by a validated layout template or the "
f"selected {label} parser. Please verify that the PDF is readable and supported."
)
return _parse_and_identify(parser, path, text)
attempted: set[type] = set()
for parser_class in _ordered_bank_candidates(text):
attempted.add(parser_class)
result = _attempt(parser_class(), path, text, "auto-detected")
if result is not None:
return result
if not (text or "").strip():
result = _attempt(HSBCParser(), path, text, "image-only-ocr")
if result is not None:
return result
# A detector may miss a new bank layout even though an existing dedicated
# parser can still extract and reconcile it. Keep this compatibility pass.
for parser_class in PARSERS:
if parser_class in attempted:
continue
result = _attempt(parser_class(), path, text, "compatibility-fallback")
if result is not None:
return result
raise ValueError(
"The bank statement format was identified, but no transaction rows could be extracted. "
"The validated template engine and all bank-specific parsers were attempted. "
"Please verify that the PDF is text-readable and that this statement layout is supported."
)