Add independent bank identity detection and APGB parser
This commit is contained in:
@@ -1,9 +1,5 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
from .idfc import IDFCFirstParser
|
||||
from .axis import AxisParser
|
||||
from .hdfc import HDFCParser
|
||||
@@ -18,12 +14,15 @@ from .yes_bank import YesBankParser
|
||||
from .city_union_bank import CityUnionBankParser
|
||||
from .bank_of_baroda import BankOfBarodaParser
|
||||
from .rbl_bank import RBLBankParser
|
||||
from .andhra_pradesh_grameena_bank import AndhraPradeshGrameenaBankParser
|
||||
from .bank_identity import apply_detected_bank_identity
|
||||
from .base import extract_text
|
||||
from .template_engine import parse_with_template
|
||||
|
||||
|
||||
PARSERS = [
|
||||
IDFCFirstParser,
|
||||
AxisParser,
|
||||
AndhraPradeshGrameenaBankParser,
|
||||
CentralBankOfIndiaParser,
|
||||
YesBankParser,
|
||||
CityUnionBankParser,
|
||||
@@ -43,6 +42,7 @@ PARSERS = [
|
||||
BANK_OPTIONS = [
|
||||
("auto", "Auto Detect"),
|
||||
("axis", "Axis Bank"),
|
||||
("andhra_pradesh_grameena_bank", "Andhra Pradesh Grameena Bank"),
|
||||
("central_bank_of_india", "Central Bank of India"),
|
||||
("yes_bank", "YES Bank"),
|
||||
("city_union_bank", "City Union Bank"),
|
||||
@@ -60,6 +60,7 @@ BANK_OPTIONS = [
|
||||
|
||||
BANK_PARSERS = {
|
||||
"axis": [AxisParser],
|
||||
"andhra_pradesh_grameena_bank": [AndhraPradeshGrameenaBankParser],
|
||||
"central_bank_of_india": [CentralBankOfIndiaParser],
|
||||
"yes_bank": [YesBankParser],
|
||||
"city_union_bank": [CityUnionBankParser],
|
||||
@@ -77,7 +78,11 @@ BANK_PARSERS = {
|
||||
|
||||
|
||||
def detect_parser(text):
|
||||
scored = sorted(((parser.detect(text), parser) for parser in PARSERS), key=lambda item: item[0], reverse=True)
|
||||
scored = sorted(
|
||||
((parser.detect(text), parser) for parser in PARSERS),
|
||||
key=lambda item: item[0],
|
||||
reverse=True,
|
||||
)
|
||||
if not scored or scored[0][0] <= 0:
|
||||
return None, 0
|
||||
return scored[0][1](), scored[0][0]
|
||||
@@ -87,115 +92,50 @@ def _selected_parser(bank_key: str, text: str):
|
||||
candidates = BANK_PARSERS.get(bank_key, [])
|
||||
if not candidates:
|
||||
return None, 0
|
||||
scored = sorted(((parser.detect(text), parser) for parser in candidates), key=lambda item: item[0], reverse=True)
|
||||
scored = sorted(
|
||||
((parser.detect(text), parser) for parser in candidates),
|
||||
key=lambda item: item[0],
|
||||
reverse=True,
|
||||
)
|
||||
if scored and scored[0][0] > 0:
|
||||
return scored[0][1](), scored[0][0]
|
||||
return None, 0
|
||||
|
||||
|
||||
def _usable_result(result) -> bool:
|
||||
if result is None:
|
||||
return False
|
||||
try:
|
||||
_meta, frame = result
|
||||
return frame is not None and not frame.empty
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _attempt(parser, path, text: str, route: str):
|
||||
try:
|
||||
result = parser.parse(path, text)
|
||||
if _usable_result(result):
|
||||
meta, frame = result
|
||||
logger.info(
|
||||
"Bank analyzer route accepted: route=%s parser=%s rows=%s file=%s",
|
||||
route, getattr(meta, "parser_name", parser.__class__.__name__), len(frame), path,
|
||||
)
|
||||
return result
|
||||
logger.warning(
|
||||
"Bank analyzer route returned no rows: route=%s parser=%s file=%s",
|
||||
route, parser.__class__.__name__, path,
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.warning(
|
||||
"Bank analyzer route failed: route=%s parser=%s file=%s error=%s",
|
||||
route, parser.__class__.__name__, path, exc, exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _template_first(path, text: str, hint: str):
|
||||
try:
|
||||
result = parse_with_template(path, text=text, bank_hint=hint)
|
||||
if _usable_result(result):
|
||||
meta, frame = result
|
||||
logger.info(
|
||||
"Bank analyzer template accepted: parser=%s rows=%s file=%s",
|
||||
getattr(meta, "parser_name", "TemplateBasedStatementParser"), len(frame), path,
|
||||
)
|
||||
return result
|
||||
logger.warning("Bank analyzer template returned no rows: file=%s", path)
|
||||
except Exception as exc:
|
||||
logger.warning(
|
||||
"Bank analyzer template rejected: file=%s error=%s", path, exc, exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _ordered_bank_candidates(text: str):
|
||||
scored = sorted(
|
||||
((parser.detect(text), parser) for parser in PARSERS),
|
||||
key=lambda item: item[0],
|
||||
reverse=True,
|
||||
)
|
||||
return [parser for score, parser in scored if score > 0]
|
||||
def _parse_and_identify(parser, path, text: str):
|
||||
meta, df = parser.parse(path, text)
|
||||
return apply_detected_bank_identity(meta, df, text)
|
||||
|
||||
|
||||
def parse_pdf(path, bank_hint: str | None = None):
|
||||
hint = (bank_hint or "auto").strip().lower()
|
||||
text = extract_text(path)
|
||||
|
||||
template_result = _template_first(path, text, hint)
|
||||
if template_result is not None:
|
||||
return template_result
|
||||
if hint == "hsbc":
|
||||
meta, df = HSBCParser().parse(path, text)
|
||||
return apply_detected_bank_identity(meta, df, text)
|
||||
|
||||
if hint and hint != "auto":
|
||||
candidates = BANK_PARSERS.get(hint, [])
|
||||
for parser_class in sorted(candidates, key=lambda cls: cls.detect(text), reverse=True):
|
||||
result = _attempt(parser_class(), path, text, f"selected:{hint}")
|
||||
if result is not None:
|
||||
return result
|
||||
label = dict(BANK_OPTIONS).get(hint, "the selected bank")
|
||||
parser, score = _selected_parser(hint, text)
|
||||
if parser is None:
|
||||
label = dict(BANK_OPTIONS).get(hint, "the selected bank")
|
||||
raise ValueError(
|
||||
f"The uploaded statement does not match the selected {label} format. "
|
||||
"Please verify the selected bank or choose Auto Detect."
|
||||
)
|
||||
return _parse_and_identify(parser, path, text)
|
||||
|
||||
parser, score = detect_parser(text)
|
||||
if parser is None and not (text or "").strip():
|
||||
# Image-only statements cannot be identified by pdftotext. HSBCParser
|
||||
# performs its own OCR identification and raises a precise mismatch error.
|
||||
try:
|
||||
meta, df = HSBCParser().parse(path, text)
|
||||
return apply_detected_bank_identity(meta, df, text)
|
||||
except ValueError:
|
||||
pass
|
||||
if parser is None:
|
||||
raise ValueError(
|
||||
f"The uploaded statement could not be parsed by a validated layout template or the "
|
||||
f"selected {label} parser. Please verify that the PDF is readable and supported."
|
||||
"Unsupported statement format. Select the bank manually or add a bank-specific parser for this statement layout."
|
||||
)
|
||||
|
||||
attempted: set[type] = set()
|
||||
for parser_class in _ordered_bank_candidates(text):
|
||||
attempted.add(parser_class)
|
||||
result = _attempt(parser_class(), path, text, "auto-detected")
|
||||
if result is not None:
|
||||
return result
|
||||
|
||||
if not (text or "").strip():
|
||||
result = _attempt(HSBCParser(), path, text, "image-only-ocr")
|
||||
if result is not None:
|
||||
return result
|
||||
|
||||
# Last-resort compatibility pass: a detector may miss a new layout even
|
||||
# though the bank-specific parser can parse its visual table correctly.
|
||||
for parser_class in PARSERS:
|
||||
if parser_class in attempted:
|
||||
continue
|
||||
result = _attempt(parser_class(), path, text, "compatibility-fallback")
|
||||
if result is not None:
|
||||
return result
|
||||
|
||||
raise ValueError(
|
||||
"The bank statement format was identified, but no transaction rows could be extracted. "
|
||||
"The template engine and all bank-specific parsers were attempted. "
|
||||
"Please verify that the PDF is text-readable and that this statement layout is supported."
|
||||
)
|
||||
|
||||
return _parse_and_identify(parser, path, text)
|
||||
|
||||
Reference in New Issue
Block a user