From 02a549bd05fdbe5168b657f34b284a30847abd14 Mon Sep 17 00:00:00 2001 From: A R R R Associates Date: Tue, 21 Jul 2026 16:45:36 +0530 Subject: [PATCH] Add HSBC bank statement parser with common OCR support --- .../bank_statement_analyzer/parsers/hsbc.py | 302 ++++++++++++++++++ .../parsers/registry.py | 15 +- 2 files changed, 316 insertions(+), 1 deletion(-) create mode 100644 app/modules/bank_statement_analyzer/parsers/hsbc.py diff --git a/app/modules/bank_statement_analyzer/parsers/hsbc.py b/app/modules/bank_statement_analyzer/parsers/hsbc.py new file mode 100644 index 0000000..1d78cec --- /dev/null +++ b/app/modules/bank_statement_analyzer/parsers/hsbc.py @@ -0,0 +1,302 @@ +from __future__ import annotations + +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +import os +import re +import shutil +import subprocess +import tempfile + +import pandas as pd + +from .base import BaseParser, StatementMeta, finalize, norm +from .common import parse_flexible_date + + +_COMPACT_DATE_RE = re.compile(r"^\s*([0-3O]\d[A-Za-z]{3}\d{4})(?:\s+|$)(.*)$", re.I) +_DECIMAL_RE = re.compile(r"(? str: + value = norm(value) + # OCR commonly reads the leading zero as the letter O. + if value and value[0].upper() == "O" and len(value) > 1 and value[1].isdigit(): + value = "0" + value[1:] + parsed = parse_flexible_date(value) + return "" if pd.isna(parsed) else parsed.strftime("%Y-%m-%d") + + +def _hsbc_amount(value: str | None) -> float | None: + if value is None: + return None + text = str(value).strip().replace(" ", "").replace("INR", "") + if not text or text == "-": + return None + # HSBC image OCR may render 84,400.00 as 84,400,00. When no dot is + # present and the final comma has two following digits, treat it as the + # decimal separator and all earlier commas as thousands separators. + if "." not in text and text.count(",") >= 1 and re.search(r",\d{2}$", text): + head, tail = text.rsplit(",", 1) + text = head.replace(",", "") + "." + tail + else: + text = text.replace(",", "") + try: + return float(text) + except ValueError: + return None + + +def _command_available(name: str) -> bool: + return shutil.which(name) is not None + + +def _run_tesseract(image_path: Path, psm: str) -> str: + environment = os.environ.copy() + environment["OMP_THREAD_LIMIT"] = "1" + result = subprocess.run( + ["tesseract", str(image_path), "stdout", "--psm", psm, "-l", "eng"], + capture_output=True, + text=True, + timeout=75, + env=environment, + ) + return result.stdout if result.returncode == 0 else "" + + +def _transaction_page_candidates(images: list[Path]) -> list[int]: + # HSBC transaction pages contain a tall ruled table. Detecting those + # vertical rules is materially faster than OCRing the recurring cover, + # legal-notice and contact pages. Pillow is already installed through + # pdfplumber, so no additional Python dependency is introduced. + from PIL import Image + import numpy as np + + line_counts: list[int] = [] + for image_path in images: + pixels = np.asarray(Image.open(image_path).convert("L")) + height, width = pixels.shape + y1, y2 = int(height * 0.15), int(height * 0.50) + dark_counts = (pixels[y1:y2:2, :] < 100).sum(axis=0) * 2 + threshold = int((y2 - y1) * 0.30) + line_counts.append(int((dark_counts > threshold).sum())) + + candidates: list[int] = [] + for index, count in enumerate(line_counts): + previous_count = line_counts[index - 1] if index > 0 else 0 + next_count = line_counts[index + 1] if index + 1 < len(line_counts) else 0 + if count >= 8: + candidates.append(index + 1) + elif count >= 5 and previous_count >= 8 and next_count <= 1: + # Short closing page at the end of a monthly statement. + candidates.append(index + 1) + return candidates + + +def _ocr_pdf(path: str | Path) -> list[tuple[int, str]]: + if not _command_available("pdftoppm") or not _command_available("tesseract"): + raise RuntimeError( + "HSBC scanned-statement analysis requires the system packages " + "poppler-utils and tesseract-ocr. Redeploy the ERP with the updated Dockerfile." + ) + + with tempfile.TemporaryDirectory(prefix="hsbc_statement_") as temp_dir: + temp = Path(temp_dir) + prefix = temp / "page" + render = subprocess.run( + [ + "pdftoppm", "-r", "110", "-gray", "-jpeg", + "-jpegopt", "quality=70", str(path), str(prefix), + ], + capture_output=True, + text=True, + timeout=180, + ) + if render.returncode != 0: + raise RuntimeError(f"Unable to render HSBC PDF for OCR: {render.stderr.strip()}") + + images = sorted(temp.glob("page-*.jpg")) + candidate_pages = _transaction_page_candidates(images) + if not candidate_pages: + raise ValueError("HSBC statement pages were found, but no transaction tables could be identified.") + + selected = [(page, images[page - 1]) for page in candidate_pages] + workers = max(1, min(4, (os.cpu_count() or 2))) + with ThreadPoolExecutor(max_workers=workers) as executor: + texts = list(executor.map(lambda item: _run_tesseract(item[1], "6"), selected)) + return [(selected[index][0], text) for index, text in enumerate(texts)] + + + +class HSBCParser(BaseParser): + bank_name = "HSBC Bank" + parser_name = "HSBCParserOCR" + + @classmethod + def detect(cls, text: str) -> float: + upper = (text or "").upper() + if "HSBC" in upper and ("STATEMENT OF ACCOUNTS" in upper or "HONGKONG AND SHANGHAI" in upper): + return 0.99 + return 0.0 + + def parse(self, path: str | Path, text: str | None = None): + # The common extraction layer now OCRs image-only and mixed PDFs for all + # banks. Reuse that text first. The targeted HSBC table-page OCR remains + # as a compatibility fallback because older deployments may call this + # parser directly with empty text. + supplied_text = str(text or "") + if supplied_text.strip(): + page_texts = supplied_text.split("\f") + pages = [(index, value) for index, value in enumerate(page_texts, start=1)] + all_text = supplied_text + else: + pages = _ocr_pdf(path) + all_text = "\n\f\n".join(page_text for _, page_text in pages) + upper_all = all_text.upper() + if "HSBC" not in upper_all: + raise ValueError("The uploaded file could not be identified as an HSBC account statement.") + + meta = StatementMeta( + bank_name=self.bank_name, + source_file=Path(path).name, + parser_name=self.parser_name, + confidence="High - OCR with balance-delta reconciliation", + ) + + account_match = re.search(r"SAVINGS\s+ACCOUNT-RES\s+([0-9][0-9\- ]{6,})", all_text, re.I) + if not account_match: + account_match = re.search(r"ACCOUNT\s+NUMBER\s+([0-9][0-9\- ]{6,})", all_text, re.I) + if account_match: + meta.account_number = re.sub(r"\s+", "", account_match.group(1)).strip("-") + + ifsc_match = re.search(r"IFSC\s+CODE\s*:?\s*([A-Z0-9]+)", all_text, re.I) + if ifsc_match: + meta.ifsc = ifsc_match.group(1).upper() + + customer_match = re.search(r"\b(MS|MRS|MR|M/S)\s+([A-Z][A-Z .]{3,60})\n", all_text) + if customer_match: + meta.customer_name = norm(f"{customer_match.group(1)} {customer_match.group(2)}").title() + + rows: list[dict] = [] + current_date = "" + narration_parts: list[str] = [] + opening_balance: float | None = None + last_closing: float | None = None + + for page_number, page_text in pages: + page_upper = page_text.upper() + is_transaction_page = ( + "BALANCE BROUGHT FORWARD" in page_upper + or "BALANCE CARRIED FORWARD" in page_upper + or "CLOSING BALANCE" in page_upper + ) and ("SAVINGS ACCOUNT" in page_upper or "TRANSACTION TURNOVER" in page_upper) + if not is_transaction_page: + continue + + for raw_line in page_text.splitlines(): + line = norm(raw_line) + if not line: + continue + upper = line.upper() + + date_match = _COMPACT_DATE_RE.match(line) + if date_match: + parsed_date = _normalise_compact_date(date_match.group(1)) + if parsed_date: + current_date = parsed_date + line = norm(date_match.group(2)) + upper = line.upper() + + if "BALANCE BROUGHT FORWARD" in upper: + amounts = [_hsbc_amount(m.group(0)) for m in _DECIMAL_RE.finditer(line)] + amounts = [value for value in amounts if value is not None] + if amounts and opening_balance is None: + opening_balance = amounts[-1] + narration_parts = [] + continue + + if "BALANCE CARRIED FORWARD" in upper: + narration_parts = [] + continue + + if "CLOSING BALANCE" in upper: + amounts = [_hsbc_amount(m.group(0)) for m in _DECIMAL_RE.finditer(line)] + amounts = [value for value in amounts if value is not None] + if amounts: + last_closing = amounts[-1] + narration_parts = [] + continue + + if any(marker in upper for marker in ("TRANSACTION TURNOVER", "TRANSACTION COUNT")): + narration_parts = [] + continue + + if upper.startswith(("DATE ", "DETAILS ", "DEPOSITS ", "WITHDRAWALS ", "PAGE ")): + continue + if "THE HONGKONG AND SHANGHAI BANKING" in upper or "WWW.HSBC" in upper: + narration_parts = [] + continue + + amount_matches = list(_DECIMAL_RE.finditer(line)) + values = [_hsbc_amount(match.group(0)) for match in amount_matches] + values = [value for value in values if value is not None] + + if len(values) >= 2 and current_date: + transaction_amount = values[-2] + balance = values[-1] + text_before_amount = line[: amount_matches[-2].start()].strip() + if text_before_amount: + narration_parts.append(text_before_amount) + narration = norm(" ".join(narration_parts)) + if narration and not any(marker in narration.upper() for marker in _SKIP_MARKERS): + reference_match = re.search(r"\b(?:UPI|HIB|CNRBR|SBIN|ICIC|HSBCN)?[A-Z0-9]{10,}\b", narration, re.I) + rows.append( + { + "transaction_date": current_date, + "value_date": current_date, + "narration": narration, + "reference_no": reference_match.group(0) if reference_match else "", + # The shared delta engine in base.finalize() assigns + # the correct side. Seed as debit so the original + # OCR amount remains available as printed_debit. + "debit": transaction_amount, + "credit": None, + "balance": balance, + "source_page": page_number, + } + ) + narration_parts = [] + else: + # Ignore table labels and footer debris while preserving the + # multi-line HSBC narration and references. + if not any(marker in upper for marker in _SKIP_MARKERS): + narration_parts.append(line) + + if not rows: + raise ValueError( + "HSBC statement was identified, but no transaction rows could be extracted. " + "Verify that the PDF pages are readable and redeploy with OCR dependencies." + ) + + meta.opening_balance = opening_balance + meta.closing_balance = last_closing + frame = finalize(pd.DataFrame(rows), meta) + if not frame.empty: + valid_dates = frame["transaction_date"].dropna() + if not valid_dates.empty: + meta.period_from = valid_dates.min().strftime("%Y-%m-%d") + meta.period_to = valid_dates.max().strftime("%Y-%m-%d") + meta.total_debit = round(float(frame["debit"].fillna(0).sum()), 2) + meta.total_credit = round(float(frame["credit"].fillna(0).sum()), 2) + if meta.closing_balance is None and pd.notna(frame["balance"].iloc[-1]): + meta.closing_balance = float(frame["balance"].iloc[-1]) + return meta, frame diff --git a/app/modules/bank_statement_analyzer/parsers/registry.py b/app/modules/bank_statement_analyzer/parsers/registry.py index 585210f..d9e71b5 100644 --- a/app/modules/bank_statement_analyzer/parsers/registry.py +++ b/app/modules/bank_statement_analyzer/parsers/registry.py @@ -3,6 +3,7 @@ from __future__ import annotations from .idfc import IDFCFirstParser from .axis import AxisParser from .hdfc import HDFCParser +from .hsbc import HSBCParser from .indian_bank import IndianBankModernParser, IndianBankLegacyParser from .indusind import IndusIndParser from .kotak import KotakParser @@ -13,6 +14,7 @@ PARSERS = [ IDFCFirstParser, AxisParser, HDFCParser, + HSBCParser, IndianBankModernParser, IndianBankLegacyParser, IndusIndParser, @@ -25,6 +27,7 @@ BANK_OPTIONS = [ ("auto", "Auto Detect"), ("axis", "Axis Bank"), ("hdfc", "HDFC Bank"), + ("hsbc", "HSBC Bank"), ("idfc", "IDFC FIRST Bank"), ("indian_bank", "Indian Bank"), ("indusind", "IndusInd Bank"), @@ -35,6 +38,7 @@ BANK_OPTIONS = [ BANK_PARSERS = { "axis": [AxisParser], "hdfc": [HDFCParser], + "hsbc": [HSBCParser], "idfc": [IDFCFirstParser], "indian_bank": [IndianBankModernParser, IndianBankLegacyParser], "indusind": [IndusIndParser], @@ -61,8 +65,10 @@ def _selected_parser(bank_key: str, text: str): def parse_pdf(path, bank_hint: str | None = None): - text = extract_text(path) hint = (bank_hint or "auto").strip().lower() + if hint == "hsbc": + return HSBCParser().parse(path, "") + text = extract_text(path) if hint and hint != "auto": parser, score = _selected_parser(hint, text) if parser is None: @@ -73,6 +79,13 @@ def parse_pdf(path, bank_hint: str | None = None): ) return parser.parse(path, text) parser, score = detect_parser(text) + if parser is None and not (text or "").strip(): + # Image-only statements cannot be identified by pdftotext. HSBCParser + # performs its own OCR identification and raises a precise mismatch error. + try: + return HSBCParser().parse(path, text) + except ValueError: + pass if parser is None: raise ValueError( "Unsupported statement format. Select the bank manually or add a bank-specific parser for this statement layout."