Fix Indian Bank extraction and add shared multi-format dates for all banks

This commit is contained in:
A R R R Associates
2026-07-16 21:56:08 +05:30
parent 43ab9c9aae
commit d9ae1519a4
10 changed files with 340 additions and 311 deletions
@@ -5,21 +5,24 @@ from pathlib import Path
import pandas as pd
from .base import *
from .common import find
from .base import BaseParser, StatementMeta, amount, extract_text, finalize, norm
from .common import DATE_TOKEN_PATTERN, date_iso, find, parse_flexible_date
def _signed_amount(value: str | None, suffix: str | None) -> float | None:
"""Return CR as positive and DR as negative."""
parsed = amount(value)
if parsed is None:
return None
return -parsed if (suffix or "").upper() == "DR" else parsed
def _parse_modern_date(value: str) -> str:
"""Normalize both '01 Apr 2025' and 'Apr 01 2025' to ISO date."""
return pd.to_datetime(value, dayfirst=True, errors="raise").strftime("%Y-%m-%d")
def _nearest_column(position: int, debit_pos: int, credit_pos: int, balance_pos: int) -> str:
distances = {
"debit": abs(position - debit_pos),
"credit": abs(position - credit_pos),
"balance": abs(position - balance_pos),
}
return min(distances, key=distances.get)
class IndianBankModernParser(BaseParser):
@@ -28,18 +31,8 @@ class IndianBankModernParser(BaseParser):
@classmethod
def detect(cls, text):
# pdfplumber's layout mode can insert multiple spaces inside headings
# (for example, 'ACCOUNT STATEMENT'). Normalize whitespace before
# matching so the same bank PDF is detected whether pdftotext is
# installed in the runtime image or the pdfplumber fallback is used.
u = norm(text).upper()
return (
0.98
if "ACCOUNT STATEMENT" in u
and "TRANSACTION DETAILS" in u
and "TOTAL CREDITS" in u
else 0
)
return 0.98 if "ACCOUNT STATEMENT" in u and "TRANSACTION DETAILS" in u and "TOTAL CREDITS" in u else 0
def parse(self, path, text=None):
text = text or extract_text(path)
@@ -50,202 +43,158 @@ class IndianBankModernParser(BaseParser):
confidence="High",
)
# Some Indian Bank statements leave these values blank. Do not let a
# value from the adjacent ACCOUNT SUMMARY column become the customer
# name merely because PDF text extraction merges the two columns.
customer = find(r"Account Holder Name[ \t]*([^\n]*)", text)
customer = norm(customer)
if customer and not re.search(
r"^(Opening Balance|Account Type|Account Number|Customer)",
customer,
re.I,
):
customer = norm(find(r"Account Holder Name[ \t]*([^\n]*)", text))
if customer and not re.search(r"^(Opening Balance|Account Type|Account Number|Customer)", customer, re.I):
meta.customer_name = customer
meta.account_number = find(r"Account Number[ \t]*([0-9X*]{4,})", text)
meta.ifsc = find(r"IFSC[ \t]*([A-Z0-9]{8,})", text)
meta.account_number = find(
r"Account Number[ \t]*([0-9X*]{4,})",
text,
)
period = re.search(rf"For period:\s*({DATE_TOKEN_PATTERN})\s*(?:-|to)\s*({DATE_TOKEN_PATTERN})", text, re.I)
if period:
meta.period_from = date_iso(period.group(1))
meta.period_to = date_iso(period.group(2))
period_match = re.search(
r"For period:\s*(\d{2}\s+[A-Za-z]{3}\s+\d{4})\s*-\s*"
r"(\d{2}\s+[A-Za-z]{3}\s+\d{4})",
text,
re.I,
)
if period_match:
meta.period_from = _parse_modern_date(period_match.group(1))
meta.period_to = _parse_modern_date(period_match.group(2))
opening = re.search(r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", text, re.I)
if opening:
meta.opening_balance = _signed_amount(opening.group(1), opening.group(2))
meta.total_credit = amount(find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text))
meta.total_debit = amount(find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text))
closing = re.search(r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", text, re.I)
if closing:
meta.closing_balance = _signed_amount(closing.group(1), closing.group(2))
opening_match = re.search(
r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?",
text,
re.I,
)
if opening_match:
meta.opening_balance = _signed_amount(
opening_match.group(1),
opening_match.group(2),
)
date_re = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})\s+(.*)$", re.I)
amount_re = re.compile(r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", re.I)
meta.total_credit = amount(
find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text)
)
meta.total_debit = amount(
find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text)
)
closing_match = re.search(
r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?",
text,
re.I,
)
if closing_match:
meta.closing_balance = _signed_amount(
closing_match.group(1),
closing_match.group(2),
)
# Indian Bank's current PDF layout is seen in both date orders,
# depending on the text extraction engine:
# 01 Apr 2025 ...
# Apr 01 2025 ...
# Accept both without changing the parser selected for older samples.
date_re = re.compile(
r"^\s*((?:\d{2}\s+[A-Za-z]{3}|[A-Za-z]{3}\s+\d{2})\s+\d{4})\s+(.*)$"
)
rows = []
current = None
rows: list[dict] = []
current: dict | None = None
page = 1
for line in text.splitlines():
if "\f" in line:
page += line.count("\f")
# Defaults match the wide first-page layout. They are replaced from
# every page header, which also supports narrower subsequent pages.
debit_pos, credit_pos, balance_pos = 56, 80, 94
for raw_line in text.splitlines():
if "\f" in raw_line:
page += raw_line.count("\f")
line = raw_line.rstrip("\n")
upper = line.upper()
if "TRANSACTION DETAILS" in upper and "DEBITS" in upper and "CREDITS" in upper and "BALANCE" in upper:
debit_pos = line.upper().find("DEBITS")
credit_pos = line.upper().find("CREDITS")
balance_pos = line.upper().rfind("BALANCE")
continue
match = date_re.match(line)
if match:
if match and not pd.isna(parse_flexible_date(match.group(1))):
if current:
rows.append(current)
transaction_date = _parse_modern_date(match.group(1))
current = {
"transaction_date": transaction_date,
"value_date": transaction_date,
"lines": [match.group(2)],
"transaction_date": date_iso(match.group(1)),
"value_date": date_iso(match.group(1)),
"first_line": line,
"detail_start": match.start(2),
"continuation": [],
"source_page": page,
"debit_pos": debit_pos,
"credit_pos": credit_pos,
"balance_pos": balance_pos,
}
elif (
current
and line.strip()
and not re.match(
r"^(Date\s+Transaction|ACCOUNT STATEMENT|Page)",
line.strip(),
re.I,
)
):
current["lines"].append(line)
continue
if not current or not line.strip():
continue
if re.match(r"^(\s*Date\s+Transaction|\s*ACCOUNT STATEMENT|\s*Indian Bank\s*\||\s*Ending Balance|\s*Total\s+INR)", line, re.I):
continue
current["continuation"].append(line.strip())
if current:
rows.append(current)
out = []
output: list[dict] = []
previous_balance = meta.opening_balance
for row in rows:
first_line = row["lines"][0]
all_text = norm(" ".join(row["lines"]))
first_line = row["first_line"]
tokens = list(amount_re.finditer(first_line))
if not tokens:
continue
# Balance is always the last INR amount and may be CR or DR.
balance_match = re.search(
r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)\s*$",
first_line,
re.I,
)
if balance_match:
balance = _signed_amount(
balance_match.group(1),
balance_match.group(2),
)
prefix = first_line[: balance_match.start()]
else:
balance = None
prefix = first_line
debit = credit = balance = None
balance_suffix = None
token_columns: list[tuple[str, re.Match]] = []
for token in tokens:
column = _nearest_column(token.start(), row["debit_pos"], row["credit_pos"], row["balance_pos"])
token_columns.append((column, token))
amount_matches = list(
re.finditer(r"INR\s*([\d,]+\.\d{2})", prefix, re.I)
)
transaction_amount = (
amount(amount_matches[-1].group(1)) if amount_matches else None
)
# The running balance is the rightmost amount and is also checked
# against the header-derived balance column.
balance_token = tokens[-1]
balance_suffix = balance_token.group(2)
balance = _signed_amount(balance_token.group(1), balance_suffix)
debit = None
credit = None
for column, token in token_columns[:-1]:
value = amount(token.group(1))
if column == "debit":
debit = value
elif column == "credit":
credit = value
if (
transaction_amount is not None
and previous_balance is not None
and balance is not None
):
if abs((previous_balance - transaction_amount) - balance) < 0.05:
debit = transaction_amount
elif abs((previous_balance + transaction_amount) - balance) < 0.05:
credit = transaction_amount
# When text extraction collapses spacing, infer the transaction side
# from the running balance rather than guessing by token position.
if debit is None and credit is None and len(tokens) >= 2:
txn = amount(tokens[-2].group(1))
if txn is not None and previous_balance is not None and balance is not None:
if abs((previous_balance - txn) - balance) < 0.05:
debit = txn
elif abs((previous_balance + txn) - balance) < 0.05:
credit = txn
# Fallback for the first row or when running-balance inference is
# unavailable. In the rendered Indian Bank table, a dash occupies
# the empty debit/credit column. Determine the side from text
# preceding the transaction amount.
if transaction_amount is not None and debit is None and credit is None:
before_amount = prefix[: amount_matches[-1].start()]
after_amount = prefix[amount_matches[-1].end() :]
# "- INR 1,000.00" means debit is empty, therefore credit.
if re.search(r"-\s*$", before_amount):
credit = transaction_amount
# "INR 1,000.00 -" means credit is empty, therefore debit.
elif re.match(r"^\s*-", after_amount):
debit = transaction_amount
else:
# Position fallback retained for extraction engines that
# preserve table spacing but omit the dash placeholder.
if amount_matches[-1].start() < 52:
debit = transaction_amount
else:
credit = transaction_amount
narration = re.split(
r"\s+INR\s*[\d,]+\.\d{2}",
all_text,
1,
flags=re.I,
)[0]
reference_no = ""
reference_match = re.search(
r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}",
all_text,
re.I,
)
if reference_match:
reference_no = reference_match.group(0)
# If the balance has no explicit DR/CR suffix, preserve continuity.
if balance is not None and previous_balance is not None:
expected = previous_balance + (credit or 0) - (debit or 0)
if abs(expected - balance) >= 0.05 and abs(expected + balance) < 0.05:
balance = -balance
if debit is None and credit is None:
continue
out.append(
first_detail_end = min((m.start() for m in tokens), default=len(first_line))
first_detail = first_line[row["detail_start"]:first_detail_end].strip()
narration_parts = [first_detail] + row["continuation"]
narration = norm(" ".join(part for part in narration_parts if part))
reference_no = ""
reference = re.search(r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", narration, re.I)
if reference:
reference_no = reference.group(0)
output.append(
{
**row,
"transaction_date": row["transaction_date"],
"value_date": row["value_date"],
"narration": narration,
"reference_no": reference_no,
"debit": debit,
"credit": credit,
"balance": balance,
"source_page": row["source_page"],
}
)
if balance is not None:
previous_balance = balance
return meta, finalize(pd.DataFrame(out), meta)
frame = finalize(pd.DataFrame(output), meta)
if not frame.empty:
# Validate and repair metadata from the parsed rows only when the
# statement summary did not supply it.
first = frame.iloc[0]
calculated_opening = (first["balance"] or 0) + (first["debit"] or 0) - (first["credit"] or 0)
if meta.opening_balance is None:
meta.opening_balance = round(calculated_opening, 2)
if meta.closing_balance is None:
meta.closing_balance = float(frame["balance"].dropna().iloc[-1])
return meta, frame
class IndianBankLegacyParser(BaseParser):
@@ -255,13 +204,7 @@ class IndianBankLegacyParser(BaseParser):
@classmethod
def detect(cls, text):
u = text.upper()
return (
0.97
if "STATEMENT OF ACCOUNT FROM" in u
and "REMITTER" in u
and "CHEQUE NO" in u
else 0
)
return 0.97 if "STATEMENT OF ACCOUNT FROM" in u and "REMITTER" in u and "CHEQUE NO" in u else 0
def parse(self, path, text=None):
text = text or extract_text(path)
@@ -272,119 +215,74 @@ class IndianBankLegacyParser(BaseParser):
confidence="Medium",
)
meta.account_number = find(r"for Account Number\s*\.?\s*([0-9X*]+)", text)
m = re.search(
r"STATEMENT OF ACCOUNT from\s*(\d{2}/\d{2}/\d{4})\s*to\s*"
r"(\d{2}/\d{2}/\d{4})",
text,
re.I,
)
if m:
meta.period_from = pd.to_datetime(
m.group(1), dayfirst=True
).strftime("%Y-%m-%d")
meta.period_to = pd.to_datetime(
m.group(2), dayfirst=True
).strftime("%Y-%m-%d")
period = re.search(rf"STATEMENT OF ACCOUNT from\s*({DATE_TOKEN_PATTERN})\s*to\s*({DATE_TOKEN_PATTERN})", text, re.I)
if period:
meta.period_from = date_iso(period.group(1))
meta.period_to = date_iso(period.group(2))
date_re = re.compile(
r"^\s*(\d{2}/\d{2})(?:/\d{4})?\s+"
r"(\d{2}/\d{2})(?:/\d{4})?\s+(.*)$"
)
# Legacy files may omit the year on each transaction. Preserve the
# previous behaviour while accepting every supported full-date format.
default_year = int(meta.period_from[:4]) if meta.period_from else None
full_date_re = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$", re.I)
short_date_re = re.compile(r"^\s*(\d{1,2}[-/.]\d{1,2})\s+(\d{1,2}[-/.]\d{1,2})\s+(.*)$")
rows = []
current = None
page = 1
for line in text.splitlines():
if "\f" in line:
page += line.count("\f")
match = date_re.match(line)
match = full_date_re.match(line) or short_date_re.match(line)
if match:
if current:
rows.append(current)
year = meta.period_from[:4] if meta.period_from else "2024"
transaction_date = match.group(1).replace(" ", "") + "/" + year
value_date = match.group(2).replace(" ", "") + "/" + year
transaction_date = date_iso(match.group(1), default_year=default_year)
value_date = date_iso(match.group(2), default_year=default_year)
current = {
"transaction_date": transaction_date,
"value_date": value_date,
"lines": [match.group(3)],
"source_page": page,
}
elif (
current
and line.strip()
and not re.match(
r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)",
line.strip(),
re.I,
)
):
elif current and line.strip() and not re.match(r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)", line.strip(), re.I):
current["lines"].append(line)
if current:
rows.append(current)
out = []
output = []
previous_balance = None
for row in rows:
first_line = row["lines"][0]
first = row["lines"][0]
all_text = norm(" ".join(row["lines"]))
balance_match = re.search(
r"([\d,]+\.\d{2})(CR|DR)\s*$",
first_line,
re.I,
)
balance_match = re.search(r"([\d,]+\.\d{2})(CR|DR)\s*$", first, re.I)
if not balance_match:
continue
balance = _signed_amount(
balance_match.group(1),
balance_match.group(2),
)
prefix = first_line[: balance_match.start()]
numbers = list(
re.finditer(r"(?<!\d)([\d,]+\.\d{2})(?!\d)", prefix)
)
transaction_amount = (
amount(numbers[-1].group(1)) if numbers else None
)
debit = None
credit = None
balance = _signed_amount(balance_match.group(1), balance_match.group(2))
prefix = first[: balance_match.start()]
numbers = list(re.finditer(r"(?<!\d)([\d,]+\.\d{2})(?!\d)", prefix))
transaction_amount = amount(numbers[-1].group(1)) if numbers else None
debit = credit = None
if transaction_amount is not None and previous_balance is not None:
if abs((previous_balance - transaction_amount) - balance) < 0.05:
debit = transaction_amount
elif abs((previous_balance + transaction_amount) - balance) < 0.05:
credit = transaction_amount
if transaction_amount is not None and debit is None and credit is None:
credit = transaction_amount if numbers[-1].start() > 70 else None
debit = transaction_amount if numbers[-1].start() <= 70 else None
reference_no = ""
reference_match = re.search(
r"(?:UPI|NEFT|IMPS|RTGS)[/A-Z0-9-]{6,}",
all_text,
re.I,
)
if reference_match:
reference_no = reference_match.group(0)
out.append(
if numbers[-1].start() > 70:
credit = transaction_amount
else:
debit = transaction_amount
if debit is None and credit is None:
continue
reference = re.search(r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", all_text, re.I)
output.append(
{
**row,
"narration": all_text,
"reference_no": reference_no,
"narration": norm(prefix[: numbers[-1].start()] + " " + " ".join(row["lines"][1:])),
"reference_no": reference.group(0) if reference else "",
"debit": debit,
"credit": credit,
"balance": balance,
}
)
previous_balance = balance
if out and meta.opening_balance is None:
first = out[0]
transaction_amount = (first.get("debit") or 0) - (
first.get("credit") or 0
)
meta.opening_balance = round(
(first["balance"] or 0) + transaction_amount,
2,
)
return meta, finalize(pd.DataFrame(out), meta)
return meta, finalize(pd.DataFrame(output), meta)