Fix Indian Bank extraction and add shared multi-format dates for all banks

This commit is contained in:
A R R R Associates
2026-07-16 21:56:08 +05:30
parent 43ab9c9aae
commit d9ae1519a4
10 changed files with 340 additions and 311 deletions
@@ -2,7 +2,7 @@ from __future__ import annotations
import re, pandas as pd
from pathlib import Path
from .base import *
from .common import find
from .common import DATE_TOKEN_PATTERN, date_iso, find, find_date_tokens
class AxisParser(BaseParser):
bank_name='Axis Bank'; parser_name='AxisParser'
@@ -14,10 +14,10 @@ class AxisParser(BaseParser):
m=re.search(r'Smart Statement Report\s*\n\s*([^\n]+)',text,re.I); meta.customer_name=norm(m.group(1)) if m else ''
meta.account_number=find(r'Statement of Account No\s*-\s*([^\s]*)',text)
meta.ifsc=find(r'IFSC:\s*([A-Z0-9]+)',text)
m=re.search(r'for period\s*\((\d{2}/\d{2}/\d{4})\s+to\s+(\d{2}/\d{2}/\d{4})\)',text,re.I)
if m: meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
m=re.search(rf'for period\s*\(({DATE_TOKEN_PATTERN})\s+to\s+({DATE_TOKEN_PATTERN})\)', text, re.I)
if m: meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
meta.opening_balance=amount(find(r'Opening Balance:\s*INR\s*([\d,]+\.\d{2})',text))
pat=re.compile(r'^\s*(\d+)\s+(\d{2}/\d{2}/\d{4})\s+(\d{2}/\d{2}/\d{4})\s+(.*)$')
pat=re.compile(rf'^\s*(\d+)\s+({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
rows=[]; cur=None; page=1
for line in text.splitlines():
if '\f' in line: page+=line.count('\f')
@@ -69,8 +69,13 @@ def finalize(df: pd.DataFrame, meta: StatementMeta) -> pd.DataFrame:
return pd.DataFrame(columns=STANDARD_COLUMNS)
for c in ['debit','credit','balance']:
df[c]=pd.to_numeric(df.get(c),errors='coerce')
from .common import parse_flexible_date
for c in ['transaction_date','value_date']:
df[c]=pd.to_datetime(df.get(c),errors='coerce',dayfirst=True)
values = df.get(c)
if values is None:
df[c] = pd.NaT
else:
df[c] = values.map(parse_flexible_date)
df['narration']=df.get('narration','').fillna('').map(norm)
df['reference_no']=df.get('reference_no','').fillna('').map(norm)
df['bank_name']=meta.bank_name
@@ -1,20 +1,157 @@
from __future__ import annotations
import re
from .base import amount, norm
from datetime import datetime
from typing import Iterable
def find(pattern,text,group=1,flags=re.I|re.M):
m=re.search(pattern,text,flags)
return norm(m.group(group)) if m else ''
import pandas as pd
def date_iso(s):
import pandas as pd
x=pd.to_datetime(s,errors='coerce',dayfirst=True)
return '' if pd.isna(x) else x.strftime('%Y-%m-%d')
from .base import norm
# Shared date support for all bank parsers. Indian bank statements and the
# other supported banks can be rendered differently by pdftotext/pdfplumber,
# so parsers must not depend on one literal date layout.
SUPPORTED_DATE_FORMATS: tuple[str, ...] = (
"%d %b %Y",
"%d %B %Y",
"%b %d %Y",
"%B %d %Y",
"%d %b, %Y",
"%d %B, %Y",
"%b %d, %Y",
"%B %d, %Y",
"%d-%b-%Y",
"%d-%B-%Y",
"%d/%b/%Y",
"%d/%B/%Y",
"%d-%m-%Y",
"%d/%m/%Y",
"%d.%m.%Y",
"%Y-%m-%d",
"%Y/%m/%d",
"%d %b %y",
"%d %B %y",
"%b %d %y",
"%B %d %y",
"%d-%b-%y",
"%d-%B-%y",
"%d/%b/%y",
"%d/%B/%y",
"%d-%m-%y",
"%d/%m/%y",
"%d.%m.%y",
)
# A permissive token used only to locate a candidate date at the start of a
# transaction line. parse_flexible_date() performs the actual validation.
DATE_TOKEN_PATTERN = (
r"(?:"
r"\d{1,2}\s+[A-Za-z]{3,9},?\s+\d{2,4}"
r"|[A-Za-z]{3,9}\s+\d{1,2},?\s+\d{2,4}"
r"|\d{1,2}[-/.]\d{1,2}[-/.]\d{2,4}"
r"|\d{4}[-/.]\d{1,2}[-/.]\d{1,2}"
r"|\d{1,2}[-/.][A-Za-z]{3,9}[-/.]\d{2,4}"
r")"
)
DATE_TOKEN_RE = re.compile(DATE_TOKEN_PATTERN, re.I)
LEADING_DATE_RE = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})(?:\s+|$)(.*)$", re.I)
def find(pattern, text, group=1, flags=re.I | re.M):
m = re.search(pattern, text, flags)
return norm(m.group(group)) if m else ""
def _normalise_date_text(value: str) -> str:
value = norm(value)
value = value.replace("", "-").replace("", "-")
value = re.sub(r"\s*,\s*", ", ", value)
value = re.sub(r"\s+", " ", value)
return value.strip(" ,")
def parse_flexible_date(value, *, default_year: int | None = None) -> pd.Timestamp | pd.NaT:
"""Parse an Indian bank date without silently swapping day and month.
Explicit formats are attempted before pandas' parser. Numeric dates are
always interpreted day-first because all supported statements are Indian
banking statements. A default year may be supplied for legacy DD/MM rows.
"""
if value is None or (isinstance(value, float) and pd.isna(value)):
return pd.NaT
if isinstance(value, (pd.Timestamp, datetime)):
return pd.Timestamp(value)
text = _normalise_date_text(str(value))
if not text:
return pd.NaT
if default_year and re.fullmatch(r"\d{1,2}[-/.]\d{1,2}", text):
text = f"{text}/{default_year}"
for fmt in SUPPORTED_DATE_FORMATS:
try:
return pd.Timestamp(datetime.strptime(text, fmt))
except ValueError:
continue
# Final guarded fallback for extraction noise. dayfirst=True is explicit
# and yearfirst is enabled only when the candidate begins with four digits.
parsed = pd.to_datetime(
text,
errors="coerce",
dayfirst=not bool(re.match(r"^\d{4}[-/.]", text)),
yearfirst=bool(re.match(r"^\d{4}[-/.]", text)),
)
return pd.NaT if pd.isna(parsed) else pd.Timestamp(parsed)
def date_iso(value, *, default_year: int | None = None) -> str:
parsed = parse_flexible_date(value, default_year=default_year)
return "" if pd.isna(parsed) else parsed.strftime("%Y-%m-%d")
def leading_dates(line: str, *, maximum: int = 2) -> tuple[list[str], str]:
"""Return up to ``maximum`` validated dates from the start of a line."""
remaining = line
dates: list[str] = []
for _ in range(maximum):
match = LEADING_DATE_RE.match(remaining)
if not match:
break
candidate = match.group(1)
if pd.isna(parse_flexible_date(candidate)):
break
dates.append(candidate)
remaining = match.group(2)
return dates, remaining
def find_date_tokens(text: str) -> list[str]:
return [m.group(0) for m in DATE_TOKEN_RE.finditer(text or "") if not pd.isna(parse_flexible_date(m.group(0)))]
def split_pages(text):
return text.split("\f")
def split_pages(text): return text.split('\f')
def infer_mode(n):
u=(n or '').upper()
for k,v in [('UPI','UPI'),('NEFT','NEFT'),('IMPS','IMPS'),('RTGS','RTGS'),('CASH DEPOSIT','Cash Deposit'),('CASH WITHDRAWAL','Cash Withdrawal'),('ATM','ATM'),('CHEQUE','Cheque'),('CHQ','Cheque'),('POS','POS'),('EDC','Card Settlement'),('ACH','ACH')]:
if k in u:return v
return 'Other'
u = (n or "").upper()
for k, v in (
("UPI", "UPI"),
("NEFT", "NEFT"),
("IMPS", "IMPS"),
("RTGS", "RTGS"),
("CASH DEPOSIT", "Cash Deposit"),
("CASH WITHDRAWAL", "Cash Withdrawal"),
("ATM", "ATM"),
("CHEQUE", "Cheque"),
("CHQ", "Cheque"),
("POS", "POS"),
("EDC", "Card Settlement"),
("ACH", "ACH"),
):
if k in u:
return v
return "Other"
@@ -2,7 +2,7 @@ from __future__ import annotations
import re, pandas as pd
from pathlib import Path
from .base import *
from .common import find
from .common import DATE_TOKEN_PATTERN, DATE_TOKEN_RE, date_iso, find
class HDFCParser(BaseParser):
bank_name='HDFC Bank'; parser_name='HDFCParser'
@@ -14,15 +14,15 @@ class HDFCParser(BaseParser):
meta.account_number=find(r'Account No\s*:?\s*([0-9X*]+)',text)
meta.customer_id=find(r'Cust ID\s*:?\s*([0-9X*]+)',text)
meta.ifsc=find(r'(?:RTGS/\s*NEFT IFSC|IFSC)\s*:?\s*([A-Z0-9]+)',text)
m=re.search(r'From\s*:\s*(\d{2}/\d{2}/\d{4})\s+To\s*:\s*(\d{2}/\d{2}/\d{4})',text,re.I)
m=re.search(rf'From\s*:\s*({DATE_TOKEN_PATTERN})\s+To\s*:\s*({DATE_TOKEN_PATTERN})', text, re.I)
if m:
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
# first non-empty line before address usually contains customer name; allow manual override in UI
m=re.search(r'\n\s{2,}([^\n:]{3,60})\s*\n.*?Address\s*:',text,re.S|re.I)
if m: meta.customer_name=norm(m.group(1).splitlines()[-1])
lines=text.splitlines(); rows=[]; cur=None; page=1
wd_pos,dep_pos,bal_pos=130,165,190
date_re=re.compile(r'^\s*(\d{2}/\d{2}/\d{2})\s+(.*)$')
date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
for line in lines:
if '\f' in line: page += line.count('\f')
if 'Withdrawal Amt.' in line and 'Deposit Amt.' in line:
@@ -37,7 +37,7 @@ class HDFCParser(BaseParser):
for r in rows:
first=r['raw_lines'][0]; body=' '.join(x.strip() for x in r['raw_lines'])
# identify value date and ref on first line
dts=list(re.finditer(r'\d{2}/\d{2}/\d{2}',first))
dts=list(DATE_TOKEN_RE.finditer(first))
if len(dts)>1: r['value_date']=dts[-1].group()
else: r['value_date']=r['transaction_date']
nums=list(re.finditer(r'(?<!\d)(?:\d{1,3}(?:,\d{3})+|\d+)\.\d{2}(?!\d)',first))
@@ -3,7 +3,7 @@ import re
import pandas as pd
from pathlib import Path
from .base import BaseParser, StatementMeta, extract_text, amount, norm, finalize, page_of_line
from .common import find, date_iso
from .common import DATE_TOKEN_PATTERN, find, date_iso
class IDFCFirstParser(BaseParser):
bank_name='IDFC FIRST Bank'; parser_name='IDFCFirstParser'
@@ -17,12 +17,12 @@ class IDFCFirstParser(BaseParser):
meta.account_number=find(r'ACCOUNT NO\s*:\s*([0-9X*]+)',text)
meta.customer_id=find(r'CUSTOMER ID\s*:\s*([0-9X*]+)',text)
meta.ifsc=find(r'IFSC\s*:\s*([A-Z0-9]+)',text)
m=re.search(r'STATEMENT PERIOD\s*:\s*(\d{4}-\d{2}-\d{2})\s+TO\s+(\d{4}-\d{2}-\d{2})',text,re.I)
if m: meta.period_from,meta.period_to=m.groups()
m=re.search(rf'STATEMENT PERIOD\s*:\s*({DATE_TOKEN_PATTERN})\s+TO\s+({DATE_TOKEN_PATTERN})', text, re.I)
if m: meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
m=re.search(r'Opening Balance\s+Total Debit\s+Total Credit\s+Closing Balance\s*\n\s*([\d,.]+)\s+([\d,.]+)\s+([\d,.]+)\s+([\d,.]+)',text,re.I)
if m: meta.opening_balance,meta.total_debit,meta.total_credit,meta.closing_balance=map(amount,m.groups())
lines=text.splitlines(); rows=[]; current=None; pos=0
pat=re.compile(r'^\s*(\d{2}-[A-Za-z]{3}-\d{4})\s+(\d{2}-[A-Za-z]{3}-\d{4})\s+(.*)$')
pat=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
for line in lines:
mm=pat.match(line)
if mm:
@@ -5,21 +5,24 @@ from pathlib import Path
import pandas as pd
from .base import *
from .common import find
from .base import BaseParser, StatementMeta, amount, extract_text, finalize, norm
from .common import DATE_TOKEN_PATTERN, date_iso, find, parse_flexible_date
def _signed_amount(value: str | None, suffix: str | None) -> float | None:
"""Return CR as positive and DR as negative."""
parsed = amount(value)
if parsed is None:
return None
return -parsed if (suffix or "").upper() == "DR" else parsed
def _parse_modern_date(value: str) -> str:
"""Normalize both '01 Apr 2025' and 'Apr 01 2025' to ISO date."""
return pd.to_datetime(value, dayfirst=True, errors="raise").strftime("%Y-%m-%d")
def _nearest_column(position: int, debit_pos: int, credit_pos: int, balance_pos: int) -> str:
distances = {
"debit": abs(position - debit_pos),
"credit": abs(position - credit_pos),
"balance": abs(position - balance_pos),
}
return min(distances, key=distances.get)
class IndianBankModernParser(BaseParser):
@@ -28,18 +31,8 @@ class IndianBankModernParser(BaseParser):
@classmethod
def detect(cls, text):
# pdfplumber's layout mode can insert multiple spaces inside headings
# (for example, 'ACCOUNT STATEMENT'). Normalize whitespace before
# matching so the same bank PDF is detected whether pdftotext is
# installed in the runtime image or the pdfplumber fallback is used.
u = norm(text).upper()
return (
0.98
if "ACCOUNT STATEMENT" in u
and "TRANSACTION DETAILS" in u
and "TOTAL CREDITS" in u
else 0
)
return 0.98 if "ACCOUNT STATEMENT" in u and "TRANSACTION DETAILS" in u and "TOTAL CREDITS" in u else 0
def parse(self, path, text=None):
text = text or extract_text(path)
@@ -50,202 +43,158 @@ class IndianBankModernParser(BaseParser):
confidence="High",
)
# Some Indian Bank statements leave these values blank. Do not let a
# value from the adjacent ACCOUNT SUMMARY column become the customer
# name merely because PDF text extraction merges the two columns.
customer = find(r"Account Holder Name[ \t]*([^\n]*)", text)
customer = norm(customer)
if customer and not re.search(
r"^(Opening Balance|Account Type|Account Number|Customer)",
customer,
re.I,
):
customer = norm(find(r"Account Holder Name[ \t]*([^\n]*)", text))
if customer and not re.search(r"^(Opening Balance|Account Type|Account Number|Customer)", customer, re.I):
meta.customer_name = customer
meta.account_number = find(r"Account Number[ \t]*([0-9X*]{4,})", text)
meta.ifsc = find(r"IFSC[ \t]*([A-Z0-9]{8,})", text)
meta.account_number = find(
r"Account Number[ \t]*([0-9X*]{4,})",
text,
)
period = re.search(rf"For period:\s*({DATE_TOKEN_PATTERN})\s*(?:-|to)\s*({DATE_TOKEN_PATTERN})", text, re.I)
if period:
meta.period_from = date_iso(period.group(1))
meta.period_to = date_iso(period.group(2))
period_match = re.search(
r"For period:\s*(\d{2}\s+[A-Za-z]{3}\s+\d{4})\s*-\s*"
r"(\d{2}\s+[A-Za-z]{3}\s+\d{4})",
text,
re.I,
)
if period_match:
meta.period_from = _parse_modern_date(period_match.group(1))
meta.period_to = _parse_modern_date(period_match.group(2))
opening = re.search(r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", text, re.I)
if opening:
meta.opening_balance = _signed_amount(opening.group(1), opening.group(2))
meta.total_credit = amount(find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text))
meta.total_debit = amount(find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text))
closing = re.search(r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", text, re.I)
if closing:
meta.closing_balance = _signed_amount(closing.group(1), closing.group(2))
opening_match = re.search(
r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?",
text,
re.I,
)
if opening_match:
meta.opening_balance = _signed_amount(
opening_match.group(1),
opening_match.group(2),
)
date_re = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})\s+(.*)$", re.I)
amount_re = re.compile(r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)?", re.I)
meta.total_credit = amount(
find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text)
)
meta.total_debit = amount(
find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text)
)
closing_match = re.search(
r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?",
text,
re.I,
)
if closing_match:
meta.closing_balance = _signed_amount(
closing_match.group(1),
closing_match.group(2),
)
# Indian Bank's current PDF layout is seen in both date orders,
# depending on the text extraction engine:
# 01 Apr 2025 ...
# Apr 01 2025 ...
# Accept both without changing the parser selected for older samples.
date_re = re.compile(
r"^\s*((?:\d{2}\s+[A-Za-z]{3}|[A-Za-z]{3}\s+\d{2})\s+\d{4})\s+(.*)$"
)
rows = []
current = None
rows: list[dict] = []
current: dict | None = None
page = 1
for line in text.splitlines():
if "\f" in line:
page += line.count("\f")
# Defaults match the wide first-page layout. They are replaced from
# every page header, which also supports narrower subsequent pages.
debit_pos, credit_pos, balance_pos = 56, 80, 94
for raw_line in text.splitlines():
if "\f" in raw_line:
page += raw_line.count("\f")
line = raw_line.rstrip("\n")
upper = line.upper()
if "TRANSACTION DETAILS" in upper and "DEBITS" in upper and "CREDITS" in upper and "BALANCE" in upper:
debit_pos = line.upper().find("DEBITS")
credit_pos = line.upper().find("CREDITS")
balance_pos = line.upper().rfind("BALANCE")
continue
match = date_re.match(line)
if match:
if match and not pd.isna(parse_flexible_date(match.group(1))):
if current:
rows.append(current)
transaction_date = _parse_modern_date(match.group(1))
current = {
"transaction_date": transaction_date,
"value_date": transaction_date,
"lines": [match.group(2)],
"transaction_date": date_iso(match.group(1)),
"value_date": date_iso(match.group(1)),
"first_line": line,
"detail_start": match.start(2),
"continuation": [],
"source_page": page,
"debit_pos": debit_pos,
"credit_pos": credit_pos,
"balance_pos": balance_pos,
}
elif (
current
and line.strip()
and not re.match(
r"^(Date\s+Transaction|ACCOUNT STATEMENT|Page)",
line.strip(),
re.I,
)
):
current["lines"].append(line)
continue
if not current or not line.strip():
continue
if re.match(r"^(\s*Date\s+Transaction|\s*ACCOUNT STATEMENT|\s*Indian Bank\s*\||\s*Ending Balance|\s*Total\s+INR)", line, re.I):
continue
current["continuation"].append(line.strip())
if current:
rows.append(current)
out = []
output: list[dict] = []
previous_balance = meta.opening_balance
for row in rows:
first_line = row["lines"][0]
all_text = norm(" ".join(row["lines"]))
first_line = row["first_line"]
tokens = list(amount_re.finditer(first_line))
if not tokens:
continue
# Balance is always the last INR amount and may be CR or DR.
balance_match = re.search(
r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)\s*$",
first_line,
re.I,
)
if balance_match:
balance = _signed_amount(
balance_match.group(1),
balance_match.group(2),
)
prefix = first_line[: balance_match.start()]
else:
balance = None
prefix = first_line
debit = credit = balance = None
balance_suffix = None
token_columns: list[tuple[str, re.Match]] = []
for token in tokens:
column = _nearest_column(token.start(), row["debit_pos"], row["credit_pos"], row["balance_pos"])
token_columns.append((column, token))
amount_matches = list(
re.finditer(r"INR\s*([\d,]+\.\d{2})", prefix, re.I)
)
transaction_amount = (
amount(amount_matches[-1].group(1)) if amount_matches else None
)
# The running balance is the rightmost amount and is also checked
# against the header-derived balance column.
balance_token = tokens[-1]
balance_suffix = balance_token.group(2)
balance = _signed_amount(balance_token.group(1), balance_suffix)
debit = None
credit = None
for column, token in token_columns[:-1]:
value = amount(token.group(1))
if column == "debit":
debit = value
elif column == "credit":
credit = value
if (
transaction_amount is not None
and previous_balance is not None
and balance is not None
):
if abs((previous_balance - transaction_amount) - balance) < 0.05:
debit = transaction_amount
elif abs((previous_balance + transaction_amount) - balance) < 0.05:
credit = transaction_amount
# When text extraction collapses spacing, infer the transaction side
# from the running balance rather than guessing by token position.
if debit is None and credit is None and len(tokens) >= 2:
txn = amount(tokens[-2].group(1))
if txn is not None and previous_balance is not None and balance is not None:
if abs((previous_balance - txn) - balance) < 0.05:
debit = txn
elif abs((previous_balance + txn) - balance) < 0.05:
credit = txn
# Fallback for the first row or when running-balance inference is
# unavailable. In the rendered Indian Bank table, a dash occupies
# the empty debit/credit column. Determine the side from text
# preceding the transaction amount.
if transaction_amount is not None and debit is None and credit is None:
before_amount = prefix[: amount_matches[-1].start()]
after_amount = prefix[amount_matches[-1].end() :]
# "- INR 1,000.00" means debit is empty, therefore credit.
if re.search(r"-\s*$", before_amount):
credit = transaction_amount
# "INR 1,000.00 -" means credit is empty, therefore debit.
elif re.match(r"^\s*-", after_amount):
debit = transaction_amount
else:
# Position fallback retained for extraction engines that
# preserve table spacing but omit the dash placeholder.
if amount_matches[-1].start() < 52:
debit = transaction_amount
else:
credit = transaction_amount
narration = re.split(
r"\s+INR\s*[\d,]+\.\d{2}",
all_text,
1,
flags=re.I,
)[0]
reference_no = ""
reference_match = re.search(
r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}",
all_text,
re.I,
)
if reference_match:
reference_no = reference_match.group(0)
# If the balance has no explicit DR/CR suffix, preserve continuity.
if balance is not None and previous_balance is not None:
expected = previous_balance + (credit or 0) - (debit or 0)
if abs(expected - balance) >= 0.05 and abs(expected + balance) < 0.05:
balance = -balance
if debit is None and credit is None:
continue
out.append(
first_detail_end = min((m.start() for m in tokens), default=len(first_line))
first_detail = first_line[row["detail_start"]:first_detail_end].strip()
narration_parts = [first_detail] + row["continuation"]
narration = norm(" ".join(part for part in narration_parts if part))
reference_no = ""
reference = re.search(r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", narration, re.I)
if reference:
reference_no = reference.group(0)
output.append(
{
**row,
"transaction_date": row["transaction_date"],
"value_date": row["value_date"],
"narration": narration,
"reference_no": reference_no,
"debit": debit,
"credit": credit,
"balance": balance,
"source_page": row["source_page"],
}
)
if balance is not None:
previous_balance = balance
return meta, finalize(pd.DataFrame(out), meta)
frame = finalize(pd.DataFrame(output), meta)
if not frame.empty:
# Validate and repair metadata from the parsed rows only when the
# statement summary did not supply it.
first = frame.iloc[0]
calculated_opening = (first["balance"] or 0) + (first["debit"] or 0) - (first["credit"] or 0)
if meta.opening_balance is None:
meta.opening_balance = round(calculated_opening, 2)
if meta.closing_balance is None:
meta.closing_balance = float(frame["balance"].dropna().iloc[-1])
return meta, frame
class IndianBankLegacyParser(BaseParser):
@@ -255,13 +204,7 @@ class IndianBankLegacyParser(BaseParser):
@classmethod
def detect(cls, text):
u = text.upper()
return (
0.97
if "STATEMENT OF ACCOUNT FROM" in u
and "REMITTER" in u
and "CHEQUE NO" in u
else 0
)
return 0.97 if "STATEMENT OF ACCOUNT FROM" in u and "REMITTER" in u and "CHEQUE NO" in u else 0
def parse(self, path, text=None):
text = text or extract_text(path)
@@ -272,119 +215,74 @@ class IndianBankLegacyParser(BaseParser):
confidence="Medium",
)
meta.account_number = find(r"for Account Number\s*\.?\s*([0-9X*]+)", text)
m = re.search(
r"STATEMENT OF ACCOUNT from\s*(\d{2}/\d{2}/\d{4})\s*to\s*"
r"(\d{2}/\d{2}/\d{4})",
text,
re.I,
)
if m:
meta.period_from = pd.to_datetime(
m.group(1), dayfirst=True
).strftime("%Y-%m-%d")
meta.period_to = pd.to_datetime(
m.group(2), dayfirst=True
).strftime("%Y-%m-%d")
period = re.search(rf"STATEMENT OF ACCOUNT from\s*({DATE_TOKEN_PATTERN})\s*to\s*({DATE_TOKEN_PATTERN})", text, re.I)
if period:
meta.period_from = date_iso(period.group(1))
meta.period_to = date_iso(period.group(2))
date_re = re.compile(
r"^\s*(\d{2}/\d{2})(?:/\d{4})?\s+"
r"(\d{2}/\d{2})(?:/\d{4})?\s+(.*)$"
)
# Legacy files may omit the year on each transaction. Preserve the
# previous behaviour while accepting every supported full-date format.
default_year = int(meta.period_from[:4]) if meta.period_from else None
full_date_re = re.compile(rf"^\s*({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$", re.I)
short_date_re = re.compile(r"^\s*(\d{1,2}[-/.]\d{1,2})\s+(\d{1,2}[-/.]\d{1,2})\s+(.*)$")
rows = []
current = None
page = 1
for line in text.splitlines():
if "\f" in line:
page += line.count("\f")
match = date_re.match(line)
match = full_date_re.match(line) or short_date_re.match(line)
if match:
if current:
rows.append(current)
year = meta.period_from[:4] if meta.period_from else "2024"
transaction_date = match.group(1).replace(" ", "") + "/" + year
value_date = match.group(2).replace(" ", "") + "/" + year
transaction_date = date_iso(match.group(1), default_year=default_year)
value_date = date_iso(match.group(2), default_year=default_year)
current = {
"transaction_date": transaction_date,
"value_date": value_date,
"lines": [match.group(3)],
"source_page": page,
}
elif (
current
and line.strip()
and not re.match(
r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)",
line.strip(),
re.I,
)
):
elif current and line.strip() and not re.match(r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)", line.strip(), re.I):
current["lines"].append(line)
if current:
rows.append(current)
out = []
output = []
previous_balance = None
for row in rows:
first_line = row["lines"][0]
first = row["lines"][0]
all_text = norm(" ".join(row["lines"]))
balance_match = re.search(
r"([\d,]+\.\d{2})(CR|DR)\s*$",
first_line,
re.I,
)
balance_match = re.search(r"([\d,]+\.\d{2})(CR|DR)\s*$", first, re.I)
if not balance_match:
continue
balance = _signed_amount(
balance_match.group(1),
balance_match.group(2),
)
prefix = first_line[: balance_match.start()]
numbers = list(
re.finditer(r"(?<!\d)([\d,]+\.\d{2})(?!\d)", prefix)
)
transaction_amount = (
amount(numbers[-1].group(1)) if numbers else None
)
debit = None
credit = None
balance = _signed_amount(balance_match.group(1), balance_match.group(2))
prefix = first[: balance_match.start()]
numbers = list(re.finditer(r"(?<!\d)([\d,]+\.\d{2})(?!\d)", prefix))
transaction_amount = amount(numbers[-1].group(1)) if numbers else None
debit = credit = None
if transaction_amount is not None and previous_balance is not None:
if abs((previous_balance - transaction_amount) - balance) < 0.05:
debit = transaction_amount
elif abs((previous_balance + transaction_amount) - balance) < 0.05:
credit = transaction_amount
if transaction_amount is not None and debit is None and credit is None:
credit = transaction_amount if numbers[-1].start() > 70 else None
debit = transaction_amount if numbers[-1].start() <= 70 else None
reference_no = ""
reference_match = re.search(
r"(?:UPI|NEFT|IMPS|RTGS)[/A-Z0-9-]{6,}",
all_text,
re.I,
)
if reference_match:
reference_no = reference_match.group(0)
out.append(
if numbers[-1].start() > 70:
credit = transaction_amount
else:
debit = transaction_amount
if debit is None and credit is None:
continue
reference = re.search(r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}", all_text, re.I)
output.append(
{
**row,
"narration": all_text,
"reference_no": reference_no,
"narration": norm(prefix[: numbers[-1].start()] + " " + " ".join(row["lines"][1:])),
"reference_no": reference.group(0) if reference else "",
"debit": debit,
"credit": credit,
"balance": balance,
}
)
previous_balance = balance
if out and meta.opening_balance is None:
first = out[0]
transaction_amount = (first.get("debit") or 0) - (
first.get("credit") or 0
)
meta.opening_balance = round(
(first["balance"] or 0) + transaction_amount,
2,
)
return meta, finalize(pd.DataFrame(out), meta)
return meta, finalize(pd.DataFrame(output), meta)
@@ -2,7 +2,7 @@ from __future__ import annotations
import re, pandas as pd
from pathlib import Path
from .base import *
from .common import find
from .common import DATE_TOKEN_PATTERN, date_iso, find
class IndusIndParser(BaseParser):
bank_name='IndusInd Bank'; parser_name='IndusIndParser'
@@ -13,13 +13,13 @@ class IndusIndParser(BaseParser):
meta.account_number=find(r'Account Number\s*:\s*([0-9X*]+)',text)
meta.customer_id=find(r'Cust\.Reln\.No\s*:\s*([0-9X*]+)',text)
meta.ifsc=find(r'IFSC Code\s*:\s*([A-Z0-9]+)',text)
m=re.search(r'Period\s*:\s*(\d{2}[/-][A-Za-z0-9]{2,3}[/-]\d{2,4})\s*(?:to|TO|-)\s*(\d{2}[/-][A-Za-z0-9]{2,3}[/-]\d{2,4})',text,re.I)
m=re.search(rf'Period\s*:\s*({DATE_TOKEN_PATTERN})\s*(?:to|-)\s*({DATE_TOKEN_PATTERN})', text, re.I)
if m:
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
meta.total_debit=amount(find(r'Total Withdrawal Amount\s*:\s*([\d,]+\.\d{2})',text))
meta.total_credit=amount(find(r'Total Deposit Amount\s*:\s*([\d,]+\.\d{2})',text))
# rows start with date and often end with amount Dr/Cr and balance Cr
date_re=re.compile(r'^\s*(\d{2}[-/]?[A-Za-z]{3}[-/]?\d{2,4}|\d{2}-\d{2}-\d{4})\s+(.*)$')
date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
rows=[]; cur=None; page=1
for line in text.splitlines():
if '\f' in line: page+=line.count('\f')
@@ -2,7 +2,7 @@ from __future__ import annotations
import re, pandas as pd
from pathlib import Path
from .base import *
from .common import find
from .common import DATE_TOKEN_PATTERN, date_iso, find
class KotakParser(BaseParser):
bank_name='Kotak Mahindra Bank'; parser_name='KotakParser'
@@ -13,10 +13,10 @@ class KotakParser(BaseParser):
text=text or extract_text(path); meta=StatementMeta(bank_name=self.bank_name,source_file=Path(path).name,parser_name=self.parser_name,confidence='High')
meta.account_number=find(r'Account\s*#\s*Variant\s*KOTAK\s*\n.*?([0-9X*]{6,})',text,flags=re.I|re.S)
meta.ifsc=find(r'IFSC\s+([A-Z0-9]+)',text)
m=re.search(r'(\d{2}\s+[A-Za-z]{3},\s*\d{4})\s*-\s*(\d{2}\s+[A-Za-z]{3},\s*\d{4})',text)
m=re.search(rf'({DATE_TOKEN_PATTERN})\s*-\s*({DATE_TOKEN_PATTERN})', text, re.I)
if m:
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
date_re=re.compile(r'^\s*(\d{2}\s+[A-Za-z]{3},\s*\d{4})\s+(.*)$')
meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
rows=[]; cur=None; page=1
for line in text.splitlines():
if '\f' in line: page+=line.count('\f')
@@ -2,7 +2,7 @@ from __future__ import annotations
import re, pandas as pd
from pathlib import Path
from .base import *
from .common import find
from .common import DATE_TOKEN_PATTERN, date_iso, find
class SBIModernParser(BaseParser):
bank_name='State Bank of India'; parser_name='SBIModernParser'
@@ -14,12 +14,12 @@ class SBIModernParser(BaseParser):
meta.customer_name=find(r'Account Name\s*:?\s*([^\n]+)',text)
meta.account_number=find(r'Account Number\s*:?\s*([0-9X*]+)',text)
meta.ifsc=find(r'IFS Code\s*:?\s*([A-Z0-9]+)',text)
m=re.search(r'Account Statement from\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+to\s+(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})',text,re.I)
m=re.search(rf'Account Statement from\s*({DATE_TOKEN_PATTERN})\s+to\s+({DATE_TOKEN_PATTERN})', text, re.I)
if m:
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
meta.opening_balance=amount(find(r'Balance as on[^\n]*\n\s*([\d,]+\.\d{2})',text))
# Format A: date details ref debit credit balance
date_re=re.compile(r'^\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+(.*)$')
date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
rows=[]; cur=None; page=1
for line in text.splitlines():
if '\f' in line: page += line.count('\f')
@@ -66,11 +66,11 @@ class SBIOtherParser(BaseParser):
meta.customer_name=find(r'Account Name\s*:\s*([^\n]+)',text)
meta.account_number=find(r'Account Number\s*:?\s*([0-9X*]+)',text)
meta.ifsc=find(r'IFS Code\s*:?\s*([A-Z0-9]+)',text)
m=re.search(r'Account Statement from\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+to\s+(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})',text,re.I)
m=re.search(rf'Account Statement from\s*({DATE_TOKEN_PATTERN})\s+to\s+({DATE_TOKEN_PATTERN})', text, re.I)
if m:
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
meta.period_from=date_iso(m.group(1)); meta.period_to=date_iso(m.group(2))
m=re.search(r'Balance as on\s+[^\n]+\n',text,re.I)
date_re=re.compile(r'^\s*(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+(\d{1,2}\s+[A-Za-z]{3}\s+\d{4})\s+(.*)$')
date_re=re.compile(rf'^\s*({DATE_TOKEN_PATTERN})\s+({DATE_TOKEN_PATTERN})\s+(.*)$', re.I)
rows=[]; cur=None; page=1
for line in text.splitlines():
if '\f' in line: page += line.count('\f')