Support current Indian Bank FY statement transaction layout
This commit is contained in:
@@ -1,113 +1,390 @@
|
||||
from __future__ import annotations
|
||||
import re, pandas as pd
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from .base import *
|
||||
from .common import find
|
||||
|
||||
|
||||
def _signed_amount(value: str | None, suffix: str | None) -> float | None:
|
||||
"""Return CR as positive and DR as negative."""
|
||||
parsed = amount(value)
|
||||
if parsed is None:
|
||||
return None
|
||||
return -parsed if (suffix or "").upper() == "DR" else parsed
|
||||
|
||||
|
||||
def _parse_modern_date(value: str) -> str:
|
||||
"""Normalize both '01 Apr 2025' and 'Apr 01 2025' to ISO date."""
|
||||
return pd.to_datetime(value, dayfirst=True, errors="raise").strftime("%Y-%m-%d")
|
||||
|
||||
|
||||
class IndianBankModernParser(BaseParser):
|
||||
bank_name='Indian Bank'; parser_name='IndianBankModernParser'
|
||||
bank_name = "Indian Bank"
|
||||
parser_name = "IndianBankModernParser"
|
||||
|
||||
@classmethod
|
||||
def detect(cls,text):
|
||||
def detect(cls, text):
|
||||
# pdfplumber's layout mode can insert multiple spaces inside headings
|
||||
# (for example, 'ACCOUNT STATEMENT'). Normalize whitespace before
|
||||
# matching so the same bank PDF is detected whether pdftotext is
|
||||
# installed in the runtime image or the pdfplumber fallback is used.
|
||||
u=norm(text).upper()
|
||||
return 0.98 if 'ACCOUNT STATEMENT' in u and 'TRANSACTION DETAILS' in u and 'TOTAL CREDITS' in u else 0
|
||||
def parse(self,path,text=None):
|
||||
text=text or extract_text(path); meta=StatementMeta(bank_name=self.bank_name,source_file=Path(path).name,parser_name=self.parser_name,confidence='High')
|
||||
u = norm(text).upper()
|
||||
return (
|
||||
0.98
|
||||
if "ACCOUNT STATEMENT" in u
|
||||
and "TRANSACTION DETAILS" in u
|
||||
and "TOTAL CREDITS" in u
|
||||
else 0
|
||||
)
|
||||
|
||||
def parse(self, path, text=None):
|
||||
text = text or extract_text(path)
|
||||
meta = StatementMeta(
|
||||
bank_name=self.bank_name,
|
||||
source_file=Path(path).name,
|
||||
parser_name=self.parser_name,
|
||||
confidence="High",
|
||||
)
|
||||
|
||||
# Some Indian Bank statements leave these values blank. Do not let a
|
||||
# value from the adjacent ACCOUNT SUMMARY column become the customer
|
||||
# name merely because PDF text extraction merges the two columns.
|
||||
customer=find(r'Account Holder Name[ \t]*([^\n]*)',text)
|
||||
customer=norm(customer)
|
||||
if customer and not re.search(r'^(Opening Balance|Account Type|Account Number|Customer)',customer,re.I):
|
||||
meta.customer_name=customer
|
||||
meta.account_number=find(r'Account Number[ \t]*([0-9X*]{4,})',text)
|
||||
m=re.search(r'For period:\s*(\d{2}\s+[A-Za-z]{3}\s+\d{4})\s*-\s*(\d{2}\s+[A-Za-z]{3}\s+\d{4})',text,re.I)
|
||||
if m:
|
||||
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
|
||||
meta.opening_balance=amount(find(r'Opening Balance\s+INR\s*([\d,]+\.\d{2})',text))
|
||||
meta.total_credit=amount(find(r'Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})',text))
|
||||
meta.total_debit=amount(find(r'Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})',text))
|
||||
meta.closing_balance=amount(find(r'Ending Balance\s+INR\s*([\d,]+\.\d{2})',text))
|
||||
date_re=re.compile(r'^\s*([A-Za-z]{3}\s+\d{2}\s+\d{4})\s+(.*)$')
|
||||
rows=[]; cur=None; page=1
|
||||
customer = find(r"Account Holder Name[ \t]*([^\n]*)", text)
|
||||
customer = norm(customer)
|
||||
if customer and not re.search(
|
||||
r"^(Opening Balance|Account Type|Account Number|Customer)",
|
||||
customer,
|
||||
re.I,
|
||||
):
|
||||
meta.customer_name = customer
|
||||
|
||||
meta.account_number = find(
|
||||
r"Account Number[ \t]*([0-9X*]{4,})",
|
||||
text,
|
||||
)
|
||||
|
||||
period_match = re.search(
|
||||
r"For period:\s*(\d{2}\s+[A-Za-z]{3}\s+\d{4})\s*-\s*"
|
||||
r"(\d{2}\s+[A-Za-z]{3}\s+\d{4})",
|
||||
text,
|
||||
re.I,
|
||||
)
|
||||
if period_match:
|
||||
meta.period_from = _parse_modern_date(period_match.group(1))
|
||||
meta.period_to = _parse_modern_date(period_match.group(2))
|
||||
|
||||
opening_match = re.search(
|
||||
r"Opening Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?",
|
||||
text,
|
||||
re.I,
|
||||
)
|
||||
if opening_match:
|
||||
meta.opening_balance = _signed_amount(
|
||||
opening_match.group(1),
|
||||
opening_match.group(2),
|
||||
)
|
||||
|
||||
meta.total_credit = amount(
|
||||
find(r"Total Credits\s+\+\s*INR\s*([\d,]+\.\d{2})", text)
|
||||
)
|
||||
meta.total_debit = amount(
|
||||
find(r"Total Debits\s+-\s*INR\s*([\d,]+\.\d{2})", text)
|
||||
)
|
||||
|
||||
closing_match = re.search(
|
||||
r"Ending Balance\s+INR\s*([\d,]+\.\d{2})\s*(CR|DR)?",
|
||||
text,
|
||||
re.I,
|
||||
)
|
||||
if closing_match:
|
||||
meta.closing_balance = _signed_amount(
|
||||
closing_match.group(1),
|
||||
closing_match.group(2),
|
||||
)
|
||||
|
||||
# Indian Bank's current PDF layout is seen in both date orders,
|
||||
# depending on the text extraction engine:
|
||||
# 01 Apr 2025 ...
|
||||
# Apr 01 2025 ...
|
||||
# Accept both without changing the parser selected for older samples.
|
||||
date_re = re.compile(
|
||||
r"^\s*((?:\d{2}\s+[A-Za-z]{3}|[A-Za-z]{3}\s+\d{2})\s+\d{4})\s+(.*)$"
|
||||
)
|
||||
|
||||
rows = []
|
||||
current = None
|
||||
page = 1
|
||||
for line in text.splitlines():
|
||||
if '\f' in line: page+=line.count('\f')
|
||||
m=date_re.match(line)
|
||||
if m:
|
||||
if cur: rows.append(cur)
|
||||
cur={'transaction_date':m.group(1),'value_date':m.group(1),'lines':[m.group(2)],'source_page':page}
|
||||
elif cur and line.strip() and not re.match(r'^(Date\s+Transaction|ACCOUNT STATEMENT|Page)',line.strip(),re.I): cur['lines'].append(line)
|
||||
if cur: rows.append(cur)
|
||||
out=[]; prev=meta.opening_balance
|
||||
for r in rows:
|
||||
first=r['lines'][0]; alltxt=norm(' '.join(r['lines']))
|
||||
vals=[(m.start(),amount(m.group(1))) for m in re.finditer(r'INR\s*([\d,]+\.\d{2})',first,re.I)]
|
||||
# columns in sample: debit ~45, credit ~65, balance ~85. Use signs/placeholders and running balance.
|
||||
bal=vals[-1][1] if vals else None; debit=credit=None
|
||||
if len(vals)>=2:
|
||||
txn=vals[-2][1]
|
||||
before=first[:re.search(r'INR\s*[\d,]+\.\d{2}',first,re.I).start()] if re.search(r'INR\s*[\d,]+\.\d{2}',first,re.I) else first
|
||||
# presence of '-' before first amount often means no debit; inspect spacing/position
|
||||
p=vals[-2][0]
|
||||
if prev is not None and bal is not None:
|
||||
if abs((prev-txn)-bal)<0.05: debit=txn
|
||||
elif abs((prev+txn)-bal)<0.05: credit=txn
|
||||
if debit is None and credit is None:
|
||||
if p<52: debit=txn
|
||||
else: credit=txn
|
||||
narr=re.split(r'\s+INR\s*[\d,]+\.\d{2}',alltxt,1,flags=re.I)[0]
|
||||
ref=''; z=re.search(r'(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}',alltxt,re.I); ref=z.group(0) if z else ''
|
||||
if debit is None and credit is None: continue
|
||||
out.append({**r,'narration':narr,'reference_no':ref,'debit':debit,'credit':credit,'balance':bal})
|
||||
if bal is not None: prev=bal
|
||||
return meta,finalize(pd.DataFrame(out),meta)
|
||||
if "\f" in line:
|
||||
page += line.count("\f")
|
||||
|
||||
match = date_re.match(line)
|
||||
if match:
|
||||
if current:
|
||||
rows.append(current)
|
||||
transaction_date = _parse_modern_date(match.group(1))
|
||||
current = {
|
||||
"transaction_date": transaction_date,
|
||||
"value_date": transaction_date,
|
||||
"lines": [match.group(2)],
|
||||
"source_page": page,
|
||||
}
|
||||
elif (
|
||||
current
|
||||
and line.strip()
|
||||
and not re.match(
|
||||
r"^(Date\s+Transaction|ACCOUNT STATEMENT|Page)",
|
||||
line.strip(),
|
||||
re.I,
|
||||
)
|
||||
):
|
||||
current["lines"].append(line)
|
||||
|
||||
if current:
|
||||
rows.append(current)
|
||||
|
||||
out = []
|
||||
previous_balance = meta.opening_balance
|
||||
|
||||
for row in rows:
|
||||
first_line = row["lines"][0]
|
||||
all_text = norm(" ".join(row["lines"]))
|
||||
|
||||
# Balance is always the last INR amount and may be CR or DR.
|
||||
balance_match = re.search(
|
||||
r"INR\s*([\d,]+\.\d{2})\s*(CR|DR)\s*$",
|
||||
first_line,
|
||||
re.I,
|
||||
)
|
||||
if balance_match:
|
||||
balance = _signed_amount(
|
||||
balance_match.group(1),
|
||||
balance_match.group(2),
|
||||
)
|
||||
prefix = first_line[: balance_match.start()]
|
||||
else:
|
||||
balance = None
|
||||
prefix = first_line
|
||||
|
||||
amount_matches = list(
|
||||
re.finditer(r"INR\s*([\d,]+\.\d{2})", prefix, re.I)
|
||||
)
|
||||
transaction_amount = (
|
||||
amount(amount_matches[-1].group(1)) if amount_matches else None
|
||||
)
|
||||
|
||||
debit = None
|
||||
credit = None
|
||||
|
||||
if (
|
||||
transaction_amount is not None
|
||||
and previous_balance is not None
|
||||
and balance is not None
|
||||
):
|
||||
if abs((previous_balance - transaction_amount) - balance) < 0.05:
|
||||
debit = transaction_amount
|
||||
elif abs((previous_balance + transaction_amount) - balance) < 0.05:
|
||||
credit = transaction_amount
|
||||
|
||||
# Fallback for the first row or when running-balance inference is
|
||||
# unavailable. In the rendered Indian Bank table, a dash occupies
|
||||
# the empty debit/credit column. Determine the side from text
|
||||
# preceding the transaction amount.
|
||||
if transaction_amount is not None and debit is None and credit is None:
|
||||
before_amount = prefix[: amount_matches[-1].start()]
|
||||
after_amount = prefix[amount_matches[-1].end() :]
|
||||
|
||||
# "- INR 1,000.00" means debit is empty, therefore credit.
|
||||
if re.search(r"-\s*$", before_amount):
|
||||
credit = transaction_amount
|
||||
# "INR 1,000.00 -" means credit is empty, therefore debit.
|
||||
elif re.match(r"^\s*-", after_amount):
|
||||
debit = transaction_amount
|
||||
else:
|
||||
# Position fallback retained for extraction engines that
|
||||
# preserve table spacing but omit the dash placeholder.
|
||||
if amount_matches[-1].start() < 52:
|
||||
debit = transaction_amount
|
||||
else:
|
||||
credit = transaction_amount
|
||||
|
||||
narration = re.split(
|
||||
r"\s+INR\s*[\d,]+\.\d{2}",
|
||||
all_text,
|
||||
1,
|
||||
flags=re.I,
|
||||
)[0]
|
||||
|
||||
reference_no = ""
|
||||
reference_match = re.search(
|
||||
r"(?:NEFT|IMPS|UPI|RTGS)[/A-Z0-9-]{6,}",
|
||||
all_text,
|
||||
re.I,
|
||||
)
|
||||
if reference_match:
|
||||
reference_no = reference_match.group(0)
|
||||
|
||||
if debit is None and credit is None:
|
||||
continue
|
||||
|
||||
out.append(
|
||||
{
|
||||
**row,
|
||||
"narration": narration,
|
||||
"reference_no": reference_no,
|
||||
"debit": debit,
|
||||
"credit": credit,
|
||||
"balance": balance,
|
||||
}
|
||||
)
|
||||
|
||||
if balance is not None:
|
||||
previous_balance = balance
|
||||
|
||||
return meta, finalize(pd.DataFrame(out), meta)
|
||||
|
||||
|
||||
class IndianBankLegacyParser(BaseParser):
|
||||
bank_name='Indian Bank'; parser_name='IndianBankLegacyParser'
|
||||
bank_name = "Indian Bank"
|
||||
parser_name = "IndianBankLegacyParser"
|
||||
|
||||
@classmethod
|
||||
def detect(cls,text):
|
||||
u=text.upper(); return 0.97 if 'STATEMENT OF ACCOUNT FROM' in u and 'REMITTER' in u and 'CHEQUE NO' in u else 0
|
||||
def parse(self,path,text=None):
|
||||
text=text or extract_text(path); meta=StatementMeta(bank_name=self.bank_name,source_file=Path(path).name,parser_name=self.parser_name,confidence='Medium')
|
||||
meta.account_number=find(r'for Account Number\s*\.?\s*([0-9X*]+)',text)
|
||||
m=re.search(r'STATEMENT OF ACCOUNT from\s*(\d{2}/\d{2}/\d{4})\s*to\s*(\d{2}/\d{2}/\d{4})',text,re.I)
|
||||
def detect(cls, text):
|
||||
u = text.upper()
|
||||
return (
|
||||
0.97
|
||||
if "STATEMENT OF ACCOUNT FROM" in u
|
||||
and "REMITTER" in u
|
||||
and "CHEQUE NO" in u
|
||||
else 0
|
||||
)
|
||||
|
||||
def parse(self, path, text=None):
|
||||
text = text or extract_text(path)
|
||||
meta = StatementMeta(
|
||||
bank_name=self.bank_name,
|
||||
source_file=Path(path).name,
|
||||
parser_name=self.parser_name,
|
||||
confidence="Medium",
|
||||
)
|
||||
meta.account_number = find(r"for Account Number\s*\.?\s*([0-9X*]+)", text)
|
||||
m = re.search(
|
||||
r"STATEMENT OF ACCOUNT from\s*(\d{2}/\d{2}/\d{4})\s*to\s*"
|
||||
r"(\d{2}/\d{2}/\d{4})",
|
||||
text,
|
||||
re.I,
|
||||
)
|
||||
if m:
|
||||
meta.period_from=pd.to_datetime(m.group(1),dayfirst=True).strftime('%Y-%m-%d'); meta.period_to=pd.to_datetime(m.group(2),dayfirst=True).strftime('%Y-%m-%d')
|
||||
date_re=re.compile(r'^\s*(\d{2}/\d{2})(?:/\d{4})?\s+(\d{2}/\d{2})(?:/\d{4})?\s+(.*)$')
|
||||
rows=[]; cur=None; page=1
|
||||
meta.period_from = pd.to_datetime(
|
||||
m.group(1), dayfirst=True
|
||||
).strftime("%Y-%m-%d")
|
||||
meta.period_to = pd.to_datetime(
|
||||
m.group(2), dayfirst=True
|
||||
).strftime("%Y-%m-%d")
|
||||
|
||||
date_re = re.compile(
|
||||
r"^\s*(\d{2}/\d{2})(?:/\d{4})?\s+"
|
||||
r"(\d{2}/\d{2})(?:/\d{4})?\s+(.*)$"
|
||||
)
|
||||
rows = []
|
||||
current = None
|
||||
page = 1
|
||||
for line in text.splitlines():
|
||||
if '\f' in line: page+=line.count('\f')
|
||||
m=date_re.match(line)
|
||||
if m:
|
||||
if cur: rows.append(cur)
|
||||
year=(meta.period_from[:4] if meta.period_from else '2024')
|
||||
td=m.group(1).replace(' ','')+'/'+year; vd=m.group(2).replace(' ','')+'/'+year
|
||||
cur={'transaction_date':td,'value_date':vd,'lines':[m.group(3)],'source_page':page}
|
||||
elif cur and line.strip() and not re.match(r'^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)',line.strip(),re.I): cur['lines'].append(line)
|
||||
if cur: rows.append(cur)
|
||||
out=[]; prev=None
|
||||
for r in rows:
|
||||
first=r['lines'][0]; alltxt=norm(' '.join(r['lines']))
|
||||
# balance is amount followed by CR/DR at far right
|
||||
mb=re.search(r'([\d,]+\.\d{2})(CR|DR)\s*$',first,re.I)
|
||||
if not mb: continue
|
||||
bal=amount(mb.group(1)); prefix=first[:mb.start()]
|
||||
nums=list(re.finditer(r'(?<!\d)([\d,]+\.\d{2})(?!\d)',prefix))
|
||||
txn=amount(nums[-1].group(1)) if nums else None
|
||||
debit=credit=None
|
||||
if txn is not None and prev is not None:
|
||||
if abs((prev-txn)-bal)<0.05: debit=txn
|
||||
elif abs((prev+txn)-bal)<0.05: credit=txn
|
||||
if txn is not None and debit is None and credit is None:
|
||||
# column location: DR before CR in legacy format
|
||||
credit=txn if nums[-1].start()>70 else None; debit=txn if nums[-1].start()<=70 else None
|
||||
narr=alltxt
|
||||
ref=''; z=re.search(r'(?:UPI|NEFT|IMPS|RTGS)[/A-Z0-9-]{6,}',alltxt,re.I); ref=z.group(0) if z else ''
|
||||
out.append({**r,'narration':narr,'reference_no':ref,'debit':debit,'credit':credit,'balance':bal})
|
||||
prev=bal
|
||||
if "\f" in line:
|
||||
page += line.count("\f")
|
||||
match = date_re.match(line)
|
||||
if match:
|
||||
if current:
|
||||
rows.append(current)
|
||||
year = meta.period_from[:4] if meta.period_from else "2024"
|
||||
transaction_date = match.group(1).replace(" ", "") + "/" + year
|
||||
value_date = match.group(2).replace(" ", "") + "/" + year
|
||||
current = {
|
||||
"transaction_date": transaction_date,
|
||||
"value_date": value_date,
|
||||
"lines": [match.group(3)],
|
||||
"source_page": page,
|
||||
}
|
||||
elif (
|
||||
current
|
||||
and line.strip()
|
||||
and not re.match(
|
||||
r"^(Value Post|Date Date|STATEMENT OF ACCOUNT|Page No)",
|
||||
line.strip(),
|
||||
re.I,
|
||||
)
|
||||
):
|
||||
current["lines"].append(line)
|
||||
if current:
|
||||
rows.append(current)
|
||||
|
||||
out = []
|
||||
previous_balance = None
|
||||
for row in rows:
|
||||
first_line = row["lines"][0]
|
||||
all_text = norm(" ".join(row["lines"]))
|
||||
balance_match = re.search(
|
||||
r"([\d,]+\.\d{2})(CR|DR)\s*$",
|
||||
first_line,
|
||||
re.I,
|
||||
)
|
||||
if not balance_match:
|
||||
continue
|
||||
balance = _signed_amount(
|
||||
balance_match.group(1),
|
||||
balance_match.group(2),
|
||||
)
|
||||
prefix = first_line[: balance_match.start()]
|
||||
numbers = list(
|
||||
re.finditer(r"(?<!\d)([\d,]+\.\d{2})(?!\d)", prefix)
|
||||
)
|
||||
transaction_amount = (
|
||||
amount(numbers[-1].group(1)) if numbers else None
|
||||
)
|
||||
debit = None
|
||||
credit = None
|
||||
if transaction_amount is not None and previous_balance is not None:
|
||||
if abs((previous_balance - transaction_amount) - balance) < 0.05:
|
||||
debit = transaction_amount
|
||||
elif abs((previous_balance + transaction_amount) - balance) < 0.05:
|
||||
credit = transaction_amount
|
||||
if transaction_amount is not None and debit is None and credit is None:
|
||||
credit = transaction_amount if numbers[-1].start() > 70 else None
|
||||
debit = transaction_amount if numbers[-1].start() <= 70 else None
|
||||
|
||||
reference_no = ""
|
||||
reference_match = re.search(
|
||||
r"(?:UPI|NEFT|IMPS|RTGS)[/A-Z0-9-]{6,}",
|
||||
all_text,
|
||||
re.I,
|
||||
)
|
||||
if reference_match:
|
||||
reference_no = reference_match.group(0)
|
||||
|
||||
out.append(
|
||||
{
|
||||
**row,
|
||||
"narration": all_text,
|
||||
"reference_no": reference_no,
|
||||
"debit": debit,
|
||||
"credit": credit,
|
||||
"balance": balance,
|
||||
}
|
||||
)
|
||||
previous_balance = balance
|
||||
|
||||
if out and meta.opening_balance is None:
|
||||
first=out[0]; txn=(first.get('debit') or 0)-(first.get('credit') or 0); meta.opening_balance=round((first['balance'] or 0)+txn,2)
|
||||
return meta,finalize(pd.DataFrame(out),meta)
|
||||
first = out[0]
|
||||
transaction_amount = (first.get("debit") or 0) - (
|
||||
first.get("credit") or 0
|
||||
)
|
||||
meta.opening_balance = round(
|
||||
(first["balance"] or 0) + transaction_amount,
|
||||
2,
|
||||
)
|
||||
|
||||
return meta, finalize(pd.DataFrame(out), meta)
|
||||
|
||||
Reference in New Issue
Block a user