Add Phase 11 bank to Tally with multi-bank contra intelligence

This commit is contained in:
A R R R Associates
2026-08-22 19:59:40 +05:30
parent 215ad2c6c6
commit 8748ee9539
16 changed files with 1140 additions and 28 deletions
+188 -1
View File
@@ -23,6 +23,8 @@ ANALYSIS_COLUMNS = [
"auto_category", "auto_nature", "auto_group", "matched_rule_id",
"matched_keyword", "suggested_ledger", "rule_confidence",
"review_required", "review_note",
"contra_pair_id", "contra_status", "contra_counter_statement_id",
"contra_counter_bank", "contra_counter_account", "contra_confidence", "contra_reason",
"category", "counterparty",
]
@@ -428,6 +430,186 @@ def enrich(df, classification_enabled: bool = True):
return _ensure_analysis_columns(x)
def _contra_normalized_ref(row) -> str:
values = [
row.get("transfer_reference"),
row.get("reference_no"),
]
for value in values:
value = re.sub(r"[^A-Z0-9]", "", _text(value).upper())
if len(value) >= 6:
return value
return ""
def _account_tail(value: str) -> str:
digits = re.sub(r"\D", "", _text(value))
return digits[-6:] if len(digits) >= 4 else digits
def _contra_candidate_score(left, right) -> tuple[int, list[str]]:
if _direction(left) == _direction(right):
return 0, []
left_account = _text(left.get("account_number"))
right_account = _text(right.get("account_number"))
if not left_account or not right_account or clean_key(left_account) == clean_key(right_account):
return 0, []
left_amount = round(_amount(left), 2)
right_amount = round(_amount(right), 2)
if left_amount <= 0 or abs(left_amount - right_amount) > 0.01:
return 0, []
left_date = pd.to_datetime(left.get("transaction_date"), errors="coerce")
right_date = pd.to_datetime(right.get("transaction_date"), errors="coerce")
if pd.isna(left_date) or pd.isna(right_date):
return 0, []
days = abs((left_date.normalize() - right_date.normalize()).days)
if days > 2:
return 0, []
score = 55
reasons = ["equal and opposite amount across different bank accounts"]
if days == 0:
score += 18
reasons.append("same transaction date")
elif days == 1:
score += 13
reasons.append("one-day settlement difference")
else:
score += 8
reasons.append("two-day settlement difference")
lref = _contra_normalized_ref(left)
rref = _contra_normalized_ref(right)
if lref and rref and lref == rref:
score += 30
reasons.append("same bank transfer reference")
ln = _text(left.get("narration")).upper()
rn = _text(right.get("narration")).upper()
own_words = ("SELF", "OWN ACCOUNT", "OWN A/C", "TRANSFER TO", "TRANSFER FROM", "INTERNAL TRANSFER")
if any(word in ln for word in own_words) or any(word in rn for word in own_words):
score += 10
reasons.append("own-account transfer wording")
ltail = _account_tail(left_account)
rtail = _account_tail(right_account)
if (ltail and ltail in rn) or (rtail and rtail in ln):
score += 18
reasons.append("counter bank account suffix appears in narration")
# Without a transfer reference or direct account evidence, do not call a generic
# same-amount movement contra merely because dates happen to align.
strong_identity = bool((lref and rref and lref == rref) or ((ltail and ltail in rn) or (rtail and rtail in ln)))
if not strong_identity and not any(word in ln for word in own_words) and not any(word in rn for word in own_words):
return 0, []
return min(100, score), reasons
def detect_interbank_contra_pairs(df: pd.DataFrame) -> pd.DataFrame:
"""Conservatively pair transfers between different uploaded accounts.
The function only pairs equal-and-opposite movements across distinct account
numbers within two days and requires transfer-reference, account-suffix or
explicit own-account wording evidence. Each transaction is used in at most
one pair.
"""
x = _ensure_analysis_columns(df)
if x.empty:
return x
for col, default in (
("contra_pair_id", ""),
("contra_status", ""),
("contra_counter_statement_id", ""),
("contra_counter_bank", ""),
("contra_counter_account", ""),
("contra_confidence", 0.0),
("contra_reason", ""),
):
if col not in x.columns:
x[col] = default
candidates = []
rows = list(x.iterrows())
for pos, (li, left) in enumerate(rows):
for ri, right in rows[pos + 1:]:
score, reasons = _contra_candidate_score(left, right)
if score >= 85:
candidates.append((score, li, ri, reasons))
candidates.sort(key=lambda item: (-item[0], item[1], item[2]))
used = set()
pair_no = 0
for score, li, ri, reasons in candidates:
if li in used or ri in used:
continue
used.add(li)
used.add(ri)
pair_no += 1
pair_id = f"CONTRA-{pair_no:04d}"
for current, other in ((li, ri), (ri, li)):
x.at[current, "contra_pair_id"] = pair_id
x.at[current, "contra_status"] = "Matched"
x.at[current, "contra_counter_statement_id"] = _text(x.at[other, "statement_id"])
x.at[current, "contra_counter_bank"] = _text(x.at[other, "bank_name"])
x.at[current, "contra_counter_account"] = _text(x.at[other, "account_number"])
x.at[current, "contra_confidence"] = float(score)
x.at[current, "contra_reason"] = "; ".join(reasons)
x.at[current, "auto_party"] = "Own Bank Transfer"
x.at[current, "auto_category"] = "Self Transfer / Contra"
x.at[current, "auto_nature"] = "Contra"
x.at[current, "auto_group"] = "Contra / Balance Sheet"
x.at[current, "suggested_ledger"] = "Other Bank Account"
x.at[current, "rule_confidence"] = float(score)
if score >= 95:
x.at[current, "review_required"] = False
x.at[current, "review_note"] = ""
else:
x.at[current, "review_required"] = True
x.at[current, "review_note"] = "Probable inter-bank contra; verify both bank accounts before posting."
return x
def interbank_contra_summary(df: pd.DataFrame) -> pd.DataFrame:
if df is None or df.empty or "contra_pair_id" not in df.columns:
return pd.DataFrame(columns=[
"contra_pair_id", "confidence", "debit_bank", "debit_account",
"credit_bank", "credit_account", "amount", "date_from", "date_to", "reason",
])
matched = df[df["contra_pair_id"].fillna("").ne("")].copy()
rows = []
for pair_id, part in matched.groupby("contra_pair_id", sort=True):
if len(part) != 2:
continue
debit = part[pd.to_numeric(part["debit"], errors="coerce").fillna(0).gt(0)]
credit = part[pd.to_numeric(part["credit"], errors="coerce").fillna(0).gt(0)]
if debit.empty or credit.empty:
continue
d = debit.iloc[0]
c = credit.iloc[0]
dates = pd.to_datetime(part["transaction_date"], errors="coerce").dropna()
rows.append({
"contra_pair_id": pair_id,
"confidence": float(part["contra_confidence"].max() or 0),
"debit_bank": d.get("bank_name", ""),
"debit_account": d.get("account_number", ""),
"credit_bank": c.get("bank_name", ""),
"credit_account": c.get("account_number", ""),
"amount": round(float(d.get("debit") or 0), 2),
"date_from": dates.min() if not dates.empty else None,
"date_to": dates.max() if not dates.empty else None,
"reason": d.get("contra_reason", ""),
})
return pd.DataFrame(rows)
def analyze_files(paths, customer_override="", account_override="", bank_hint="auto", classification_enabled=True):
metas = []
frames = []
@@ -447,6 +629,7 @@ def analyze_files(paths, customer_override="", account_override="", bank_hint="a
frames.append(df)
combined = pd.concat(frames, ignore_index=True) if frames else pd.DataFrame()
all_df = _ensure_analysis_columns(enrich(combined, classification_enabled))
all_df = detect_interbank_contra_pairs(all_df)
# IMPORTANT ACCOUNTING CONTROL:
# Duplicate detection is advisory only. A bank may legitimately contain two
@@ -598,7 +781,10 @@ def _workbook_columns(df):
"statement_id", "transaction_date", "value_date", "narration", "transfer_bank_code", "transfer_reference",
"transfer_comment", "auto_party", "party_match_method", "party_match_confidence",
"auto_category", "auto_nature", "auto_group", "matched_rule_id", "matched_keyword",
"suggested_ledger", "rule_confidence", "review_required", "mode", "direction", "debit", "credit",
"suggested_ledger", "rule_confidence", "review_required",
"contra_pair_id", "contra_status", "contra_counter_statement_id", "contra_counter_bank",
"contra_counter_account", "contra_confidence", "contra_reason",
"mode", "direction", "debit", "credit",
"balance", "reference_no", "bank_name", "customer_name", "account_number", "source_file",
"source_page", "parser_name", "exact_duplicate", "possible_duplicate", "duplicate_group_id",
"duplicate_reason", "duplicate_confidence", "review_note",
@@ -900,6 +1086,7 @@ def export_excel(output, metas, all_df, unique_df, financial_year="", selected_b
exact_export.to_excel(writer, sheet_name="Exact Duplicates", index=False)
possible_export.to_excel(writer, sheet_name="Possible Duplicates", index=False)
duplicate_summary(all_df).to_excel(writer, sheet_name="Duplicate Summary", index=False)
interbank_contra_summary(unique_df).to_excel(writer, sheet_name="Interbank Contra Matches", index=False)
# Masters first so validation ranges exist.
max_master = max(len(categories), len(parties), len(natures), len(groups), 1)