Add Phase 7 GSTR-2B purchase intelligence

This commit is contained in:
A R R R Associates
2026-08-22 14:31:24 +05:30
parent 7e1b9c24ef
commit 7283f45f43
8 changed files with 1199 additions and 0 deletions
+307
View File
@@ -0,0 +1,307 @@
from __future__ import annotations
import csv
import hashlib
import io
import re
from datetime import date, datetime
from pathlib import Path
from typing import Iterable
from openpyxl import load_workbook
MAX_UPLOAD_BYTES = 25 * 1024 * 1024
HEADER_ALIASES = {
"supplier_gstin": {
"gstin of supplier", "supplier gstin", "gstin", "gstin/uin of supplier",
"gstin / uin of supplier", "ctin",
},
"supplier_name": {
"trade/legal name", "trade legal name", "supplier name", "legal name",
"trade name", "name of supplier", "supplier trade name",
},
"invoice_number": {
"invoice number", "invoice no", "invoice no.", "document number",
"document no", "doc no", "inum",
},
"invoice_date": {
"invoice date", "document date", "doc date", "invoice dt", "idt",
},
"invoice_type": {
"invoice type", "document type", "type", "inv type",
},
"invoice_value": {
"invoice value", "document value", "invoice amount", "total invoice value",
"val",
},
"place_of_supply": {
"place of supply", "pos", "place of supply (state/ut)", "place of supply state/ut",
},
"reverse_charge": {
"supply attract reverse charge", "reverse charge", "rcm", "reverse charge applicable",
},
"taxable_value": {
"taxable value", "taxable amount", "txval",
},
"igst": {
"integrated tax", "integrated tax amount", "igst", "igst amount", "iamt",
},
"cgst": {
"central tax", "central tax amount", "cgst", "cgst amount", "camt",
},
"sgst": {
"state/ut tax", "state / ut tax", "state tax", "sgst", "utgst",
"state/ut tax amount", "sgst amount", "samt",
},
"cess": {
"cess", "cess amount", "csamt",
},
"itc_availability": {
"itc availability", "itc available", "itc available for the tax period",
"itc eligibility", "eligibility for itc",
},
"hsn_code": {
"hsn", "hsn code", "hsn/sac", "hsn / sac", "hsn/sac code",
},
"description_text": {
"description", "item description", "product description", "goods/service description",
"description of goods/services", "description of goods / services",
},
}
def _norm_header(value) -> str:
text = str(value or "").strip().lower()
text = text.replace("\n", " ").replace("\r", " ")
text = re.sub(r"[_\-]+", " ", text)
text = re.sub(r"[^\w/(). ]+", " ", text)
text = re.sub(r"\s+", " ", text).strip()
return text
ALIAS_LOOKUP = {
alias: key
for key, aliases in HEADER_ALIASES.items()
for alias in aliases
}
def _canonical_header(value) -> str | None:
norm = _norm_header(value)
if norm in ALIAS_LOOKUP:
return ALIAS_LOOKUP[norm]
# Conservative contains matching only for longer aliases.
for alias, key in ALIAS_LOOKUP.items():
if len(alias) >= 10 and (norm.startswith(alias) or alias in norm):
return key
return None
def _text(value) -> str:
if value is None:
return ""
if isinstance(value, float) and value.is_integer():
return str(int(value))
return str(value).strip()
def _money(value) -> float:
if value in (None, ""):
return 0.0
if isinstance(value, (int, float)):
return float(value)
text = str(value).strip().replace(",", "").replace("₹", "")
text = re.sub(r"^\((.*)\)$", r"-\1", text)
try:
return float(text)
except Exception:
return 0.0
def _date_text(value) -> str:
if value in (None, ""):
return ""
if isinstance(value, datetime):
return value.date().isoformat()
if isinstance(value, date):
return value.isoformat()
text = str(value).strip()
candidates = (
"%d-%m-%Y", "%d/%m/%Y", "%Y-%m-%d", "%d-%b-%Y", "%d %b %Y",
"%d.%m.%Y", "%m/%d/%Y",
)
for fmt in candidates:
try:
return datetime.strptime(text, fmt).date().isoformat()
except Exception:
pass
# Preserve original if portal format is unfamiliar; duplicate identity remains stable.
return text[:20]
def _gstin(value) -> str:
return re.sub(r"[^A-Z0-9]", "", str(value or "").upper())[:15]
def _hsn(value) -> str:
text = re.sub(r"\D", "", str(value or ""))
return text[:8]
def _document_type(invoice_type: str, invoice_number: str) -> str:
text = f"{invoice_type} {invoice_number}".upper()
if "CREDIT" in text or "CR NOTE" in text or "CREDIT NOTE" in text:
return "credit_note"
if "DEBIT" in text or "DR NOTE" in text or "DEBIT NOTE" in text:
return "debit_note"
return "invoice"
def _row_hash(row: dict) -> str:
raw = "|".join([
row.get("supplier_gstin", ""),
row.get("supplier_name", ""),
row.get("invoice_number", ""),
row.get("invoice_date", ""),
row.get("document_type", ""),
f"{row.get('taxable_value', 0):.2f}",
f"{row.get('igst', 0):.2f}",
f"{row.get('cgst', 0):.2f}",
f"{row.get('sgst', 0):.2f}",
row.get("hsn_code", ""),
row.get("description_text", ""),
])
return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest()
def _find_header(rows: list[list], max_scan: int = 40):
best = None
for index, values in enumerate(rows[:max_scan]):
mapping = {}
for col, value in enumerate(values):
key = _canonical_header(value)
if key and key not in mapping:
mapping[key] = col
score = sum(1 for key in ("supplier_gstin", "invoice_number", "invoice_date", "taxable_value") if key in mapping)
if score >= 3 and ("supplier_gstin" in mapping or "supplier_name" in mapping):
if best is None or score > best[0]:
best = (score, index, mapping)
return best
def _normalized_record(values: list, mapping: dict[str, int], *, sheet: str, row_number: int):
def get(key):
idx = mapping.get(key)
return values[idx] if idx is not None and idx < len(values) else None
supplier_gstin = _gstin(get("supplier_gstin"))
supplier_name = _text(get("supplier_name"))
invoice_number = _text(get("invoice_number"))
invoice_date = _date_text(get("invoice_date"))
invoice_type = _text(get("invoice_type"))
if not supplier_gstin and not supplier_name:
return None
if not invoice_number or not invoice_date:
return None
row = {
"supplier_gstin": supplier_gstin,
"supplier_name": supplier_name,
"invoice_number": invoice_number[:160],
"invoice_date": invoice_date,
"document_type": _document_type(invoice_type, invoice_number),
"invoice_type": invoice_type[:80],
"invoice_value": _money(get("invoice_value")),
"place_of_supply": _text(get("place_of_supply"))[:120],
"reverse_charge": _text(get("reverse_charge"))[:20],
"taxable_value": _money(get("taxable_value")),
"igst": _money(get("igst")),
"cgst": _money(get("cgst")),
"sgst": _money(get("sgst")),
"cess": _money(get("cess")),
"itc_availability": _text(get("itc_availability"))[:80],
"hsn_code": _hsn(get("hsn_code")),
"description_text": _text(get("description_text"))[:4000],
"source_sheet": sheet[:160],
"source_row_number": int(row_number),
}
if not row["invoice_value"]:
row["invoice_value"] = (
row["taxable_value"] + row["igst"] + row["cgst"] + row["sgst"] + row["cess"]
)
row["source_row_hash"] = _row_hash(row)
return row
def _iter_xlsx(content: bytes):
workbook = load_workbook(io.BytesIO(content), read_only=True, data_only=True)
try:
for ws in workbook.worksheets:
# Summary sheets usually won't pass header detection.
rows = [list(row) for row in ws.iter_rows(values_only=True)]
if not rows:
continue
found = _find_header(rows)
if not found:
continue
_, header_index, mapping = found
for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2):
record = _normalized_record(values, mapping, sheet=ws.title, row_number=idx)
if record:
yield record
finally:
workbook.close()
def _decode_csv(content: bytes) -> str:
for encoding in ("utf-8-sig", "utf-8", "cp1252", "latin-1"):
try:
return content.decode(encoding)
except UnicodeDecodeError:
pass
return content.decode("utf-8", "replace")
def _iter_csv(content: bytes):
text = _decode_csv(content)
sample = text[:8192]
try:
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
except Exception:
dialect = csv.excel
rows = [list(row) for row in csv.reader(io.StringIO(text), dialect)]
found = _find_header(rows)
if not found:
return
_, header_index, mapping = found
for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2):
record = _normalized_record(values, mapping, sheet="CSV", row_number=idx)
if record:
yield record
def parse_gstr2b(content: bytes, filename: str):
if not content:
raise ValueError("Uploaded GSTR-2B file is empty.")
if len(content) > MAX_UPLOAD_BYTES:
raise ValueError("GSTR-2B upload exceeds the 25 MB limit.")
suffix = Path(filename or "").suffix.lower()
if suffix in {".xlsx", ".xlsm"}:
records = list(_iter_xlsx(content))
elif suffix in {".csv", ".txt"}:
records = list(_iter_csv(content))
else:
raise ValueError("Upload an .xlsx, .xlsm or .csv GSTR-2B file.")
if not records:
raise ValueError(
"No GSTR-2B invoice rows were detected. The file must contain supplier GSTIN/name, "
"invoice number, invoice date and taxable value columns."
)
return records
def file_sha256(content: bytes) -> str:
return hashlib.sha256(content).hexdigest()