Add Phase 7 GSTR-2B purchase intelligence
This commit is contained in:
@@ -0,0 +1,307 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import hashlib
|
||||
import io
|
||||
import re
|
||||
from datetime import date, datetime
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
from openpyxl import load_workbook
|
||||
|
||||
|
||||
MAX_UPLOAD_BYTES = 25 * 1024 * 1024
|
||||
|
||||
HEADER_ALIASES = {
|
||||
"supplier_gstin": {
|
||||
"gstin of supplier", "supplier gstin", "gstin", "gstin/uin of supplier",
|
||||
"gstin / uin of supplier", "ctin",
|
||||
},
|
||||
"supplier_name": {
|
||||
"trade/legal name", "trade legal name", "supplier name", "legal name",
|
||||
"trade name", "name of supplier", "supplier trade name",
|
||||
},
|
||||
"invoice_number": {
|
||||
"invoice number", "invoice no", "invoice no.", "document number",
|
||||
"document no", "doc no", "inum",
|
||||
},
|
||||
"invoice_date": {
|
||||
"invoice date", "document date", "doc date", "invoice dt", "idt",
|
||||
},
|
||||
"invoice_type": {
|
||||
"invoice type", "document type", "type", "inv type",
|
||||
},
|
||||
"invoice_value": {
|
||||
"invoice value", "document value", "invoice amount", "total invoice value",
|
||||
"val",
|
||||
},
|
||||
"place_of_supply": {
|
||||
"place of supply", "pos", "place of supply (state/ut)", "place of supply state/ut",
|
||||
},
|
||||
"reverse_charge": {
|
||||
"supply attract reverse charge", "reverse charge", "rcm", "reverse charge applicable",
|
||||
},
|
||||
"taxable_value": {
|
||||
"taxable value", "taxable amount", "txval",
|
||||
},
|
||||
"igst": {
|
||||
"integrated tax", "integrated tax amount", "igst", "igst amount", "iamt",
|
||||
},
|
||||
"cgst": {
|
||||
"central tax", "central tax amount", "cgst", "cgst amount", "camt",
|
||||
},
|
||||
"sgst": {
|
||||
"state/ut tax", "state / ut tax", "state tax", "sgst", "utgst",
|
||||
"state/ut tax amount", "sgst amount", "samt",
|
||||
},
|
||||
"cess": {
|
||||
"cess", "cess amount", "csamt",
|
||||
},
|
||||
"itc_availability": {
|
||||
"itc availability", "itc available", "itc available for the tax period",
|
||||
"itc eligibility", "eligibility for itc",
|
||||
},
|
||||
"hsn_code": {
|
||||
"hsn", "hsn code", "hsn/sac", "hsn / sac", "hsn/sac code",
|
||||
},
|
||||
"description_text": {
|
||||
"description", "item description", "product description", "goods/service description",
|
||||
"description of goods/services", "description of goods / services",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _norm_header(value) -> str:
|
||||
text = str(value or "").strip().lower()
|
||||
text = text.replace("\n", " ").replace("\r", " ")
|
||||
text = re.sub(r"[_\-]+", " ", text)
|
||||
text = re.sub(r"[^\w/(). ]+", " ", text)
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
return text
|
||||
|
||||
|
||||
ALIAS_LOOKUP = {
|
||||
alias: key
|
||||
for key, aliases in HEADER_ALIASES.items()
|
||||
for alias in aliases
|
||||
}
|
||||
|
||||
|
||||
def _canonical_header(value) -> str | None:
|
||||
norm = _norm_header(value)
|
||||
if norm in ALIAS_LOOKUP:
|
||||
return ALIAS_LOOKUP[norm]
|
||||
# Conservative contains matching only for longer aliases.
|
||||
for alias, key in ALIAS_LOOKUP.items():
|
||||
if len(alias) >= 10 and (norm.startswith(alias) or alias in norm):
|
||||
return key
|
||||
return None
|
||||
|
||||
|
||||
def _text(value) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, float) and value.is_integer():
|
||||
return str(int(value))
|
||||
return str(value).strip()
|
||||
|
||||
|
||||
def _money(value) -> float:
|
||||
if value in (None, ""):
|
||||
return 0.0
|
||||
if isinstance(value, (int, float)):
|
||||
return float(value)
|
||||
text = str(value).strip().replace(",", "").replace("₹", "")
|
||||
text = re.sub(r"^\((.*)\)$", r"-\1", text)
|
||||
try:
|
||||
return float(text)
|
||||
except Exception:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _date_text(value) -> str:
|
||||
if value in (None, ""):
|
||||
return ""
|
||||
if isinstance(value, datetime):
|
||||
return value.date().isoformat()
|
||||
if isinstance(value, date):
|
||||
return value.isoformat()
|
||||
text = str(value).strip()
|
||||
candidates = (
|
||||
"%d-%m-%Y", "%d/%m/%Y", "%Y-%m-%d", "%d-%b-%Y", "%d %b %Y",
|
||||
"%d.%m.%Y", "%m/%d/%Y",
|
||||
)
|
||||
for fmt in candidates:
|
||||
try:
|
||||
return datetime.strptime(text, fmt).date().isoformat()
|
||||
except Exception:
|
||||
pass
|
||||
# Preserve original if portal format is unfamiliar; duplicate identity remains stable.
|
||||
return text[:20]
|
||||
|
||||
|
||||
def _gstin(value) -> str:
|
||||
return re.sub(r"[^A-Z0-9]", "", str(value or "").upper())[:15]
|
||||
|
||||
|
||||
def _hsn(value) -> str:
|
||||
text = re.sub(r"\D", "", str(value or ""))
|
||||
return text[:8]
|
||||
|
||||
|
||||
def _document_type(invoice_type: str, invoice_number: str) -> str:
|
||||
text = f"{invoice_type} {invoice_number}".upper()
|
||||
if "CREDIT" in text or "CR NOTE" in text or "CREDIT NOTE" in text:
|
||||
return "credit_note"
|
||||
if "DEBIT" in text or "DR NOTE" in text or "DEBIT NOTE" in text:
|
||||
return "debit_note"
|
||||
return "invoice"
|
||||
|
||||
|
||||
def _row_hash(row: dict) -> str:
|
||||
raw = "|".join([
|
||||
row.get("supplier_gstin", ""),
|
||||
row.get("supplier_name", ""),
|
||||
row.get("invoice_number", ""),
|
||||
row.get("invoice_date", ""),
|
||||
row.get("document_type", ""),
|
||||
f"{row.get('taxable_value', 0):.2f}",
|
||||
f"{row.get('igst', 0):.2f}",
|
||||
f"{row.get('cgst', 0):.2f}",
|
||||
f"{row.get('sgst', 0):.2f}",
|
||||
row.get("hsn_code", ""),
|
||||
row.get("description_text", ""),
|
||||
])
|
||||
return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest()
|
||||
|
||||
|
||||
def _find_header(rows: list[list], max_scan: int = 40):
|
||||
best = None
|
||||
for index, values in enumerate(rows[:max_scan]):
|
||||
mapping = {}
|
||||
for col, value in enumerate(values):
|
||||
key = _canonical_header(value)
|
||||
if key and key not in mapping:
|
||||
mapping[key] = col
|
||||
score = sum(1 for key in ("supplier_gstin", "invoice_number", "invoice_date", "taxable_value") if key in mapping)
|
||||
if score >= 3 and ("supplier_gstin" in mapping or "supplier_name" in mapping):
|
||||
if best is None or score > best[0]:
|
||||
best = (score, index, mapping)
|
||||
return best
|
||||
|
||||
|
||||
def _normalized_record(values: list, mapping: dict[str, int], *, sheet: str, row_number: int):
|
||||
def get(key):
|
||||
idx = mapping.get(key)
|
||||
return values[idx] if idx is not None and idx < len(values) else None
|
||||
|
||||
supplier_gstin = _gstin(get("supplier_gstin"))
|
||||
supplier_name = _text(get("supplier_name"))
|
||||
invoice_number = _text(get("invoice_number"))
|
||||
invoice_date = _date_text(get("invoice_date"))
|
||||
invoice_type = _text(get("invoice_type"))
|
||||
|
||||
if not supplier_gstin and not supplier_name:
|
||||
return None
|
||||
if not invoice_number or not invoice_date:
|
||||
return None
|
||||
|
||||
row = {
|
||||
"supplier_gstin": supplier_gstin,
|
||||
"supplier_name": supplier_name,
|
||||
"invoice_number": invoice_number[:160],
|
||||
"invoice_date": invoice_date,
|
||||
"document_type": _document_type(invoice_type, invoice_number),
|
||||
"invoice_type": invoice_type[:80],
|
||||
"invoice_value": _money(get("invoice_value")),
|
||||
"place_of_supply": _text(get("place_of_supply"))[:120],
|
||||
"reverse_charge": _text(get("reverse_charge"))[:20],
|
||||
"taxable_value": _money(get("taxable_value")),
|
||||
"igst": _money(get("igst")),
|
||||
"cgst": _money(get("cgst")),
|
||||
"sgst": _money(get("sgst")),
|
||||
"cess": _money(get("cess")),
|
||||
"itc_availability": _text(get("itc_availability"))[:80],
|
||||
"hsn_code": _hsn(get("hsn_code")),
|
||||
"description_text": _text(get("description_text"))[:4000],
|
||||
"source_sheet": sheet[:160],
|
||||
"source_row_number": int(row_number),
|
||||
}
|
||||
if not row["invoice_value"]:
|
||||
row["invoice_value"] = (
|
||||
row["taxable_value"] + row["igst"] + row["cgst"] + row["sgst"] + row["cess"]
|
||||
)
|
||||
row["source_row_hash"] = _row_hash(row)
|
||||
return row
|
||||
|
||||
|
||||
def _iter_xlsx(content: bytes):
|
||||
workbook = load_workbook(io.BytesIO(content), read_only=True, data_only=True)
|
||||
try:
|
||||
for ws in workbook.worksheets:
|
||||
# Summary sheets usually won't pass header detection.
|
||||
rows = [list(row) for row in ws.iter_rows(values_only=True)]
|
||||
if not rows:
|
||||
continue
|
||||
found = _find_header(rows)
|
||||
if not found:
|
||||
continue
|
||||
_, header_index, mapping = found
|
||||
for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2):
|
||||
record = _normalized_record(values, mapping, sheet=ws.title, row_number=idx)
|
||||
if record:
|
||||
yield record
|
||||
finally:
|
||||
workbook.close()
|
||||
|
||||
|
||||
def _decode_csv(content: bytes) -> str:
|
||||
for encoding in ("utf-8-sig", "utf-8", "cp1252", "latin-1"):
|
||||
try:
|
||||
return content.decode(encoding)
|
||||
except UnicodeDecodeError:
|
||||
pass
|
||||
return content.decode("utf-8", "replace")
|
||||
|
||||
|
||||
def _iter_csv(content: bytes):
|
||||
text = _decode_csv(content)
|
||||
sample = text[:8192]
|
||||
try:
|
||||
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
|
||||
except Exception:
|
||||
dialect = csv.excel
|
||||
rows = [list(row) for row in csv.reader(io.StringIO(text), dialect)]
|
||||
found = _find_header(rows)
|
||||
if not found:
|
||||
return
|
||||
_, header_index, mapping = found
|
||||
for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2):
|
||||
record = _normalized_record(values, mapping, sheet="CSV", row_number=idx)
|
||||
if record:
|
||||
yield record
|
||||
|
||||
|
||||
def parse_gstr2b(content: bytes, filename: str):
|
||||
if not content:
|
||||
raise ValueError("Uploaded GSTR-2B file is empty.")
|
||||
if len(content) > MAX_UPLOAD_BYTES:
|
||||
raise ValueError("GSTR-2B upload exceeds the 25 MB limit.")
|
||||
suffix = Path(filename or "").suffix.lower()
|
||||
if suffix in {".xlsx", ".xlsm"}:
|
||||
records = list(_iter_xlsx(content))
|
||||
elif suffix in {".csv", ".txt"}:
|
||||
records = list(_iter_csv(content))
|
||||
else:
|
||||
raise ValueError("Upload an .xlsx, .xlsm or .csv GSTR-2B file.")
|
||||
if not records:
|
||||
raise ValueError(
|
||||
"No GSTR-2B invoice rows were detected. The file must contain supplier GSTIN/name, "
|
||||
"invoice number, invoice date and taxable value columns."
|
||||
)
|
||||
return records
|
||||
|
||||
|
||||
def file_sha256(content: bytes) -> str:
|
||||
return hashlib.sha256(content).hexdigest()
|
||||
Reference in New Issue
Block a user