from __future__ import annotations import csv import hashlib import io import re from datetime import date, datetime from pathlib import Path from typing import Iterable from openpyxl import load_workbook MAX_UPLOAD_BYTES = 25 * 1024 * 1024 HEADER_ALIASES = { "supplier_gstin": { "gstin of supplier", "supplier gstin", "gstin", "gstin/uin of supplier", "gstin / uin of supplier", "ctin", }, "supplier_name": { "trade/legal name", "trade legal name", "supplier name", "legal name", "trade name", "name of supplier", "supplier trade name", }, "invoice_number": { "invoice number", "invoice no", "invoice no.", "document number", "document no", "doc no", "inum", }, "invoice_date": { "invoice date", "document date", "doc date", "invoice dt", "idt", }, "invoice_type": { "invoice type", "document type", "type", "inv type", }, "invoice_value": { "invoice value", "document value", "invoice amount", "total invoice value", "val", }, "place_of_supply": { "place of supply", "pos", "place of supply (state/ut)", "place of supply state/ut", }, "reverse_charge": { "supply attract reverse charge", "reverse charge", "rcm", "reverse charge applicable", }, "taxable_value": { "taxable value", "taxable amount", "txval", }, "igst": { "integrated tax", "integrated tax amount", "igst", "igst amount", "iamt", }, "cgst": { "central tax", "central tax amount", "cgst", "cgst amount", "camt", }, "sgst": { "state/ut tax", "state / ut tax", "state tax", "sgst", "utgst", "state/ut tax amount", "sgst amount", "samt", }, "cess": { "cess", "cess amount", "csamt", }, "itc_availability": { "itc availability", "itc available", "itc available for the tax period", "itc eligibility", "eligibility for itc", }, "hsn_code": { "hsn", "hsn code", "hsn/sac", "hsn / sac", "hsn/sac code", }, "description_text": { "description", "item description", "product description", "goods/service description", "description of goods/services", "description of goods / services", }, } def _norm_header(value) -> str: text = str(value or "").strip().lower() text = text.replace("\n", " ").replace("\r", " ") text = re.sub(r"[_\-]+", " ", text) text = re.sub(r"[^\w/(). ]+", " ", text) text = re.sub(r"\s+", " ", text).strip() return text ALIAS_LOOKUP = { alias: key for key, aliases in HEADER_ALIASES.items() for alias in aliases } def _canonical_header(value) -> str | None: norm = _norm_header(value) if norm in ALIAS_LOOKUP: return ALIAS_LOOKUP[norm] # Conservative contains matching only for longer aliases. for alias, key in ALIAS_LOOKUP.items(): if len(alias) >= 10 and (norm.startswith(alias) or alias in norm): return key return None def _text(value) -> str: if value is None: return "" if isinstance(value, float) and value.is_integer(): return str(int(value)) return str(value).strip() def _money(value) -> float: if value in (None, ""): return 0.0 if isinstance(value, (int, float)): return float(value) text = str(value).strip().replace(",", "").replace("₹", "") text = re.sub(r"^\((.*)\)$", r"-\1", text) try: return float(text) except Exception: return 0.0 def _date_text(value) -> str: if value in (None, ""): return "" if isinstance(value, datetime): return value.date().isoformat() if isinstance(value, date): return value.isoformat() text = str(value).strip() candidates = ( "%d-%m-%Y", "%d/%m/%Y", "%Y-%m-%d", "%d-%b-%Y", "%d %b %Y", "%d.%m.%Y", "%m/%d/%Y", ) for fmt in candidates: try: return datetime.strptime(text, fmt).date().isoformat() except Exception: pass # Preserve original if portal format is unfamiliar; duplicate identity remains stable. return text[:20] def _gstin(value) -> str: return re.sub(r"[^A-Z0-9]", "", str(value or "").upper())[:15] def _hsn(value) -> str: text = re.sub(r"\D", "", str(value or "")) return text[:8] def _document_type(invoice_type: str, invoice_number: str) -> str: text = f"{invoice_type} {invoice_number}".upper() if "CREDIT" in text or "CR NOTE" in text or "CREDIT NOTE" in text: return "credit_note" if "DEBIT" in text or "DR NOTE" in text or "DEBIT NOTE" in text: return "debit_note" return "invoice" def _row_hash(row: dict) -> str: raw = "|".join([ row.get("supplier_gstin", ""), row.get("supplier_name", ""), row.get("invoice_number", ""), row.get("invoice_date", ""), row.get("document_type", ""), f"{row.get('taxable_value', 0):.2f}", f"{row.get('igst', 0):.2f}", f"{row.get('cgst', 0):.2f}", f"{row.get('sgst', 0):.2f}", row.get("hsn_code", ""), row.get("description_text", ""), ]) return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest() def _find_header(rows: list[list], max_scan: int = 40): best = None for index, values in enumerate(rows[:max_scan]): mapping = {} for col, value in enumerate(values): key = _canonical_header(value) if key and key not in mapping: mapping[key] = col score = sum(1 for key in ("supplier_gstin", "invoice_number", "invoice_date", "taxable_value") if key in mapping) if score >= 3 and ("supplier_gstin" in mapping or "supplier_name" in mapping): if best is None or score > best[0]: best = (score, index, mapping) return best def _normalized_record(values: list, mapping: dict[str, int], *, sheet: str, row_number: int): def get(key): idx = mapping.get(key) return values[idx] if idx is not None and idx < len(values) else None supplier_gstin = _gstin(get("supplier_gstin")) supplier_name = _text(get("supplier_name")) invoice_number = _text(get("invoice_number")) invoice_date = _date_text(get("invoice_date")) invoice_type = _text(get("invoice_type")) if not supplier_gstin and not supplier_name: return None if not invoice_number or not invoice_date: return None row = { "supplier_gstin": supplier_gstin, "supplier_name": supplier_name, "invoice_number": invoice_number[:160], "invoice_date": invoice_date, "document_type": _document_type(invoice_type, invoice_number), "invoice_type": invoice_type[:80], "invoice_value": _money(get("invoice_value")), "place_of_supply": _text(get("place_of_supply"))[:120], "reverse_charge": _text(get("reverse_charge"))[:20], "taxable_value": _money(get("taxable_value")), "igst": _money(get("igst")), "cgst": _money(get("cgst")), "sgst": _money(get("sgst")), "cess": _money(get("cess")), "itc_availability": _text(get("itc_availability"))[:80], "hsn_code": _hsn(get("hsn_code")), "description_text": _text(get("description_text"))[:4000], "source_sheet": sheet[:160], "source_row_number": int(row_number), } if not row["invoice_value"]: row["invoice_value"] = ( row["taxable_value"] + row["igst"] + row["cgst"] + row["sgst"] + row["cess"] ) row["source_row_hash"] = _row_hash(row) return row def _iter_xlsx(content: bytes): workbook = load_workbook(io.BytesIO(content), read_only=True, data_only=True) try: for ws in workbook.worksheets: # Summary sheets usually won't pass header detection. rows = [list(row) for row in ws.iter_rows(values_only=True)] if not rows: continue found = _find_header(rows) if not found: continue _, header_index, mapping = found for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2): record = _normalized_record(values, mapping, sheet=ws.title, row_number=idx) if record: yield record finally: workbook.close() def _decode_csv(content: bytes) -> str: for encoding in ("utf-8-sig", "utf-8", "cp1252", "latin-1"): try: return content.decode(encoding) except UnicodeDecodeError: pass return content.decode("utf-8", "replace") def _iter_csv(content: bytes): text = _decode_csv(content) sample = text[:8192] try: dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|") except Exception: dialect = csv.excel rows = [list(row) for row in csv.reader(io.StringIO(text), dialect)] found = _find_header(rows) if not found: return _, header_index, mapping = found for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2): record = _normalized_record(values, mapping, sheet="CSV", row_number=idx) if record: yield record def parse_gstr2b(content: bytes, filename: str): if not content: raise ValueError("Uploaded GSTR-2B file is empty.") if len(content) > MAX_UPLOAD_BYTES: raise ValueError("GSTR-2B upload exceeds the 25 MB limit.") suffix = Path(filename or "").suffix.lower() if suffix in {".xlsx", ".xlsm"}: records = list(_iter_xlsx(content)) elif suffix in {".csv", ".txt"}: records = list(_iter_csv(content)) else: raise ValueError("Upload an .xlsx, .xlsm or .csv GSTR-2B file.") if not records: raise ValueError( "No GSTR-2B invoice rows were detected. The file must contain supplier GSTIN/name, " "invoice number, invoice date and taxable value columns." ) return records def file_sha256(content: bytes) -> str: return hashlib.sha256(content).hexdigest()