308 lines
9.9 KiB
Python
308 lines
9.9 KiB
Python
from __future__ import annotations
|
|
|
|
import csv
|
|
import hashlib
|
|
import io
|
|
import re
|
|
from datetime import date, datetime
|
|
from pathlib import Path
|
|
from typing import Iterable
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
|
|
MAX_UPLOAD_BYTES = 25 * 1024 * 1024
|
|
|
|
HEADER_ALIASES = {
|
|
"supplier_gstin": {
|
|
"gstin of supplier", "supplier gstin", "gstin", "gstin/uin of supplier",
|
|
"gstin / uin of supplier", "ctin",
|
|
},
|
|
"supplier_name": {
|
|
"trade/legal name", "trade legal name", "supplier name", "legal name",
|
|
"trade name", "name of supplier", "supplier trade name",
|
|
},
|
|
"invoice_number": {
|
|
"invoice number", "invoice no", "invoice no.", "document number",
|
|
"document no", "doc no", "inum",
|
|
},
|
|
"invoice_date": {
|
|
"invoice date", "document date", "doc date", "invoice dt", "idt",
|
|
},
|
|
"invoice_type": {
|
|
"invoice type", "document type", "type", "inv type",
|
|
},
|
|
"invoice_value": {
|
|
"invoice value", "document value", "invoice amount", "total invoice value",
|
|
"val",
|
|
},
|
|
"place_of_supply": {
|
|
"place of supply", "pos", "place of supply (state/ut)", "place of supply state/ut",
|
|
},
|
|
"reverse_charge": {
|
|
"supply attract reverse charge", "reverse charge", "rcm", "reverse charge applicable",
|
|
},
|
|
"taxable_value": {
|
|
"taxable value", "taxable amount", "txval",
|
|
},
|
|
"igst": {
|
|
"integrated tax", "integrated tax amount", "igst", "igst amount", "iamt",
|
|
},
|
|
"cgst": {
|
|
"central tax", "central tax amount", "cgst", "cgst amount", "camt",
|
|
},
|
|
"sgst": {
|
|
"state/ut tax", "state / ut tax", "state tax", "sgst", "utgst",
|
|
"state/ut tax amount", "sgst amount", "samt",
|
|
},
|
|
"cess": {
|
|
"cess", "cess amount", "csamt",
|
|
},
|
|
"itc_availability": {
|
|
"itc availability", "itc available", "itc available for the tax period",
|
|
"itc eligibility", "eligibility for itc",
|
|
},
|
|
"hsn_code": {
|
|
"hsn", "hsn code", "hsn/sac", "hsn / sac", "hsn/sac code",
|
|
},
|
|
"description_text": {
|
|
"description", "item description", "product description", "goods/service description",
|
|
"description of goods/services", "description of goods / services",
|
|
},
|
|
}
|
|
|
|
|
|
def _norm_header(value) -> str:
|
|
text = str(value or "").strip().lower()
|
|
text = text.replace("\n", " ").replace("\r", " ")
|
|
text = re.sub(r"[_\-]+", " ", text)
|
|
text = re.sub(r"[^\w/(). ]+", " ", text)
|
|
text = re.sub(r"\s+", " ", text).strip()
|
|
return text
|
|
|
|
|
|
ALIAS_LOOKUP = {
|
|
alias: key
|
|
for key, aliases in HEADER_ALIASES.items()
|
|
for alias in aliases
|
|
}
|
|
|
|
|
|
def _canonical_header(value) -> str | None:
|
|
norm = _norm_header(value)
|
|
if norm in ALIAS_LOOKUP:
|
|
return ALIAS_LOOKUP[norm]
|
|
# Conservative contains matching only for longer aliases.
|
|
for alias, key in ALIAS_LOOKUP.items():
|
|
if len(alias) >= 10 and (norm.startswith(alias) or alias in norm):
|
|
return key
|
|
return None
|
|
|
|
|
|
def _text(value) -> str:
|
|
if value is None:
|
|
return ""
|
|
if isinstance(value, float) and value.is_integer():
|
|
return str(int(value))
|
|
return str(value).strip()
|
|
|
|
|
|
def _money(value) -> float:
|
|
if value in (None, ""):
|
|
return 0.0
|
|
if isinstance(value, (int, float)):
|
|
return float(value)
|
|
text = str(value).strip().replace(",", "").replace("₹", "")
|
|
text = re.sub(r"^\((.*)\)$", r"-\1", text)
|
|
try:
|
|
return float(text)
|
|
except Exception:
|
|
return 0.0
|
|
|
|
|
|
def _date_text(value) -> str:
|
|
if value in (None, ""):
|
|
return ""
|
|
if isinstance(value, datetime):
|
|
return value.date().isoformat()
|
|
if isinstance(value, date):
|
|
return value.isoformat()
|
|
text = str(value).strip()
|
|
candidates = (
|
|
"%d-%m-%Y", "%d/%m/%Y", "%Y-%m-%d", "%d-%b-%Y", "%d %b %Y",
|
|
"%d.%m.%Y", "%m/%d/%Y",
|
|
)
|
|
for fmt in candidates:
|
|
try:
|
|
return datetime.strptime(text, fmt).date().isoformat()
|
|
except Exception:
|
|
pass
|
|
# Preserve original if portal format is unfamiliar; duplicate identity remains stable.
|
|
return text[:20]
|
|
|
|
|
|
def _gstin(value) -> str:
|
|
return re.sub(r"[^A-Z0-9]", "", str(value or "").upper())[:15]
|
|
|
|
|
|
def _hsn(value) -> str:
|
|
text = re.sub(r"\D", "", str(value or ""))
|
|
return text[:8]
|
|
|
|
|
|
def _document_type(invoice_type: str, invoice_number: str) -> str:
|
|
text = f"{invoice_type} {invoice_number}".upper()
|
|
if "CREDIT" in text or "CR NOTE" in text or "CREDIT NOTE" in text:
|
|
return "credit_note"
|
|
if "DEBIT" in text or "DR NOTE" in text or "DEBIT NOTE" in text:
|
|
return "debit_note"
|
|
return "invoice"
|
|
|
|
|
|
def _row_hash(row: dict) -> str:
|
|
raw = "|".join([
|
|
row.get("supplier_gstin", ""),
|
|
row.get("supplier_name", ""),
|
|
row.get("invoice_number", ""),
|
|
row.get("invoice_date", ""),
|
|
row.get("document_type", ""),
|
|
f"{row.get('taxable_value', 0):.2f}",
|
|
f"{row.get('igst', 0):.2f}",
|
|
f"{row.get('cgst', 0):.2f}",
|
|
f"{row.get('sgst', 0):.2f}",
|
|
row.get("hsn_code", ""),
|
|
row.get("description_text", ""),
|
|
])
|
|
return hashlib.sha256(raw.encode("utf-8", "ignore")).hexdigest()
|
|
|
|
|
|
def _find_header(rows: list[list], max_scan: int = 40):
|
|
best = None
|
|
for index, values in enumerate(rows[:max_scan]):
|
|
mapping = {}
|
|
for col, value in enumerate(values):
|
|
key = _canonical_header(value)
|
|
if key and key not in mapping:
|
|
mapping[key] = col
|
|
score = sum(1 for key in ("supplier_gstin", "invoice_number", "invoice_date", "taxable_value") if key in mapping)
|
|
if score >= 3 and ("supplier_gstin" in mapping or "supplier_name" in mapping):
|
|
if best is None or score > best[0]:
|
|
best = (score, index, mapping)
|
|
return best
|
|
|
|
|
|
def _normalized_record(values: list, mapping: dict[str, int], *, sheet: str, row_number: int):
|
|
def get(key):
|
|
idx = mapping.get(key)
|
|
return values[idx] if idx is not None and idx < len(values) else None
|
|
|
|
supplier_gstin = _gstin(get("supplier_gstin"))
|
|
supplier_name = _text(get("supplier_name"))
|
|
invoice_number = _text(get("invoice_number"))
|
|
invoice_date = _date_text(get("invoice_date"))
|
|
invoice_type = _text(get("invoice_type"))
|
|
|
|
if not supplier_gstin and not supplier_name:
|
|
return None
|
|
if not invoice_number or not invoice_date:
|
|
return None
|
|
|
|
row = {
|
|
"supplier_gstin": supplier_gstin,
|
|
"supplier_name": supplier_name,
|
|
"invoice_number": invoice_number[:160],
|
|
"invoice_date": invoice_date,
|
|
"document_type": _document_type(invoice_type, invoice_number),
|
|
"invoice_type": invoice_type[:80],
|
|
"invoice_value": _money(get("invoice_value")),
|
|
"place_of_supply": _text(get("place_of_supply"))[:120],
|
|
"reverse_charge": _text(get("reverse_charge"))[:20],
|
|
"taxable_value": _money(get("taxable_value")),
|
|
"igst": _money(get("igst")),
|
|
"cgst": _money(get("cgst")),
|
|
"sgst": _money(get("sgst")),
|
|
"cess": _money(get("cess")),
|
|
"itc_availability": _text(get("itc_availability"))[:80],
|
|
"hsn_code": _hsn(get("hsn_code")),
|
|
"description_text": _text(get("description_text"))[:4000],
|
|
"source_sheet": sheet[:160],
|
|
"source_row_number": int(row_number),
|
|
}
|
|
if not row["invoice_value"]:
|
|
row["invoice_value"] = (
|
|
row["taxable_value"] + row["igst"] + row["cgst"] + row["sgst"] + row["cess"]
|
|
)
|
|
row["source_row_hash"] = _row_hash(row)
|
|
return row
|
|
|
|
|
|
def _iter_xlsx(content: bytes):
|
|
workbook = load_workbook(io.BytesIO(content), read_only=True, data_only=True)
|
|
try:
|
|
for ws in workbook.worksheets:
|
|
# Summary sheets usually won't pass header detection.
|
|
rows = [list(row) for row in ws.iter_rows(values_only=True)]
|
|
if not rows:
|
|
continue
|
|
found = _find_header(rows)
|
|
if not found:
|
|
continue
|
|
_, header_index, mapping = found
|
|
for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2):
|
|
record = _normalized_record(values, mapping, sheet=ws.title, row_number=idx)
|
|
if record:
|
|
yield record
|
|
finally:
|
|
workbook.close()
|
|
|
|
|
|
def _decode_csv(content: bytes) -> str:
|
|
for encoding in ("utf-8-sig", "utf-8", "cp1252", "latin-1"):
|
|
try:
|
|
return content.decode(encoding)
|
|
except UnicodeDecodeError:
|
|
pass
|
|
return content.decode("utf-8", "replace")
|
|
|
|
|
|
def _iter_csv(content: bytes):
|
|
text = _decode_csv(content)
|
|
sample = text[:8192]
|
|
try:
|
|
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
|
|
except Exception:
|
|
dialect = csv.excel
|
|
rows = [list(row) for row in csv.reader(io.StringIO(text), dialect)]
|
|
found = _find_header(rows)
|
|
if not found:
|
|
return
|
|
_, header_index, mapping = found
|
|
for idx, values in enumerate(rows[header_index + 1:], start=header_index + 2):
|
|
record = _normalized_record(values, mapping, sheet="CSV", row_number=idx)
|
|
if record:
|
|
yield record
|
|
|
|
|
|
def parse_gstr2b(content: bytes, filename: str):
|
|
if not content:
|
|
raise ValueError("Uploaded GSTR-2B file is empty.")
|
|
if len(content) > MAX_UPLOAD_BYTES:
|
|
raise ValueError("GSTR-2B upload exceeds the 25 MB limit.")
|
|
suffix = Path(filename or "").suffix.lower()
|
|
if suffix in {".xlsx", ".xlsm"}:
|
|
records = list(_iter_xlsx(content))
|
|
elif suffix in {".csv", ".txt"}:
|
|
records = list(_iter_csv(content))
|
|
else:
|
|
raise ValueError("Upload an .xlsx, .xlsm or .csv GSTR-2B file.")
|
|
if not records:
|
|
raise ValueError(
|
|
"No GSTR-2B invoice rows were detected. The file must contain supplier GSTIN/name, "
|
|
"invoice number, invoice date and taxable value columns."
|
|
)
|
|
return records
|
|
|
|
|
|
def file_sha256(content: bytes) -> str:
|
|
return hashlib.sha256(content).hexdigest()
|