from __future__ import annotations import csv import io import logging import os from datetime import date, datetime from pathlib import Path from typing import Iterable from openpyxl import load_workbook from sqlalchemy import select from sqlalchemy.orm import Session from app.models.throughput import ProductionThroughput, ThroughputProduct logger = logging.getLogger("data_entry_app.throughput") PRODUCTION_SHEET = "Production" NAMES_SHEET = "Names" # The historical throughput export. Bundled into the image under input_data/ so # the seed can import it on a fresh deployment (e.g. a new Postgres volume). WORKBOOK_FILENAME = "Operations Throughput.xlsx" # Anything at or above this kg/bag is treated as a bulka batch, not a per-bag count. _BULKA_BAG_SIZE_THRESHOLD = 100.0 def normalise_staff_name(value: object) -> str | None: if value is None: return None text = str(value).strip() if not text: return None # Collapse internal whitespace, title-case for consistency. cleaned = " ".join(text.split()) return cleaned def calculate_kg(quantity: float | None, quantity_type: str, bag_size: float | None) -> float: if quantity is None: return 0.0 if quantity_type == "kg": return float(quantity) if bag_size is None: return 0.0 return float(quantity) * float(bag_size) def qa_passed(entry: ProductionThroughput) -> bool: return bool(entry.scales_checked and entry.label_correct and entry.bag_sealed and entry.pallet_good_condition) def serialize_entry(entry: ProductionThroughput) -> dict: return { "id": entry.id, "tenant_id": entry.tenant_id, "production_date": entry.production_date, "product_id": entry.product_id, "product_name_snapshot": entry.product_name_snapshot, "bag_size": entry.bag_size, "scales_checked": entry.scales_checked, "label_correct": entry.label_correct, "bag_sealed": entry.bag_sealed, "pallet_good_condition": entry.pallet_good_condition, "for_order": entry.for_order, "for_stock": entry.for_stock, "job_number": entry.job_number, "stock_quantity": entry.stock_quantity, "sample_box_no": entry.sample_box_no, "test_weight_1": entry.test_weight_1, "test_weight_2": entry.test_weight_2, "test_weight_3": entry.test_weight_3, "test_weight_4": entry.test_weight_4, "test_weight_5": entry.test_weight_5, "quantity": entry.quantity, "quantity_type": entry.quantity_type, "calculated_kg": entry.calculated_kg, "staff_name": entry.staff_name, "notes": entry.notes, "qa_passed": qa_passed(entry), "created_by": entry.created_by, "created_at": entry.created_at, "updated_at": entry.updated_at, } def _coerce_bool(value: object) -> bool: if isinstance(value, bool): return value if value is None: return True if isinstance(value, (int, float)): return bool(value) text = str(value).strip().lower() if text in {"yes", "y", "true", "1", "pass", "ok", "x", "checked"}: return True if text in {"no", "n", "false", "0", "fail"}: return False return True def _coerce_float(value: object) -> float | None: if value is None or value == "": return None if isinstance(value, bool): return float(value) if isinstance(value, (int, float)): return float(value) text = str(value).strip().replace(",", "") if not text: return None try: return float(text) except ValueError: return None def _coerce_text(value: object) -> str | None: if value is None: return None text = str(value).strip() if not text or text.lower() in {"#value!", "#n/a", "n/a"}: return None return text def _coerce_date(value: object) -> date | None: if value is None: return None if isinstance(value, datetime): return value.date() if isinstance(value, date): return value text = str(value).strip() if not text: return None for fmt in ("%Y-%m-%d", "%d/%m/%Y", "%m/%d/%Y"): try: return datetime.strptime(text, fmt).date() except ValueError: continue return None def _infer_bulka_default(name: str, bag_size: float | None) -> bool: lowered = name.lower() if "bulka" in lowered: return True if bag_size is None: return False return bag_size >= _BULKA_BAG_SIZE_THRESHOLD def import_names_sheet(db: Session, workbook, tenant_id: str) -> tuple[int, int]: """Upsert product master from the Names sheet. Returns (created, updated).""" if NAMES_SHEET not in workbook.sheetnames: return (0, 0) ws = workbook[NAMES_SHEET] existing: dict[tuple[str, str | None], ThroughputProduct] = {} by_item: dict[str, ThroughputProduct] = {} by_name: dict[str, ThroughputProduct] = {} for product in db.scalars( select(ThroughputProduct).where(ThroughputProduct.tenant_id == tenant_id) ).all(): if product.item_id: by_item[str(product.item_id)] = product by_name[product.name.lower()] = product created = 0 updated = 0 for row in ws.iter_rows(min_row=2, values_only=True): if not row: continue name = _coerce_text(row[0] if len(row) > 0 else None) if not name: continue item_id_raw = row[1] if len(row) > 1 else None item_id = None if item_id_raw is not None: if isinstance(item_id_raw, float) and item_id_raw.is_integer(): item_id = str(int(item_id_raw)) else: item_id = _coerce_text(item_id_raw) product = (by_item.get(item_id) if item_id else None) or by_name.get(name.lower()) if product is None: product = ThroughputProduct( tenant_id=tenant_id, item_id=item_id, name=name, default_bag_size=None, is_bulka_default="bulka" in name.lower(), active=True, notes="Imported from Operations Throughput.xlsx", ) db.add(product) created += 1 if item_id: by_item[item_id] = product by_name[name.lower()] = product else: if item_id and not product.item_id: product.item_id = item_id if name and product.name != name: product.name = name updated += 1 db.flush() return (created, updated) def import_production_sheet(db: Session, workbook, tenant_id: str) -> tuple[int, int]: """Import the Production sheet. Returns (imported, skipped).""" if PRODUCTION_SHEET not in workbook.sheetnames: return (0, 0) ws = workbook[PRODUCTION_SHEET] # Header row is row 3 in the sheet (rows 1 and 2 are display banners). products_by_name: dict[str, ThroughputProduct] = { product.name.lower(): product for product in db.scalars( select(ThroughputProduct).where(ThroughputProduct.tenant_id == tenant_id) ).all() } bag_size_seen: dict[int, list[float]] = {} imported = 0 skipped = 0 for row in ws.iter_rows(min_row=4, values_only=True): if not row or len(row) < 15: skipped += 1 continue production_date = _coerce_date(row[0]) product_name = _coerce_text(row[1]) if production_date is None or not product_name: skipped += 1 continue bag_size = _coerce_float(row[2]) scales = _coerce_bool(row[3]) label = _coerce_bool(row[4]) sealed = _coerce_bool(row[5]) pallet = _coerce_bool(row[6]) sample_box = _coerce_text(row[7]) tw1 = _coerce_float(row[8]) tw2 = _coerce_float(row[9]) tw3 = _coerce_float(row[10]) tw4 = _coerce_float(row[11]) tw5 = _coerce_float(row[12]) quantity = _coerce_float(row[13]) or 0.0 staff = normalise_staff_name(row[14]) notes = _coerce_text(row[15]) if len(row) > 15 else None # Infer quantity_type: bulka-style rows have a blank or very large bag size. if bag_size is None or bag_size >= _BULKA_BAG_SIZE_THRESHOLD or "bulka" in product_name.lower(): quantity_type = "kg" else: quantity_type = "bags" product = products_by_name.get(product_name.lower()) if product is None: product = ThroughputProduct( tenant_id=tenant_id, item_id=None, name=product_name, default_bag_size=bag_size, is_bulka_default=_infer_bulka_default(product_name, bag_size), active=True, notes="Auto-created during Operations Throughput import", ) db.add(product) db.flush() products_by_name[product_name.lower()] = product if product.id is not None and bag_size is not None and bag_size > 0: bag_size_seen.setdefault(product.id, []).append(bag_size) calculated = calculate_kg(quantity, quantity_type, bag_size) entry = ProductionThroughput( tenant_id=tenant_id, production_date=production_date, product_id=product.id, product_name_snapshot=product_name, bag_size=bag_size, scales_checked=scales, label_correct=label, bag_sealed=sealed, pallet_good_condition=pallet, sample_box_no=sample_box, test_weight_1=tw1, test_weight_2=tw2, test_weight_3=tw3, test_weight_4=tw4, test_weight_5=tw5, quantity=quantity, quantity_type=quantity_type, calculated_kg=calculated, staff_name=staff, notes=notes, created_by="workbook-import", ) db.add(entry) imported += 1 # Backfill default_bag_size on products that don't have one but appear in entries. for product_id, sizes in bag_size_seen.items(): product = db.get(ThroughputProduct, product_id) if product and product.default_bag_size is None: # Use the most common bag size seen. common = max(set(sizes), key=sizes.count) product.default_bag_size = common if not product.is_bulka_default: product.is_bulka_default = _infer_bulka_default(product.name, common) db.flush() return (imported, skipped) def import_workbook(db: Session, workbook_path: Path, tenant_id: str) -> dict: workbook = load_workbook(workbook_path, data_only=True) products_created, products_updated = import_names_sheet(db, workbook, tenant_id) entries_imported, entries_skipped = import_production_sheet(db, workbook, tenant_id) return { "products_created": products_created, "products_updated": products_updated, "entries_imported": entries_imported, "entries_skipped": entries_skipped, } def workbook_candidates() -> Iterable[Path]: repo_root = Path(__file__).resolve().parents[3] cwd = Path.cwd() env_value = os.getenv("THROUGHPUT_WORKBOOK_PATH") env_path = Path(env_value.strip()) if isinstance(env_value, str) and env_value.strip() else None # input_data/ is where the workbook is bundled in the image; in the # container the working directory is /app, so cwd/input_data resolves it. candidates = [ env_path, repo_root / "input_data" / WORKBOOK_FILENAME, cwd / "input_data" / WORKBOOK_FILENAME, Path("/app") / "input_data" / WORKBOOK_FILENAME, Path("/srv/lean101-clients") / "input_data" / WORKBOOK_FILENAME, repo_root / WORKBOOK_FILENAME, repo_root.parent / WORKBOOK_FILENAME, cwd / WORKBOOK_FILENAME, Path("/srv/lean101-clients") / WORKBOOK_FILENAME, Path("/app") / WORKBOOK_FILENAME, ] seen: set[str] = set() ordered: list[Path] = [] for candidate in candidates: if candidate is None: continue key = str(candidate) if key in seen: continue seen.add(key) ordered.append(candidate) return ordered def resolve_workbook_path() -> Path | None: for candidate in workbook_candidates(): if candidate.exists(): return candidate return None # ── Ad-hoc CSV / spreadsheet upload import ────────────────────────────────── # Lets an operator upload their own CSV or .xlsx of packing runs (from Settings # → Import) and have every row saved as a throughput entry. Unlike the bundled # workbook seed above, this is column-header driven so the file can be a simple # hand-built sheet rather than the exact "Operations Throughput.xlsx" layout. # Maps the column headers we accept (normalised: lower-cased, spaces/dashes → # single spaces) onto the canonical field used internally. Several aliases per # field so a human-built sheet "just works". _HEADER_ALIASES: dict[str, str] = { "date": "date", "production date": "date", "production_date": "date", "product": "product", "product name": "product", "product_name": "product", "product name snapshot": "product", "name": "product", "item id": "item_id", "item_id": "item_id", "itemid": "item_id", "sku": "item_id", "quantity": "quantity", "qty": "quantity", "packed": "quantity", "quantity packed": "quantity", "amount": "quantity", "quantity type": "quantity_type", "type": "quantity_type", "unit": "quantity_type", "packed as": "quantity_type", "bag size": "bag_size", "bag_size": "bag_size", "kg per bag": "bag_size", "kg/bag": "bag_size", "bagsize": "bag_size", "staff": "staff_name", "staff name": "staff_name", "packed by": "staff_name", "operator": "staff_name", "for order": "for_order", "order": "for_order", "for stock": "for_stock", "stock": "for_stock", "job number": "job_number", "job": "job_number", "job no": "job_number", "order number": "job_number", "stock quantity": "stock_quantity", "stock qty": "stock_quantity", "sample box no": "sample_box_no", "sample box": "sample_box_no", "scales checked": "scales_checked", "scales": "scales_checked", "label correct": "label_correct", "label": "label_correct", "bag sealed": "bag_sealed", "sealed": "bag_sealed", "pallet good condition": "pallet_good_condition", "pallet": "pallet_good_condition", "notes": "notes", "note": "notes", "comment": "notes", "comments": "notes", } # How many row-level errors we collect before truncating, to keep the response # (and the toast) sane on a badly-formed file. _MAX_REPORTED_ERRORS = 50 def _normalise_header(raw: object) -> str | None: if raw is None: return None key = " ".join(str(raw).strip().lower().replace("-", " ").replace("_", " ").split()) if not key: return None if key in _HEADER_ALIASES: return _HEADER_ALIASES[key] # Test weights: "test weight 1".."test weight 5" (and "tw1" style). for n in range(1, 6): if key in {f"test weight {n}", f"tw{n}", f"test {n}"}: return f"test_weight_{n}" return None def _coerce_quantity_type(value: object) -> str | None: if value is None: return None text = str(value).strip().lower() if not text: return None if text in {"bag", "bags", "b"}: return "bags" if text in {"kg", "kgs", "kilogram", "kilograms", "bulka", "bulk"}: return "kg" return None def _read_tabular_file(filename: str, content: bytes) -> tuple[list[str | None], list[tuple]]: """Return (headers, data_rows). Detects CSV vs .xlsx by extension/content.""" lowered = (filename or "").lower() is_excel = lowered.endswith((".xlsx", ".xlsm", ".xls")) if is_excel: workbook = load_workbook(io.BytesIO(content), data_only=True, read_only=True) ws = workbook.active rows = [tuple(r) for r in ws.iter_rows(values_only=True)] workbook.close() else: text = None for encoding in ("utf-8-sig", "utf-8", "latin-1"): try: text = content.decode(encoding) break except UnicodeDecodeError: continue if text is None: raise ValueError("Could not decode the file as text. Save it as UTF-8 CSV or .xlsx.") # Sniff the delimiter (comma/semicolon/tab) but fall back to comma. sample = text[:4096] try: dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") except csv.Error: dialect = csv.excel rows = [tuple(r) for r in csv.reader(io.StringIO(text), dialect)] # Find the first row that has at least one recognised header; treat it as # the header row and everything after as data. for index, row in enumerate(rows): if any(_normalise_header(cell) is not None for cell in row): return list(row), rows[index + 1 :] return [], [] def import_entries_from_file( db: Session, *, filename: str, content: bytes, tenant_id: str, created_by: str | None, ) -> dict: """Parse an uploaded CSV/spreadsheet and persist each row as a throughput entry. Products are matched by item_id then name, and auto-created when not found so every entry stays linked. Returns a summary with row-level errors. """ headers, data_rows = _read_tabular_file(filename, content) if not headers: raise ValueError( "No recognised columns found. The file needs a header row with at " "least Date, Product and Quantity columns." ) # Map canonical field name → column index. First occurrence wins. field_index: dict[str, int] = {} for col, raw in enumerate(headers): field = _normalise_header(raw) if field and field not in field_index: field_index[field] = col for required in ("date", "product", "quantity"): if required not in field_index: raise ValueError( f"Missing required '{required}' column. Required columns are " "Date, Product and Quantity." ) def cell(row: tuple, field: str) -> object: idx = field_index.get(field) if idx is None or idx >= len(row): return None return row[idx] # Index existing products for matching (by item_id and by lower-cased name). by_item: dict[str, ThroughputProduct] = {} by_name: dict[str, ThroughputProduct] = {} for product in db.scalars( select(ThroughputProduct).where(ThroughputProduct.tenant_id == tenant_id) ).all(): if product.item_id: by_item[str(product.item_id)] = product by_name[product.name.lower()] = product imported = 0 skipped = 0 products_created = 0 errors: list[str] = [] def note_error(message: str) -> None: if len(errors) < _MAX_REPORTED_ERRORS: errors.append(message) for offset, row in enumerate(data_rows): # Sheet/file row number for human-friendly error messages (header = 1). line_no = offset + 2 if not row or all(value is None or str(value).strip() == "" for value in row): continue production_date = _coerce_date(cell(row, "date")) product_name = _coerce_text(cell(row, "product")) quantity = _coerce_float(cell(row, "quantity")) if production_date is None: skipped += 1 note_error(f"Row {line_no}: missing or invalid date.") continue if not product_name: skipped += 1 note_error(f"Row {line_no}: missing product name.") continue if quantity is None or quantity < 0: skipped += 1 note_error(f"Row {line_no}: missing or invalid quantity.") continue bag_size = _coerce_float(cell(row, "bag_size")) quantity_type = _coerce_quantity_type(cell(row, "quantity_type")) if quantity_type is None: # Infer: bulka-style rows have a blank or very large bag size. if bag_size is None or bag_size >= _BULKA_BAG_SIZE_THRESHOLD or "bulka" in product_name.lower(): quantity_type = "kg" else: quantity_type = "bags" if quantity_type == "bags" and (bag_size is None or bag_size <= 0): skipped += 1 note_error(f"Row {line_no}: bag size is required when packed as bags.") continue item_id_raw = cell(row, "item_id") item_id = None if item_id_raw is not None: if isinstance(item_id_raw, float) and item_id_raw.is_integer(): item_id = str(int(item_id_raw)) else: item_id = _coerce_text(item_id_raw) product = (by_item.get(item_id) if item_id else None) or by_name.get(product_name.lower()) if product is None: product = ThroughputProduct( tenant_id=tenant_id, item_id=item_id, name=product_name, default_bag_size=bag_size, is_bulka_default=_infer_bulka_default(product_name, bag_size), active=True, notes="Auto-created during throughput import", ) db.add(product) db.flush() products_created += 1 if item_id: by_item[item_id] = product by_name[product_name.lower()] = product for_order = _coerce_bool(cell(row, "for_order")) if field_index.get("for_order") is not None else False for_stock = _coerce_bool(cell(row, "for_stock")) if field_index.get("for_stock") is not None else False stock_quantity = _coerce_float(cell(row, "stock_quantity")) if for_stock else None calculated = calculate_kg(quantity, quantity_type, bag_size) entry = ProductionThroughput( tenant_id=tenant_id, production_date=production_date, product_id=product.id, product_name_snapshot=product_name, bag_size=bag_size, scales_checked=_coerce_bool(cell(row, "scales_checked")), label_correct=_coerce_bool(cell(row, "label_correct")), bag_sealed=_coerce_bool(cell(row, "bag_sealed")), pallet_good_condition=_coerce_bool(cell(row, "pallet_good_condition")), for_order=for_order, for_stock=for_stock, job_number=_coerce_text(cell(row, "job_number")) if for_order else None, stock_quantity=stock_quantity, sample_box_no=_coerce_text(cell(row, "sample_box_no")), test_weight_1=_coerce_float(cell(row, "test_weight_1")), test_weight_2=_coerce_float(cell(row, "test_weight_2")), test_weight_3=_coerce_float(cell(row, "test_weight_3")), test_weight_4=_coerce_float(cell(row, "test_weight_4")), test_weight_5=_coerce_float(cell(row, "test_weight_5")), quantity=quantity, quantity_type=quantity_type, calculated_kg=calculated, staff_name=normalise_staff_name(cell(row, "staff_name")), notes=_coerce_text(cell(row, "notes")), created_by=created_by or "csv-import", ) db.add(entry) imported += 1 if imported == 0 and products_created == 0: # Nothing landed — don't leave a half-open transaction. db.rollback() else: db.commit() return { "entries_imported": imported, "entries_skipped": skipped, "products_created": products_created, "errors": errors, }