Files
data-entry-app/backend/app/services/throughput_service.py
T

745 lines
26 KiB
Python
Raw Normal View History

2026-05-31 20:19:44 +12:00
from __future__ import annotations
import csv
import io
2026-05-31 20:19:44 +12:00
import logging
2026-06-02 15:41:53 +12:00
import os
2026-06-17 21:55:04 +12:00
import re
2026-05-31 20:19:44 +12:00
from datetime import date, datetime
from pathlib import Path
from typing import Iterable
from openpyxl import load_workbook
from sqlalchemy import select
from sqlalchemy.orm import Session
from app.models.throughput import ProductionThroughput, ThroughputProduct
logger = logging.getLogger("data_entry_app.throughput")
PRODUCTION_SHEET = "Production"
NAMES_SHEET = "Names"
2026-06-02 15:41:53 +12:00
# The historical throughput export. Bundled into the image under input_data/ so
# the seed can import it on a fresh deployment (e.g. a new Postgres volume).
WORKBOOK_FILENAME = "Operations Throughput.xlsx"
2026-05-31 20:19:44 +12:00
# Anything at or above this kg/bag is treated as a bulka batch, not a per-bag count.
_BULKA_BAG_SIZE_THRESHOLD = 100.0
def normalise_staff_name(value: object) -> str | None:
if value is None:
return None
text = str(value).strip()
if not text:
return None
# Collapse internal whitespace, title-case for consistency.
cleaned = " ".join(text.split())
return cleaned
def calculate_kg(quantity: float | None, quantity_type: str, bag_size: float | None) -> float:
if quantity is None:
return 0.0
if quantity_type == "kg":
return float(quantity)
if bag_size is None:
return 0.0
return float(quantity) * float(bag_size)
def qa_passed(entry: ProductionThroughput) -> bool:
return bool(entry.scales_checked and entry.label_correct and entry.bag_sealed and entry.pallet_good_condition)
def serialize_entry(entry: ProductionThroughput) -> dict:
return {
"id": entry.id,
"tenant_id": entry.tenant_id,
"production_date": entry.production_date,
"product_id": entry.product_id,
"product_name_snapshot": entry.product_name_snapshot,
"bag_size": entry.bag_size,
"scales_checked": entry.scales_checked,
"label_correct": entry.label_correct,
"bag_sealed": entry.bag_sealed,
"pallet_good_condition": entry.pallet_good_condition,
2026-06-02 15:41:53 +12:00
"for_order": entry.for_order,
"for_stock": entry.for_stock,
"job_number": entry.job_number,
"stock_quantity": entry.stock_quantity,
2026-05-31 20:19:44 +12:00
"sample_box_no": entry.sample_box_no,
"test_weight_1": entry.test_weight_1,
"test_weight_2": entry.test_weight_2,
"test_weight_3": entry.test_weight_3,
"test_weight_4": entry.test_weight_4,
"test_weight_5": entry.test_weight_5,
"quantity": entry.quantity,
"quantity_type": entry.quantity_type,
"calculated_kg": entry.calculated_kg,
"staff_name": entry.staff_name,
"notes": entry.notes,
"qa_passed": qa_passed(entry),
"created_by": entry.created_by,
"created_at": entry.created_at,
"updated_at": entry.updated_at,
}
def _coerce_bool(value: object) -> bool:
if isinstance(value, bool):
return value
if value is None:
return True
if isinstance(value, (int, float)):
return bool(value)
text = str(value).strip().lower()
if text in {"yes", "y", "true", "1", "pass", "ok", "x", "checked"}:
return True
if text in {"no", "n", "false", "0", "fail"}:
return False
return True
2026-06-15 10:13:02 +12:00
def _coerce_import_bool(value: object, *, default: bool = False) -> bool:
"""Conservative boolean parsing for ad-hoc imports.
Uploaded CSV/XLSX rows often leave destination columns blank, or use text
like "stock" / "order" elsewhere in the row. Those should not silently
become True. Only explicit truthy markers opt in.
"""
if isinstance(value, bool):
return value
if value is None:
return default
if isinstance(value, (int, float)):
return bool(value)
text = str(value).strip().lower()
if not text:
return default
if text in {"yes", "y", "true", "1", "pass", "ok", "x", "checked"}:
return True
if text in {"no", "n", "false", "0", "fail"}:
return False
return default
2026-05-31 20:19:44 +12:00
def _coerce_float(value: object) -> float | None:
if value is None or value == "":
return None
if isinstance(value, bool):
return float(value)
if isinstance(value, (int, float)):
return float(value)
text = str(value).strip().replace(",", "")
if not text:
return None
try:
return float(text)
except ValueError:
return None
def _coerce_text(value: object) -> str | None:
if value is None:
return None
text = str(value).strip()
if not text or text.lower() in {"#value!", "#n/a", "n/a"}:
return None
return text
2026-06-17 21:55:04 +12:00
# Default slash-date preference. The app is Australian, so an ambiguous
# "x/y/z" is read day-first unless a column is detected as month-first.
_DAY_FIRST_FORMATS = ("%Y-%m-%d", "%d/%m/%Y", "%m/%d/%Y")
_MONTH_FIRST_FORMATS = ("%Y-%m-%d", "%m/%d/%Y", "%d/%m/%Y")
_SLASH_DATE_RE = re.compile(r"^\s*(\d{1,2})[/-](\d{1,2})[/-](\d{2,4})\s*$")
def _coerce_date(
value: object, formats: tuple[str, ...] = _DAY_FIRST_FORMATS
) -> date | None:
2026-05-31 20:19:44 +12:00
if value is None:
return None
if isinstance(value, datetime):
return value.date()
if isinstance(value, date):
return value
text = str(value).strip()
if not text:
return None
2026-06-17 21:55:04 +12:00
for fmt in formats:
2026-05-31 20:19:44 +12:00
try:
return datetime.strptime(text, fmt).date()
except ValueError:
continue
return None
2026-06-17 21:55:04 +12:00
def _detect_slash_date_formats(values: Iterable[object]) -> tuple[str, ...]:
"""Inspect every slash/dash date in a column and decide whether the file is
day-first (D/M/Y) or month-first (M/D/Y), so all rows parse consistently.
A first component > 12 proves day-first; a second component > 12 proves
month-first. If only month-first evidence exists we switch to M/D/Y;
otherwise we keep the Australian day-first default.
"""
day_first = False
month_first = False
for value in values:
if value is None or isinstance(value, (datetime, date)):
continue
match = _SLASH_DATE_RE.match(str(value))
if not match:
continue
first, second = int(match.group(1)), int(match.group(2))
if first > 12:
day_first = True
elif second > 12:
month_first = True
if month_first and not day_first:
return _MONTH_FIRST_FORMATS
return _DAY_FIRST_FORMATS
2026-05-31 20:19:44 +12:00
def _infer_bulka_default(name: str, bag_size: float | None) -> bool:
lowered = name.lower()
if "bulka" in lowered:
return True
if bag_size is None:
return False
return bag_size >= _BULKA_BAG_SIZE_THRESHOLD
def import_names_sheet(db: Session, workbook, tenant_id: str) -> tuple[int, int]:
"""Upsert product master from the Names sheet. Returns (created, updated)."""
if NAMES_SHEET not in workbook.sheetnames:
return (0, 0)
ws = workbook[NAMES_SHEET]
existing: dict[tuple[str, str | None], ThroughputProduct] = {}
by_item: dict[str, ThroughputProduct] = {}
by_name: dict[str, ThroughputProduct] = {}
for product in db.scalars(
select(ThroughputProduct).where(ThroughputProduct.tenant_id == tenant_id)
).all():
if product.item_id:
by_item[str(product.item_id)] = product
by_name[product.name.lower()] = product
created = 0
updated = 0
for row in ws.iter_rows(min_row=2, values_only=True):
if not row:
continue
name = _coerce_text(row[0] if len(row) > 0 else None)
if not name:
continue
item_id_raw = row[1] if len(row) > 1 else None
item_id = None
if item_id_raw is not None:
if isinstance(item_id_raw, float) and item_id_raw.is_integer():
item_id = str(int(item_id_raw))
else:
item_id = _coerce_text(item_id_raw)
product = (by_item.get(item_id) if item_id else None) or by_name.get(name.lower())
if product is None:
product = ThroughputProduct(
tenant_id=tenant_id,
item_id=item_id,
name=name,
default_bag_size=None,
is_bulka_default="bulka" in name.lower(),
active=True,
notes="Imported from Operations Throughput.xlsx",
)
db.add(product)
created += 1
if item_id:
by_item[item_id] = product
by_name[name.lower()] = product
else:
if item_id and not product.item_id:
product.item_id = item_id
if name and product.name != name:
product.name = name
updated += 1
db.flush()
return (created, updated)
def import_production_sheet(db: Session, workbook, tenant_id: str) -> tuple[int, int]:
"""Import the Production sheet. Returns (imported, skipped)."""
if PRODUCTION_SHEET not in workbook.sheetnames:
return (0, 0)
ws = workbook[PRODUCTION_SHEET]
# Header row is row 3 in the sheet (rows 1 and 2 are display banners).
products_by_name: dict[str, ThroughputProduct] = {
product.name.lower(): product
for product in db.scalars(
select(ThroughputProduct).where(ThroughputProduct.tenant_id == tenant_id)
).all()
}
bag_size_seen: dict[int, list[float]] = {}
imported = 0
skipped = 0
for row in ws.iter_rows(min_row=4, values_only=True):
if not row or len(row) < 15:
skipped += 1
continue
production_date = _coerce_date(row[0])
product_name = _coerce_text(row[1])
if production_date is None or not product_name:
skipped += 1
continue
bag_size = _coerce_float(row[2])
scales = _coerce_bool(row[3])
label = _coerce_bool(row[4])
sealed = _coerce_bool(row[5])
pallet = _coerce_bool(row[6])
sample_box = _coerce_text(row[7])
tw1 = _coerce_float(row[8])
tw2 = _coerce_float(row[9])
tw3 = _coerce_float(row[10])
tw4 = _coerce_float(row[11])
tw5 = _coerce_float(row[12])
quantity = _coerce_float(row[13]) or 0.0
staff = normalise_staff_name(row[14])
notes = _coerce_text(row[15]) if len(row) > 15 else None
# Infer quantity_type: bulka-style rows have a blank or very large bag size.
if bag_size is None or bag_size >= _BULKA_BAG_SIZE_THRESHOLD or "bulka" in product_name.lower():
quantity_type = "kg"
else:
quantity_type = "bags"
product = products_by_name.get(product_name.lower())
if product is None:
product = ThroughputProduct(
tenant_id=tenant_id,
item_id=None,
name=product_name,
default_bag_size=bag_size,
is_bulka_default=_infer_bulka_default(product_name, bag_size),
active=True,
notes="Auto-created during Operations Throughput import",
)
db.add(product)
db.flush()
products_by_name[product_name.lower()] = product
if product.id is not None and bag_size is not None and bag_size > 0:
bag_size_seen.setdefault(product.id, []).append(bag_size)
calculated = calculate_kg(quantity, quantity_type, bag_size)
entry = ProductionThroughput(
tenant_id=tenant_id,
production_date=production_date,
product_id=product.id,
product_name_snapshot=product_name,
bag_size=bag_size,
scales_checked=scales,
label_correct=label,
bag_sealed=sealed,
pallet_good_condition=pallet,
sample_box_no=sample_box,
test_weight_1=tw1,
test_weight_2=tw2,
test_weight_3=tw3,
test_weight_4=tw4,
test_weight_5=tw5,
quantity=quantity,
quantity_type=quantity_type,
calculated_kg=calculated,
staff_name=staff,
notes=notes,
created_by="workbook-import",
)
db.add(entry)
imported += 1
# Backfill default_bag_size on products that don't have one but appear in entries.
for product_id, sizes in bag_size_seen.items():
product = db.get(ThroughputProduct, product_id)
if product and product.default_bag_size is None:
# Use the most common bag size seen.
common = max(set(sizes), key=sizes.count)
product.default_bag_size = common
if not product.is_bulka_default:
product.is_bulka_default = _infer_bulka_default(product.name, common)
db.flush()
return (imported, skipped)
def import_workbook(db: Session, workbook_path: Path, tenant_id: str) -> dict:
workbook = load_workbook(workbook_path, data_only=True)
products_created, products_updated = import_names_sheet(db, workbook, tenant_id)
entries_imported, entries_skipped = import_production_sheet(db, workbook, tenant_id)
return {
"products_created": products_created,
"products_updated": products_updated,
"entries_imported": entries_imported,
"entries_skipped": entries_skipped,
}
def workbook_candidates() -> Iterable[Path]:
repo_root = Path(__file__).resolve().parents[3]
2026-06-02 15:41:53 +12:00
cwd = Path.cwd()
env_value = os.getenv("THROUGHPUT_WORKBOOK_PATH")
env_path = Path(env_value.strip()) if isinstance(env_value, str) and env_value.strip() else None
# input_data/ is where the workbook is bundled in the image; in the
# container the working directory is /app, so cwd/input_data resolves it.
2026-05-31 20:19:44 +12:00
candidates = [
2026-06-02 15:41:53 +12:00
env_path,
repo_root / "input_data" / WORKBOOK_FILENAME,
cwd / "input_data" / WORKBOOK_FILENAME,
Path("/app") / "input_data" / WORKBOOK_FILENAME,
Path("/srv/lean101-clients") / "input_data" / WORKBOOK_FILENAME,
repo_root / WORKBOOK_FILENAME,
repo_root.parent / WORKBOOK_FILENAME,
cwd / WORKBOOK_FILENAME,
Path("/srv/lean101-clients") / WORKBOOK_FILENAME,
Path("/app") / WORKBOOK_FILENAME,
2026-05-31 20:19:44 +12:00
]
seen: set[str] = set()
ordered: list[Path] = []
for candidate in candidates:
2026-06-02 15:41:53 +12:00
if candidate is None:
continue
2026-05-31 20:19:44 +12:00
key = str(candidate)
if key in seen:
continue
seen.add(key)
ordered.append(candidate)
return ordered
def resolve_workbook_path() -> Path | None:
for candidate in workbook_candidates():
if candidate.exists():
return candidate
return None
# ── Ad-hoc CSV / spreadsheet upload import ──────────────────────────────────
# Lets an operator upload their own CSV or .xlsx of packing runs (from Settings
# → Import) and have every row saved as a throughput entry. Unlike the bundled
# workbook seed above, this is column-header driven so the file can be a simple
# hand-built sheet rather than the exact "Operations Throughput.xlsx" layout.
# Maps the column headers we accept (normalised: lower-cased, spaces/dashes →
# single spaces) onto the canonical field used internally. Several aliases per
# field so a human-built sheet "just works".
_HEADER_ALIASES: dict[str, str] = {
"date": "date",
"production date": "date",
"production_date": "date",
"product": "product",
"product name": "product",
"product_name": "product",
"product name snapshot": "product",
"name": "product",
"item id": "item_id",
"item_id": "item_id",
"itemid": "item_id",
"sku": "item_id",
"quantity": "quantity",
"qty": "quantity",
"packed": "quantity",
"quantity packed": "quantity",
"amount": "quantity",
"quantity type": "quantity_type",
"type": "quantity_type",
"unit": "quantity_type",
"packed as": "quantity_type",
"bag size": "bag_size",
"bag_size": "bag_size",
"kg per bag": "bag_size",
"kg/bag": "bag_size",
"bagsize": "bag_size",
"staff": "staff_name",
"staff name": "staff_name",
"packed by": "staff_name",
"operator": "staff_name",
"for order": "for_order",
"order": "for_order",
"for stock": "for_stock",
"stock": "for_stock",
"job number": "job_number",
"job": "job_number",
"job no": "job_number",
"order number": "job_number",
"stock quantity": "stock_quantity",
"stock qty": "stock_quantity",
"sample box no": "sample_box_no",
"sample box": "sample_box_no",
"scales checked": "scales_checked",
"scales": "scales_checked",
"label correct": "label_correct",
"label": "label_correct",
"bag sealed": "bag_sealed",
"sealed": "bag_sealed",
"pallet good condition": "pallet_good_condition",
"pallet": "pallet_good_condition",
"notes": "notes",
"note": "notes",
"comment": "notes",
"comments": "notes",
}
# How many row-level errors we collect before truncating, to keep the response
# (and the toast) sane on a badly-formed file.
_MAX_REPORTED_ERRORS = 50
def _normalise_header(raw: object) -> str | None:
if raw is None:
return None
key = " ".join(str(raw).strip().lower().replace("-", " ").replace("_", " ").split())
if not key:
return None
if key in _HEADER_ALIASES:
return _HEADER_ALIASES[key]
# Test weights: "test weight 1".."test weight 5" (and "tw1" style).
for n in range(1, 6):
if key in {f"test weight {n}", f"tw{n}", f"test {n}"}:
return f"test_weight_{n}"
return None
def _coerce_quantity_type(value: object) -> str | None:
if value is None:
return None
text = str(value).strip().lower()
if not text:
return None
if text in {"bag", "bags", "b"}:
return "bags"
if text in {"kg", "kgs", "kilogram", "kilograms", "bulka", "bulk"}:
return "kg"
return None
def _read_tabular_file(filename: str, content: bytes) -> tuple[list[str | None], list[tuple]]:
"""Return (headers, data_rows). Detects CSV vs .xlsx by extension/content."""
lowered = (filename or "").lower()
is_excel = lowered.endswith((".xlsx", ".xlsm", ".xls"))
if is_excel:
workbook = load_workbook(io.BytesIO(content), data_only=True, read_only=True)
ws = workbook.active
rows = [tuple(r) for r in ws.iter_rows(values_only=True)]
workbook.close()
else:
text = None
for encoding in ("utf-8-sig", "utf-8", "latin-1"):
try:
text = content.decode(encoding)
break
except UnicodeDecodeError:
continue
if text is None:
raise ValueError("Could not decode the file as text. Save it as UTF-8 CSV or .xlsx.")
# Sniff the delimiter (comma/semicolon/tab) but fall back to comma.
sample = text[:4096]
try:
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
except csv.Error:
dialect = csv.excel
rows = [tuple(r) for r in csv.reader(io.StringIO(text), dialect)]
# Find the first row that has at least one recognised header; treat it as
# the header row and everything after as data.
for index, row in enumerate(rows):
if any(_normalise_header(cell) is not None for cell in row):
return list(row), rows[index + 1 :]
return [], []
def import_entries_from_file(
db: Session,
*,
filename: str,
content: bytes,
tenant_id: str,
created_by: str | None,
) -> dict:
"""Parse an uploaded CSV/spreadsheet and persist each row as a throughput
entry. Products are matched by item_id then name, and auto-created when not
found so every entry stays linked. Returns a summary with row-level errors.
"""
headers, data_rows = _read_tabular_file(filename, content)
if not headers:
raise ValueError(
"No recognised columns found. The file needs a header row with at "
"least Date, Product and Quantity columns."
)
# Map canonical field name → column index. First occurrence wins.
field_index: dict[str, int] = {}
for col, raw in enumerate(headers):
field = _normalise_header(raw)
if field and field not in field_index:
field_index[field] = col
for required in ("date", "product", "quantity"):
if required not in field_index:
raise ValueError(
f"Missing required '{required}' column. Required columns are "
"Date, Product and Quantity."
)
def cell(row: tuple, field: str) -> object:
idx = field_index.get(field)
if idx is None or idx >= len(row):
return None
return row[idx]
2026-06-17 21:55:04 +12:00
# Decide the slash-date order once for the whole file so ambiguous values
# like "12/9/2025" follow the same convention as the unambiguous ones.
date_formats = _detect_slash_date_formats(cell(row, "date") for row in data_rows)
# Index existing products for matching (by item_id and by lower-cased name).
by_item: dict[str, ThroughputProduct] = {}
by_name: dict[str, ThroughputProduct] = {}
for product in db.scalars(
select(ThroughputProduct).where(ThroughputProduct.tenant_id == tenant_id)
).all():
if product.item_id:
by_item[str(product.item_id)] = product
by_name[product.name.lower()] = product
imported = 0
skipped = 0
products_created = 0
errors: list[str] = []
def note_error(message: str) -> None:
if len(errors) < _MAX_REPORTED_ERRORS:
errors.append(message)
for offset, row in enumerate(data_rows):
# Sheet/file row number for human-friendly error messages (header = 1).
line_no = offset + 2
if not row or all(value is None or str(value).strip() == "" for value in row):
continue
2026-06-17 21:55:04 +12:00
production_date = _coerce_date(cell(row, "date"), date_formats)
product_name = _coerce_text(cell(row, "product"))
quantity = _coerce_float(cell(row, "quantity"))
if production_date is None:
skipped += 1
note_error(f"Row {line_no}: missing or invalid date.")
continue
if not product_name:
skipped += 1
note_error(f"Row {line_no}: missing product name.")
continue
if quantity is None or quantity < 0:
skipped += 1
note_error(f"Row {line_no}: missing or invalid quantity.")
continue
bag_size = _coerce_float(cell(row, "bag_size"))
quantity_type = _coerce_quantity_type(cell(row, "quantity_type"))
if quantity_type is None:
# Infer: bulka-style rows have a blank or very large bag size.
if bag_size is None or bag_size >= _BULKA_BAG_SIZE_THRESHOLD or "bulka" in product_name.lower():
quantity_type = "kg"
else:
quantity_type = "bags"
if quantity_type == "bags" and (bag_size is None or bag_size <= 0):
skipped += 1
note_error(f"Row {line_no}: bag size is required when packed as bags.")
continue
item_id_raw = cell(row, "item_id")
item_id = None
if item_id_raw is not None:
if isinstance(item_id_raw, float) and item_id_raw.is_integer():
item_id = str(int(item_id_raw))
else:
item_id = _coerce_text(item_id_raw)
product = (by_item.get(item_id) if item_id else None) or by_name.get(product_name.lower())
if product is None:
product = ThroughputProduct(
tenant_id=tenant_id,
item_id=item_id,
name=product_name,
default_bag_size=bag_size,
is_bulka_default=_infer_bulka_default(product_name, bag_size),
active=True,
notes="Auto-created during throughput import",
)
db.add(product)
db.flush()
products_created += 1
if item_id:
by_item[item_id] = product
by_name[product_name.lower()] = product
2026-06-15 10:13:02 +12:00
for_order = _coerce_import_bool(cell(row, "for_order")) if field_index.get("for_order") is not None else False
for_stock = _coerce_import_bool(cell(row, "for_stock")) if field_index.get("for_stock") is not None else False
stock_quantity = _coerce_float(cell(row, "stock_quantity")) if for_stock else None
calculated = calculate_kg(quantity, quantity_type, bag_size)
entry = ProductionThroughput(
tenant_id=tenant_id,
production_date=production_date,
product_id=product.id,
product_name_snapshot=product_name,
bag_size=bag_size,
scales_checked=_coerce_bool(cell(row, "scales_checked")),
label_correct=_coerce_bool(cell(row, "label_correct")),
bag_sealed=_coerce_bool(cell(row, "bag_sealed")),
pallet_good_condition=_coerce_bool(cell(row, "pallet_good_condition")),
for_order=for_order,
for_stock=for_stock,
job_number=_coerce_text(cell(row, "job_number")) if for_order else None,
stock_quantity=stock_quantity,
sample_box_no=_coerce_text(cell(row, "sample_box_no")),
test_weight_1=_coerce_float(cell(row, "test_weight_1")),
test_weight_2=_coerce_float(cell(row, "test_weight_2")),
test_weight_3=_coerce_float(cell(row, "test_weight_3")),
test_weight_4=_coerce_float(cell(row, "test_weight_4")),
test_weight_5=_coerce_float(cell(row, "test_weight_5")),
quantity=quantity,
quantity_type=quantity_type,
calculated_kg=calculated,
staff_name=normalise_staff_name(cell(row, "staff_name")),
notes=_coerce_text(cell(row, "notes")),
created_by=created_by or "csv-import",
)
db.add(entry)
imported += 1
if imported == 0 and products_created == 0:
# Nothing landed — don't leave a half-open transaction.
db.rollback()
else:
db.commit()
return {
"entries_imported": imported,
"entries_skipped": skipped,
"products_created": products_created,
"errors": errors,
}