#!/usr/bin/env python3
"""Match a buyer's messy order sheet against your product catalog. Draft only; a person approves.

    python3 normalize.py buyer-order.csv catalog.csv                  # prints the draft, grouped by store
    python3 normalize.py buyer-order.csv catalog.csv --out draft.csv  # also writes a CSV you can review in a sheet
    python3 normalize.py --self-test                                  # checks the matcher on the bundled files

buyer-order.csv columns: store, item, qty        (qty may be "2 cs", "1 case", "24", "12 ea")
catalog.csv columns:     sku, name, brand, category, size, unit, case_size

Every output line gets a status:
    match    confidence >= MATCH   -> fine to put on the draft
    review   REVIEW..MATCH         -> show the rep the top candidates, they pick
    unknown  < REVIEW              -> the buyer wants something you do not sell, or a new abbreviation; ask them

Nothing here calls a model or writes to any system. It is the deterministic half of order intake;
the lesson (04-03) shows where a model helps with the leftovers. Version 1.0, tested 2026-09-17,
Python 3.11, standard library only. CC0 1.0.
"""
import argparse
import csv
import re
import sys
from collections import defaultdict
from difflib import SequenceMatcher
from pathlib import Path

# ---- tune these on your own orders, then rerun ------------------------------------------
MATCH = 0.80
REVIEW = 0.50
# Buyer shorthand -> words in your catalog names. Add your customers' habits here.
ALIASES = {
    "bd": "blue dream",
    "g41": "gelato 41",
    "sd": "sour diesel",
    "sh": "sleepy hollow",
    "lr": "live resin",
    "disty": "distillate",
    "cart": "cart", "carts": "cart",
    "preroll": "pre-roll", "prerolls": "pre-roll", "pre-rolls": "pre-roll", "pr": "pre-roll",
    "eighth": "3.5g", "eighths": "3.5g",
    "1/2 oz": "14g", "half oz": "14g", "half ounce": "14g",
    "half gram": "0.5g", "1/2 gram": "0.5g", "1/2g": "0.5g",
    "5pk": "5-pack", "5 pk": "5-pack", "5 pack": "5-pack",
    "gummy": "gummies",
}
# -------------------------------------------------------------------------------------------

SIZE_RE = re.compile(r"(\d+(?:\.\d+)?)\s*(mg|ml|g)\b")
CASE_RE = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*(cs|case|cases|cse)\s*$", re.I)
EACH_RE = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*(ea|each|units?|pcs?)?\s*$", re.I)


def norm(text):
    t = text.lower().strip()
    # apply multi-word aliases first, then single tokens
    for k in sorted(ALIASES, key=len, reverse=True):
        t = re.sub(rf"(?<![a-z0-9]){re.escape(k)}(?![a-z0-9])", ALIASES[k], t)
    t = re.sub(r"(\d+(?:\.\d+)?)\s+(mg|ml|g)\b", r"\1\2", t)       # "3.5 g" -> "3.5g"
    t = re.sub(r"(?<=\d)\s*:\s*(?=\d)", ":", t)                    # "1 : 1" -> "1:1"
    t = re.sub(r"[^a-z0-9:.\- ]+", " ", t)
    return re.sub(r"\s+", " ", t).strip()


def sizes(text):
    return {(float(n), u) for n, u in SIZE_RE.findall(text)}


def bare_size(text):
    """'BD 3.5' has a number with no unit; return it so it can meet a catalog 3.5g."""
    m = re.search(r"(?<![\d.:])(\d+(?:\.\d+)?)(?![\d.]*\s*(?:mg|ml|g|pk|pack|:))\s*$", text)
    return float(m.group(1)) if m else None


def score(buyer, product):
    b, p = norm(buyer), norm(product["name"])
    bt, pt = tokens(b), tokens(p)
    overlap = len(bt & pt) / max(1, len(bt))                 # how much of what the buyer said is in the name
    seq = SequenceMatcher(None, b, p).ratio()
    s = 0.65 * overlap + 0.35 * seq
    bs, ps = sizes(b), sizes(p)
    if bs and ps:
        s += 0.10 if bs & ps else -0.35                     # a stated size that disagrees is a different product
    elif not bs and ps:
        bare = bare_size(b)
        if bare is not None:
            s += 0.05 if (bare, product["unit"]) in ps else -0.30
    return max(0.0, min(1.0, round(s, 3)))


def tokens(text):
    """Word set for overlap; '3.5g' becomes '3.5' and 'g' so a bare '3.5' can meet it."""
    return set(re.sub(r"(\d)(mg|ml|g)\b", r"\1 \2", text).split()) - {"g", "mg", "ml"}


def parse_qty(raw, case_size):
    raw = str(raw).strip()
    m = CASE_RE.match(raw)
    if m:
        n = float(m.group(1))
        return (n * case_size if case_size else None), f"{n:g} case(s) x {case_size or '?'}"
    m = EACH_RE.match(raw)
    if m:
        return float(m.group(1)), "each"
    return None, f"could not read quantity '{raw}'"


def load(path):
    with open(path, newline="", encoding="utf-8-sig") as f:
        return [{k.strip().lower(): (v or "").strip() for k, v in r.items()} for r in csv.DictReader(f)]


def normalize(order_rows, catalog):
    for p in catalog:
        p["case_size"] = int(p["case_size"]) if p.get("case_size") else None
    out = []
    for r in order_rows:
        ranked = sorted(((score(r["item"], p), p) for p in catalog), key=lambda t: -t[0])
        best, prod = ranked[0]
        # a close runner-up means the buyer's words fit two products; do not auto-match
        if len(ranked) > 1 and best - ranked[1][0] < 0.08 and best >= REVIEW:
            status = "review"
        else:
            status = "match" if best >= MATCH else "review" if best >= REVIEW else "unknown"
        qty, how = parse_qty(r["qty"], prod["case_size"] if status != "unknown" else None)
        if qty is None and status == "match":
            status = "review"
        out.append({
            "store": r["store"], "raw_item": r["item"], "raw_qty": r["qty"],
            "sku": prod["sku"] if status != "unknown" else "",
            "catalog_name": prod["name"] if status != "unknown" else "",
            "qty_units": f"{qty:g}" if qty is not None else "", "qty_note": how,
            "confidence": f"{best:.2f}", "status": status,
            "candidates": " | ".join(f"{p['sku']} {s:.2f}" for s, p in ranked[:3]),
        })
    return out


def print_draft(rows):
    by_store = defaultdict(list)
    for r in rows:
        by_store[r["store"]].append(r)
    for store, lines in by_store.items():
        print(f"\n== {store} ==")
        for r in lines:
            flag = {"match": " ", "review": "?", "unknown": "!"}[r["status"]]
            name = r["catalog_name"] or "(no catalog match)"
            print(f"{flag} {r['confidence']}  {name:<36} {r['qty_units'] or '?':>6} ea   <- '{r['raw_item']}' x '{r['raw_qty']}'  [{r['qty_note']}]")
    counts = defaultdict(int)
    for r in rows:
        counts[r["status"]] += 1
    print(f"\n{counts['match']} matched, {counts['review']} to review, {counts['unknown']} unknown. Nothing has been created anywhere.")


def self_test():
    here = Path(__file__).resolve().parent
    rows = normalize(load(here / "buyer-order.csv"), load(here / "catalog.csv"))
    got = {(r["store"].split(" - ")[1], r["raw_item"]): r for r in rows}

    def expect(store, item, sku, status, qty=None):
        r = got[(store, item)]
        assert r["status"] == status, (item, r["status"], r["confidence"], r["candidates"])
        if sku:
            assert r["sku"] == sku, (item, r["sku"], r["candidates"])
        if qty is not None:
            assert r["qty_units"] == qty, (item, r["qty_units"], r["qty_note"])

    expect("Downtown", "BD 3.5", "FL-BD-35", "match", "48")               # 2 cases x 24
    expect("Downtown", "Blue Dream 1/2 oz bags", "FL-BD-14", "match", "24")
    expect("Downtown", "gelato 41 eighths", "FL-G41-35", "match", "24")
    expect("Downtown", "Live resin papaya carts 1g", "VP-LR-PAP-1", "match", "50")
    expect("Eastside", "papaya LR cart half gram", "VP-LR-PAP-05", "match", "25")
    expect("Eastside", "Citrus gummies 100", "ED-CB-100", "match", "120")
    expect("Eastside", "SH tincture 2:1", "TN-SH-21-30", "match", "12")
    expect("Eastside", "Wedding cake disty 1g", "VP-DS-WC-1", "match", "20")
    expect("Northgate", "G41 prerolls", "PR-G41-1", "match", "100")
    expect("Northgate", "BD 5pk prerolls", "PR-BD-5PK", "match", "40")
    expect("Northgate", "Strawberry rosin", "CN-RS-SB-1", "match", "20")
    expect("Northgate", "Purple Punch 3.5", None, "unknown")               # not in catalog
    expect("Northgate", "gummies", None, "review")                         # three gummy SKUs; a person picks
    assert got[("Downtown", "midnight mint gummies")]["sku"] == "ED-MM-100"
    assert got[("Downtown", "midnight mint gummies")]["status"] in ("match", "review")
    print("self-test passed: 11 clean matches, 1 unknown, ambiguous rows held for review, case math right")


def main():
    ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    ap.add_argument("order", nargs="?")
    ap.add_argument("catalog", nargs="?")
    ap.add_argument("--out", help="write the draft as CSV here")
    ap.add_argument("--self-test", action="store_true")
    a = ap.parse_args()
    if a.self_test:
        return self_test()
    if not (a.order and a.catalog):
        ap.error("give buyer-order.csv and catalog.csv, or --self-test")
    rows = normalize(load(a.order), load(a.catalog))
    print_draft(rows)
    if a.out:
        with open(a.out, "w", newline="", encoding="utf-8") as f:
            w = csv.DictWriter(f, fieldnames=list(rows[0].keys()))
            w.writeheader()
            w.writerows(rows)
        print(f"wrote {a.out}")


if __name__ == "__main__":
    main()
