#!/usr/bin/env python3
"""Clean a product CSV before it goes to a menu platform or a bulk import. No AI in this file.

    python3 product-data-cleanup.py sample-products.csv --out clean.csv
    python3 product-data-cleanup.py sample-products.csv --out clean.csv --image-base https://cdn.example.com/products/
    python3 product-data-cleanup.py sample-products.csv --check-copy          # only the description check
    python3 product-data-cleanup.py --self-test

What it fixes (and logs, one line per change):
    sku          uppercase, trimmed, internal spaces removed; duplicates reported, never silently merged
    category     mapped to one spelling via CATEGORY_MAP (edit it)
    unit         g / mg / ml / ea from the twenty ways people type them
    size         pulled out of the name ("3.5g Jar" -> 3.5) when the size cell is blank
    price        currency symbols and commas stripped; two decimals
    image_url    image_file joined to --image-base, lowercased extension; missing files listed
    description  checked against the forbidden patterns in menu-copy-rules.md; hits are reported, text is NOT rewritten

Version 1.0, tested 2026-09-17, Python 3.11, standard library only. CC0 1.0.
"""
import argparse
import csv
import re
import sys
from collections import Counter
from pathlib import Path

# ---- edit these for your catalog and your state ---------------------------------------------
CATEGORY_MAP = {
    "flower": "Flower", "flowers": "Flower", "bud": "Flower",
    "pre-roll": "Pre-Roll", "preroll": "Pre-Roll", "prerolls": "Pre-Roll", "pre-rolls": "Pre-Roll", "joints": "Pre-Roll",
    "edible": "Edible", "edibles": "Edible", "gummies": "Edible",
    "vape": "Vape", "vapes": "Vape", "cartridge": "Vape", "carts": "Vape",
    "concentrate": "Concentrate", "concentrates": "Concentrate", "extract": "Concentrate",
    "tincture": "Tincture", "tinctures": "Tincture",
    "topical": "Topical", "topicals": "Topical",
}
UNIT_MAP = {
    "g": "g", "gram": "g", "grams": "g", "gr": "g", "gm": "g",
    "mg": "mg", "milligram": "mg", "milligrams": "mg",
    "ml": "ml", "milliliter": "ml", "millilitre": "ml",
    "ea": "ea", "each": "ea", "unit": "ea", "units": "ea", "pc": "ea", "pcs": "ea", "count": "ea", "ct": "ea",
}
# Same patterns as the table in menu-copy-rules.md. A hit is a stop, not a suggestion.
COPY_RULES = {
    "condition": r"\b(pain|anxiety|anxious|insomnia|stress|nausea|inflammation|inflammatory|ptsd|arthritis|depress|seizure|epilep|appetite|migraine|cancer|tumou?r|glaucoma|adhd|chronic)\w*",
    "claims_help": r"\b(help(s|ing)? (you |with )?|reliev|treat(s|ing|ment)?\b|cur(e|es|ing)\b|heal|sooth|eas(e|es|ing) (your|the)|alleviat|remed|medicin|therapeut)",
    "promised_effect": r"\b(you('ll| will) feel|guarantee|melts? (away|stress)|knock(s|ed)? you|energi[sz]e(s|d)? you|calm(s|ed)? you|makes? you)",
    "sleep_wellness": r"\b(sleep aid|help(s)? you sleep|fall asleep|wellness|healthy|health)\b",
    "superlative_safety": r"\b(safe(ly|st)?|non-?addictive|no side effects|strongest|most potent)\b",
    "minors": r"\b(candy|cartoon|kids?|children|toy|school)\b",
    "dosing": r"\b(take (one|two|three|\d+) (for|to)|double (the )?dose|dose up)\b",
}
# ----------------------------------------------------------------------------------------------

SIZE_IN_NAME = re.compile(r"(\d+(?:\.\d+)?)\s*(mg|ml|g)\b", re.I)


def clean_sku(v, log, row_no):
    new = re.sub(r"\s+", "", v).upper()
    if new != v:
        log.append(f"row {row_no}: sku '{v}' -> '{new}'")
    return new


def clean_category(v, log, row_no):
    new = CATEGORY_MAP.get(v.strip().lower(), v.strip())
    if new != v:
        log.append(f"row {row_no}: category '{v}' -> '{new}'")
    return new


def clean_unit(v, log, row_no):
    key = v.strip().lower().rstrip(".")
    if not key:
        return ""
    new = UNIT_MAP.get(key)
    if new is None:
        log.append(f"row {row_no}: unit '{v}' not recognised; left as is. Add it to UNIT_MAP")
        return v.strip()
    if new != v:
        log.append(f"row {row_no}: unit '{v}' -> '{new}'")
    return new


def clean_size(size, unit, name, log, row_no):
    s = size.strip()
    m = re.fullmatch(r"(\d+(?:\.\d+)?)\s*(mg|ml|g)?", s, re.I)
    if m:
        if m.group(2) and not unit:
            unit = m.group(2).lower()
            log.append(f"row {row_no}: unit '{unit}' taken from size cell '{s}'")
        if m.group(1) != s:
            log.append(f"row {row_no}: size '{s}' -> '{m.group(1)}'")
        return m.group(1), unit
    if not s:
        m = SIZE_IN_NAME.search(name)
        if m:
            log.append(f"row {row_no}: size blank; '{m.group(1)}{m.group(2).lower()}' taken from name")
            return m.group(1), unit or m.group(2).lower()
        log.append(f"row {row_no}: size blank and not in name; fill by hand")
        return "", unit
    log.append(f"row {row_no}: size '{s}' not understood; left as is")
    return s, unit


def clean_price(v, log, row_no):
    s = v.strip()
    if not s:
        log.append(f"row {row_no}: price blank")
        return ""
    num = re.sub(r"[^\d.\-]", "", s)
    try:
        new = f"{float(num):.2f}"
    except ValueError:
        log.append(f"row {row_no}: price '{s}' not a number; left as is")
        return s
    if new != s:
        log.append(f"row {row_no}: price '{s}' -> '{new}'")
    return new


def image_url(fname, base, log, row_no, existing):
    f = fname.strip()
    if not f:
        log.append(f"row {row_no}: no image file")
        return ""
    stem, dot, ext = f.rpartition(".")
    f2 = f"{stem}{dot}{ext.lower()}" if dot else f
    if existing is not None and f2 not in existing and f not in existing:
        log.append(f"row {row_no}: image '{f}' not found in --image-dir")
    if f2 != f:
        log.append(f"row {row_no}: image '{f}' -> '{f2}'")
    return (base.rstrip("/") + "/" + f2) if base else f2


def check_copy(text, row_no, hits):
    found = []
    for rule, pat in COPY_RULES.items():
        m = re.search(pat, text, re.I)
        if m:
            found.append((rule, m.group(0)))
    if found:
        hits.append((row_no, found))
    return found


def clean(rows, image_base="", image_dir=None):
    log, hits = [], []
    existing = {p.name for p in Path(image_dir).iterdir()} if image_dir else None
    out = []
    for i, r in enumerate(rows, start=2):  # row 1 is the header
        r = {k.strip().lower(): (v or "") for k, v in r.items()}
        sku = clean_sku(r.get("sku", ""), log, i)
        unit = clean_unit(r.get("unit", ""), log, i)
        size, unit = clean_size(r.get("size", ""), unit, r.get("name", ""), log, i)
        row = {
            "sku": sku,
            "name": re.sub(r"\s+", " ", r.get("name", "")).strip(),
            "category": clean_category(r.get("category", ""), log, i),
            "size": size,
            "unit": unit,
            "price": clean_price(r.get("price", ""), log, i),
            "image_url": image_url(r.get("image_file", r.get("image_url", "")), image_base, log, i, existing),
            "description": r.get("description", "").strip(),
            "copy_flags": "",
        }
        found = check_copy(row["description"], i, hits)
        row["copy_flags"] = "; ".join(f"{rule}: '{word}'" for rule, word in found)
        if not row["description"]:
            log.append(f"row {i}: description blank")
        out.append(row)
    dupes = [s for s, n in Counter(r["sku"] for r in out).items() if n > 1 and s]
    for d in dupes:
        log.append(f"DUPLICATE sku {d} appears {sum(1 for r in out if r['sku'] == d)} times; fix by hand before import")
    return out, log, hits, dupes


def self_test():
    here = Path(__file__).resolve().parent
    with open(here / "sample-products.csv", newline="", encoding="utf-8-sig") as f:
        rows = list(csv.DictReader(f))
    out, log, hits, dupes = clean(rows, image_base="https://cdn.example.com/products/")
    by = {r["sku"]: r for r in out}
    assert by["FL-BD-35"]["unit"] == "g" and by["FL-BD-35"]["price"] == "35.00"
    assert by["FL-BD-14"]["size"] == "14" and by["FL-BD-14"]["unit"] == "g" and by["FL-BD-14"]["category"] == "Flower"
    assert by["FL-BD-14"]["image_url"].endswith("blue_dream_14.jpg")
    assert by["FL-G41-35"]["size"] == "3.5" and by["FL-G41-35"]["unit"] == "g"
    assert by["PR-G41-1"]["unit"] == "g" and by["PR-G41-1"]["image_url"] == ""
    assert by["ED-CB-100"]["category"] == "Edible"
    assert by["FL-SD-35"]["price"] == "32.00"
    assert dupes == ["VP-LR-PAP-1"], dupes
    flagged = {r["sku"] for r in out if r["copy_flags"]}
    assert flagged == {"FL-G41-35", "ED-MM-100", "ED-CB-100", "VP-DS-WC-1", "TN-SH-21-30"}, flagged
    assert "condition" in by["FL-G41-35"]["copy_flags"]          # "stress"
    assert "promised_effect" in by["ED-MM-100"]["copy_flags"]    # "melts stress"
    assert "minors" in by["ED-CB-100"]["copy_flags"]             # "kid", "candy"
    assert "superlative_safety" in by["VP-DS-WC-1"]["copy_flags"]
    assert "claims_help" in by["TN-SH-21-30"]["copy_flags"]      # "treats"
    clean_rows = {"FL-BD-35", "FL-BD-14", "PR-G41-1", "VP-LR-PAP-1", "CN-RS-SB-1"}
    assert not any(by[s]["copy_flags"] for s in clean_rows), [(s, by[s]["copy_flags"]) for s in clean_rows if by[s]["copy_flags"]]
    print(f"self-test passed: {len(log)} changes logged, {len(hits)} descriptions flagged, 1 duplicate sku caught, clean copy left alone")


def main():
    ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    ap.add_argument("csv", nargs="?")
    ap.add_argument("--out", help="write the cleaned CSV here")
    ap.add_argument("--image-base", default="", help="URL prefix your image files will be hosted under")
    ap.add_argument("--image-dir", help="local folder of image files, to report missing ones")
    ap.add_argument("--check-copy", action="store_true", help="only report description rule hits")
    ap.add_argument("--self-test", action="store_true")
    a = ap.parse_args()
    if a.self_test:
        return self_test()
    if not a.csv:
        ap.error("give a CSV, or --self-test")
    with open(a.csv, newline="", encoding="utf-8-sig") as f:
        rows = list(csv.DictReader(f))
    out, log, hits, dupes = clean(rows, a.image_base, a.image_dir)
    if a.check_copy:
        for row_no, found in hits:
            print(f"row {row_no}: " + "; ".join(f"{rule} -> '{word}'" for rule, word in found))
        print(f"\n{len(hits)} of {len(out)} descriptions hit a rule. Each one is a stop: rewrite or drop.")
        return
    for line in log:
        print(line)
    print(f"\n{len(out)} rows, {len(log)} changes, {len(hits)} descriptions flagged, {len(dupes)} duplicate sku(s)")
    if a.out:
        with open(a.out, "w", newline="", encoding="utf-8") as f:
            w = csv.DictWriter(f, fieldnames=list(out[0].keys()))
            w.writeheader()
            w.writerows(out)
        print(f"wrote {a.out}. Rows with copy_flags are not ready for a menu.")
    if dupes:
        sys.exit(2)


if __name__ == "__main__":
    main()
