#!/usr/bin/env python3
"""csv-head.py: look at a CSV before you paste it into an AI.

Prints the header row, row count, column count, the first N rows, blank-cell
counts per column, columns that look like they hold personal data, and a rough
token estimate for the whole file. Offline. Standard library only. No network.

    python3 csv-head.py export.csv            # headers, 3 sample rows, counts, token estimate
    python3 csv-head.py export.csv -n 5       # 5 sample rows
    python3 csv-head.py export.csv --paste    # print a paste-ready block (header + N rows) and stop
    python3 csv-head.py --self-test           # runs the checks on a built-in sample

Token estimate: words cost about one token per 4-5 letters, but numbers cost a
token per ~3 digits and every comma, quote and 24-character Metrc tag costs
more. A package export is mostly tags and numbers, so it lands near 2
characters per token, not the 4 people quote for prose. The estimator counts
letters, digits and punctuation separately and prints a +/-15% range.
Calibrated 2026-09-17 against tiktoken (cl100k_base and o200k_base) on three
synthetic exports (tag-heavy, prose-heavy, mixed): see README for numbers.

From Distru's No Bullshit AI Course, lesson 03-02. CC0 1.0. Copy, edit, share.
"""
import argparse
import csv
import io
import re
import sys
from collections import Counter

PII_HINTS = re.compile(
    r"(customer|patient|first_?name|last_?name|full_?name|email|phone|mobile|"
    r"address|street|zip|postal|dob|birth|ssn|license_?number|driver|badge|"
    r"employee_?id|card|medical_?id|mmj)",
    re.I,
)
TAG_LIKE = re.compile(r"^1A[0-9A-F]{22}$", re.I)  # Metrc package/plant tag: 24 chars, 1A prefix


def analyze(text: str, sample_rows: int = 3):
    text = text.lstrip("﻿")  # Excel BOM
    dialect = csv.Sniffer().sniff(text[:4096], delimiters=",;\t|") if text.strip() else csv.excel
    reader = csv.reader(io.StringIO(text), dialect)
    rows = [r for r in reader if any(cell.strip() for cell in r)]
    if not rows:
        return {"error": "empty file"}
    header, data = rows[0], rows[1:]
    ncols = len(header)
    ragged = sum(1 for r in data if len(r) != ncols)
    blanks = Counter()
    for r in data:
        for i, cell in enumerate(r[:ncols]):
            if not cell.strip():
                blanks[header[i]] += 1
    pii = [h for h in header if PII_HINTS.search(h)]
    tag_cols = []
    for i, h in enumerate(header):
        vals = [r[i].strip() for r in data[:50] if i < len(r) and r[i].strip()]
        if vals and sum(1 for v in vals if TAG_LIKE.match(v)) / len(vals) > 0.8:
            tag_cols.append(h)
    chars = len(text)
    est = estimate_tokens(text)
    return {
        "delimiter": dialect.delimiter,
        "header": header,
        "ncols": ncols,
        "nrows": len(data),
        "ragged": ragged,
        "blanks": dict(blanks),
        "pii": pii,
        "tag_cols": tag_cols,
        "chars": chars,
        "tokens_low": int(est * 0.85),
        "tokens_high": int(est * 1.15),
        "sample": data[:sample_rows],
    }


def estimate_tokens(text: str) -> int:
    """Rough token count without a tokenizer. Letters ~5/token, digits ~2.6/token,
    each punctuation mark ~1, whitespace mostly merges into neighbours.
    Constants fitted 2026-09-17 against tiktoken; max error 7% on three CSV shapes."""
    n = 0.0
    for m in re.finditer(r"[A-Za-z]+|[0-9]+|[^A-Za-z0-9\s]|\s+", text):
        s = m.group()
        if s[0].isalpha():
            n += max(1, len(s) / 5)
        elif s[0].isdigit():
            n += max(1, len(s) / 2.6)
        elif s.isspace():
            n += 0.1 * len(s)
        else:
            n += 1
    return int(n)


def paste_block(text: str, n: int) -> str:
    text = text.lstrip("﻿")
    lines = [l for l in text.splitlines() if l.strip()]
    return "\n".join(lines[: n + 1])


def report(a: dict, path: str) -> str:
    if "error" in a:
        return f"{path}: {a['error']}"
    out = [f"file:        {path}"]
    out.append(f"delimiter:   {a['delimiter']!r}")
    out.append(f"columns:     {a['ncols']}")
    out.append(f"data rows:   {a['nrows']}" + (f"   ({a['ragged']} ragged rows: wrong column count, fix before pasting)" if a["ragged"] else ""))
    out.append(f"size:        {a['chars']:,} characters  ->  roughly {a['tokens_low']:,} to {a['tokens_high']:,} tokens")
    out.append("")
    out.append("header:")
    for h in a["header"]:
        flags = []
        if a["blanks"].get(h):
            flags.append(f"{a['blanks'][h]} blank")
        if h in a["pii"]:
            flags.append("LOOKS PERSONAL: redact or drop before pasting")
        if h in a["tag_cols"]:
            flags.append("Metrc tags: keep as text, never let a spreadsheet reformat")
        out.append(f"  {h}" + (f"   [{'; '.join(flags)}]" if flags else ""))
    out.append("")
    out.append(f"first {len(a['sample'])} rows:")
    for r in a["sample"]:
        out.append("  " + a["delimiter"].join(r))
    out.append("")
    fits = "fits in one chat message on any current model" if a["tokens_high"] < 60_000 else \
        "large: upload as a file, or ask for the formula/script instead of the answer"
    out.append(f"verdict:     {fits}")
    if a["nrows"] > 30:
        out.append("             more than ~30 rows: for anything numeric, ask for the formula or script, not the number")
    return "\n".join(out)


SAMPLE = """tag,product,quantity,unit,status,customer_name,last_activity
1A4FF0100000022000001234,Blue Dream 3.5g,120,ea,Active,Green Leaf Dispensary,2026-09-10
1A4FF0100000022000001235,OG Kush 1g preroll,,ea,Active,Acme Farms,2026-09-11
1A4FF0100000022000001236,Bulk trim,4536.2,g,Active,,2026-09-12
"""


def self_test() -> int:
    a = analyze(SAMPLE)
    checks = {
        "header parsed": a["header"][0] == "tag" and a["ncols"] == 7,
        "row count": a["nrows"] == 3,
        "blank counted": a["blanks"].get("quantity") == 1 and a["blanks"].get("customer_name") == 1,
        "pii flagged": a["pii"] == ["customer_name"],
        "tag column found": a["tag_cols"] == ["tag"],
        "token range sane": a["tokens_low"] < a["tokens_high"] and 80 < a["tokens_low"] < 300,
        "paste block": paste_block(SAMPLE, 2).count("\n") == 2,
    }
    for name, ok in checks.items():
        print(("ok   " if ok else "FAIL ") + name)
    return 0 if all(checks.values()) else 1


def main() -> int:
    p = argparse.ArgumentParser(description=__doc__.splitlines()[0])
    p.add_argument("file", nargs="?", help="CSV file to inspect")
    p.add_argument("-n", type=int, default=3, help="sample rows to show (default 3)")
    p.add_argument("--paste", action="store_true", help="print header + N rows as a paste-ready block")
    p.add_argument("--self-test", action="store_true", help="run built-in checks and exit")
    args = p.parse_args()
    if args.self_test:
        return self_test()
    if not args.file:
        p.print_help()
        return 2
    with open(args.file, encoding="utf-8", errors="replace", newline="") as f:
        text = f.read()
    if args.paste:
        print(paste_block(text, args.n))
        return 0
    print(report(analyze(text, args.n), args.file))
    return 0


if __name__ == "__main__":
    sys.exit(main())
