#!/usr/bin/env python3
"""Pull packages, orders or products out of Distru and save them as CSV.

    python3 export.py --resource orders --dry-run          # prints the exact requests; no key, no network
    python3 export.py --resource orders --since 2026-09-01  # live; needs DISTRU_API_TOKEN
    python3 export.py --resource packages --out packages.csv --max-pages 3

Standard library only. Follows `next_page` until it is null (Distru's pagination
contract). Flattens each record to dotted columns (`company.name`), lists become
JSON strings. Never writes to Distru: GET only.

Verified against Distru's public API docs, September 2026:
  base URL        https://app.distru.com/public/v1
  auth            Authorization: Bearer <token>  (Settings -> Integrations -> Distru API; keys expire after one year)
  pagination      response is {"data": [...], "next_page": <url or null>}; follow next_page as given
  datetime filter inclusive comma range: updated_datetime=<after>,<before>  (packages: inserted_datetime)
  rate limits     only /pdf endpoints (20/min, 1000/day); everything else unlimited today. Be polite anyway.
  errors          {"errors": [{"message": ..., "pointer": [...]}]}; 400 401 403 404 429 only, never 422

CC0 1.0. Copy it, change it, no attribution needed. Version 1.0.0 (2026-09-17).
"""
import argparse
import csv
import json
import os
import sys
import time
import urllib.error
import urllib.parse
import urllib.request

BASE_URL = "https://app.distru.com/public/v1"
TOKEN_ENV = "DISTRU_API_TOKEN"

# Which datetime filter each list endpoint accepts for "changed/created since".
# orders and products take updated_datetime; the packages list takes inserted_datetime.
RESOURCES = {
    "packages": {"path": "/packages", "since_param": "inserted_datetime"},
    "orders": {"path": "/orders", "since_param": "updated_datetime"},
    "products": {"path": "/products", "since_param": "updated_datetime"},
}

# Politeness pause between pages. Distru does not rate-limit list endpoints today; it may later.
PAGE_PAUSE_SECONDS = 0.2


def headers(token):
    return {
        "Authorization": f"Bearer {token}",
        "Content-Type": "application/json",
        "Accept": "application/json",
    }


def masked(token):
    return "Bearer " + ("<" + TOKEN_ENV + ">" if not token else token[:4] + "..." + token[-4:])


def first_url(resource, since):
    spec = RESOURCES[resource]
    url = BASE_URL + spec["path"]
    if since:
        # Inclusive range, "after," means on or after. Distru wants ISO 8601 UTC.
        after = since if "T" in since else since + "T00:00:00.000000Z"
        url += "?" + urllib.parse.urlencode({spec["since_param"]: after + ","})
    return url


def describe_request(url, token):
    print("GET " + url)
    for k, v in headers(token).items():
        print(f"  {k}: {masked(token) if k == 'Authorization' else v}")


def fetch(url, token):
    req = urllib.request.Request(url, headers=headers(token), method="GET")
    try:
        with urllib.request.urlopen(req, timeout=60) as resp:
            return json.loads(resp.read().decode("utf-8"))
    except urllib.error.HTTPError as e:
        body = e.read().decode("utf-8", errors="replace")
        try:
            errors = json.loads(body).get("errors", [])
            detail = "; ".join(f"{'/'.join(map(str, x.get('pointer', [])))}: {x.get('message')}" for x in errors)
        except Exception:
            detail = body[:300]
        if e.code == 429:
            wait = int(e.headers.get("Retry-After", "5"))
            print(f"429 rate limited, waiting {wait}s", file=sys.stderr)
            time.sleep(wait)
            return fetch(url, token)
        sys.exit(f"HTTP {e.code} from Distru: {detail}")


def walk(resource, since, token, max_pages, dry_run):
    """Yield every record. Follows next_page exactly as the server returns it."""
    url = first_url(resource, since)
    page = 0
    while url and (max_pages is None or page < max_pages):
        page += 1
        if dry_run:
            describe_request(url, token)
            if page == 1:
                print("  -> then follow response['next_page'] exactly as given, until it is null")
            return
        body = fetch(url, token)
        data = body.get("data", [])
        print(f"page {page}: {len(data)} {resource}", file=sys.stderr)
        for row in data:
            yield row
        url = body.get("next_page")
        if url:
            time.sleep(PAGE_PAUSE_SECONDS)


def flatten(obj, prefix=""):
    """{'company': {'name': 'x'}} -> {'company.name': 'x'}; lists -> JSON strings."""
    out = {}
    if isinstance(obj, dict):
        for k, v in obj.items():
            key = f"{prefix}{k}"
            if isinstance(v, dict):
                out.update(flatten(v, key + "."))
            elif isinstance(v, list):
                out[key] = json.dumps(v, ensure_ascii=False)
            else:
                out[key] = v
    return out


def write_csv(rows, path):
    flat = [flatten(r) for r in rows]
    columns = []
    seen = set()
    for r in flat:
        for k in r:
            if k not in seen:
                seen.add(k)
                columns.append(k)
    # id first, then everything else in first-seen order
    if "id" in columns:
        columns.remove("id")
        columns.insert(0, "id")
    with open(path, "w", newline="", encoding="utf-8") as f:
        w = csv.DictWriter(f, fieldnames=columns, extrasaction="ignore")
        w.writeheader()
        for r in flat:
            w.writerow(r)
    return len(flat), len(columns)


def main():
    ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    ap.add_argument("--resource", choices=sorted(RESOURCES), required=True)
    ap.add_argument("--since", help="ISO date or datetime; keeps records changed (orders, products) or created (packages) on or after it")
    ap.add_argument("--out", help="CSV path (default <resource>.csv)")
    ap.add_argument("--max-pages", type=int, help="stop after this many pages (handy for a first look)")
    ap.add_argument("--dry-run", action="store_true", help="print the exact requests and exit; no network, no key needed")
    args = ap.parse_args()

    token = os.environ.get(TOKEN_ENV, "")
    if not args.dry_run and not token:
        sys.exit(f"Set {TOKEN_ENV} (Distru: Settings -> Integrations -> Distru API). Or add --dry-run to see the requests first.")

    if args.dry_run:
        print(f"DRY RUN: {args.resource} -> {args.out or args.resource + '.csv'}")
        list(walk(args.resource, args.since, token, args.max_pages, dry_run=True))
        print("No request was sent.")
        return

    rows = list(walk(args.resource, args.since, token, args.max_pages, dry_run=False))
    out = args.out or f"{args.resource}.csv"
    n, cols = write_csv(rows, out)
    print(f"wrote {n} rows x {cols} columns to {out}")


if __name__ == "__main__":
    main()
