#!/usr/bin/env python3
"""ad_intel.py — competitor ad videos → transcript → teardown → artifacts + a review page.

Our own tool, no third-party API. Input is a `cards.json` from `adlib_fetch.py` (the
public Meta Ad Library, fetched with the cloak browser), one public video URL (TikTok /
Instagram / Facebook, via yt-dlp), or a local mp4. Per ad: whisper transcribes the audio
(primed with the card's own text), Gemini reads the video WITH its sound plus the
transcript and the card text under the framework in `references/teardown-framework.md`,
and one artifact is written per ad. Across a set, a teardown clusters the angles, ranks
the longest-running and most-repeated ads, and keeps a verbatim hooks swipe file.

Usage:
  ad_intel.py --project <projects/slug> (--cards <cards.json> | --url <video url> | --file <mp4>)
              [--brand <name>] [--our-brand <name>] [--max 12] [--model gemini-3.5-flash]
              [--whisper-model small] [--language en] [--dry-run]
  ad_intel.py --project <projects/slug> --record <result.json> [--brand <name>]
Exit 0 written · 2 bad input · 3 no Gemini key · 4 ffmpeg / whisper / yt-dlp / the model failed
     · 5 unsupported URL (Ad Library pages go through adlib_fetch.py)

Artifacts: <project>/artifacts/ad-intel/<id>.json (one per ad, schema ad-intel),
           <project>/artifacts/ad-intel/<brand>-teardown.json (schema ad-intel-teardown);
page:      <project>/build/ad-intel/<brand>.html (self-contained; frame grabs inline).
`--record` writes an artifact from a result you already have (no whisper, no model) — the
schema round-trip test uses it. `--dry-run` prints what would be sent and returns before
any key is read.
"""
from __future__ import annotations

import argparse
import base64
import collections
import datetime as dt
import hashlib
import html
import json
import os
import re
import subprocess
import sys
from urllib.parse import urlparse

HERE = os.path.dirname(os.path.abspath(__file__))
REPO = os.path.abspath(os.path.join(HERE, "..", "..", ".."))
RAP_SCRIPTS = os.path.join(REPO, "skills", "rap-avatar-mv", "scripts")
BREAKDOWN_SCRIPTS = os.path.join(REPO, "skills", "dance-breakdown", "scripts")
FRAMEWORK = os.path.join(HERE, "..", "references", "teardown-framework.md")
FFMPEG = os.environ.get("FFMPEG_FULL") or os.environ.get("FFMPEG_BIN") or "ffmpeg"
FFPROBE = os.environ.get("FFPROBE_BIN") or "ffprobe"
CATEGORIES = ("problem", "mechanism", "proof", "offer", "urgency", "trust")
SEGMENTS = ("hook", "setup", "claim", "evidence", "payoff", "cta", "other")
ANALYSIS_LISTS = ("structure", "claims", "proof", "objections", "angles")
ANALYSIS_STRINGS = ("summary", "hook", "offer", "cta", "visual_pattern", "ugc_script")
INLINE_BUDGET = 19 * 1024 * 1024        # Gemini inline data cap is ~20 MB per request
VIDEO_HOSTS = ("tiktok.com", "vm.tiktok.com", "instagram.com", "facebook.com", "fb.watch",
               "youtube.com", "youtu.be")
DEFAULT_MODEL = "gemini-3.5-flash"

for p in (RAP_SCRIPTS, BREAKDOWN_SCRIPTS):
    if p not in sys.path:
        sys.path.insert(0, p)


def say(msg: str) -> None:
    print(f"ad_intel: {msg}", file=sys.stderr)


def die(msg: str, code: int) -> None:
    say(msg)
    raise SystemExit(code)


def now_iso() -> str:
    return dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds")


def sha256_file(path: str) -> str:
    h = hashlib.sha256()
    with open(path, "rb") as fh:
        for chunk in iter(lambda: fh.read(1 << 20), b""):
            h.update(chunk)
    return h.hexdigest()


def slugify(s: str) -> str:
    return "-".join(w for w in re.sub(r"[^a-z0-9]+", "-", (s or "").lower()).split("-") if w)[:60] or "ads"


def load_prompt() -> str:
    text = open(FRAMEWORK, encoding="utf-8").read()
    m = re.search(r"<!-- prompt:start -->\n(.*?)<!-- prompt:end -->", text, re.S)
    if not m:
        die(f"{FRAMEWORK} has no prompt:start/prompt:end block", 2)
    return m.group(1).strip()


def probe_seconds(path: str) -> float:
    r = subprocess.run([FFPROBE, "-v", "error", "-show_entries", "format=duration",
                        "-of", "csv=p=0", path], capture_output=True, text=True)
    try:
        return round(float(r.stdout.strip()), 2)
    except ValueError:
        return 0.0


def landing_domain(url: str | None) -> str | None:
    if not url:
        return None
    try:
        host = urlparse(url).hostname or ""
    except ValueError:
        return None
    return host[4:] if host.startswith("www.") else host or None


def parse_started(text: str | None) -> str | None:
    """'22 May 2026' → '2026-05-22' (the Ad Library's date wording); None when unparseable."""
    if not text:
        return None
    for fmt in ("%d %b %Y", "%b %d, %Y", "%d %B %Y", "%B %d, %Y"):
        try:
            return dt.datetime.strptime(text.strip(), fmt).date().isoformat()
        except ValueError:
            continue
    return None


def url_kind(url: str) -> str:
    host = (urlparse(url).hostname or "").lower()
    path = urlparse(url).path or ""
    if host.endswith("facebook.com") and path.startswith("/ads/library"):
        return "ad-library-page"
    if any(host == h or host.endswith("." + h) for h in VIDEO_HOSTS):
        return "video"
    return "unsupported"


# ---------------------------------------------------------------------------
# media
# ---------------------------------------------------------------------------
def encode_for_model(src: str, dst: str) -> str:
    """A small mp4 WITH audio for the inline request (breakdown.cut_preview mutes; an ad's
    words are the point). Two passes of shrinking, then a named refusal — never a silent
    frames-only downgrade."""
    for scale, fps, crf in ((640, 12, 28), (480, 8, 32)):
        subprocess.run([FFMPEG, "-v", "error", "-y", "-i", src,
                        "-vf", f"scale={scale}:-2,fps={fps}", "-c:v", "libx264", "-crf", str(crf),
                        "-pix_fmt", "yuv420p", "-c:a", "aac", "-b:a", "64k", "-ac", "1",
                        "-movflags", "+faststart", dst], check=True, capture_output=True)
        if os.path.getsize(dst) <= INLINE_BUDGET:
            return dst
    die(f"{src} is still over the inline budget at 480p/8fps — trim the ad first", 4)
    return dst


def frame_grab(src: str, dst: str, at: float = 1.0) -> str | None:
    r = subprocess.run([FFMPEG, "-v", "error", "-y", "-ss", f"{at:.2f}", "-i", src, "-frames:v", "1",
                        "-vf", "scale=480:-2", "-q:v", "4", dst], capture_output=True)
    return dst if r.returncode == 0 and os.path.isfile(dst) else None


def download_url(url: str, dest: str) -> None:
    import ingest_source  # rap lane's yt-dlp argv (bestvideo+bestaudio, mp4 merge)
    r = subprocess.run(ingest_source.ytdlp_argv(url, dest, None), capture_output=True, text=True)
    if r.returncode != 0 or not os.path.isfile(dest):
        die(f"yt-dlp failed for {url}: {r.stderr.strip()[-500:]}", 4)


def transcribe(media: str, cache_dir: str, prompt: str | None, model: str, language: str) -> dict:
    import transcribe_clips as tc
    doc = tc.whisper_words(media, model, cache_dir, language=language, prompt=prompt or None)
    segments = [{"start": round(float(s.get("start", 0)), 2), "end": round(float(s.get("end", 0)), 2),
                 "text": (s.get("text") or "").strip()} for s in (doc.get("segments") or [])]
    return {"text": (doc.get("text") or "").strip(), "segments": segments,
            "wordCount": len(doc.get("words") or []), "noSpeech": not (doc.get("words") or []),
            "language": language, "whisperModel": model}


# ---------------------------------------------------------------------------
# the model
# ---------------------------------------------------------------------------
def card_context(source: dict, our_brand: str | None) -> str:
    fields = [("page", source.get("pageName")), ("primary text", source.get("primaryText")),
              ("headline", source.get("headline")), ("description", source.get("description")),
              ("call to action", source.get("cta")), ("landing domain", source.get("landingDomain")),
              ("started running", source.get("startedOn")),
              ("versions of this creative", source.get("versions"))]
    lines = [f"{k}: {v}" for k, v in fields if v not in (None, "", 0)]
    lines.append(f"the operator's brand (for ugc_script): {our_brand or 'our product'}")
    return "\n".join(lines)


def request_parts(prompt: str, media_b64: str | None, transcript: dict, source: dict, our_brand: str | None) -> list[dict]:
    parts = []
    if media_b64 is not None:
        parts.append({"inlineData": {"mimeType": "video/mp4", "data": media_b64}})
    parts.append({"text": prompt + "\n\nTRANSCRIPT (whisper, word-timed):\n"
                  + json.dumps(transcript.get("segments") or [], ensure_ascii=False)
                  + "\n\nADVERTISER TEXT BESIDE THE VIDEO:\n" + card_context(source, our_brand)})
    return parts


def normalise_analysis(raw: dict) -> tuple[dict, list[str]]:
    """Every key present, lists are lists, angle categories in the enum — drop what is
    not, with a warning; never relabel."""
    warnings: list[str] = []
    if not isinstance(raw, dict):
        return {}, ["the model returned no object"]
    out: dict = {}
    for k in ANALYSIS_STRINGS:
        v = raw.get(k)
        out[k] = v.strip() if isinstance(v, str) else ("" if v is None else json.dumps(v, ensure_ascii=False))
    for k in ANALYSIS_LISTS:
        v = raw.get(k)
        out[k] = v if isinstance(v, list) else ([] if v is None else [v])
    angles = []
    for a in out["angles"]:
        if isinstance(a, dict) and isinstance(a.get("angle"), str) and a.get("category") in CATEGORIES:
            angles.append({"angle": a["angle"].strip(), "category": a["category"]})
        else:
            warnings.append(f"dropped an angle outside the vocabulary: {json.dumps(a, ensure_ascii=False)[:120]}")
    out["angles"] = angles
    structure = []
    for s in out["structure"]:
        if not isinstance(s, dict):
            continue
        seg = s.get("segment") if s.get("segment") in SEGMENTS else "other"
        try:
            structure.append({"segment": seg, "start": round(float(s.get("start", 0)), 2),
                              "end": round(float(s.get("end", 0)), 2), "note": str(s.get("note") or "")})
        except (TypeError, ValueError):
            warnings.append("dropped a structure row with non-numeric timing")
    out["structure"] = structure
    for k in ("claims", "proof", "objections"):
        out[k] = [str(x) for x in out[k]]
    return out, warnings


def analyse(keys: list[str], model: str, prompt: str, media_gemini: str, transcript: dict,
            source: dict, our_brand: str | None, cooldown: dict) -> tuple[dict, list[str]]:
    import breakdown
    b64 = base64.b64encode(open(media_gemini, "rb").read()).decode()
    parts = request_parts(prompt, b64, transcript, source, our_brand)
    try:
        raw = breakdown.call_model(keys, model, parts, cooldown, max_output_tokens=16384)
    except Exception as e:  # noqa: BLE001 — named, then exit 4
        die(f"the model failed: {e}", 4)
    return normalise_analysis(raw)


def synthesize(keys: list[str], model: str, teardown: dict, cooldown: dict) -> dict:
    import breakdown
    ads = [{"id": a["id"], "startedOn": a.get("startedOn"), "versions": a.get("versions"),
            "hook": a.get("hook"), "angles": a.get("angles"), "offer": a.get("offer"),
            "summary": a.get("summary")} for a in teardown["ads"]]
    text = ("You read per-ad teardowns of ONE advertiser's active paid social ads. Answer with one "
            "JSON object: {\"positioning\": one sentence on how this advertiser positions itself; "
            "\"appearsToBeTesting\": array of short strings — what they seem to be A/B testing "
            "(differing hooks on the same offer, many versions of one creative, offers that vary); "
            "\"testsForUs\": array of {\"test\", \"why\", \"ads\": [ids]} — the three angles our brand "
            "should test first, each tied to the ad ids that carry it}. Rules: an active ad is not "
            "a winning ad; repetition (versions, age) is the only signal; invent no numbers.\n\n"
            + json.dumps({"brand": teardown["brand"], "ads": ads}, ensure_ascii=False))
    try:
        raw = breakdown.call_model(keys, model, [{"text": text}], cooldown, max_output_tokens=8192)
    except Exception as e:  # noqa: BLE001
        say(f"synthesis failed ({e}); the teardown keeps its computed parts only")
        return {}
    if not isinstance(raw, dict):
        return {}
    return {"positioning": str(raw.get("positioning") or ""),
            "appearsToBeTesting": [str(x) for x in (raw.get("appearsToBeTesting") or []) if x],
            "testsForUs": [t for t in (raw.get("testsForUs") or []) if isinstance(t, dict)]}


# ---------------------------------------------------------------------------
# artifacts
# ---------------------------------------------------------------------------
def source_from_card(card: dict, cards_url: str | None) -> dict:
    return {"kind": "meta-ad-library", "url": f"https://www.facebook.com/ads/library/?id={card['libraryId']}",
            "libraryId": card.get("libraryId"), "status": card.get("status"),
            "startedOn": parse_started(card.get("startedOn")), "startedOnText": card.get("startedOn"),
            "versions": int(card.get("versions") or 1), "pageName": card.get("pageName"),
            "pageUrl": card.get("pageUrl"), "primaryText": card.get("primaryText"),
            "headline": card.get("headline"), "description": card.get("description"),
            "cta": card.get("cta"), "landingUrl": card.get("landingUrl"),
            "landingDomain": landing_domain(card.get("landingUrl")), "searchUrl": cards_url}


def rel_to(project: str, path: str) -> str:
    try:
        return os.path.relpath(path, project) if os.path.commonpath([project, path]) == project else path
    except ValueError:
        return path


def write_ad(project: str, ad: dict) -> str:
    import raplib
    path = os.path.join(project, "artifacts", "ad-intel", f"{ad['id']}.json")
    raplib.write_json_atomic(path, ad)
    return path


def build_teardown(brand: str, ads: list[dict], our_brand: str | None) -> dict:
    rows = []
    for ad in ads:
        s, a = ad["source"], ad["analysis"]
        rows.append({"id": ad["id"], "pageName": s.get("pageName"), "startedOn": s.get("startedOn"),
                     "versions": s.get("versions") or 1, "hook": a.get("hook"), "summary": a.get("summary"),
                     "offer": a.get("offer"), "cta": a.get("cta"), "angles": a.get("angles") or [],
                     "landingDomain": s.get("landingDomain")})
    clusters = collections.OrderedDict((c, collections.OrderedDict()) for c in CATEGORIES)
    for r in rows:
        for ang in r["angles"]:
            key = ang["angle"].strip().lower()
            clusters[ang["category"]].setdefault(key, {"angle": ang["angle"], "ads": []})
            if r["id"] not in clusters[ang["category"]][key]["ads"]:
                clusters[ang["category"]][key]["ads"].append(r["id"])
    angle_clusters = []
    for cat, m in clusters.items():
        items = sorted(m.values(), key=lambda x: -len(x["ads"]))
        angle_clusters.append({"category": cat, "count": sum(len(i["ads"]) for i in items), "angles": items})
    by_age = sorted(rows, key=lambda r: (r["startedOn"] or "9999-99-99"))
    by_versions = sorted(rows, key=lambda r: -(r["versions"] or 1))
    return {"schemaVersion": 1, "brand": brand, "ourBrand": our_brand, "generatedAt": now_iso(),
            "adsAnalyzed": len(rows), "ads": rows, "angleClusters": angle_clusters,
            "hooks": [{"id": r["id"], "hook": r["hook"], "startedOn": r["startedOn"], "versions": r["versions"]} for r in rows],
            "longestRunning": [r["id"] for r in by_age[:5]],
            "mostRepeated": [r["id"] for r in by_versions[:5] if (r["versions"] or 1) > 1],
            "positioning": "", "appearsToBeTesting": [], "testsForUs": [],
            "rules": ["an active ad is not a winning ad", "repetition (versions, age) is the signal",
                      "no spend, CTR or conversion figures are invented",
                      "hooks are verbatim; angles and objections are readings"]}


# ---------------------------------------------------------------------------
# the page
# ---------------------------------------------------------------------------
def _e(s) -> str:
    return html.escape("" if s is None else str(s))


def render_page(teardown: dict, ads: list[dict], thumbs: dict[str, str | None]) -> str:
    css = """
    body{font:15px/1.5 -apple-system,Segoe UI,Helvetica,Arial,sans-serif;margin:0;background:#f4f5f7;color:#1c2229}
    main{max-width:1180px;margin:0 auto;padding:28px 20px 80px}
    h1{font-size:1.7rem;margin:0 0 4px} h2{font-size:1.15rem;margin:36px 0 10px;border-top:1px solid #d9dde3;padding-top:14px}
    .meta{color:#5c6773;font-size:.9rem} .pill{display:inline-block;padding:1px 8px;border-radius:999px;font-size:.72rem;letter-spacing:.03em;margin:0 4px 4px 0;background:#e6eef7;color:#1f4e79}
    .pill.problem{background:#fbe6e3;color:#8a2c1f}.pill.mechanism{background:#e7f1e6;color:#25562a}.pill.proof{background:#e4eef8;color:#1e4a7a}.pill.offer{background:#fff0d6;color:#7a4c00}.pill.urgency{background:#f7e3f3;color:#6b1f5e}.pill.trust{background:#e6f4f2;color:#155a52}
    table{border-collapse:collapse;width:100%;background:#fff;font-size:.9rem} th,td{text-align:left;vertical-align:top;padding:8px 10px;border-bottom:1px solid #e3e7ec} th{font-size:.72rem;text-transform:uppercase;letter-spacing:.06em;color:#5c6773}
    .grid{display:grid;grid-template-columns:repeat(auto-fill,minmax(340px,1fr));gap:16px}
    .card{background:#fff;border:1px solid #d9dde3;border-radius:6px;overflow:hidden} .card img{width:100%;display:block;background:#111;aspect-ratio:16/9;object-fit:contain}
    .card .body{padding:12px 14px} .hook{font-weight:600;margin:6px 0} .small{font-size:.82rem;color:#5c6773}
    details{margin-top:8px} summary{cursor:pointer;color:#1f4e79;font-size:.85rem} pre{white-space:pre-wrap;font:13px/1.45 ui-monospace,Menlo,monospace;background:#f7f8fa;padding:8px;border-radius:4px}
    .rules{background:#fff8e6;border:1px solid #f0d9a0;padding:10px 14px;font-size:.88rem}
    """
    t = teardown
    parts = [f"<!doctype html><meta charset='utf-8'><title>Ad intel · {_e(t['brand'])}</title><style>{css}</style><main>",
             f"<h1>{_e(t['brand'])} · competitor ad teardown</h1>",
             f"<div class='meta'>{t['adsAnalyzed']} active ad(s) from the public Meta Ad Library · generated {_e(t['generatedAt'])}"
             + (f" · our brand: {_e(t['ourBrand'])}" if t.get('ourBrand') else "") + "</div>",
             "<p class='rules'>" + " · ".join(_e(r) for r in t["rules"]) + "</p>"]
    if t.get("positioning"):
        parts.append(f"<h2>Positioning (a reading)</h2><p>{_e(t['positioning'])}</p>")
    parts.append("<h2>Angles, clustered</h2><table><tr><th>Category</th><th>Ads</th><th>Angles (ads carrying it)</th></tr>")
    for c in t["angleClusters"]:
        if not c["angles"]:
            continue
        items = "<br>".join(f"{_e(a['angle'])} <span class='small'>({', '.join(_e(x) for x in a['ads'])})</span>" for a in c["angles"])
        parts.append(f"<tr><td><span class='pill {c['category']}'>{c['category']}</span></td><td>{c['count']}</td><td>{items}</td></tr>")
    parts.append("</table>")
    if t.get("appearsToBeTesting"):
        parts.append("<h2>What they appear to be testing</h2><ul>" + "".join(f"<li>{_e(x)}</li>" for x in t["appearsToBeTesting"]) + "</ul>")
    if t.get("testsForUs"):
        parts.append("<h2>Tests for us</h2><ol>" + "".join(
            f"<li><b>{_e(x.get('test'))}</b> — {_e(x.get('why'))} <span class='small'>({', '.join(_e(i) for i in (x.get('ads') or []))})</span></li>"
            for x in t["testsForUs"]) + "</ol>")
    parts.append("<h2>Hooks, verbatim</h2><ul>")
    for h in t["hooks"]:
        parts.append(f"<li>“{_e(h['hook'])}” <span class='small'>— {_e(h['id'])}, since {_e(h['startedOn'] or '?')}, {h['versions']} version(s)</span></li>")
    parts.append("</ul><h2>The ads</h2><div class='grid'>")
    by_id = {a["id"]: a for a in ads}
    for r in t["ads"]:
        ad = by_id.get(r["id"], {})
        s, a = ad.get("source", {}), ad.get("analysis", {})
        thumb = thumbs.get(r["id"])
        img = f"<img src='{thumb}' alt=''>" if thumb else "<div class='small' style='padding:20px'>no frame grab</div>"
        pills = "".join(f"<span class='pill {x['category']}'>{_e(x['category'])}: {_e(x['angle'])}</span>" for x in a.get("angles") or [])
        parts.append(
            f"<div class='card'>{img}<div class='body'>"
            f"<div class='small'>{_e(s.get('pageName'))} · since {_e(s.get('startedOn') or s.get('startedOnText') or '?')} · {s.get('versions') or 1} version(s)"
            f" · <a href='{_e(s.get('url'))}'>{_e(r['id'])}</a></div>"
            f"<div class='hook'>“{_e(a.get('hook'))}”</div>"
            f"<div>{pills}</div>"
            f"<div class='small'>CTA: {_e(a.get('cta') or s.get('cta') or '—')} · offer: {_e(a.get('offer') or '—')} · lands on {_e(s.get('landingDomain') or '—')}</div>"
            f"<p>{_e(a.get('summary'))}</p><div class='small'>{_e(a.get('visual_pattern'))}</div>"
            f"<details><summary>claims, proof, objections</summary><pre>{_e(json.dumps({k: a.get(k) for k in ('claims', 'proof', 'objections')}, ensure_ascii=False, indent=1))}</pre></details>"
            f"<details><summary>transcript</summary><pre>{_e(ad.get('transcript', {}).get('text'))}</pre></details>"
            f"<details><summary>our UGC script (a reading, in our voice)</summary><pre>{_e(a.get('ugc_script'))}</pre></details>"
            f"</div></div>")
    parts.append("</div></main>")
    return "\n".join(parts)


# ---------------------------------------------------------------------------
# main
# ---------------------------------------------------------------------------
def collect_inputs(args, project: str, work: str) -> tuple[list[dict], str]:
    """[{id, media, source}], brand."""
    items = []
    if args.cards:
        cards_doc = json.load(open(args.cards, encoding="utf-8"))
        cards_dir = os.path.dirname(os.path.abspath(args.cards))
        for c in cards_doc.get("cards") or []:
            if not c.get("videoFile"):
                continue
            items.append({"id": c["libraryId"], "media": os.path.join(cards_dir, c["videoFile"]),
                          "source": source_from_card(c, cards_doc.get("url"))})
        names = collections.Counter(i["source"].get("pageName") for i in items if i["source"].get("pageName"))
        brand = args.brand or (names.most_common(1)[0][0] if names else cards_doc.get("query") or "ads")
    elif args.url:
        kind = url_kind(args.url)
        if kind == "ad-library-page":
            die("an Ad Library page is fetched by adlib_fetch.py (it needs the browser); pass its cards.json here", 5)
        if kind != "video":
            die(f"unsupported host for --url: {urlparse(args.url).hostname} (TikTok, Instagram, Facebook video, YouTube)", 5)
        uid = hashlib.sha256(args.url.encode()).hexdigest()[:12]
        items.append({"id": f"url-{uid}", "media": os.path.join(work, "dl", f"url-{uid}.mp4"),
                      "source": {"kind": "url", "url": args.url, "host": urlparse(args.url).hostname}, "download": args.url})
        brand = args.brand or "ads"
    else:
        f = os.path.abspath(args.file)
        if not os.path.isfile(f):
            die(f"--file not found: {f}", 2)
        items.append({"id": slugify(os.path.splitext(os.path.basename(f))[0]), "media": f,
                      "source": {"kind": "file", "url": None, "file": f}})
        brand = args.brand or "ads"
    return items[: args.max], brand


def run_record(args, project: str) -> int:
    rec = json.load(open(args.record, encoding="utf-8"))
    if not isinstance(rec, dict) or not rec.get("id") or not isinstance(rec.get("analysis"), dict):
        die("--record needs a JSON object with at least {id, analysis}", 2)
    analysis, warnings = normalise_analysis(rec["analysis"])
    ad = {"schemaVersion": 1, "id": str(rec["id"]), "generatedAt": now_iso(), "model": "recorded",
          "source": rec.get("source") or {"kind": "file", "url": None},
          "media": rec.get("media") or {"file": None, "sha256": None, "seconds": None},
          "transcript": rec.get("transcript") or {"text": "", "segments": [], "wordCount": 0, "noSpeech": True,
                                                  "language": "en", "whisperModel": "recorded"},
          "analysis": analysis, "warnings": warnings + ["recorded: no whisper, no model call"]}
    path = write_ad(project, ad)
    out = {"artifact": path}
    if args.brand:
        td = build_teardown(args.brand, [ad], args.our_brand)
        import raplib
        tpath = os.path.join(project, "artifacts", "ad-intel", f"{slugify(args.brand)}-teardown.json")
        raplib.write_json_atomic(tpath, td)
        out["teardown"] = tpath
    print(json.dumps(out))
    return 0


def main() -> int:
    ap = argparse.ArgumentParser(description="competitor ad videos → transcript → teardown → artifacts + page")
    ap.add_argument("--project", required=True, help="path to projects/<slug>")
    src = ap.add_mutually_exclusive_group()
    src.add_argument("--cards", help="cards.json from adlib_fetch.py")
    src.add_argument("--url", help="one public video URL (TikTok / Instagram / Facebook / YouTube)")
    src.add_argument("--file", help="one local mp4")
    src.add_argument("--record", help="a result JSON to write as an artifact (no whisper, no model)")
    ap.add_argument("--brand", help="the advertiser (default: the cards' page name)")
    ap.add_argument("--our-brand", help="the operator's brand the ugc_script is written for")
    ap.add_argument("--max", type=int, default=12)
    ap.add_argument("--model", default=DEFAULT_MODEL)
    ap.add_argument("--whisper-model", default="small")
    ap.add_argument("--language", default=os.environ.get("AD_LANGUAGE") or "en")
    ap.add_argument("--dry-run", action="store_true", help="print the request shape; read no key, call nothing")
    args = ap.parse_args()

    project = os.path.abspath(args.project)
    if not os.path.isdir(project):
        die(f"project directory not found: {project}", 2)
    if not (args.cards or args.url or args.file or args.record):
        ap.error("one of --cards, --url, --file, --record is required")
    if args.record:
        return run_record(args, project)

    work = os.path.join(project, "build", "ad-intel")
    os.makedirs(work, exist_ok=True)
    items, brand = collect_inputs(args, project, work)
    if not items:
        die("nothing to analyse (no card with a downloaded video)", 2)
    prompt = load_prompt()

    if args.dry_run:
        sample = items[0]
        parts = request_parts(prompt, "<base64 mp4 with audio>", {"segments": []}, sample["source"], args.our_brand)
        print(json.dumps({"brand": brand, "ourBrand": args.our_brand, "model": args.model,
                          "ads": [{"id": i["id"], "media": i["media"], "kind": i["source"].get("kind")} for i in items],
                          "whisper": {"model": args.whisper_model, "language": args.language, "primedWith": "the card's primary text + headline"},
                          "request": {"parts": [{"inlineData": {"mimeType": p["inlineData"]["mimeType"]}} if "inlineData" in p
                                                else {"text": p["text"][:400] + "…"} for p in parts]},
                          "artifacts": [os.path.join(project, "artifacts", "ad-intel", f"{i['id']}.json") for i in items],
                          "teardown": os.path.join(project, "artifacts", "ad-intel", f"{slugify(brand)}-teardown.json"),
                          "page": os.path.join(work, f"{slugify(brand)}.html")}, indent=1, ensure_ascii=False))
        return 0

    import breakdown
    keys = breakdown.api_keys()
    if not keys:
        die("no Gemini key (GEMINI_API_KEYS / GOOGLE_API_KEYS / GOOGLE_API_KEY, or .env.local)", 3)
    cooldown: dict = {}
    cache = os.path.join(work, "whisper-cache")
    thumbs: dict[str, str | None] = {}
    ads: list[dict] = []
    for i in items:
        if i.get("download"):
            os.makedirs(os.path.dirname(i["media"]), exist_ok=True)
            if not os.path.isfile(i["media"]):
                say(f"{i['id']}: downloading {i['download']}")
                download_url(i["download"], i["media"])
        if not os.path.isfile(i["media"]):
            die(f"{i['id']}: media missing at {i['media']}", 2)
        src = i["source"]
        prime = " ".join(x for x in (src.get("pageName"), src.get("primaryText"), src.get("headline")) if x) or None
        say(f"{i['id']}: whisper ({args.whisper_model}, {args.language})")
        try:
            transcript = transcribe(i["media"], cache, prime, args.whisper_model, args.language)
        except SystemExit:
            raise
        except Exception as e:  # noqa: BLE001
            die(f"{i['id']}: whisper failed: {e}", 4)
        small = os.path.join(work, "gemini", f"{i['id']}.mp4")
        os.makedirs(os.path.dirname(small), exist_ok=True)
        try:
            encode_for_model(i["media"], small)
        except subprocess.CalledProcessError as e:
            die(f"{i['id']}: ffmpeg failed: {e.stderr.decode(errors='replace')[-400:] if e.stderr else e}", 4)
        say(f"{i['id']}: {args.model}")
        analysis, warnings = analyse(keys, args.model, prompt, small, transcript, src, args.our_brand, cooldown)
        os.makedirs(os.path.join(work, "thumbs"), exist_ok=True)
        jpg = frame_grab(i["media"], os.path.join(work, "thumbs", f"{i['id']}.jpg"))
        thumbs[i["id"]] = ("data:image/jpeg;base64," + base64.b64encode(open(jpg, "rb").read()).decode()) if jpg else None
        ad = {"schemaVersion": 1, "id": i["id"], "generatedAt": now_iso(), "model": args.model,
              "source": src,
              "media": {"file": rel_to(project, i["media"]), "sha256": sha256_file(i["media"]),
                        "seconds": probe_seconds(i["media"])},
              "transcript": transcript, "analysis": analysis, "warnings": warnings}
        path = write_ad(project, ad)
        say(f"{i['id']}: hook “{analysis.get('hook', '')[:80]}” → {rel_to(project, path)}")
        ads.append(ad)

    teardown = build_teardown(brand, ads, args.our_brand)
    if len(ads) >= 2:
        teardown.update(synthesize(keys, args.model, teardown, cooldown))
    import raplib
    tpath = os.path.join(project, "artifacts", "ad-intel", f"{slugify(brand)}-teardown.json")
    raplib.write_json_atomic(tpath, teardown)
    page = os.path.join(work, f"{slugify(brand)}.html")
    with open(page + ".tmp", "w", encoding="utf-8") as fh:
        fh.write(render_page(teardown, ads, thumbs))
    os.replace(page + ".tmp", page)
    print(json.dumps({"ads": len(ads), "teardown": tpath, "page": page,
                      "artifacts": [os.path.join(project, "artifacts", "ad-intel", f"{a['id']}.json") for a in ads]}))
    return 0


if __name__ == "__main__":
    sys.exit(main())
