#!/usr/bin/env python3
"""answer_plan.py — read a torn-down ad's SHAPE and write the contract for our answer to it.

Plan-only: nothing is rendered and nothing is spent. The format is read from the reference,
not assumed — aspect from ffprobe of the fetched mp4, target length from its seconds, beats
from the teardown's `structure[]` (hook / claims / payoff / cta), cutaways from the
`visual_pattern` and the structure notes, captions as the social-ad default. The script is
the teardown's `ugc_script` expanded to ONE line per beat (≤ 22 words, our voice, no line
copied) by Gemini; the operator edits the lines on the contract page before any render.

Still prompts never contain the words phone / photograph / screenshot / mock-up: GB obeys a
named medium like a prop (a "photograph" came back as a print on a grey wall; "shot on a
phone" put the scene inside a phone).

usage: answer_plan.py --project <p> --ad <libraryId> [--our-brand X] [--beats 6] [--voice he|she]
                      [--route "veo-useapi fast"] [--credits-per-beat 10] [--model gemini-3.5-flash] [--dry-run]
→ <project>/artifacts/ad-intel/answer/<id>/plan.json and <project>/build/ad-intel/answer-<id>.html
Exit 0 written · 2 bad input · 3 no Gemini key · 4 the model's script broke a rule (which beat is named).
"""
from __future__ import annotations

import argparse
import datetime as dt
import html
import json
import os
import re
import subprocess
import sys

HERE = os.path.dirname(os.path.abspath(__file__))
REPO = os.path.abspath(os.path.join(HERE, "..", "..", ".."))
BREAKDOWN_SCRIPTS = os.path.join(REPO, "skills", "dance-breakdown", "scripts")
if BREAKDOWN_SCRIPTS not in sys.path:
    sys.path.insert(0, BREAKDOWN_SCRIPTS)
sys.path.insert(0, HERE)

FFPROBE = os.environ.get("FFPROBE_BIN") or "ffprobe"
DEFAULT_MODEL = "gemini-3.5-flash"
MAX_WORDS = 22
BEAT_SECONDS = 8            # one Veo take per beat
BANNED_STILL_WORDS = ("phone", "photograph", "screenshot", "mock-up", "mockup")
DEFAULT_BEATS = ["hook", "tip1", "tip2", "tip3", "payoff", "cta"]


def say(msg: str) -> None:
    print(f"answer_plan: {msg}", file=sys.stderr)


def die(msg: str, code: int) -> None:
    say(msg); raise SystemExit(code)


def now_iso() -> str:
    return dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds")


def _e(s) -> str:
    return html.escape("" if s is None else str(s))


def probe(path: str) -> dict:
    r = subprocess.run([FFPROBE, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height:format=duration",
                        "-of", "json", path], capture_output=True, text=True)
    try:
        d = json.loads(r.stdout or "{}")
        st = (d.get("streams") or [{}])[0]
        return {"width": int(st.get("width") or 0), "height": int(st.get("height") or 0),
                "seconds": round(float((d.get("format") or {}).get("duration") or 0), 2)}
    except (ValueError, TypeError):
        return {"width": 0, "height": 0, "seconds": 0.0}


def aspect_of(w: int, h: int) -> str:
    if not w or not h:
        return "9:16"
    r = w / h
    return "9:16" if r < 0.7 else "16:9" if r > 1.4 else "1:1"


# --------------------------------------------------------------------------
# the shape: beats, cutaways, captions — READ from the teardown
# --------------------------------------------------------------------------

def beats_from_structure(structure: list[dict], n: int) -> list[dict]:
    """hook → claims → payoff → cta, from the reference's own segments; padded/capped to n."""
    segs = [s for s in structure if isinstance(s, dict) and s.get("segment")]
    ids: list[dict] = []
    claims = 0
    for s in segs:
        seg = s["segment"]
        if seg == "hook" and not any(b["id"] == "hook" for b in ids):
            ids.append({"id": "hook", "from": s})
        elif seg in ("claim", "evidence", "setup"):
            claims += 1; ids.append({"id": f"tip{claims}", "from": s})
        elif seg == "payoff" and not any(b["id"] == "payoff" for b in ids):
            ids.append({"id": "payoff", "from": s})
        elif seg == "cta" and not any(b["id"] == "cta" for b in ids):
            ids.append({"id": "cta", "from": s})
    if not ids:
        ids = [{"id": i, "from": None} for i in DEFAULT_BEATS]
    if not any(b["id"] == "hook" for b in ids):
        ids.insert(0, {"id": "hook", "from": None})
    if not any(b["id"] == "cta" for b in ids):
        ids.append({"id": "cta", "from": None})
    # cap to n: keep hook first and cta last, drop the last tips / payoff in between
    while len(ids) > n:
        mid = [k for k, b in enumerate(ids) if b["id"] not in ("hook", "cta")]
        if not mid:
            break
        ids.pop(mid[-1])
    while len(ids) < n:
        k = len([b for b in ids if b["id"].startswith("tip")]) + 1
        ids.insert(len(ids) - 1, {"id": f"tip{k}", "from": None})
    return ids


def wants_cutaways(visual_pattern: str, structure: list[dict]) -> bool:
    text = " ".join([visual_pattern or ""] + [str(s.get("note") or "") for s in structure if isinstance(s, dict)]).lower()
    return any(w in text for w in ("screen", "recording", "notes", "app", "graphic", "overlay", "b-roll", "broll", "insert", "cutaway", "split"))


def noun_of(line: str) -> str:
    """The last content word of the line — where the cutaway lands when the take reaches it."""
    stop = {"the", "a", "an", "it", "in", "on", "to", "for", "of", "and", "so", "is", "are", "my", "your", "with", "me", "you", "i", "we", "that", "this"}
    words = [w for w in re.split(r"[^a-z0-9']+", line.lower()) if w and w not in stop]
    return words[-1] if words else ""


# --------------------------------------------------------------------------
# the script — Gemini expands the teardown's ugc_script to one line per beat
# --------------------------------------------------------------------------

def script_request(ad: dict, beats: list[dict], our_brand: str) -> str:
    a = ad.get("analysis") or {}
    lines = [
        f"You write the SPOKEN script for a short vertical social ad for {our_brand}, answering a competitor's ad with the same SHAPE.",
        "Rules: one line per beat, at most 22 plain words each, one sentence per line (a two-sentence line gets half-skipped by the",
        "video model), first person, conversational, no claim we cannot make, NO line copied from the competitor. Return ONLY JSON:",
        '{"beats":[{"id":"<beat id>","line":"<the spoken line>"}]}',
        "",
        f"Beats, in order (keep the ids): {json.dumps([{'id': b['id'], 'their_note': (b.get('from') or {}).get('note')} for b in beats])}",
        f"Their hook (for the SHAPE only, do not copy): {a.get('hook')!r}",
        f"Their summary: {a.get('summary')!r}",
        f"Their offer / CTA: {a.get('offer')!r} / {a.get('cta')!r}",
        f"Our seed script (the teardown's draft, expand and tighten it): {a.get('ugc_script')!r}",
    ]
    return "\n".join(lines)


def check_lines(beats: list[dict], answer: dict) -> list[dict]:
    got = {b.get("id"): (b.get("line") or "").strip() for b in (answer.get("beats") or []) if isinstance(b, dict)}
    out = []
    for b in beats:
        line = got.get(b["id"])
        if not line:
            die(f"the model returned no line for beat {b['id']}", 4)
        n = len(line.split())
        if n > MAX_WORDS:
            die(f"beat {b['id']} is {n} words (max {MAX_WORDS}): {line!r}", 4)
        out.append({**b, "line": line, "words": n})
    return out


# --------------------------------------------------------------------------
# the stills and the storyboard prompts
# --------------------------------------------------------------------------

def presenter_frame(aspect: str, visual_pattern: str, voice: str) -> str:
    who = "a man in his early thirties" if voice == "he" else "a woman in her early thirties"
    setting = ("seated at a wooden desk with a large podcast microphone on a boom arm, black over-ear headphones, plain dark t-shirt, "
               "dark studio behind with warm bokeh string lights and soft blue ambient light")
    if "outdoor" in (visual_pattern or "").lower() or "street" in (visual_pattern or "").lower():
        setting = "outdoors on a quiet street in soft daylight, plain jacket, handheld framing"
    elif "car" in (visual_pattern or "").lower():
        setting = "in the driver's seat of a parked car, daylight through the windows, plain jacket"
    return (f"Vertical {aspect} image filling the whole frame edge to edge, medium close-up of {who}, short dark hair, clear-framed glasses, "
            f"{setting}, speaking to camera with an open friendly expression, both hands resting on the desk, nothing in the hands, "
            "no logos, no text anywhere, natural skin.")


def cutaway_prompt(aspect: str, beat: dict, brand: str) -> str:
    n = beat["id"][-1] if beat["id"].startswith("tip") else "1"
    text = beat["line"]
    return (f"Vertical {aspect} image, a clean light-mode app screen for {brand} that illustrates this line — “{text}” — "
            f"with a short heading of at most five words and a small {brand} wordmark, no other text, no people, crisp UI, fills the frame edge to edge. (tip {n})")


def storyboard_prompt(voice: str, line: str) -> str:
    frame = ("Vertical phone-shot video, medium close-up, the presenter with short dark hair, clear-framed glasses and black over-ear headphones "
             "sits at a wooden desk behind a podcast microphone, dark studio with warm bokeh lights behind, talking straight to camera, "
             "natural UGC energy, small hand gestures, stays seated and in frame the whole time, no on-screen text, room tone only.")
    return f"{frame} {'He' if voice == 'he' else 'She'} says: \"{line}\""


def banned(prompt: str) -> list[str]:
    """Whole words only: "headphones" is a prop we WANT; "phone" is the medium we do not."""
    low = prompt.lower()
    return [w for w in BANNED_STILL_WORDS if re.search(r"(?<![a-z-])" + re.escape(w) + r"(?![a-z])", low)]


# --------------------------------------------------------------------------
# the contract page
# --------------------------------------------------------------------------

def render(plan: dict) -> str:
    css = """body{font:15px/1.5 -apple-system,Segoe UI,Helvetica,Arial,sans-serif;margin:0;background:#f4f5f7;color:#1c2229}main{max-width:1100px;margin:0 auto;padding:26px 20px 70px}
h1{font-size:1.6rem;margin:0 0 6px}h2{font-size:1.1rem;margin:30px 0 10px;border-top:1px solid #d9dde3;padding-top:12px}.meta{color:#5a6570}
table{border-collapse:collapse;width:100%;font-size:.92rem}th,td{text-align:left;vertical-align:top;padding:8px 10px;border-bottom:1px solid #d9dde3}th{font-size:.74rem;letter-spacing:.06em;text-transform:uppercase;color:#5a6570}
.line{font-weight:600}.p{font-family:ui-monospace,Menlo,monospace;font-size:.78rem;color:#3b4650;white-space:pre-wrap}.cost{background:#fff;border:1px solid #d9dde3;padding:12px 16px;display:inline-block}
.warn{background:#fbf0d9;color:#8a5a00;padding:8px 12px;border-radius:4px;display:inline-block}"""
    r = plan["reference"]; t = plan["target"]
    rows = "".join(
        f"<tr><td>{i}</td><td><b>{_e(b['id'])}</b></td><td class='line'>“{_e(b['line'])}”<br><span class='meta'>{b['words']} words</span></td>"
        f"<td>{_e(b.get('cutaway') or '—')}</td><td class='p'>{_e(b['storyboardPrompt'])}</td></tr>"
        for i, b in enumerate(plan["beats"]))
    stills = "".join(f"<tr><td><b>{_e(k)}</b></td><td class='p'>{_e(v)}</td></tr>" for k, v in plan["stills"].items())
    return f"""<!doctype html><meta charset='utf-8'><title>Answer contract · {_e(plan['brand'])} {_e(plan['id'])}</title><style>{css}</style><main>
<h1>The render contract: {_e(plan['ourBrand'])}'s answer to {_e(plan['brand'])}'s ad</h1>
<p class='meta'>Reference: Library ID {_e(plan['id'])} · {_e(r['aspect'])} · {r['seconds']} s · {_e(' → '.join(s.get('segment', '') for s in r['structure']) or 'no structure read')}
 · cutaways: {'yes' if t['cutaways'] else 'no'} · captions: {_e(t['captions'])}. Ours mirrors that shape with our presenter and our script; nothing of theirs is reused.</p>
<div class='cost'><b>Route</b> {_e(plan['route'])} · <b>{len(plan['beats'])} beats × {plan['creditsPerBeat']} = {plan['totalCredits']} credits</b> · target {t['seconds']} s in {t['aspect']}</div>
<p class='warn'>Spend gate: nothing renders until the operator says go. Edit any line first.</p>
<h2>The beats — the exact storyboard prompt per scene (the line is last)</h2>
<table><thead><tr><th>#</th><th>beat</th><th>line (our voice)</th><th>cutaway</th><th>storyboard prompt</th></tr></thead><tbody>{rows}</tbody></table>
<h2>The stills to make first (GB, 9:16, one master for every beat)</h2>
<table><thead><tr><th>still</th><th>prompt</th></tr></thead><tbody>{stills}</tbody></table>
<h2>How it is checked</h2>
<p>Scene 0 first; a 1 fps filmstrip must show the same presenter and desk for all {BEAT_SECONDS} s and whisper (primed with the line) must hear ≥ 80 % of it; only then the rest. A refusal changes one variable (the line first). Assembly QC on the encoded file: 1080×1920, −14 LUFS ±1, captions on a mid-frame grab.</p>
</main>"""


def main() -> int:
    ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
    ap.add_argument("--project", required=True)
    ap.add_argument("--ad", required=True, help="the Library ID (the per-ad artifact under artifacts/ad-intel/)")
    ap.add_argument("--our-brand")
    ap.add_argument("--beats", type=int, default=6)
    ap.add_argument("--voice", choices=("he", "she"), default="he")
    ap.add_argument("--route", default="veo-useapi fast (image-to-video from one master still, native speech)")
    ap.add_argument("--credits-per-beat", type=int, default=10)
    ap.add_argument("--model", default=DEFAULT_MODEL)
    ap.add_argument("--dry-run", action="store_true", help="print the request; read no key; write nothing")
    a = ap.parse_args()

    project = os.path.abspath(a.project)
    ad_path = os.path.join(project, "artifacts", "ad-intel", f"{a.ad}.json")
    if not os.path.isfile(ad_path):
        die(f"no per-ad artifact at {ad_path} — run ad_intel.py first", 2)
    ad = json.load(open(ad_path, encoding="utf-8"))
    analysis = ad.get("analysis") or {}
    if not analysis.get("ugc_script") and not analysis.get("hook"):
        die(f"{ad_path} has no analysis to answer", 2)
    teardowns = sorted((f for f in os.listdir(os.path.join(project, "artifacts", "ad-intel")) if f.endswith("-teardown.json")), reverse=True)
    brand = (ad.get("source") or {}).get("pageName") or "the competitor"
    our_brand = a.our_brand
    if not our_brand and teardowns:
        our_brand = (json.load(open(os.path.join(project, "artifacts", "ad-intel", teardowns[0]))).get("ourBrand"))
    our_brand = our_brand or "us"

    media = ad.get("media") or {}
    ref_file = media.get("file")
    ref_abs = ref_file if ref_file and os.path.isabs(ref_file) else (os.path.join(project, ref_file) if ref_file else "")
    pr = probe(ref_abs) if ref_abs and os.path.isfile(ref_abs) else {"width": 0, "height": 0, "seconds": float(media.get("seconds") or 0)}
    structure = [s for s in (analysis.get("structure") or []) if isinstance(s, dict)]
    aspect = aspect_of(pr["width"], pr["height"])
    beats = beats_from_structure(structure, a.beats)
    cut = wants_cutaways(analysis.get("visual_pattern") or "", structure)
    target_seconds = min(len(beats) * BEAT_SECONDS, round(pr["seconds"]) or len(beats) * BEAT_SECONDS)

    request = script_request(ad, beats, our_brand)
    if a.dry_run:
        print(json.dumps({"dryRun": True, "reference": {"aspect": aspect, "seconds": pr["seconds"], "beats": [b["id"] for b in beats], "cutaways": cut},
                          "request": request[:1200], "model": a.model}, indent=1))
        return 0
    import breakdown
    keys = breakdown.api_keys()
    if not keys:
        die("no Gemini key (GEMINI_API_KEYS / GOOGLE_API_KEY, or .env.local)", 3)
    raw = breakdown.call_model(keys, a.model, [{"text": request}], {}, max_output_tokens=4096)
    if not isinstance(raw, dict):
        die(f"the model returned {type(raw).__name__}, not an object", 4)
    lines = check_lines(beats, raw)

    plan_beats = []
    for b in lines:
        is_tip = b["id"].startswith("tip")
        plan_beats.append({
            "id": b["id"], "line": b["line"], "words": b["words"],
            "cutaway": b["id"] if (cut and is_tip) else None,
            "noun": noun_of(b["line"]) if (cut and is_tip) else None,
            "storyboardPrompt": storyboard_prompt(a.voice, b["line"]),
            "theirSegment": (b.get("from") or {}).get("segment"), "theirNote": (b.get("from") or {}).get("note"),
        })
    stills = {"master": presenter_frame(aspect, analysis.get("visual_pattern") or "", a.voice)}
    for b in plan_beats:
        if b["cutaway"]:
            stills[b["cutaway"]] = cutaway_prompt(aspect, b, our_brand)
    for k, v in stills.items():
        bad = banned(v)
        if bad:
            die(f"still prompt {k} contains a banned prop word {bad} — GB obeys it as a prop", 4)

    plan = {
        "schemaVersion": 1, "id": a.ad, "brand": brand, "ourBrand": our_brand, "generatedAt": now_iso(), "model": a.model,
        "reference": {"file": ref_file, "aspect": aspect, "width": pr["width"], "height": pr["height"], "seconds": pr["seconds"],
                      "structure": structure, "visualPattern": analysis.get("visual_pattern"), "hook": analysis.get("hook")},
        "target": {"aspect": aspect, "seconds": target_seconds, "beatSeconds": BEAT_SECONDS, "cutaways": cut,
                   "captions": "burned, large, phrase-level (the social-ad default; --no-subs at assembly to drop)"},
        "voice": a.voice, "beats": plan_beats, "stills": stills,
        "route": a.route, "creditsPerBeat": a.credits_per_beat, "totalCredits": a.credits_per_beat * len(plan_beats),
        "next": [
            "make the stills: gen-image --backend gobananas --aspect " + aspect + " --kind prop|screen --confirm-spend (one master, one per cutaway)",
            "vclaw video init/brief (--aspect-ratio " + aspect + "), set-execution-profile --veo-model fast, project.json standingRenderRules:false",
            "vclaw video storyboard --scene <storyboardPrompt> ... (one per beat, in order); assets --asset image:<master>:<i> for every beat",
            "produce --scene 0 --confirm-spend, QC (filmstrip + whisper), then the rest; assemble_answer.py --script this plan.json",
        ],
    }
    out_dir = os.path.join(project, "artifacts", "ad-intel", "answer", a.ad); os.makedirs(out_dir, exist_ok=True)
    plan_path = os.path.join(out_dir, "plan.json")
    tmp = plan_path + ".tmp"
    with open(tmp, "w", encoding="utf-8") as fh:
        json.dump(plan, fh, indent=1)
    os.replace(tmp, plan_path)
    page_dir = os.path.join(project, "build", "ad-intel"); os.makedirs(page_dir, exist_ok=True)
    page_path = os.path.join(page_dir, f"answer-{a.ad}.html")
    with open(page_path, "w", encoding="utf-8") as fh:
        fh.write(render(plan))
    print(json.dumps({"plan": plan_path, "page": page_path, "beats": len(plan_beats), "aspect": aspect, "totalCredits": plan["totalCredits"]}))
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
