#!/usr/bin/env python3
"""Capture how the Higgsfield UI uploads a VIDEO, by watching you do it once.

WHY THIS EXISTS
---------------
higgs-cloak solved image upload and never solved video upload. Its own
`provision_assets.py` says so, verbatim:

    VOICES — DOCUMENTED, not yet automated. In every confirmed-working run, the
    cloned-voice mp4 was uploaded via the Higgsfield UI "Upload media" drawer…
    We have NOT reverse-engineered a pure-API path that produces a voice media
    usable as @Video N… That exact request shape was never captured (the UI did
    it). Capturing it (HAR the drawer upload of one voice mp4) is the remaining
    work.

`grep -r 'video/mp4'` across that entire repo returns nothing. Every working
voice run REPLAYED media ids that a human had produced by drag-and-drop. That is
fine for a cartoon series reusing five character voices; it does not scale to a
rap MV needing a different 8-second voice slice per window.

So: drive the real UI once, record exactly what it sends, and read the answer off
the wire instead of guessing it.

THREE UNKNOWNS THIS ANSWERS
---------------------------
1. `surface` + mimetype for a video on POST /media/batch. Known-good for PNG is
   surface="seedance_2"; the video equivalent was never observed.
2. Whether /media/batch returns a DICT for video where it returns a LIST for
   images — `upload_image` indexes `[0]`, which would crash on a dict.
3. The finalize shape that makes the media come back as type "video_input"
   rather than the "media_input" images produce. That type is what lets the
   prompt cite it as @Video N.

RUNNING IT
----------
    ./.venv/bin/python bootstrap/capture_voice_upload.py --minutes 15

A real Chrome window opens on the logged-in profile. Then, by hand:

    1. Go to the video tool and open the "Upload media" drawer.
    2. Drag in ONE voice .mp4 (>= 640x640 — ours are 720x1280, so fine).
    3. Attach it as a reference and start ONE generation.

The result does not matter. Only the requests do. Everything is written
incrementally, so killing it early still leaves usable data.

SAFETY
------
Read-only with respect to our code: it observes, it never submits on its own.
Credential values (authorization, cookie, clerk headers, session tokens) are
redacted at write time — never on disk, per the repo's no-secrets rule. Header
NAMES are kept, because which headers are present is part of the contract.
"""

from __future__ import annotations

import argparse
import json
import os
import re
import sys
import time

HERE = os.path.dirname(os.path.abspath(__file__))
ENGINE_DIR = os.path.dirname(HERE)
sys.path.insert(0, ENGINE_DIR)

# Reuse the engine's own constants so this cannot drift from what the adapter
# actually talks to.
from adapter import API_BASE, APP_URL, DEFAULT_PROFILE  # noqa: E402

# What counts as worth recording.
#
# The first cut of this matched any path containing "/assets", which also matches
# `assets.higgsfield.ai/tanstack/assets/*.css` — the app's static bundle. That
# buried the four calls that matter under ~200 CSS/JS fetches and made the
# recorder pull multi-MB JS bodies. Anchor on the API HOST instead of on path
# words, and never record a static-asset host.
STATIC_HOSTS = ("assets.higgsfield.ai", "fonts.g", "cdn.", "analytics.",
                "sentry.", "segment.", "posthog", "intercom", "stripe.com")
STATIC_EXT = re.compile(r"\.(css|js|mjs|woff2?|ttf|png|jpe?g|gif|svg|ico|map)(\?|$)", re.I)

# Paths on the API host we care about.
API_PATHS = re.compile(r"/(media|jobs|reference-elements|upload|wallet|unlim)", re.I)

# A signed storage PUT is the middle step of the upload and lives on a different
# host (S3/GCS/R2), so the host check alone would miss it.
SIGNED_UPLOAD = re.compile(
    r"(amazonaws\.com|storage\.googleapis\.com|r2\.cloudflarestorage\.com|blob\.core\.windows\.net)", re.I
)


# Pure analytics/marketing. Excluded by host because they are xhr/fetch too.
NOISE_HOSTS = ("googletagmanager", "google-analytics", "firstpromoter",
               "sentry.", "segment.", "posthog", "intercom", "doubleclick",
               "facebook.", "instagram.", "hotjar", "mixpanel", "amplitude")


def is_interesting(url, method, resource_type=None):
    """Record by REQUEST TYPE, not by guessed host.

    The previous version allowlisted `fnf.higgsfield.ai` — the host the ADAPTER
    talks to. But the web app's own API host is not in the page HTML (it is
    inside the JS bundle), so allowlisting it was a guess, and it captured
    nothing in ten minutes of an open app. Guessing is exactly the wrong
    approach when missing the single upload the user performs costs another
    manual round-trip.

    Every API call a browser app makes is xhr or fetch; stylesheets, scripts,
    fonts and images are their own resource types. So keying on the type both
    excludes the static flood (structurally, not by pattern) and cannot miss an
    API call regardless of which host serves it.
    """
    lowered = url.lower()
    if any(h in lowered for h in NOISE_HOSTS):
        return False
    # This app is TanStack, which PRELOADS its JS modules via fetch() — so the
    # static bundle arrives typed as "fetch", not "script", and the type check
    # alone let ~800 module fetches back in. Excluding the asset host and file
    # extensions closes that without re-introducing a host allowlist for the
    # calls we actually want.
    if any(h in lowered for h in STATIC_HOSTS) or STATIC_EXT.search(lowered):
        return False
    if resource_type in ("xhr", "fetch"):
        return True
    # The signed upload PUT carrying the actual bytes can be typed as "other".
    if SIGNED_UPLOAD.search(lowered) and method in ("PUT", "POST"):
        return True
    # Belt and braces: a media/jobs path on any host, whatever its type.
    if API_PATHS.search(lowered) and not STATIC_EXT.search(lowered) \
            and not any(h in lowered for h in STATIC_HOSTS):
        return True
    return False

# Redacted by NAME, keeping the name visible. Which headers the UI sends is part
# of the contract we are reverse-engineering; their values are secrets.
SECRET_HEADERS = {
    "authorization", "cookie", "set-cookie", "x-clerk-db-jwt",
    "x-clerk-session", "proxy-authorization", "x-api-key",
}
# JWTs and Clerk/session tokens that appear in BODIES rather than headers.
SECRET_PATTERNS = [
    (re.compile(r"eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]+"), "<JWT_REDACTED>"),
    (re.compile(r"(sess|client|__session)_[A-Za-z0-9]{16,}"), "<SESSION_REDACTED>"),
]
# Signed-upload URLs carry the credential in the QUERY STRING. The path shape is
# what we need; the signature is not.
#
# TWO BUGS FIXED HERE, both found in review after the first real capture:
#
# 1. The value class was `[^&]*`, which also matches `"`, `}` and newlines — so on
#    a JSON body it consumed everything from the signature to the next `&` or EOF.
#    That silently DESTROYED the presign response, the single most important
#    exchange this tool exists to record. It fails safe (over-redacts, never
#    under-redacts) which is why it went unnoticed: the 2026-08-07 capture is
#    truncated mid-string, and the fields we needed survived only because
#    X-Amz-Signature happened to be the last parameter. Excluding quote and
#    whitespace keeps the redaction and leaves the JSON parseable.
#
# 2. Only AWS SigV4 names were covered. `storage.googleapis.com` is explicitly in
#    SIGNED_UPLOAD below, so a GCS-backed presign IS captured — and its
#    `X-Goog-Signature` would have been written verbatim. Same for
#    access_token / refresh_token / id_token, which the bare `token` alternative
#    did not match (it required `[?&]token=` exactly). Latent, not live: Higgsfield
#    uses CloudFront + SigV4. Closed anyway, because "the provider we use today
#    doesn't do that" is not a security property.
# Names are listed EXPLICITLY rather than pattern-matched. A fuzzy
# `[\w-]*(sig|key|token)[\w-]*` also swallows `X-Amz-SignedHeaders`, whose value
# (`content-type;host`) is precisely what proved the PUT must send
# `Content-Type: video/mp4`. Over-redacting the contract defeats the tool as
# surely as under-redacting leaks a secret.
SIGNED_PARAMS = re.compile(
    r"([?&]("
    r"X-Amz-Signature|X-Amz-Credential|X-Amz-Security-Token|"
    r"X-Goog-Signature|X-Goog-Credential|GoogleAccessId|"
    r"Signature|AWSAccessKeyId|"
    r"access[_-]?token|refresh[_-]?token|id[_-]?token|"
    r"api[_-]?key|token|key|sig"
    r")=)[^&\"'\s]*",
    re.I,
)


def scrub_text(s):
    """Strip credential material from an arbitrary string."""
    if not s:
        return s
    for pattern, replacement in SECRET_PATTERNS:
        s = pattern.sub(replacement, s)
    return SIGNED_PARAMS.sub(r"\1<REDACTED>", s)


def scrub_headers(headers):
    return {
        k: ("<REDACTED>" if k.lower() in SECRET_HEADERS else scrub_text(v))
        for k, v in (headers or {}).items()
    }


def body_preview(raw, limit=20000):
    """Decode a body for the record, keeping JSON structured where possible.

    Video uploads are megabytes of binary; recording that is useless and would
    bloat the capture past what a human or a spec generator can read. We keep
    the SHAPE — enough to see the fields — and note what was elided.
    """
    if raw is None:
        return None
    if isinstance(raw, bytes):
        if len(raw) > limit:
            return {"_binary": True, "_bytes": len(raw),
                    "_note": "body elided — size is the point, not the content"}
        try:
            raw = raw.decode("utf-8")
        except UnicodeDecodeError:
            return {"_binary": True, "_bytes": len(raw)}
    if len(raw) > limit:
        return {"_truncated": True, "_chars": len(raw), "head": scrub_text(raw[:limit])}
    try:
        return json.loads(scrub_text(raw))
    except (json.JSONDecodeError, TypeError):
        return scrub_text(raw)


def main():
    ap = argparse.ArgumentParser(description=__doc__,
                                 formatter_class=argparse.RawDescriptionHelpFormatter)
    ap.add_argument("--minutes", type=float, default=15.0,
                    help="how long to record before writing and closing (default 15)")
    ap.add_argument("--profile", default=DEFAULT_PROFILE,
                    help="cloak profile to open (default: the engine's live one)")
    ap.add_argument("--out", default=os.path.join(HERE, "voice-upload-capture.json"),
                    help="where to write the capture")
    ap.add_argument("--har", default=os.path.join(HERE, "voice-upload-capture.har"),
                    help="HAR for `/printing-press --har <path> --name higgsfield`")
    args = ap.parse_args()

    entries = []
    pending = {}

    def note(msg):
        print(f"[capture] {msg}", flush=True)

    def flush():
        """Write after every interesting exchange, so an early kill still helps."""
        with open(args.out, "w") as fh:
            json.dump({"apiBase": API_BASE, "appUrl": APP_URL,
                       "capturedEntries": len(entries), "entries": entries}, fh, indent=2)
        har = {"log": {"version": "1.2",
                       "creator": {"name": "capture_voice_upload", "version": "1"},
                       "entries": [_har_entry(e) for e in entries]}}
        with open(args.har, "w") as fh:
            json.dump(har, fh, indent=2)

    def _har_entry(e):
        return {
            "startedDateTime": e["startedDateTime"],
            "time": 0,
            "request": {
                "method": e["method"], "url": e["url"], "httpVersion": "HTTP/1.1",
                "cookies": [], "queryString": [],
                "headers": [{"name": k, "value": v}
                            for k, v in e.get("requestHeaders", {}).items()],
                "headersSize": -1, "bodySize": -1,
                "postData": ({"mimeType": e.get("requestContentType") or "application/json",
                              "text": json.dumps(e["requestBody"])
                              if not isinstance(e["requestBody"], str) else e["requestBody"]}
                             if e.get("requestBody") is not None else None),
            },
            "response": {
                "status": e.get("status", 0), "statusText": "", "httpVersion": "HTTP/1.1",
                "cookies": [], "headers": [{"name": k, "value": v}
                                           for k, v in e.get("responseHeaders", {}).items()],
                "content": {"size": -1,
                            "mimeType": e.get("responseContentType") or "application/json",
                            "text": json.dumps(e.get("responseBody"))},
                "redirectURL": "", "headersSize": -1, "bodySize": -1,
            },
            "cache": {}, "timings": {"send": 0, "wait": 0, "receive": 0},
        }

    note(f"opening {args.profile}")
    note("a real Chrome window will appear — it is NOT headless, that is deliberate")

    from cloakbrowser import launch_persistent_context
    ctx = launch_persistent_context(args.profile, headless=False,
                                    viewport={"width": 1440, "height": 900})
    page = ctx.pages[0] if ctx.pages else ctx.new_page()

    def on_request(req):
        if not is_interesting(req.url, req.method, req.resource_type):
            return
        try:
            post = req.post_data_buffer
        except Exception:
            post = None
        pending[req] = {
            "startedDateTime": time.strftime("%Y-%m-%dT%H:%M:%S"),
            "method": req.method,
            "url": scrub_text(req.url),
            "resourceType": req.resource_type,
            "requestHeaders": scrub_headers(req.headers),
            "requestContentType": (req.headers or {}).get("content-type"),
            "requestBody": body_preview(post),
        }
        note(f"-> {req.method} {scrub_text(req.url)[:120]}")

    def on_response(resp):
        entry = pending.pop(resp.request, None)
        if entry is None:
            return
        entry["status"] = resp.status
        entry["responseHeaders"] = scrub_headers(resp.headers)
        entry["responseContentType"] = (resp.headers or {}).get("content-type")
        try:
            entry["responseBody"] = body_preview(resp.body())
        except Exception as exc:
            entry["responseBody"] = {"_unavailable": str(exc)}
        entries.append(entry)
        note(f"<- {resp.status} {scrub_text(resp.url)[:100]}")
        flush()

    # Listeners attach BEFORE navigation so the app's own startup calls (which
    # include the media listing shapes) are captured too.
    page.on("request", on_request)
    page.on("response", on_response)

    page.goto(APP_URL, wait_until="domcontentloaded", timeout=60000)

    deadline = time.time() + args.minutes * 60
    note("=" * 68)
    note("NOW, IN THE BROWSER WINDOW:")
    note("  1. open the 'Upload media' drawer")
    note("  2. drag in ONE voice .mp4")
    note("  3. attach it as a reference and start ONE generation")
    note("The output does not matter. Only the requests do.")
    note(f"Recording for {args.minutes:g} minutes, writing to:")
    note(f"  {args.out}")
    note("=" * 68)

    try:
        while time.time() < deadline:
            page.wait_for_timeout(5000)
    except KeyboardInterrupt:
        note("interrupted — writing what we have")
    except Exception as exc:
        # A closed window is a normal way to end the capture, not a failure.
        note(f"page ended ({type(exc).__name__}) — writing what we have")

    flush()
    note(f"captured {len(entries)} exchanges -> {args.out}")
    note(f"HAR -> {args.har}")

    # What the whole exercise was for. Surfaced here so the answer is visible
    # without reading the JSON.
    media = [e for e in entries if "/media" in e["url"]]
    if media:
        note("")
        note("MEDIA CALLS (the ones that matter):")
        for e in media:
            shape = type(e.get("responseBody")).__name__
            note(f"  {e['method']:6} {e['url'][:88]}  -> {e.get('status')} ({shape})")
    else:
        note("")
        note("NO /media calls captured — the upload probably did not happen.")
        note("Re-run and make sure you actually drop a file into the drawer.")

    try:
        ctx.close()
    except Exception:
        pass


if __name__ == "__main__":
    main()
