#!/usr/bin/env python3
"""adlib_fetch.py — the public Meta Ad Library → ad cards + their videos, with OUR browser.

No third-party API, no login, no cookies. The Ad Library search page is public and
renders every card's Library ID, start date, page name, primary text, headline, CTA,
landing link and a signed public-CDN mp4. This script opens that page in the cloak
browser the seedance-direct engine already ships (cloakbrowser over Playwright, in
`engines/seedance-direct/.venv`), evaluates `lib/adlib_extract.js` in the page, and
downloads each card's video through the same session. `ad_intel.py` reads the result.

Usage:
  adlib_fetch.py (--query <brand or keyword> | --advertiser <page name> | --id <libraryId>)
                 [--country GB] [--media video|all] [--max 30] [--out <dir>]
                 [--headed] [--dry-run]
Exit 0 · 2 bad inputs · 3 the engine venv is missing (run engines/seedance-direct/bootstrap.sh)
     · 4 the page gated us or the DOM changed (a login wall, or zero "Library ID" nodes)

Its own profile, ALWAYS: `HIGGS_VCLAW_ADLIB_PROFILE` (default
`$VCLAW_WORKSPACE/tools/ad-intel/.cloak-profile`), never the logged-in Higgsfield profile —
Chromium allows one instance per profile, so sharing it would collide with a running render,
and the Ad Library must never see the Higgsfield session's cookies. Headless by default;
`HIGGS_VCLAW_HEADLESS=0` (or --headed) opens a window.

What it never does: log in, import cookies, click through anything, bypass a gate. If the
page shows a login wall or no card, it says so and exits 4 — the DOM changed or Meta gated
the page; nothing is guessed.
"""
from __future__ import annotations

import argparse
import datetime as dt
import hashlib
import importlib.util
import json
import os
import sys
import re
import time
import types
from urllib.parse import quote

HERE = os.path.dirname(os.path.abspath(__file__))
REPO = os.path.abspath(os.path.join(HERE, "..", "..", ".."))
ENGINE_DIR = os.environ.get("HIGGS_VCLAW_ENGINE_DIR") or os.path.join(REPO, "engines", "seedance-direct")
ENGINE_PY = os.path.join(ENGINE_DIR, ".venv", "bin", "python")
EXTRACT_JS = os.path.join(HERE, "lib", "adlib_extract.js")
AD_LIBRARY = "https://www.facebook.com/ads/library/"
REFERER = "https://www.facebook.com/"
CARD_FIELDS = ["libraryId", "status", "startedOn", "versions", "pageName", "pageUrl", "primaryText",
               "headline", "description", "cta", "landingUrl", "videoUrl", "posterUrl", "imageUrl"]
WAIT_FOR_CARDS_S = 45
SCROLL_ROUNDS_MAX = 40
SCROLL_SETTLE_S = 2.5


def say(msg: str) -> None:
    print(f"adlib_fetch: {msg}", file=sys.stderr)


def workspace_root() -> str:
    return (os.environ.get("VCLAW_WORKSPACE") or os.environ.get("VIDEOCLAW_WORKSPACE")
            or os.path.join(os.path.expanduser("~"), "videoclaw"))


def resolve_profile(explicit: str | None = None) -> str:
    """The profile this script may use. Never the engine's Higgsfield profile."""
    p = explicit or os.environ.get("HIGGS_VCLAW_ADLIB_PROFILE") \
        or os.path.join(workspace_root(), "tools", "ad-intel", ".cloak-profile")
    p = os.path.abspath(os.path.expanduser(p))
    forbidden = [os.path.abspath(os.path.join(ENGINE_DIR, ".cloak-profile"))]
    if os.environ.get("HIGGS_VCLAW_PROFILE"):
        forbidden.append(os.path.abspath(os.path.expanduser(os.environ["HIGGS_VCLAW_PROFILE"])))
    for f in forbidden:
        if os.path.realpath(p) == os.path.realpath(f):
            raise SystemExit(
                f"adlib_fetch: refusing to use the Higgsfield profile ({p}); set "
                f"HIGGS_VCLAW_ADLIB_PROFILE to a profile of its own")
    return p


def search_url(query: str | None, library_id: str | None, country: str, media: str) -> str:
    if library_id:
        return f"{AD_LIBRARY}?id={quote(library_id)}"
    url = (f"{AD_LIBRARY}?active_status=active&ad_type=all&country={quote(country)}"
           f"&q={quote(query or '')}&search_type=keyword_unordered")
    if media == "video":
        url += "&media_type=video"
    return url


def unwrap_landing(href: str | None) -> str | None:
    """`https://l.facebook.com/l.php?u=<encoded>&h=…` → the destination; anything else as is."""
    if not href:
        return None
    from urllib.parse import parse_qs, unquote, urlparse
    try:
        u = urlparse(href)
    except ValueError:
        return href
    if (u.hostname or "").endswith("l.facebook.com") and u.path == "/l.php":
        target = parse_qs(u.query).get("u", [None])[0]
        return unquote(target) if target else href
    return href


def parse_card_lines(lines: list[str]) -> dict:
    """The text fields of one card from its innerText lines — the python twin of the
    walk in lib/adlib_extract.js (same order, same rules), so a fixture can pin both."""
    ls = [l.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "").strip() for l in lines]
    ls = [l for l in ls if l]
    text = "\n".join(ls)
    lib = re.search(r"Library ID: (\d+)", text)
    started = re.search(r"Started running on ([^\n]+)", text)
    versions = re.search(r"(\d+) ads use this creative", text)
    status = ls[0] if ls and ls[0] in ("Active", "Inactive") else None
    sponsored_at = ls.index("Sponsored") if "Sponsored" in ls else -1
    page_name = ls[sponsored_at - 1] if sponsored_at > 0 else None
    timer_at = next((i for i, l in enumerate(ls) if i > sponsored_at and re.match(r"^\d+:\d\d / \d+:\d\d$", l)), -1)
    body_end = timer_at if timer_at > 0 else len(ls)
    primary = "\n".join(ls[sponsored_at + 1:body_end]) if sponsored_at >= 0 else None
    after = ls[timer_at + 1:] if timer_at > 0 else []
    cta = after[-1] if after and len(after[-1]) <= 40 else None
    headline = after[0] if len(after) > 1 else None
    description = "\n".join(after[1:-1]) if len(after) > 2 else None
    return {"libraryId": lib.group(1) if lib else None, "status": status,
            "startedOn": started.group(1).strip() if started else None,
            "versions": int(versions.group(1)) if versions else 1, "pageName": page_name,
            "primaryText": primary, "headline": headline, "description": description, "cta": cta}


def dedupe_cards(cards: list[dict]) -> list[dict]:
    seen, out = set(), []
    for c in cards:
        k = c.get("libraryId")
        if k and k not in seen:
            seen.add(k)
            out.append(c)
    return out


def slugify(s: str) -> str:
    return "".join(c if c.isalnum() else "-" for c in s.lower()).strip("-")[:60] or "ads"


def load_engine():
    """engines/seedance-direct/adapter.py, imported for its launch helpers only (stdlib at
    import; the way preflight_higgs.load_adapter() does it)."""
    path = os.path.join(ENGINE_DIR, "adapter.py")
    spec = importlib.util.spec_from_file_location("seedance_direct_adapter", path)
    mod = importlib.util.module_from_spec(spec)
    spec.loader.exec_module(mod)
    for name in ("_clear_stale_singleton", "_profile_seed", "HiggsSession"):
        if not hasattr(mod, name):
            raise SystemExit(f"adlib_fetch: {path} has no {name}; the engine moved under us")
    return mod


def sha256_file(path: str) -> str:
    h = hashlib.sha256()
    with open(path, "rb") as fh:
        for chunk in iter(lambda: fh.read(1 << 20), b""):
            h.update(chunk)
    return h.hexdigest()


def write_json_atomic(path: str, doc: dict) -> None:
    tmp = f"{path}.tmp.{os.getpid()}"
    os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
    with open(tmp, "w", encoding="utf-8") as fh:
        json.dump(doc, fh, indent=1, ensure_ascii=False)
    os.replace(tmp, path)


def ensure_engine_python(argv: list[str]) -> None:
    """Re-exec under the engine venv (cloakbrowser + playwright live there)."""
    if os.environ.get("ADLIB_FETCH_IN_ENGINE") == "1":
        return
    if not os.path.isfile(ENGINE_PY):
        say(f"the cloak browser's venv is missing at {ENGINE_PY}")
        say(f"run: bash {os.path.join(ENGINE_DIR, 'bootstrap.sh')}   (installs cloakbrowser + playwright chromium)")
        raise SystemExit(3)
    os.environ["ADLIB_FETCH_IN_ENGINE"] = "1"
    os.execv(ENGINE_PY, [ENGINE_PY, os.path.abspath(__file__)] + argv)


class AdLibrarySession:
    """A cloak-browser session on the ad-intel profile. The launch is the engine's own
    code (its singleton clear, its fingerprint seed, its cross-process profile lock)."""

    def __init__(self, engine, profile: str, headless: bool, diag_dir: str = "."):
        from cloakbrowser import launch_persistent_context  # engine venv only
        self.engine = engine
        self.diag_dir = diag_dir
        self.holder = types.SimpleNamespace(_lock_fd=None)
        os.makedirs(os.path.dirname(profile.rstrip("/")) or ".", exist_ok=True)  # the lock sits beside the profile
        self.holder._lock_fd = engine.HiggsSession._acquire_profile_lock(self.holder, profile)
        engine._clear_stale_singleton(profile)
        os.makedirs(profile, exist_ok=True)
        # bypass_csp: facebook.com's Content-Security-Policy forbids 'unsafe-eval', and
        # Playwright delivers every evaluate() through eval — without this the walk cannot
        # run at all (EvalError on the first wait_for_function, 2026-09-06). It only lets OUR
        # extractor run in the page; it reads nothing the page does not already show.
        self.ctx = launch_persistent_context(
            profile, headless=headless, viewport={"width": 1280, "height": 900},
            args=[f"--fingerprint={engine._profile_seed(profile)}"], bypass_csp=True,
        )
        self.page = self.ctx.pages[0] if self.ctx.pages else self.ctx.new_page()

    def close(self) -> None:
        try:
            self.ctx.close()
        except Exception:  # noqa: BLE001 — closing is best effort
            pass
        self.engine.HiggsSession._release_profile_lock(self.holder)

    def open_library(self, url: str, want: int) -> int:
        self.page.goto(url, wait_until="domcontentloaded", timeout=60000)
        try:
            self.page.wait_for_function(
                "document.body && /Library ID: \\d+/.test(document.body.innerText)",
                timeout=WAIT_FOR_CARDS_S * 1000)
        except Exception as e:  # noqa: BLE001 — timed out: diagnose, do not guess
            say(f"wait failed: {type(e).__name__}: {str(e)[:300]}")
            try:
                say(f"page now: {self.page.url} | title {self.page.title()!r}")
                say("text head: " + self.page.evaluate("document.body ? document.body.innerText.slice(0, 400).replace(/\\n+/g, ' | ') : '(no body)'"))
                self.page.screenshot(path=os.path.join(self.diag_dir, "adlib-fetch-failed.png"))
                say(f"screenshot: {os.path.join(self.diag_dir, 'adlib-fetch-failed.png')}")
            except Exception:  # noqa: BLE001
                pass
            wall = self.page.evaluate(
                "!!document.querySelector('input[name=email], form[action*=login]')")
            if wall:
                say("the Ad Library showed a LOGIN WALL — this script never logs in; try again "
                    "later or from another network")
            else:
                say("no 'Library ID' text appeared within the wait — the DOM changed, the query "
                    "has no results, or the page was gated")
            raise SystemExit(4)
        count = self.count_cards()
        stalled = 0
        for _ in range(SCROLL_ROUNDS_MAX):
            if count >= want:
                break
            self.page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
            time.sleep(SCROLL_SETTLE_S)
            new = self.count_cards()
            stalled = stalled + 1 if new == count else 0
            count = new
            if stalled >= 3:
                break
        return count

    def count_cards(self) -> int:
        return int(self.page.evaluate("(document.body.innerText.match(/Library ID: \\d+/g) || []).length"))

    def extract(self) -> list[dict]:
        js = open(EXTRACT_JS, encoding="utf-8").read()
        return self.page.evaluate(js)

    def download(self, url: str, dest: str) -> dict:
        """The video through the session's own request context (its cookies, its UA),
        with the Referer the CDN expects. The signed URL expires; the file does not."""
        resp = self.ctx.request.get(url, headers={"Referer": REFERER}, timeout=120000)
        if not resp.ok:
            return {"ok": False, "status": resp.status}
        body = resp.body()
        if len(body) < 10_000 or b"ftyp" not in body[:64]:
            return {"ok": False, "status": resp.status, "reason": "not an mp4"}
        tmp = dest + ".part"
        with open(tmp, "wb") as fh:
            fh.write(body)
        os.replace(tmp, dest)
        return {"ok": True, "status": resp.status, "bytes": len(body)}


def main() -> int:
    ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
    who = ap.add_mutually_exclusive_group()
    who.add_argument("--query", help="brand or keyword, searched across all advertisers")
    who.add_argument("--advertiser", help="page name; searches it and keeps only that page's cards")
    who.add_argument("--id", dest="library_id", help="one Library ID (the ?id= page)")
    ap.add_argument("--country", default="GB")
    ap.add_argument("--media", choices=("video", "all"), default="video")
    ap.add_argument("--max", type=int, default=30)
    ap.add_argument("--out", help="default $VCLAW_WORKSPACE/tools/ad-intel/fetch/<query>")
    ap.add_argument("--profile", help="override HIGGS_VCLAW_ADLIB_PROFILE")
    ap.add_argument("--headed", action="store_true")
    ap.add_argument("--dry-run", action="store_true", help="print the URL, profile and the extractor's contract; open nothing")
    ap.add_argument("--test-extract", metavar="HTML", help="load a saved page/card HTML into the browser and print what lib/adlib_extract.js reads from it (no network)")
    args = ap.parse_args()

    if args.test_extract:
        ensure_engine_python(sys.argv[1:])
        engine = load_engine()
        profile = resolve_profile(args.profile)
        sess = AdLibrarySession(engine, profile, headless=True)
        try:
            # a saved page: no network at all — its CDN URLs are placeholders that would
            # otherwise hold the load event open until the timeout
            sess.ctx.route("http://**", lambda route: route.abort())
            sess.ctx.route("https://**", lambda route: route.abort())
            sess.page.goto("file://" + os.path.abspath(args.test_extract), wait_until="domcontentloaded", timeout=30000)
            cards = sess.extract()
        finally:
            sess.close()
        print(json.dumps(cards, ensure_ascii=False))
        return 0

    if not (args.query or args.advertiser or args.library_id):
        ap.error("one of --query, --advertiser, --id is required")
    query = args.query or args.advertiser
    label = args.library_id or query
    url = search_url(query, args.library_id, args.country, args.media)
    profile = resolve_profile(args.profile)
    out = os.path.abspath(args.out or os.path.join(workspace_root(), "tools", "ad-intel", "fetch", slugify(label)))
    headless = not args.headed and os.environ.get("HIGGS_VCLAW_HEADLESS", "1") != "0"

    if args.dry_run:
        print(json.dumps({"url": url, "profile": profile, "engineVenv": ENGINE_PY,
                          "engineVenvPresent": os.path.isfile(ENGINE_PY), "out": out,
                          "headless": headless, "cardFields": CARD_FIELDS,
                          "downloads": "each card's videoUrl via the session, Referer facebook.com",
                          "neverDoes": ["log in", "import cookies", "use the Higgsfield profile"]}, indent=1))
        return 0

    ensure_engine_python(sys.argv[1:])
    engine = load_engine()
    os.makedirs(out, exist_ok=True)
    say(f"opening {url}")
    sess = AdLibrarySession(engine, profile, headless, diag_dir=out)
    try:
        seen = sess.open_library(url, args.max)
        say(f"{seen} card(s) on the page after scrolling")
        cards = sess.extract()
        if args.advertiser:
            needle = args.advertiser.lower()
            cards = [c for c in cards if (c.get("pageName") or "").lower().find(needle) >= 0]
        if args.library_id:
            cards = [c for c in cards if c.get("libraryId") == args.library_id] or cards[:1]
        if args.media == "video":
            cards = [c for c in cards if c.get("videoUrl")]
        cards = cards[: args.max]
        if not cards:
            say("the page rendered but the extractor found no card — the DOM changed or the "
                "filter left nothing; nothing was guessed")
            return 4
        got = 0
        for c in cards:
            c.pop("lines", None)
            if not c.get("videoUrl"):
                continue
            dest = os.path.join(out, f"{c['libraryId']}.mp4")
            side = dest + ".sha256"
            if os.path.isfile(dest) and os.path.isfile(side):
                c["videoFile"] = os.path.basename(dest)
                c["videoSha256"] = open(side, encoding="utf-8").read().strip()
                got += 1
                continue
            r = sess.download(c["videoUrl"], dest)
            if r.get("ok"):
                sha = sha256_file(dest)
                open(side, "w", encoding="utf-8").write(sha + "\n")
                c["videoFile"] = os.path.basename(dest)
                c["videoSha256"] = sha
                got += 1
            else:
                c["videoDownload"] = r
                say(f"{c['libraryId']}: video download failed ({r})")
    finally:
        sess.close()

    doc = {"schemaVersion": 1, "fetchedAt": dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds"),
           "query": query, "advertiser": args.advertiser, "libraryId": args.library_id,
           "country": args.country, "media": args.media, "url": url, "cards": cards}
    path = os.path.join(out, "cards.json")
    write_json_atomic(path, doc)
    print(json.dumps({"cards": path, "count": len(cards), "videos": got, "out": out}))
    return 0


if __name__ == "__main__":
    sys.exit(main())
