#!/usr/bin/env python3
"""Lint a .brand pack BEFORE anything renders — the cheap half of the design gate.

Every check here exists because a real defect reached a rendered master and was
caught late by the fresh-eyes design reviewer (a ~10 min, post-render pass).
These are the subset that are decidable from the pack alone, in under a second,
so the tool refuses to render wrong instead of a human catching it afterwards.

  E1 tone enum      MEET_ROW tone is a semantic slot (blue|white|ink), never a
                    brand colour. "green" used to crash the engine mid-stills.
  E2 QR agreement   the printed label, the encoded URL and the SPOKEN outro URL
                    must all name the same place. (Shipped once with the label
                    saying apple.com/ai while the code opened a different page.)
  E3 acronym casing an initialism written uppercase somewhere and lowercase on a
                    card reads as a typo in the same mono face ("human vs ai"
                    under a headline reading "A meter for AI").
  W1 echo           a statement subtitle that restates its own caption, or a MY
                    TAKE catch that restates a beat caption, wastes the frame.
  E4 card overflow  cmd_card centres mono 42 in a 690px card and does NOT clip.
                    Past 674px the glyphs reach the rounded corners (error);
                    past the 606px padding budget it is merely tight (advisory).
  W2 icon overflow  icon_h is a HEIGHT; a wide glyph (aspect > ~2:1) at a normal
                    height overshoots the column and collides with the cards.
                    Needs the sliced icons, so it only runs post-slice.

Usage:  lint_pack.py [<pack-path>] [--icons <dir>] [--strict]
        (pack path defaults to $BRAND_PACK)
Exit:   0 clean (warnings allowed), 1 errors — or warnings too under --strict.
"""
import os
import re
import sys

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from packlib import parse_pack, word_at, not_a_brand_film  # noqa: E402

# geometry (mirrors engine.py: 16x9 design space, icons at LX, cards w=690 @ RX)
LX, RX, CARD_W, EDGE, GUTTER = 470, 1245, 690, 48, 48
ICON_MAX_W = 2 * min(LX - EDGE, (RX - CARD_W / 2 - GUTTER) - LX)  # -> 764
CARD_MONO_PT, CARD_PAD = 42, 42                # cmd_card: mono(42), 42px each side
CARD_TEXT_W = CARD_W - 2 * CARD_PAD            # -> 606 design px: the intended padding
CARD_EDGE_W = CARD_W - 2 * 8                   # -> 674: past this the glyphs hit the corner

TONES = {"blue", "white", "ink"}
MONTHS = (
    "January", "February", "March", "April", "May", "June", "July",
    "August", "September", "October", "November", "December",
)
STOP = {
    "a", "an", "the", "is", "are", "was", "it", "its", "and", "or", "but", "of",
    "to", "in", "on", "at", "for", "with", "that", "this", "you", "your", "not",
    "now", "new", "one", "be", "by", "as", "from", "has", "have", "just",
}


def norm_url(u):
    """Compare URLs by what a viewer would read: no scheme, no www, no trailing
    slash, no hyphens (a spoken 'A-I' is the same place as 'ai')."""
    u = u.strip().lower()
    u = re.sub(r"^[a-z]+://", "", u)
    u = re.sub(r"^www\.", "", u)
    return u.rstrip("/").replace("-", "")


def spoken_to_url(line):
    """'apple dot com slash A-I' -> 'apple.com/ai' so a spoken outro can be
    compared with the QR it is promising."""
    s = " " + line.strip().lower() + " "
    s = s.replace(" dot ", ".").replace(" slash ", "/")
    return s.replace("-", "").strip()


def words(s):
    return {w for w in re.findall(r"[a-z]+", s.lower()) if w not in STOP and len(w) > 2}


def overlap(a, b):
    wa, wb = words(a), words(b)
    if not wa or not wb:
        return 0.0
    return len(wa & wb) / min(len(wa), len(wb))


def main():
    args = [a for a in sys.argv[1:]]
    strict = "--strict" in args
    args = [a for a in args if a != "--strict"]
    icons_dir = None
    if "--icons" in args:
        i = args.index("--icons")
        icons_dir = args[i + 1] if i + 1 < len(args) else None
        del args[i:i + 2]
    pack_path = args[0] if args else os.environ.get("BRAND_PACK")
    if not pack_path:
        sys.exit("lint: no pack path (pass one or set BRAND_PACK)")

    os.environ.setdefault("BRAND_PROJECT", os.getcwd())
    P = parse_pack(pack_path)
    skip = not_a_brand_film(pack_path, P)
    if skip:
        print(f"pack lint SKIPPED: {skip}")
        return 0
    errors, warns = [], []

    beats = sorted(
        (k for k in P if re.fullmatch(r"BEAT_\d+", k)),
        key=lambda k: int(k.split("_")[1]),
    )

    # ---- collect every viewer-visible string, keyed by where it came from ----
    fields = []       # (label, text)
    captions = []     # (label, caption) — for the echo check
    for k in beats:
        f = P[k].split("|")
        if len(f) not in (7, 8):
            # do NOT skip quietly: a `~` typed where a `|` belongs silently turns
            # a card into a subtitle, and every later check then reads the wrong
            # field. Seen in the wild (microsoft-mai BEAT_8, 2026-07-25).
            errors.append(
                f"{k}: {len(f)} fields, needs 7 (or 8 with icon_dx) — a `~` typed "
                f"where a `|` belongs turns a card into a subtitle: {P[k][:70]}…"
            )
            continue
        pill, c1, c2, cap = f[2], f[3], f[4], f[5]
        # the KICKER is uppercased by the engine at render, so lowercase there is
        # correct by design — label it so the casing checks can skip it.
        for i, part in enumerate(pill.split("~")):
            fields.append((f"{k} kicker" if i == 0 else f"{k} pill", part))
        fields += [(f"{k} card1", c1), (f"{k} card2", c2), (f"{k} caption", cap)]
        captions.append((k, cap))
    for k in sorted(P):
        if re.fullmatch(r"MEET_ROW_\d+", k):
            r = P[k].split("|")
            fields.append((f"{k} value", r[1] if len(r) > 1 else ""))
            if len(r) > 4:
                fields.append((f"{k} subtitle", r[4]))
    for k in ("TAGLINE", "MEET_SUBTITLE", "OUTRO_CAPTION", "MYTAKE_VERDICT_1",
              "MYTAKE_VERDICT_2", "MYTAKE_BEST_FOR", "MYTAKE_CATCH",
              "MYTAKE_CAPTION"):
        if P.get(k):
            fields.append((k, P[k]))

    # ---- E7 glyph coverage: every character must exist in the face that draws it
    # A REAL BRAND FACE IS NOT A SYSTEM FONT. The fleet's early packs used
    # HelveticaNeue.ttc, which covers everything, so nothing ever tripped this.
    # The moment a pack uses the brand's OWN webfont the coverage gets narrow:
    # cdnjs ships FT Kunst Grotesk from cloudflare.com, and its cmap has NO
    # U+2014, U+2013, U+00B7, U+2019 — or even '%'. An em dash in OUTRO_CAPTION
    # therefore rendered as a .notdef TOFU BOX dead-centre in the last held
    # frame, the one viewers freeze on to scan the QR. Four automated checks and
    # a full QC pass had already cleared that cut; a human caught it by eye.
    # Mono (Apercu) had every one of those glyphs, which is why the SAME
    # characters were fine on cards — so the face each field is drawn in decides
    # the verdict, and this check must follow that split, not the pack order.
    def _cmap(path):
        if not path or not os.path.exists(path):
            return None
        try:
            from fontTools.ttLib import TTFont
            font = TTFont(path, fontNumber=0, lazy=True)
            cm = set()
            for t in font["cmap"].tables:
                cm |= set(t.cmap.keys())
            return cm or None
        except Exception:
            return None      # unreadable/collection face: skip rather than false-alarm

    _sans = _cmap(P.get("FONT_SANS_REGULAR") or P.get("FONT_SANS", ""))
    _mono = _cmap(P.get("FONT_MONO", ""))
    # cmd_card draws card1/card2 in mono(42); the MEET row subtitle is mono(18).
    # Everything else in `fields` — kickers, pills, captions, MEET values,
    # TAGLINE, MYTAKE_*, OUTRO_CAPTION — goes through F(), the sans.
    _is_mono = re.compile(r"(card[12]|MEET_ROW_\d+ subtitle)$")
    for label, val in fields:
        cm = _mono if _is_mono.search(label) else _sans
        if not cm or not val:
            continue
        face = "FONT_MONO" if _is_mono.search(label) else "FONT_SANS"
        # a STATEMENT pill carries MARKUP the engine consumes and never draws:
        # `^` splits the stacked lines and `*...*` wraps the ACCENT line. Checking
        # those as glyphs fails every statement beat in the fleet, so strip them.
        text = val.replace("^", "").replace("*", "") if label.endswith("pill") else val
        missing = sorted({c for c in text if c != " " and ord(c) not in cm})
        if missing:
            shown = " ".join(f"{c!r} (U+{ord(c):04X})" for c in missing)
            errors.append(
                f"{label}: {face} has no glyph for {shown} — it renders as a "
                f".notdef TOFU BOX in the delivered frame. Use a character the face "
                f"has (a plain '-' for a dash), or move the text to the other face."
            )

    # ---- E1 tone enum -------------------------------------------------------
    for k in sorted(P):
        if re.fullmatch(r"MEET_ROW_\d+", k):
            r = P[k].split("|")
            tone = (r[2] if len(r) > 2 else "").strip()
            if tone not in TONES:
                errors.append(
                    f"{k}: tone '{tone}' is not a semantic slot "
                    f"({'|'.join(sorted(TONES))}) — it is a card STYLE, not a brand colour"
                )

    # ---- E2 QR label / URL / spoken outro all name the same place -----------
    qr_url, qr_label = P.get("QR_URL", ""), P.get("QR_URL_LABEL", "")
    if qr_url and qr_label:
        if norm_url(qr_label) != norm_url(qr_url):
            errors.append(
                f"QR mismatch: label '{qr_label}' but the code encodes '{qr_url}' "
                f"— a viewer scans a different page than the one printed"
            )
    n_vo = len([k for k in P if re.fullmatch(r"VO_\d+", k)])
    outro_vo = P.get(f"VO_{n_vo - 1}", "") if n_vo else ""
    if qr_url and outro_vo:
        if norm_url(qr_url) not in spoken_to_url(outro_vo):
            errors.append(
                f"QR vs narration: the outro says \"{outro_vo.strip()}\" but the QR "
                f"encodes '{qr_url}' — voice, label and code must agree"
            )

    # ---- E3 acronym casing --------------------------------------------------
    def strip_urls(t):
        return " ".join(w for w in t.split() if "/" not in w and not re.search(r"\w\.\w", w))

    # kickers render uppercase, so they can never be the lowercase half of a split
    cased = [(lbl, t) for lbl, t in fields if not lbl.endswith("kicker")]

    acronyms = set()
    for _, t in fields:
        acronyms |= set(re.findall(r"\b[A-Z]{2,5}\b", t))
    for a in sorted(acronyms):
        low = a.lower()
        hits = [lbl for lbl, t in cased if re.search(rf"\b{re.escape(low)}\b", strip_urls(t))]
        if hits:
            up = sorted({lbl for lbl, t in fields if re.search(rf"\b{re.escape(a)}\b", t)})
            errors.append(
                f"acronym casing: '{a}' is uppercase in {up[0]} but lowercase "
                f"in {', '.join(hits[:3])} — same face, reads as a typo"
            )

    # ---- E4 proper-noun casing (months, the brand's own name) ---------------
    # House style: proper nouns keep their casing. A card reading "weights: july
    # 27" beside an outro row reading "weights July 27" is the same defect class
    # as the acronym split (kimi-k3 BEAT_8, 2026-07-25). Only flagged when the
    # capitalised form appears elsewhere, so a deliberately all-lowercase pack
    # voice is never a finding.
    # Months only. A brand's own name is deliberately lowercase in some packs'
    # mono voice ("gemini api · ai studio"), so it is a style choice, not a
    # defect; months are unambiguous proper nouns.
    for noun in MONTHS:
        low = noun.lower()
        low_hits = [lbl for lbl, t in cased
                    if re.search(rf"\b{re.escape(low)}\b", strip_urls(t))]
        up_hits = [lbl for lbl, t in fields if re.search(rf"\b{re.escape(noun)}\b", t)]
        if low_hits and up_hits:
            errors.append(
                f"proper-noun casing: '{noun}' is capitalised in {up_hits[0]} but "
                f"lowercase in {', '.join(low_hits[:3])} — proper nouns keep their casing"
            )

    # ---- W1 echo: a line that restates the line beside it -------------------
    for k in beats:
        f = P[k].split("|")
        if len(f) < 7:
            continue
        pill = f[2].split("~")
        if len(pill) >= 3 and "^" in pill[1]:            # statement beat
            sub, cap = pill[2], f[5]
            if overlap(sub, cap) >= 0.6:
                warns.append(
                    f"{k}: statement subtitle and caption say the same thing "
                    f"(\"{sub}\" / \"{cap}\") — both are on screen at once"
                )
    catch = P.get("MYTAKE_CATCH", "")
    for k, cap in captions:
        if catch and overlap(catch, cap) >= 0.6:
            warns.append(
                f"MYTAKE_CATCH restates {k}'s caption (\"{catch}\" / \"{cap}\") "
                f"— the verdict should add a new facet"
            )

    # ---- W2 icon overflow (post-slice only) ---------------------------------
    if icons_dir and os.path.isdir(icons_dir):
        try:
            from PIL import Image
        except ImportError:
            Image = None
        if Image:
            for k in beats:
                f = P[k].split("|")
                if len(f) < 7:
                    continue
                name, h = f[0].strip(), float(f[1])
                path = os.path.join(icons_dir, f"{name}.png")
                if not os.path.exists(path):
                    continue
                im = Image.open(path)
                w = h * im.width / im.height
                if w > ICON_MAX_W:
                    fit = int(ICON_MAX_W * im.height / im.width)
                    errors.append(
                        f"{k}: icon '{name}' is {im.width/im.height:.2f}:1 wide, so "
                        f"icon_h={h:.0f} renders {w:.0f}px across (max {ICON_MAX_W:.0f} "
                        f"before it collides with the cards) — icon_h is a HEIGHT, so "
                        f"a wide glyph needs a much smaller one: {fit} is the widest "
                        f"that fits, and it will still read large"
                    )

    # ---- chat bubbles only render on the plain card path ---------------------
    # feature_beat reaches chat_card from ONE of its three card paths. A STATEMENT
    # beat ('^' in the pill) and a RICH beat ('~' in either card) call cmd_card
    # raw, so a '>'/'<' card there renders as a single-line mono command with the
    # prefix showing, past the 26-char ceiling the bubble budget below skips.
    for k in beats:
        parts = [p.strip() for p in P[k].split("|")]
        if len(parts) not in (7, 8):
            continue
        cards = (("card1", parts[3]), ("card2", parts[4]))
        where = ("a STATEMENT beat ('^' in the pill)" if "^" in parts[2] else
                 "a RICH beat ('~' in a card)" if any("~" in c for _, c in cards) else "")
        for slot, text in cards:
            if text[:1] not in (">", "<"):
                continue
            if where:
                errors.append(
                    f"{k} {slot}: a chat bubble in {where} renders as a raw mono "
                    f"command, prefix and all — bubbles only render on a plain feature beat")
                continue
            # CHAT BUBBLES wrap onto multiple lines in the sans face, so the mono
            # single-line ceiling does not apply. They get their own budget: past
            # ~110 characters the bubble grows tall enough to collide with the
            # other card. Checked here, not under E4, because E4 needs FONT_MONO
            # and the engine renders bubbles without it.
            body = text[1:].strip()
            if not body:
                errors.append(f"{k} {slot}: an empty chat bubble says nothing")
            elif len(body) > 110:
                errors.append(
                    f"{k} {slot}: chat bubble is {len(body)} characters — past "
                    f"110 it wraps tall enough to collide with the other card. "
                    f"Split it or shorten it.")
            elif len(body) > 82:
                warns.append(
                    f"{k} {slot}: chat bubble is {len(body)} characters — it will "
                    f"wrap to three lines; 82 keeps it to two")

    # ---- POSTER_UNDERLINE must name a word the poster actually sets ----------
    # The engine draws nothing when the word is missing — a silent miss of the
    # opening hook — and only poster mode (POSTER_LINE_1) draws the lockup at all.
    ul = P.get("POSTER_UNDERLINE", "").strip()
    if ul:
        if not P.get("POSTER_LINE_1", "").strip():
            errors.append("POSTER_UNDERLINE is set but POSTER_LINE_1 is not — only the poster "
                          "opening draws the underline, so nothing would render")
        elif word_at(P.get("POSTER_LINE_2", ""), ul) < 0:
            errors.append(f"POSTER_UNDERLINE {ul!r} is not a whole word of POSTER_LINE_2 "
                          f"{P.get('POSTER_LINE_2', '')!r} — the engine would draw no stroke")

    # ---- E4 card text overflow ----------------------------------------------
    # cmd_card centres mono 42 in a 690px card and does NOT clip. Thresholds are
    # derived from the shipped fleet, not the design intent: the widest card text
    # that ever shipped is 668px (mistral BEAT_8), so the padding budget of 606px
    # is a TARGET, not a limit — three delivered films sit above it and read fine.
    # Past CARD_EDGE_W the glyphs reach the rounded corner, which nothing shipped
    # does. Caught while writing "autonomous research: limited" (708px) into it.
    # PRODUCT cards (a `~` in the field) are a different layout with its own
    # metrics, so they are skipped here.
    mono_path = P.get("FONT_MONO", "")
    if mono_path and os.path.exists(mono_path):
        try:
            from PIL import ImageFont
            adv = ImageFont.truetype(mono_path, CARD_MONO_PT).getlength("M")
        except Exception:
            adv = 0
        if adv:
            for k in beats:
                parts = [p.strip() for p in P[k].split("|")]
                if len(parts) not in (7, 8):
                    continue
                for slot, text in (("card1", parts[3]), ("card2", parts[4])):
                    if "~" in text:
                        continue
                    if text[:1] in (">", "<"):
                        continue  # chat bubbles: budget checked above, not mono metrics
                    w_px = adv * len(text)
                    if w_px > CARD_EDGE_W:
                        errors.append(
                            f"{k} {slot}: \"{text}\" is {len(text)} characters "
                            f"({w_px:.0f}px) in a {CARD_W:.0f}px card — cmd_card does not "
                            f"clip, so the text reaches the rounded corners. "
                            f"{int(CARD_EDGE_W / adv)} characters is the hard ceiling, "
                            f"{int(CARD_TEXT_W / adv)} keeps the intended padding"
                        )
                    elif w_px > CARD_TEXT_W:
                        warns.append(
                            f"{k} {slot}: \"{text}\" is {w_px:.0f}px against the "
                            f"{CARD_TEXT_W:.0f}px padding budget — inside the card but "
                            f"tight ({int(CARD_TEXT_W / adv)} characters keeps the air)"
                        )

    # ---- E5 MY TAKE card overflow -------------------------------------------
    # mytake_beat centres BEST FOR / THE CATCH values in sans-medium 34 inside a 690px
    # card and does not wrap or clip. The recipe's "<=7 words" is not a width: "Strictly
    # enforces its own opinionated workflows" is six words and 800px, and it ran past both
    # card edges on the first autonomous Linear film (2026-08-29). Two ceilings, both hard:
    # the measured glyph width (what the viewer sees) and a character count — the spacing
    # QC estimates text at len * 34 * 0.62 against the card, and flags overlap above ~37
    # characters whatever the real glyphs measure, so a pack must satisfy the estimate too.
    MYTAKE_PT, MYTAKE_CARD_W, MYTAKE_PAD, MYTAKE_MAX_CHARS = 34, 690, 28, 36
    sans_med = P.get("FONT_SANS_MEDIUM", "") or P.get("FONT_SANS", "")
    adv_med = 0
    if sans_med and os.path.exists(sans_med):
        try:
            from PIL import ImageFont
            _font_med = ImageFont.truetype(sans_med, MYTAKE_PT)
            adv_med = 1
        except Exception:
            adv_med = 0
    for k in ("MYTAKE_BEST_FOR", "MYTAKE_CATCH"):
        text = (P.get(k, "") or "").strip()
        if not text:
            continue
        w_px = _font_med.getlength(text) if adv_med else len(text) * MYTAKE_PT * 0.5
        budget = MYTAKE_CARD_W - 2 * MYTAKE_PAD
        if w_px > budget or len(text) > MYTAKE_MAX_CHARS:
            errors.append(
                f"{k}: \"{text}\" is {len(text)} characters ({w_px:.0f}px) in a "
                f"{MYTAKE_CARD_W}px card — mytake_beat does not wrap or clip, so it runs past "
                f"the edges. Keep it under {MYTAKE_MAX_CHARS} characters and {budget}px "
                f"(about {int(budget / max(w_px / max(len(text), 1), 1))} of these characters)"
            )

    for w in warns:
        print(f"lint WARN  {w}")
    for e in errors:
        print(f"lint ERROR {e}")
    if errors or (strict and warns):
        sys.exit(1)
    print(f"pack lint OK: {len(beats)} beats, {len(warns)} advisory")


if __name__ == "__main__":
    main()
