#!/usr/bin/env python3
"""Scrape REAL brand tokens from a live site into a pack-ready report.

The palette must come from the brand's own CSS, never from memory or a
summarizer (a fetch-and-summarize pass reported Buzz had "no distinctive
palette" when its CSS defined five exact custom properties). This tool fetches
the page + its stylesheets and reports: CSS custom properties with colour
values, the most frequent hex colours, font-family declarations, gradients,
and icon/mark links — as a `.brand` snippet to review and paste.

It is a REPORT tool by design: choosing which hex is GROUND vs ACCENT is the
operator's call at the tokens gate, not a heuristic's.

Usage: python3 scrape_brand_tokens.py <url>
"""
import re
import sys
import urllib.parse
import urllib.request

UA = {"User-Agent": "Mozilla/5.0 (brand-token scrape; operator-reviewed)"}


def fetch(url, limit=800_000):
    req = urllib.request.Request(url, headers=UA)
    with urllib.request.urlopen(req, timeout=25) as r:
        return r.read(limit).decode("utf-8", "replace")


def main():
    if len(sys.argv) < 2:
        sys.exit("usage: scrape_brand_tokens.py <url>")
    url = sys.argv[1]
    html = fetch(url)
    css_blobs = [html]
    for href in re.findall(r'<link[^>]+rel="stylesheet"[^>]+href="([^"]+)"', html) \
            + re.findall(r'<link[^>]+href="([^"]+\.css[^"]*)"', html):
        try:
            css_blobs.append(fetch(urllib.parse.urljoin(url, href)))
        except Exception as e:  # noqa: BLE001 — report and continue
            print(f"# note: could not fetch {href}: {e}")
    blob = "\n".join(css_blobs)

    print(f"# tokens scraped from {url} — REVIEW before pasting into the pack\n")

    hexes = re.findall(r"#[0-9a-fA-F]{6}\b", blob)
    counts = {}
    for h in hexes:
        counts[h.lower()] = counts.get(h.lower(), 0) + 1
    print("# top hex colours (count):")
    for h, n in sorted(counts.items(), key=lambda kv: -kv[1])[:15]:
        print(f"#   {h}  x{n}")

    print("\n# CSS custom properties carrying colours (the real token names):")
    for m in sorted(set(re.findall(
            r"(--[a-z0-9-]+)\s*:\s*(#[0-9a-fA-F]{3,8}|rgba?\([^)]*\))", blob)))[:30]:
        print(f"#   {m[0]}: {m[1]}")

    print("\n# gradients:")
    for g in sorted(set(re.findall(r"linear-gradient\([^;{}]{0,140}", blob)))[:6]:
        print(f"#   {g}")

    print("\n# font families:")
    fams = sorted(set(re.findall(r"font-family\s*:\s*([^;}]{0,80})", blob)))
    for f in fams[:10]:
        print(f"#   {f.strip()}")

    print("\n# icon / mark links (download the real SVG — never redraw a mark):")
    for link in sorted(set(re.findall(
            r'(?:src|href)="([^"]*(?:icon|logo|mark|favicon)[^"]*)"', html)))[:8]:
        print(f"#   {urllib.parse.urljoin(url, link)}")

    print("\n# pack snippet draft — assign roles yourself, then verify on screen:")
    top = [h for h, _ in sorted(counts.items(), key=lambda kv: -kv[1])]
    print(f'COLOR_GROUND="{(top[0] if top else "#??????").upper()}"')
    print(f'COLOR_INK="{(top[1] if len(top) > 1 else "#??????").upper()}"')
    print('COLOR_SECONDARY="#??????"   # the pale offset layer')
    print('COLOR_ACCENT="#??????"      # ONE point of emphasis per frame')
    print('COLOR_CARD="#FFFFFF"')


if __name__ == "__main__":
    main()
