#!/usr/bin/env python3
"""
Smoke-scan the consumer repo to produce a deterministic <=6KB text blob that
captures enough product signal for the AI prompt to generate ASO-optimized
App Store metadata without reading every file.

Sources (in fixed order, each hard-capped):
  * README.md / readme.md  (2048B)  human pitch
  * $SCHEME/Info.plist      (512B)  display name + capability keys
  * Package.swift / Podfile(1024B)  SDK fingerprint (Firebase, RevenueCat, ...)
  * top 3 *.swift by size  (1536B)  class / view names reveal feature set
  * Bundle id + scheme      ( 128B)  naming signal
  * *.entitlements          ( 256B)  capability keys (VPN, iCloud, ...)

Total hard cap: ~6KB. Walk is scoped to $GITHUB_WORKSPACE (or os.getcwd());
paths resolved against that root are re-checked to prevent traversal.
Fail-open: any uncaught error -> warning, "<unavailable>" blob, exit 0.

Environment:
  GITHUB_WORKSPACE  -- repo root (falls back to cwd)
  BUNDLE_ID         -- app bundle id (from auto_detect or ci.config.yaml)
  SCHEME            -- Xcode scheme name
"""

from __future__ import annotations

import os
import plistlib
import re
import sys
from pathlib import Path

from metadata_constants import warn


README_CAP = 2048
INFO_PLIST_CAP = 512
DEPS_CAP = 1024
SWIFT_TOTAL_CAP = 1536
SWIFT_PER_FILE_CAP = 500
NAMING_CAP = 128
ENTITLEMENTS_CAP = 256


def _root() -> Path:
    workspace = os.environ.get("GITHUB_WORKSPACE", "").strip() or os.getcwd()
    return Path(workspace).resolve()


def _safe_path(root: Path, candidate: Path) -> Path | None:
    """Return resolved candidate if it is inside root, else None.

    Guards against symlinked or traversal-crafted paths reaching outside
    the workspace. Non-existent paths return None.
    """
    try:
        resolved = candidate.resolve()
    except OSError:
        return None
    try:
        resolved.relative_to(root)
    except ValueError:
        return None
    if not resolved.exists():
        return None
    return resolved


def _read_capped(path: Path, cap: int) -> str:
    try:
        raw = path.read_bytes()[:cap]
        return raw.decode("utf-8", errors="replace").strip()
    except OSError as exc:
        warn(f"scanner: read failed for {path.name}: {exc!r}")
        return ""


def read_readme(root: Path) -> str:
    for name in ("README.md", "readme.md", "README.MD", "Readme.md"):
        p = _safe_path(root, root / name)
        if p:
            return _read_capped(p, README_CAP)
    return ""


def _scheme_root(root: Path, scheme: str) -> Path | None:
    if not scheme:
        return None
    for candidate in (root / scheme, root / "Sources" / scheme, root / "Sources"):
        p = _safe_path(root, candidate)
        if p and p.is_dir():
            return p
    return None


def read_info_plist(root: Path, scheme: str) -> str:
    search_dirs = [d for d in (_scheme_root(root, scheme), root) if d]
    for base in search_dirs:
        plist = _safe_path(base, base / "Info.plist")
        if plist:
            return _extract_plist_summary(plist)
    for plist in sorted(root.rglob("Info.plist"))[:5]:
        safe = _safe_path(root, plist)
        if safe:
            summary = _extract_plist_summary(safe)
            if summary:
                return summary
    return ""


def _extract_plist_summary(path: Path) -> str:
    try:
        with open(path, "rb") as f:
            data = plistlib.load(f)
    except (OSError, plistlib.InvalidFileException, ValueError) as exc:
        warn(f"scanner: plist parse failed for {path.name}: {exc!r}")
        return ""
    if not isinstance(data, dict):
        return ""
    display = data.get("CFBundleDisplayName") or data.get("CFBundleName") or ""
    version = data.get("CFBundleShortVersionString") or ""
    perms = sorted(k for k in data if k.startswith("NS") and "UsageDescription" in k)
    lines = []
    if display:
        lines.append(f"display_name={display}")
    if version:
        lines.append(f"version={version}")
    if perms:
        lines.append("permissions=" + ",".join(perms))
    return "\n".join(lines)[:INFO_PLIST_CAP]


def read_dependencies(root: Path) -> str:
    pkg_swift = _safe_path(root, root / "Package.swift")
    if pkg_swift:
        return _grep_lines(pkg_swift, r"\.package\s*\(", DEPS_CAP)
    podfile = _safe_path(root, root / "Podfile")
    if podfile:
        return _grep_lines(podfile, r"^\s*pod\s+", DEPS_CAP)
    return ""


def _grep_lines(path: Path, pattern: str, cap: int) -> str:
    try:
        text = path.read_text(encoding="utf-8", errors="replace")
    except OSError as exc:
        warn(f"scanner: read failed for {path.name}: {exc!r}")
        return ""
    regex = re.compile(pattern)
    matches = [line.strip() for line in text.splitlines() if regex.search(line)]
    return "\n".join(matches)[:cap]


def read_swift_files(root: Path, scheme: str) -> str:
    base = _scheme_root(root, scheme) or root
    swift_paths: list[Path] = []
    for p in base.rglob("*.swift"):
        safe = _safe_path(root, p)
        if not safe or "Tests" in safe.parts:
            continue
        try:
            safe.stat()  # confirm readable; sort key below will re-stat
        except OSError:
            continue
        swift_paths.append(safe)
    swift_paths.sort(key=lambda p: (-p.stat().st_size, p.name))
    chunks: list[str] = []
    used = 0
    for p in swift_paths[:3]:
        if used >= SWIFT_TOTAL_CAP:
            break
        snippet = _read_capped(p, SWIFT_PER_FILE_CAP)
        if not snippet:
            continue
        block = f"--- {p.name} ---\n{snippet}"
        if used + len(block) > SWIFT_TOTAL_CAP:
            block = block[: SWIFT_TOTAL_CAP - used]
        chunks.append(block)
        used += len(block) + 1
    return "\n".join(chunks)


def read_entitlements(root: Path) -> str:
    for ent in sorted(root.rglob("*.entitlements"))[:3]:
        safe = _safe_path(root, ent)
        if not safe:
            continue
        try:
            with open(safe, "rb") as f:
                data = plistlib.load(f)
        except (OSError, plistlib.InvalidFileException, ValueError):
            continue
        if isinstance(data, dict) and data:
            return "\n".join(sorted(data))[:ENTITLEMENTS_CAP]
    return ""


def build_blob() -> str:
    root = _root()
    bundle_id = os.environ.get("BUNDLE_ID", "").strip()
    scheme = os.environ.get("SCHEME", "").strip()

    naming = []
    if bundle_id:
        naming.append(f"BUNDLE_ID: {bundle_id}")
    if scheme:
        naming.append(f"SCHEME: {scheme}")
    naming_blob = "\n".join(naming)[:NAMING_CAP]

    sections = [
        ("===APP CONTEXT===", naming_blob),
        ("===README===", read_readme(root)),
        ("===INFO_PLIST===", read_info_plist(root, scheme)),
        ("===DEPENDENCIES===", read_dependencies(root)),
        ("===SWIFT_FILES===", read_swift_files(root, scheme)),
        ("===ENTITLEMENTS===", read_entitlements(root)),
    ]
    out_parts: list[str] = []
    for header, content in sections:
        out_parts.extend((header, content or "<unavailable>"))
    out_parts.append("===END===")
    return "\n".join(out_parts)


def main() -> int:
    try:
        blob = build_blob()
    except SystemExit:
        raise
    except Exception as exc:
        warn(f"app context scanner failed (non-fatal): {exc!r}")
        print("===APP CONTEXT===\n<unavailable>\n===END===")
        return 0
    print(blob)
    return 0


if __name__ == "__main__":
    sys.exit(main())
