"""
DSL 校验器 - 对 Video DSL v1alpha1 做结构校验与标准化。

模块结构（T6 重构后）：

* 细粒度 ``_check_*`` 检查器：每个函数检查一个独立维度，返回
  ``list[ValidationError]``。
* 三个 preset 聚合器，对应历史上 gen-script / render-video / 完整版三种
  使用场景：
    - ``validate_structural(dsl)``  ⇐ gen-script 历史调用点
    - ``validate_integrity(dsl)``   ⇐ render-video 历史调用点
    - ``validate_dsl(dsl)``         ⇐ 完整版（含 capabilities-driven 语言约束、字段枚举等）
  历史的两个 caller 现在 thin-wrap 对应 preset，**所检规则与历史完全等价**，
  以保证不引入任何"原本能过、现在被拒"的行为变化。

模块以前同时存在三份算法（在 gen_script.py、render_video.py 与本文件内），
``lineSyncSlides`` 注册表查询、items 数对齐、模板专属语言约束等代码在三份
之间已经静默漂移。统一到本文件后，所有维度只有一份实现。所有模板差异通过
``template.json`` 的 ``capabilities`` 字段声明，本文件读取并应用 —— 新增需要特殊
处理的模板时**不需要再改 remixmate 代码**。
"""

import difflib
import re
from typing import Iterable, Optional

VALID_VERSIONS = {"v1alpha1"}
VALID_RATIOS = {"16:9", "9:16", "1:1", "4:3", "3:4", "21:9"}
VALID_RESOLUTIONS = {"480p", "720p", "1080p", "4k"}
VALID_PURPOSES = {"opening", "intro", "point", "example", "explanation", "transition", "highlight", "cta", "ending"}
VALID_LAYOUTS = {"full-visual", "split-left-right", "split-top-bottom", "picture-in-picture", "text-overlay", "avatar-with-bg", "kenburns"}
VALID_ASSET_TYPES = {"image", "video", "audio", "avatar", "subtitle", "bgm"}
VALID_ASSET_SOURCES = {"existing", "gen-image", "gen-video", "gen-voice", "gen-digital-human"}

# 由 Composition 消费、而非 slide 组件消费的字段。前八个与模板侧
# utils/slidePayload.ts 的 FRAME_LEVEL_KEYS 同源 ——「整包态」binder 会把它们混进
# templateData，出现在那里是正当的，不该被当成未声明字段。highlightMap 同理（模板
# 一般在 slideSchemas._shared 里声明它，这里兜底没声明的情况）。
_SLIDE_FRAME_LEVEL_KEYS = frozenset({
    "slideId",
    "titleText",
    "subtitleText",
    "watermarkText",
    "narrationAssetId",
    "narrationSrc",
    "background",
    "templateData",
    "highlightMap",
})

# CJK 汉字范围（简繁通用）+ 常见中文标点
_CJK_PATTERN = re.compile(r"[一-鿿㐀-䶿＀-￯　-〿]")


# ════════════════════════════════════════════════════════════════════════════
# Error type
# ════════════════════════════════════════════════════════════════════════════


class ValidationError:
    def __init__(self, path: str, message: str, severity: str = "error"):
        self.path = path
        self.message = message
        self.severity = severity

    def __str__(self) -> str:
        return f"[{self.severity.upper()}] {self.path}: {self.message}"

    def to_dict(self) -> dict:
        return {"path": self.path, "message": self.message, "severity": self.severity}


def errors_as_strings(errors: Iterable[ValidationError]) -> list[str]:
    """Convert ValidationError objects to ``"path: message"`` strings.

    Designed for the two historical callers (gen_script / render_video) that
    print ``f"   - {err}"`` after a header line. We drop the ``[ERROR]`` /
    ``[WARNING]`` prefix here because the prior callers' output didn't have
    it; the call site already announced "validation failed".
    """
    return [f"{e.path}: {e.message}" for e in errors]


# ════════════════════════════════════════════════════════════════════════════
# Helpers
# ════════════════════════════════════════════════════════════════════════════


def _pick_template_id(dsl: dict) -> Optional[str]:
    """Pick the templateId from DSL meta (single source of truth).

    ``renderHints.templatePreference`` was the legacy location and is no
    longer generated by gen_script. We still tolerate it during a
    transition period, but new authors should write ``meta.templateId``.
    """
    meta = dsl.get("meta") or {}
    if isinstance(meta, dict):
        tid = meta.get("templateId")
        if isinstance(tid, str) and tid:
            return tid
    hints = dsl.get("renderHints") or {}
    prefs = hints.get("templatePreference") or []
    if isinstance(prefs, list) and prefs:
        return prefs[0]
    return hints.get("templateId") if isinstance(hints, dict) else None


def _line_sync_slides_for_template(template_id: Optional[str]) -> dict[str, str]:
    """Look up ``lineSyncSlides`` for a template from registry.json.

    新增多项卡片幻灯片只需在 template.json 加一行 ``lineSyncSlides[slideId] = key``，
    不再需要改 remixmate。registry 读取失败（环境缺 registry_loader、HTTP 不通等）
    时返回空 dict，调用方将跳过相关校验。

    O(1) lookup via ``registry_loader.get_template``; safe to call once per
    scene at 100+ templates without scanning the full list.
    """
    tpl = _get_template_cfg(template_id)
    if not tpl:
        return {}
    spec = tpl.get("lineSyncSlides") or {}
    return {k: v for k, v in spec.items() if isinstance(v, str)}


def _get_template_cfg(template_id: Optional[str]) -> Optional[dict]:
    """Indexed registry lookup with the legacy import fallback.

    Returns None silently when:
      - ``template_id`` is empty
      - ``registry_loader`` cannot be imported (e.g. unit tests outside the
        shared-lib sys.path)
      - the registry can't be loaded (network/file/etc.)
      - the template isn't in the visible set (e.g. beta gate hides it)

    最后那一条是**静默跳过整套模板契约校验**的入口 —— payload 必填字段、版式字段名、
    逐句对齐全部形同不存在。它由 ``_check_template_status_gate`` 单独报一条错来兜底,
    所以这里保持返回 None 不变(否则每个 ``_check_*`` 都会为同一件事各报一遍)。
    """
    if not template_id:
        return None
    try:
        from registry_loader import get_template  # type: ignore
    except ImportError:
        return None
    try:
        return get_template(template_id)
    except Exception:
        return None


def _check_template_status_gate(dsl: dict) -> list[ValidationError]:
    """模板存在、却被状态门控挡在可见集合外 → 一条 error。

    为什么必须报错而不是继续:下面所有读 ``_get_template_cfg`` 的检查(payload 必填、
    slideSchemas 字段名、lineSyncSlides)拿到 None 之后都是 ``return []`` —— 于是一个
    beta 模板的 DSL 能"零错误"通过校验,再一路烧掉 TTS 与渲染积分产出空壳画面。
    contracts 测试之所以没发现这一点,是因为它们 patch 掉了 ``_get_template_cfg``,
    生产里的门控在测试里从不生效。
    """
    template_id = _pick_template_id(dsl)
    if not template_id:
        return []
    try:
        from registry_loader import gated_status  # type: ignore
    except ImportError:
        return []
    try:
        status = gated_status(template_id)
    except Exception:
        # registry 不可达等问题由各自的加载路径报,这里不重复成第二条噪声。
        return []
    if not status:
        return []
    return [ValidationError(
        "meta.templateId",
        f"template '{template_id}' exists but its status is '{status}', so it is hidden "
        f"from this process — every template-specific contract check (payload required "
        f"fields, slide schemas, line-sync) would be skipped instead of enforced. "
        f"Set ENABLE_BETA_TEMPLATES=1 to admit beta templates, or promote the template "
        f"to stable.",
    )]


def _contains_cjk(text: str) -> bool:
    return bool(text) and bool(_CJK_PATTERN.search(text))


# ════════════════════════════════════════════════════════════════════════════
# Granular checks
# ════════════════════════════════════════════════════════════════════════════


def _check_version(dsl: dict) -> list[ValidationError]:
    if dsl.get("version") not in VALID_VERSIONS:
        return [ValidationError(
            "version",
            f"must be one of {sorted(VALID_VERSIONS)}, got '{dsl.get('version')}'",
        )]
    return []


def _check_meta_presence(dsl: dict) -> list[ValidationError]:
    """gen-script 历史:meta 必须存在且至少有 title。"""
    errors: list[ValidationError] = []
    meta = dsl.get("meta")
    if not isinstance(meta, dict):
        errors.append(ValidationError("meta", "required and must be an object"))
        return errors
    if not meta.get("title"):
        errors.append(ValidationError("meta.title", "required"))
    return errors


def _check_global_presence(dsl: dict) -> list[ValidationError]:
    """gen-script 历史:global 必须存在(只检查存在性)。"""
    if not isinstance(dsl.get("global"), dict):
        return [ValidationError("global", "required and must be an object")]
    return []


def _check_global_enums(dsl: dict) -> list[ValidationError]:
    """完整版额外:global.aspectRatio / resolution 必须是已知枚举。"""
    errors: list[ValidationError] = []
    g = dsl.get("global")
    if isinstance(g, dict):
        ratio = g.get("aspectRatio", "16:9")
        if ratio not in VALID_RATIOS:
            errors.append(ValidationError(
                "global.aspectRatio", f"must be one of {sorted(VALID_RATIOS)}",
            ))
        res = g.get("resolution", "1080p")
        if res not in VALID_RESOLUTIONS:
            errors.append(ValidationError(
                "global.resolution", f"must be one of {sorted(VALID_RESOLUTIONS)}",
            ))
    return errors


def _check_scenes_min(dsl: dict) -> list[ValidationError]:
    """Scenes 至少 1。两个 preset 共用。"""
    scenes = dsl.get("scenes", [])
    if not isinstance(scenes, list) or len(scenes) == 0:
        return [ValidationError("scenes", "must contain at least 1 scene")]
    return []


def _check_scenes_max(dsl: dict) -> list[ValidationError]:
    """Scenes 上限 20。仅 gen-script 历史规则集 + 完整版用——render-video
    历史不查上限,合并时若同时启用此检查会"原本能过的 DSL 现在被拒",
    故只在 ``validate_structural`` / ``validate_dsl`` 启用。
    """
    scenes = dsl.get("scenes", [])
    if isinstance(scenes, list) and len(scenes) > 20:
        return [ValidationError(
            "scenes", f"must contain at most 20 scenes, got {len(scenes)}",
        )]
    return []


def _check_scene_required_fields(dsl: dict) -> list[ValidationError]:
    """gen-script 历史:每个 scene 必须有 id 和 purpose。"""
    errors: list[ValidationError] = []
    for i, scene in enumerate(dsl.get("scenes", []) or []):
        if not scene.get("id"):
            errors.append(ValidationError(f"scenes[{i}].id", "required"))
        if not scene.get("purpose"):
            errors.append(ValidationError(f"scenes[{i}].purpose", "required"))
    return errors


def _check_scene_id_uniqueness(dsl: dict) -> list[ValidationError]:
    """完整版额外:scene.id 唯一性。"""
    errors: list[ValidationError] = []
    seen: set = set()
    for i, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id")
        if sid and sid in seen:
            errors.append(ValidationError(
                f"scenes[{i}].id", f"duplicate scene id '{sid}'",
            ))
        elif sid:
            seen.add(sid)
    return errors


def _check_scene_enum_warnings(dsl: dict) -> list[ValidationError]:
    """完整版额外:scene.purpose / scene.layout 枚举(severity=warning)。"""
    errors: list[ValidationError] = []
    for i, scene in enumerate(dsl.get("scenes", []) or []):
        purpose = scene.get("purpose")
        if purpose and purpose not in VALID_PURPOSES:
            errors.append(ValidationError(
                f"scenes[{i}].purpose",
                f"must be one of {sorted(VALID_PURPOSES)}", "warning",
            ))
        layout = scene.get("layout")
        if layout and layout not in VALID_LAYOUTS:
            errors.append(ValidationError(
                f"scenes[{i}].layout",
                f"must be one of {sorted(VALID_LAYOUTS)}", "warning",
            ))
    return errors


def _check_asset_duplicates(dsl: dict) -> list[ValidationError]:
    """render-video 历史:assets 数组里 assetId 重复检查。"""
    errors: list[ValidationError] = []
    counts: dict[str, int] = {}
    for a in dsl.get("assets", []) or []:
        aid = a.get("assetId", "")
        counts[aid] = counts.get(aid, 0) + 1
    for aid, cnt in counts.items():
        if cnt > 1:
            errors.append(ValidationError(
                f"assets[{aid}]", f"duplicate assetId '{aid}' (appears {cnt} times)",
            ))
    return errors


def _check_asset_enums(dsl: dict) -> list[ValidationError]:
    """完整版额外:asset.type / asset.source 枚举。"""
    errors: list[ValidationError] = []
    for i, asset in enumerate(dsl.get("assets", []) or []):
        atype = asset.get("type")
        if atype and atype not in VALID_ASSET_TYPES:
            errors.append(ValidationError(
                f"assets[{i}].type", f"must be one of {sorted(VALID_ASSET_TYPES)}",
            ))
        asource = asset.get("source")
        if asource and asource not in VALID_ASSET_SOURCES:
            errors.append(ValidationError(
                f"assets[{i}].source", f"must be one of {sorted(VALID_ASSET_SOURCES)}",
            ))
    return errors


def _check_scene_asset_refs(dsl: dict) -> list[ValidationError]:
    """render-video 历史:scene 引用的 assetRef 必须在 assets[] 声明过。

    包括 customPayload / templateData 中的嵌套 assetRef 与 global 引用。
    """
    errors: list[ValidationError] = []
    declared = {
        a.get("assetId") for a in (dsl.get("assets") or []) if a.get("assetId")
    }
    from .payload_contract import asset_references
    for key in ("scenes", "global"):
        for path, ref in asset_references(dsl.get(key), key):
            if ref not in declared:
                errors.append(ValidationError(path, "assetRef not found in assets[]"))
    return errors


def _check_assetbindings_deprecation(dsl: dict) -> list[ValidationError]:
    """完整版额外:assetBindings 引用未知 id 时给 warning + 提示已弃用。"""
    errors: list[ValidationError] = []
    declared = {
        a.get("assetId") for a in (dsl.get("assets") or []) if a.get("assetId")
    }
    for i, scene in enumerate(dsl.get("scenes", []) or []):
        for ref in scene.get("assetBindings", []) or []:
            if ref not in declared:
                errors.append(ValidationError(
                    f"scenes[{i}].assetBindings",
                    f"references unknown assetId '{ref}' (note: assetBindings is deprecated, "
                    "use visuals.background.assetRef + audio.narration.assetRef instead)",
                    "warning",
                ))
    return errors


def _check_narration_items_text_coexist(dsl: dict) -> list[ValidationError]:
    """分行旁白与整段 ``narration.text`` 并存 → warning(text 会被整个忽略)。

    渲染侧是 **items 优先、text 完全忽略**:``extract_narration_lines`` 一旦拿到
    items,TTS 文本就是 ``"\\n".join(items)``,字幕也改走 ``split_subtitle_from_lines``,
    ``narration.text`` 既不喂 TTS 也不切字幕。

    为什么值得报:gen_script 的骨架**总是**写一条 ``narration.text``,agent 之后为
    逐行高亮补 items 时不会删它,于是两者并存几乎是这类模板的默认结局。留着的那
    一份是死数据 —— 它曾经在旁白编辑面板里被当成"可以单独改、单独重新生成"的一
    行摆出来(实测一个场景因此显示三行,其中一行是另外两行拼起来的),改它不影响
    成片,生成它照样扣积分。

    只报 warning 不拒绝:两者并存不会让画面出错,它只是让人误以为 text 还有用。
    """
    errors: list[ValidationError] = []
    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id", f"scenes[{idx}]")
        narration = (scene.get("audio") or {}).get("narration") or {}
        if not isinstance(narration, dict):
            continue
        items = narration.get("items")
        has_items = isinstance(items, list) and any(
            (isinstance(x, str) and x.strip()) or isinstance(x, dict) for x in items
        )
        has_intro_outro = any(
            isinstance(narration.get(k), str) and narration[k].strip()
            for k in ("intro", "outro")
        )
        text = narration.get("text")
        if not (has_items or has_intro_outro):
            continue
        if not (isinstance(text, str) and text.strip()):
            continue
        errors.append(ValidationError(
            f"scenes[{idx}].audio.narration.text",
            f"scene '{sid}': this scene uses per-line narration (intro/items/outro), so "
            "narration.text is ignored end to end — it is neither sent to TTS nor split "
            "into subtitles. Leaving it here makes it look like editable copy that no "
            "longer affects the video. Drop the text field and keep the lines.",
            severity="warning",
        ))
    return errors


def _check_narration_items_count(dsl: dict) -> list[ValidationError]:
    """render-video / 完整版:结构化 narration items 数与 templateData 的对应数组等长。

    哪些 slideId 走该校验,由模板的 ``lineSyncSlides`` 字段声明(P2.2 起);
    registry 不可用或模板没声明则跳过。两个 preset 共用本检查器,统一来源
    避免 render-video 与完整版的算法继续漂移。
    """
    errors: list[ValidationError] = []
    template_id = _pick_template_id(dsl)
    line_sync_slides = _line_sync_slides_for_template(template_id)

    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id", f"scenes[{idx}]")
        narration = (scene.get("audio") or {}).get("narration") or {}
        items = narration.get("items")
        if not items or not isinstance(items, list):
            continue
        custom = scene.get("customPayload") or {}
        slide_id = custom.get("slideId") or (custom.get("templateData") or {}).get("slideId")
        tdata = custom.get("templateData") or scene.get("templateData") or {}
        items_key = line_sync_slides.get(slide_id or "")
        if not items_key:
            continue
        card_list = tdata.get(items_key)
        if not isinstance(card_list, list):
            continue
        # render-video 历史只在 actual!=expected 时报;完整版也是同样语义。
        actual = len([x for x in items if (isinstance(x, str) and x.strip()) or isinstance(x, dict)])
        expected = len(card_list)
        if actual != expected:
            errors.append(ValidationError(
                f"scenes[{idx}].audio.narration.items",
                f"scene '{sid}': narration.items has {actual} entries "
                f"but templateData.{items_key} has {expected} — counts must match "
                "so each card lines up with one narration line",
            ))
    return errors


# gen_script 写进骨架的「作者还没选版式」哨兵。
#
# 两处必须是同一个字面量,而 gen-script 与 template-registry 是两个独立 skill 目录、
# 没有共享模块可以 import,所以这里复制一份。改动需同步 gen_script.py 的
# ``SLIDE_ID_SENTINEL``(那边的注释也指回这里)。
SLIDE_ID_SENTINEL = "__CHOOSE_SLIDE__"


def _check_slide_id_chosen(dsl: dict) -> list[ValidationError]:
    """多版式模板的每个场景都必须显式选一个 ``customPayload.slideId``。

    为什么需要这道门禁:slideId 缺失(或还是哨兵)时,渲染端会静默回落到
    ``DefaultSlide`` —— 渲染成功、零日志、退出码 0,画面只剩一行居中标题。这是整条
    管线里最安静的失效方式,看起来"像是模板本来就长这样",只有付完渲染的钱才会发现。

    **判定「是不是多版式模板」不写死模板 id**,而是看模板自己声明的
    ``customPayloadSchema.slideId.enum`` 有没有超过一个取值 —— 有得挑才要求挑。
    单版式模板(枚举只有一项或压根没声明)不拦:那里没有可选的东西,报错只会变成
    修不掉的常驻噪音。

    报错文案直接把该模板全部可选 slideId 列出来,因为收到这条错误的调用方多半
    正是那个"不知道有哪些版式"的 agent。
    """
    tpl = _get_template_cfg(_pick_template_id(dsl))
    if not tpl:
        return []
    schema = tpl.get("customPayloadSchema")
    if not isinstance(schema, dict):
        return []
    slide_schema = schema.get("slideId")
    enum = slide_schema.get("enum") if isinstance(slide_schema, dict) else None
    if not isinstance(enum, list) or len(enum) <= 1:
        return []

    tid = tpl.get("templateId", "<unknown>")
    choices = ", ".join(str(x) for x in enum)
    errors: list[ValidationError] = []
    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id", f"scenes[{idx}]")
        custom = scene.get("customPayload") or {}
        slide_id = custom.get("slideId") or (custom.get("templateData") or {}).get("slideId")
        if isinstance(slide_id, str) and slide_id and slide_id != SLIDE_ID_SENTINEL:
            continue
        reason = (
            "still holds the gen_script placeholder"
            if slide_id == SLIDE_ID_SENTINEL
            else "has no slideId"
        )
        errors.append(ValidationError(
            f"scenes[{idx}].customPayload.slideId",
            f"scene '{sid}' {reason} — template '{tid}' registers several layouts and "
            "one must be chosen per scene, otherwise the whole templateData is dropped "
            "and the frame silently renders a single centred title.\n"
            f"     Pick the one that matches this scene's information shape: {choices}",
        ))
    return errors


def _check_slide_cross_field(dsl: dict) -> list[ValidationError]:
    """字段之间的一致性 —— 单看任何一个字段都合法,凑在一起才不成立。

    schema 管不了这一类:它逐字段校验类型与枚举,表达不了"这个数必须落在另一个字段的
    长度内"。三条规则都**静态可判定**,而且失败形态一致 —— 画面少东西、零报错:

      · ``annotations[].line`` 指向不存在的代码行 → 那句旁白期间没有任何一行点亮
      · ``highlightMap`` 的值越过卡片数组 → 那一行全屏变暗、无一张点亮
      · 黑板版 ``highlight`` 不是 ``title`` 的子串 → 标题静默退回单色（模板 llmHint
        自己写明了这个行为，正因为写明了才更该校验）

    与 `_check_slide_payload_fields` 分开是因为那个函数按 schema 逐层递归,这里要的是
    跨字段视角;混在一起会让两边都难读。
    """
    tpl = _get_template_cfg(_pick_template_id(dsl))
    if not tpl:
        return []
    schemas = tpl.get("slideSchemas")
    if not isinstance(schemas, dict):
        return []
    line_sync = {k: v for k, v in (tpl.get("lineSyncSlides") or {}).items()
                 if isinstance(v, str)}

    errors: list[ValidationError] = []

    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id", f"scenes[{idx}]")
        custom = scene.get("customPayload") or {}
        if not isinstance(custom, dict):
            continue
        slide_id = custom.get("slideId") or (custom.get("templateData") or {}).get("slideId")
        if not isinstance(slide_id, str) or slide_id not in schemas:
            continue
        tdata = custom.get("templateData")
        if not isinstance(tdata, dict):
            tdata = scene.get("templateData") if isinstance(scene.get("templateData"), dict) else {}
        if not tdata:
            continue
        base = f"scenes[{idx}].customPayload.templateData"

        # ① 注解行号 ≤ 代码行数。`line` 是 1 起算的（契约里写明了）。
        code = tdata.get("code")
        annotations = tdata.get("annotations")
        if isinstance(code, str) and isinstance(annotations, list):
            total = len(code.split("\n"))
            for i, note in enumerate(annotations):
                if not isinstance(note, dict):
                    continue
                line = note.get("line")
                if not isinstance(line, int) or isinstance(line, bool):
                    continue
                if line < 1 or line > total:
                    errors.append(ValidationError(
                        f"{base}.annotations[{i}].line",
                        f"scene '{sid}': annotation points at line {line} but the code block has "
                        f"{total} line(s) — while that narration line plays, no code line lights "
                        "up at all. Line numbers are 1-based.",
                    ))

        # ② highlightMap 的值必须落在卡片数组里。
        hmap = tdata.get("highlightMap")
        cards_key = line_sync.get(slide_id)
        cards = tdata.get(cards_key) if cards_key else None
        if isinstance(hmap, dict) and isinstance(cards, list):
            for k, v in hmap.items():
                if not isinstance(v, int) or isinstance(v, bool):
                    continue
                if v < 0 or v >= len(cards):
                    errors.append(ValidationError(
                        f"{base}.highlightMap.{k}",
                        f"scene '{sid}': highlightMap sends subtitle line {k} to card index {v}, "
                        f"but templateData.{cards_key} only has {len(cards)} entries — during that "
                        "line every card goes dim and none lights up.",
                    ))

        # ③ 黑板版的 highlight 必须是 title 的子串。字段是否存在由 schema 决定,
        #    这里不写死模板名 —— 没有声明 highlight 的模板天然跳过。
        if "highlight" in ((schemas.get(slide_id) or {}).get("properties") or {}):
            highlight = tdata.get("highlight")
            title = tdata.get("title")
            if (
                isinstance(highlight, str)
                and highlight.strip()
                and isinstance(title, str)
                and highlight not in title
            ):
                errors.append(ValidationError(
                    f"{base}.highlight",
                    f"scene '{sid}': highlight '{highlight}' is not a substring of title "
                    f"'{title}' — the two-colour split silently falls back to a single colour.",
                    severity="warning",
                ))

    return errors


def _check_line_sync_narration(dsl: dict) -> list[ValidationError]:
    """走逐行高亮的版式,旁白必须是**分行**的,否则整屏一个卡片都不会点亮。

    为什么必须拦:这是"画面看起来没渲染完"的主路径,而且全链路零日志。链条是——
    旁白只给了整段 ``text``(单卡版式的合法写法)→ 绑定不写 ``narrationLineCount``
    → ``render_video`` 的 ``_auto_highlight_map`` 在第一个 if 就 return → 没有
    ``highlightMap`` → 模板侧 ``resolveHighlightMap`` 同样拿不到行数 →
    ``activeHighlightIndex`` 恒为 -1 → **整场所有卡片停在未点亮态**。
    连 ``_auto_highlight_map`` 那条 warn 日志都打不出来,因为它在更前面就返回了。

    哪些版式受约束由模板的 ``lineSyncSlides`` 声明,不写死模板名/版式名。

    两个逃生阀(命中任一即放行,都是模板自己声明的能力,不是本函数的特例):
      - ``templateData.highlightMap`` —— 作者显式给了映射表,那他知道自己在做什么;
      - schema 里标了 ``x-staticHighlight`` 的字段(如 timeline-axis 的
        ``progressIndex``、timeline / column-compare 的 ``highlightIndex``)——
        这些字段的存在本身就是"本屏走静态高亮、不跟旁白走"的声明。
    """
    tpl = _get_template_cfg(_pick_template_id(dsl))
    if not tpl:
        return []
    line_sync = {k: v for k, v in (tpl.get("lineSyncSlides") or {}).items()
                 if isinstance(v, str)}
    if not line_sync:
        return []
    schemas = tpl.get("slideSchemas") if isinstance(tpl.get("slideSchemas"), dict) else {}

    errors: list[ValidationError] = []

    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id", f"scenes[{idx}]")
        custom = scene.get("customPayload") or {}
        if not isinstance(custom, dict):
            continue
        slide_id = custom.get("slideId") or (custom.get("templateData") or {}).get("slideId")
        if not isinstance(slide_id, str) or slide_id not in line_sync:
            continue

        narration = (scene.get("audio") or {}).get("narration") or {}
        items = narration.get("items")
        if isinstance(items, list) and any(
            (isinstance(x, str) and x.strip()) or isinstance(x, dict) for x in items
        ):
            continue  # 有分行旁白，数量对不对交给 _check_narration_items_count

        tdata = custom.get("templateData")
        if not isinstance(tdata, dict):
            tdata = scene.get("templateData") if isinstance(scene.get("templateData"), dict) else {}

        if isinstance(tdata.get("highlightMap"), dict) and tdata["highlightMap"]:
            continue

        static_keys = [
            k for k, sub in ((schemas.get(slide_id) or {}).get("properties") or {}).items()
            if isinstance(sub, dict) and sub.get("x-staticHighlight") is True
        ]
        if any(tdata.get(k) is not None for k in static_keys):
            continue

        cards_key = line_sync[slide_id]
        escape = (
            f" If this scene is meant to stay static, set one of: {', '.join(sorted(static_keys))}."
            if static_keys else ""
        )
        errors.append(ValidationError(
            f"scenes[{idx}].audio.narration.items",
            f"scene '{sid}': layout '{slide_id}' lights its {cards_key} one at a time, driven by "
            "one narration line per card — but this scene's narration is a single block with no "
            "'items'. Without per-line narration nothing ever lights up: every card stays in the "
            "dimmed state for the whole scene and the frame looks half-rendered, with no error "
            f"and no log. Split the narration into one line per entry of templateData.{cards_key} "
            "under audio.narration.items." + escape,
        ))

    return errors


def _check_slide_payload_fields(dsl: dict) -> list[ValidationError]:
    """逐版式校验 ``customPayload.templateData`` 的字段名。

    为什么需要这道门禁:多版式模板里「slideId 决定 templateData 吃哪些字段」这条
    契约,此前只存在于模板的 React 源码里 —— 而写 DSL 的 agent 只能看到 registry
    下发的 ``template.json``。字段名猜错(把 ``name`` 写成 ``title``、把 ``detail``
    写成 ``description``)时组件读到 undefined,要么渲染出内置的占位文案、要么整块
    空白,**渲染成功、零日志、退出码 0**。真实案例:一份 7 场景的 DSL 里 5 个场景
    中招,TTS 和渲染的钱全花完才在成片里看出来。

    契约现在写在 ``template.json`` 的 ``slideSchemas`` 里(键 = slideId,值 = 该版式
    的 JSON Schema),由 template-library 的 ``check:slide-schemas`` 与组件源码双向
    对账。本函数按它核对每个场景,**未声明的字段名带 "did you mean" 提示** ——
    收到这条错误的多半正是那个猜错了名字的 agent,直接把正确的名字给它。

    没有声明 ``slideSchemas`` 的模板(全部单版式模板)一律跳过,不产生噪音。
    """
    tpl = _get_template_cfg(_pick_template_id(dsl))
    if not tpl:
        return []
    schemas = tpl.get("slideSchemas")
    if not isinstance(schemas, dict):
        return []

    tid = tpl.get("templateId", "<unknown>")
    shared = ((schemas.get("_shared") or {}).get("properties") or {})
    known_slides = [k for k in schemas if not k.startswith("_")]

    errors: list[ValidationError] = []

    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        sid = scene.get("id", f"scenes[{idx}]")
        custom = scene.get("customPayload") or {}
        if not isinstance(custom, dict):
            continue
        data = custom.get("templateData")
        if not isinstance(data, dict):
            data = scene.get("templateData")
        if not isinstance(data, dict):
            continue

        slide_id = custom.get("slideId") or data.get("slideId")
        if not isinstance(slide_id, str) or not slide_id:
            continue  # 「没选版式」由 _check_slide_id_chosen 负责报

        schema = schemas.get(slide_id)
        if not isinstance(schema, dict):
            # 版式名写错 —— 同样静默回落到 DefaultSlide,和没写一样致命。
            errors.append(ValidationError(
                f"scenes[{idx}].customPayload.slideId",
                f"scene '{sid}' uses slideId '{slide_id}', which template '{tid}' does not "
                "register — the frame silently falls back to a single centred title.\n"
                f"     Available layouts: {', '.join(sorted(known_slides))}",
            ))
            continue

        errors += _walk_payload(
            data,
            schema,
            path=f"scenes[{idx}].customPayload.templateData",
            scene=sid,
            extra_allowed=shared,
            # frame 级字段由 Composition 消费,「整包态」binder 会把它们混进
            # templateData —— 出现在这里是正当的,不算未声明字段。
            skip_keys=_SLIDE_FRAME_LEVEL_KEYS,
        )

    return errors


def _check_custom_payload_fields(dsl: dict) -> list[ValidationError]:
    """单版式模板:按 ``customPayloadSchema`` 逐字段核对 ``customPayload``。

    与 `_check_slide_payload_fields` 是同一件事的两种形态 —— 多版式模板的契约按 slideId
    分支(``slideSchemas``),单版式模板只有一份(``customPayloadSchema``)。多版式模板在
    这里**跳过**,避免同一份数据被两套规则各报一遍。

    为什么值得做:在此之前,10 个模板里只有 2 个的字段名会被校验,其余 8 个写错字段名
    照样一路渲染到成片。契约本来就在 ``template.json`` 里躺着,只是没人读。
    """
    tpl = _get_template_cfg(_pick_template_id(dsl))
    if not tpl:
        return []
    from .payload_contract import required_payload_errors
    errors: list[ValidationError] = []
    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        custom = scene.get("customPayload")
        custom = dict(custom) if isinstance(custom, dict) else {}
        if not custom.get("templateData") and isinstance(scene.get("templateData"), dict):
            custom["templateData"] = scene["templateData"]
        errors += [ValidationError(f"scenes[{idx}].customPayload", message)
                   for message in required_payload_errors(tpl, custom)]
    if isinstance(tpl.get("slideSchemas"), dict):
        return errors  # 多版式字段走 _check_slide_payload_fields
    cps = tpl.get("customPayloadSchema")
    if not isinstance(cps, dict) or not cps:
        return errors

    root = {"type": "object", "properties": cps}
    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        custom = scene.get("customPayload")
        if not isinstance(custom, dict) or not custom:
            continue
        errors += _walk_payload(
            custom,
            root,
            path=f"scenes[{idx}].customPayload",
            scene=scene.get("id", f"scenes[{idx}]"),
        )
    return errors


def _walk_payload(
    data: dict,
    schema: dict,
    *,
    path: str,
    scene: str,
    extra_allowed: Optional[dict] = None,
    skip_keys: frozenset = frozenset(),
    fallback_desc: Optional[str] = None,
) -> list[ValidationError]:
    """沿 schema 递归下钻,逐层做「未声明字段 + 缺必填 + 空数组」核对。

    只下钻**描述了形状的**层：schema 节点没有 ``properties`` 就停(那是作者显式开的
    开放包,如 ppt-to-video 由抽取脚本生成的 ``slide.nodes[]``),不去猜它的字段。

    ⚠️ 这里刻意**不做**类型 / enum / 区间校验 —— 那些由 template-library 的
    `check:payload` 在仓库侧对 dsl-example 做,已有一份完整实现。运行时这份只管
    "字段名对不对、该给的给没给",两边职责不重叠,也就不需要在 Python 里再养一份
    JSON Schema 引擎(这个仓库已经被"同一算法两份实现"坑过不止一次)。
    """
    errors = _check_payload_object(
        data,
        schema,
        path=path,
        scene=scene,
        extra_allowed=extra_allowed,
        skip_keys=skip_keys,
        fallback_desc=fallback_desc,
    )

    own_desc = schema.get("description") if isinstance(schema.get("description"), str) else None

    for key, sub in (schema.get("properties") or {}).items():
        if not isinstance(sub, dict):
            continue
        value = data.get(key)
        if sub.get("type") == "object" and sub.get("properties") and isinstance(value, dict):
            errors += _walk_payload(
                value, sub, path=f"{path}.{key}", scene=scene,
                fallback_desc=own_desc or fallback_desc,
            )
        elif sub.get("type") == "array" and isinstance(value, list):
            item_schema = sub.get("items")
            if not isinstance(item_schema, dict) or not item_schema.get("properties"):
                continue
            for i, item in enumerate(value):
                if isinstance(item, dict):
                    errors += _walk_payload(
                        item, item_schema, path=f"{path}.{key}[{i}]", scene=scene,
                        fallback_desc=own_desc or fallback_desc,
                    )

    return errors


def _check_payload_object(
    data: dict,
    schema: dict,
    *,
    path: str,
    scene: str,
    extra_allowed: Optional[dict] = None,
    skip_keys: frozenset = frozenset(),
    fallback_desc: Optional[str] = None,
) -> list[ValidationError]:
    """一层对象的「未声明字段 + 缺必填字段 + 空数组」核对。数组元素与顶层共用这一份。"""
    props = schema.get("properties")
    if not isinstance(props, dict):
        return []
    allowed = set(props) | set(extra_allowed or {})
    merged_props = {**(extra_allowed or {}), **props}

    errors: list[ValidationError] = []

    for key in data:
        if key in allowed or key in skip_keys:
            continue
        errors.append(ValidationError(
            f"{path}.{key}",
            f"scene '{scene}': the slide does not read '{key}' — the value is silently "
            f"dropped and that slot renders empty (or shows the component's placeholder)."
            + _suggest_field(key, merged_props, schema, fallback_desc),
        ))

    for key in schema.get("required") or []:
        if _is_blank(data.get(key)):
            errors.append(ValidationError(
                f"{path}.{key}",
                f"scene '{scene}': required field '{key}' is missing or empty — without it "
                "the slide renders its built-in placeholder instead of your content.",
            ))

    # 空数组 = 空屏。这些数组就是版式的画面主体,给了个 [] 与没给没有区别,
    # 但 required 判定不出来(键确实在)。用 schema 的 minItems 表达,不写死键名。
    for key, sub in props.items():
        if not isinstance(sub, dict) or sub.get("type") != "array":
            continue
        min_items = sub.get("minItems")
        value = data.get(key)
        if isinstance(min_items, int) and isinstance(value, list) and len(value) < min_items:
            errors.append(ValidationError(
                f"{path}.{key}",
                f"scene '{scene}': '{key}' has {len(value)} entries but this layout needs at "
                f"least {min_items} — the array is what the slide draws, an empty one renders "
                "a blank frame.",
            ))

    return errors


def _is_blank(value: object) -> bool:
    """必填判定：``None`` 与纯空白字符串都算没给。

    为什么空串也算:组件里普遍写成 ``props.title || "占位文案"``,而 ``""`` 是 falsy ——
    写空串拿到的**同样是占位文案**,不是空白。也就是说空串从来就不是"有意留空"的表达,
    只是另一种漏填。放行它等于给静默失效留一道后门。
    """
    if value is None:
        return True
    return isinstance(value, str) and not value.strip()


def _suggest_field(
    wrong: str,
    props: dict,
    schema: dict,
    fallback_desc: Optional[str] = None,
) -> str:
    """把写错的字段名换成一句能照着改的提示。三级降级。

    ① **在各字段的 description 里搜「不是 <wrong>」**。写契约时易错点已经写进了描述
       （``卡片标题（⚠️ 键名是 name，不是 title）``），直接拿错的名字去搜即可。实测对
       真实事故里那 6 个错误 6/6 命中,而 difflib 是 0/6 —— 因为 title↔name、
       description↔detail 是**同义词**而非拼写错误,字符串相似度对它们天然无效。
       这一级还自带维护性:提示与契约同处一地,改字段名时顺手就改了。
    ② difflib —— 留给真正的拼写错误（``titel`` → ``title``）。
    ③ 列出全部合法字段,并附上**该版式自己的 description**。后者回答的是"那我该用哪个
       版式"：demo-concept-overview 的描述里就写着「卡片不支持描述文字,要配说明改用
       feature-grid」,而这正是写错的人真正需要的下一步。
    """
    for name, sub in props.items():
        desc = (sub or {}).get("description") if isinstance(sub, dict) else None
        if not isinstance(desc, str):
            continue
        # 「不是 title」/「不是 title，」/「不是 `title`」都要认。
        if re.search(rf"不是\s*[`'\"]?{re.escape(wrong)}[`'\"]?(?![\w-])", desc):
            return f" This layout calls it '{name}' — {desc}"

    hit = difflib.get_close_matches(wrong, sorted(props), n=1, cutoff=0.6)
    if hit:
        return f" Did you mean '{hit[0]}'?"

    tail = f" Declared fields: {', '.join(sorted(props))}."
    # 走到这一级说明「改个名字」解决不了 —— 多半是这个版式压根没有这种能力。
    # 把版式自己的说明带出来，那里通常写着该换成哪个版式（demo-concept-overview 的
    # 描述就写着「卡片不支持描述文字，每项要配一句说明时改用 feature-grid」）。
    # 嵌套 item 层没有自己的 description，所以由调用方把版式级的那句传进来。
    slide_desc = schema.get("description") or fallback_desc
    if isinstance(slide_desc, str) and slide_desc.strip():
        tail += f" About this layout: {slide_desc}"
    return tail


def _check_narration_language_strict(dsl: dict) -> list[ValidationError]:
    """完整版:模板若声明 ``capabilities.narrationLanguageStrict`` 则强约束旁白语言。

    Generic capability-driven replacement for what used to be a hardcoded
    ``picture-book-en`` CJK check. Templates declare which language the
    narration MUST be in (currently supported: ``"en"`` rejects any CJK
    character; ``"zh"`` is a placeholder reserved for future symmetric
    enforcement — we don't yet flag latin-only narration on zh templates
    because gen-script regularly mixes brand names in English).

    Detection points are unchanged from the legacy check:
      - every ``gen-voice`` asset's ``payload.text``
      - every scene's ``audio.narration.text``
    """
    tpl = _get_template_cfg(_pick_template_id(dsl))
    if not tpl:
        return []
    strict = (tpl.get("capabilities") or {}).get("narrationLanguageStrict")
    if strict != "en":
        # Today only the "narration must be English" case is enforced;
        # adding "zh" / other languages goes here without touching callers.
        return []
    tid = tpl.get("templateId", "<unknown>")
    msg = (
        f"narration on template '{tid}' must be English only (no CJK characters); "
        "put the Chinese translation in textLayers[role=subheadline].content instead"
    )
    errors: list[ValidationError] = []
    for i, asset in enumerate(dsl.get("assets", []) or []):
        if asset.get("source") != "gen-voice":
            continue
        voice_text = ((asset.get("payload") or {}).get("text")) or ""
        if _contains_cjk(voice_text):
            errors.append(ValidationError(f"assets[{i}].payload.text", msg))
    for idx, scene in enumerate(dsl.get("scenes", []) or []):
        narration = (scene.get("audio") or {}).get("narration") or {}
        narration_text = narration.get("text")
        if isinstance(narration_text, str) and _contains_cjk(narration_text):
            errors.append(ValidationError(
                f"scenes[{idx}].audio.narration.text", msg,
            ))
    return errors


# ════════════════════════════════════════════════════════════════════════════
# Aggregator presets
# ════════════════════════════════════════════════════════════════════════════


def validate_structural(dsl: dict) -> list[ValidationError]:
    """gen-script 历史规则集 — 仅检查 DSL 整体结构 / 必填字段。

    与历史 ``gen_script.validate_dsl`` 同集合:
      version + meta(title) + global(presence) + scene 数 1–20 + 每场 id/purpose。

    不检查:asset 一致性、enum 枚举、capabilities.narrationLanguageStrict——保持与历史等价。
    """
    errors: list[ValidationError] = []
    errors += _check_version(dsl)
    errors += _check_meta_presence(dsl)
    errors += _check_global_presence(dsl)
    errors += _check_scenes_min(dsl)
    errors += _check_scenes_max(dsl)
    errors += _check_scene_required_fields(dsl)
    return errors


def validate_integrity(dsl: dict) -> list[ValidationError]:
    """render-video 历史规则集 — 关注 asset 引用一致性 + 数量对齐。

    与历史 ``render_video.validate_dsl`` 同集合:
      version + scene 数 + assetId 去重 + assetRef 可解析 + 结构化 narration items 数。

    不检查:meta.title、global 必填、enum 枚举、capabilities.narrationLanguageStrict——保持等价。
    """
    errors: list[ValidationError] = []
    errors += _check_version(dsl)
    errors += _check_scenes_min(dsl)
    # NOTE: render-video 历史只查下限,不查 ≤20 上限——合并时若同启用上限会
    # 把"超 20 场但其它都对"的 DSL 从「通过」改成「拒绝」,违反"重构不影响功能"。
    errors += _check_asset_duplicates(dsl)
    errors += _check_scene_asset_refs(dsl)
    errors += _check_narration_items_count(dsl)
    errors += _check_narration_items_text_coexist(dsl)
    # 新增(非历史规则):多版式模板必须逐场景选定 slideId。放进 integrity 集合是因为
    # render-video 与 prepare_video_assets 都走这条 —— 而"没选版式"正是要在烧掉渲染
    # 积分之前拦住的东西。单版式模板不受影响,见 _check_slide_id_chosen。
    errors += _check_slide_id_chosen(dsl)
    # 必须排在下面那批模板契约检查**之前**:模板被状态门控挡住时它们会集体空转,
    # 于是"零错误通过"其实是"一条都没查"。见 _check_template_status_gate。
    errors += _check_template_status_gate(dsl)
    # 同理:字段名写错同样是「渲染成功但画面是空壳」,必须在烧掉 TTS / 渲染积分
    # 之前拦住。只对声明了 slideSchemas 的模板生效,见 _check_slide_payload_fields。
    errors += _check_slide_payload_fields(dsl)
    errors += _check_custom_payload_fields(dsl)
    errors += _check_line_sync_narration(dsl)
    errors += _check_slide_cross_field(dsl)
    return errors


def validate_dsl(dsl: dict) -> list[ValidationError]:
    """完整 DSL 校验:结构 + 一致性 + 枚举 + capabilities.narrationLanguageStrict。

    历史上 dsl_validator 暴露的就是这版,无外部 caller(三个真 caller 都
    走 ``validate_structural`` / ``validate_integrity``)。本函数面向想要"全
    部规则"的新代码 —— 例如 CI 校验或人工 ``--validate-strict`` 入口。
    """
    errors: list[ValidationError] = []
    errors += _check_version(dsl)
    errors += _check_meta_presence(dsl)
    errors += _check_global_presence(dsl)
    errors += _check_global_enums(dsl)
    errors += _check_scenes_min(dsl)
    errors += _check_scenes_max(dsl)
    errors += _check_scene_required_fields(dsl)
    errors += _check_scene_id_uniqueness(dsl)
    errors += _check_scene_enum_warnings(dsl)
    errors += _check_asset_duplicates(dsl)
    errors += _check_asset_enums(dsl)
    errors += _check_scene_asset_refs(dsl)
    errors += _check_assetbindings_deprecation(dsl)
    errors += _check_narration_items_count(dsl)
    errors += _check_narration_items_text_coexist(dsl)
    errors += _check_narration_language_strict(dsl)
    errors += _check_slide_id_chosen(dsl)
    errors += _check_template_status_gate(dsl)
    errors += _check_slide_payload_fields(dsl)
    errors += _check_custom_payload_fields(dsl)
    errors += _check_line_sync_narration(dsl)
    errors += _check_slide_cross_field(dsl)
    return errors
