#!/usr/bin/env python3
"""Write the delivered master's subtitle track.

Every platform upload wants an .srt beside the .mp4, and it was being hand-typed
per film — which means hand-timed, from a timeline that already exists in code.
The engine owns BEAT_STARTS and mix.py owns the placement rule (clip i lands at
BEAT_STARTS[i] + 0.3s, time-compressed to its beat budget when it overruns), so
this derives both instead of restating them. One source of truth for the timeline.

Cue text is the pack's VO_N verbatim — the VO was written to fit the beats, so the
subtitle and the narration are the same sentence by construction.

Env: BRAND_PACK, BRAND_PROJECT.
Usage: make_srt.py [<out-path>]   (defaults to final/videos/<slug>.srt)
"""
import importlib.util
import os
import subprocess
import sys

PROJ = os.environ.get("BRAND_PROJECT", "")
if not PROJ:
    sys.exit("make_srt: set BRAND_PROJECT")

# capture the caller's args BEFORE handing argv to the engine, which reads argv[1]
# as its aspect (the first version of this script wrote its SRT to a file named
# "16x9" because it read argv back afterwards)
_ARGS = [a for a in sys.argv[1:] if not a.startswith("-")]
_here = os.path.dirname(os.path.abspath(__file__))
sys.argv = ["make_srt", "16x9"]
_spec = importlib.util.spec_from_file_location("_engine", os.path.join(_here, "engine.py"))
E = importlib.util.module_from_spec(_spec)
sys.modules[_spec.name] = E
_spec.loader.exec_module(E)

LEAD = 0.3                  # mix.py: VO clip i lands at BEAT_STARTS[i] + LEAD
TAIL_GUARD = 0.2            # never let a cue touch the next beat's first frame
STARTS = list(E.BEAT_STARTS) + [E.TOTAL - E.OUTRO_D]
DURS = list(E.BEAT_DURS) + [E.OUTRO_D]
SLUG = os.path.basename(PROJ.rstrip("/"))


def probe(path):
    r = subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration",
                        "-of", "csv=p=0", path], capture_output=True, text=True)
    return float(r.stdout.strip())


def stamp(t):
    ms = int(round(t * 1000))
    h, ms = divmod(ms, 3600_000)
    m, ms = divmod(ms, 60_000)
    s, ms = divmod(ms, 1000)
    return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"


def main():
    out_path = _ARGS[0] if _ARGS else f"{PROJ}/final/videos/{SLUG}.srt"
    n_vo = len([k for k in E.P if k.startswith("VO_")])
    if n_vo != len(STARTS):
        sys.exit(f"make_srt: {n_vo} VO lines for {len(STARTS)} slots — the pack and the "
                 f"timeline disagree, so the cues would drift")
    cues = []
    for i in range(n_vo):
        text = E.P.get(f"VO_{i}", "").strip()
        clip = f"{PROJ}/artifacts/audio/dialogue-{i}-narrator.wav"
        if not text or not os.path.exists(clip):
            sys.exit(f"make_srt: no text or no audio for VO_{i}")
        budget = DURS[i] - 0.5
        # mix.py time-compresses an overrun to exactly its budget, so the cue's
        # on-screen length is the SHORTER of the raw clip and that budget
        length = min(probe(clip), budget)
        start = STARTS[i] + LEAD
        end = min(start + length, STARTS[i] + DURS[i] - TAIL_GUARD)
        cues.append((start, end, text))

    with open(out_path, "w", encoding="utf-8") as fh:
        for n, (start, end, text) in enumerate(cues, 1):
            fh.write(f"{n}\n{stamp(start)} --> {stamp(end)}\n{text}\n\n")
    print(f"srt: {out_path} ({len(cues)} cues, {stamp(cues[-1][1])} last out)")


main()
