/** * Stitch keystone for the assemble stage (sub-slice 3h). * * Source of truth (ported VERBATIM): * - `skills/video-replicator/scripts/stitch_bunty.py` — the bunty stitch. Uses * the concat **demuxer** path (`ffmpeg -y -f concat -safe 0 -i concat.txt * -c copy output`) — a single ffmpeg invocation regardless of segment count, * chosen deliberately so it survives the sandbox's per-session FFmpeg limit. * - `skills/video-replicator/scripts/ffmpeg_wrapper.py:concat_via_filter` — the * concat **filter** fallback (`-filter_complex * "[0:v][0:a]...concat=n=N:v=1:a=1[outv][outa]"` + re-encode). Used for 8+ * segments where demuxer-accumulated AV drift across boundaries matters, OR * when a segment has incompatible codec params and the demuxer rejects it. * - `skills/video-replicator/scripts/assembly_utils.py:add_background_music` — * the music-bed mix (loops the music under the narration at low volume with a * tail fade-out, via amix). * * bunty (stitch_bunty.py) and nex (nex_assemble.py) collapse into ONE * parameterized stitch driven by brand-profile-derived knobs: * - bunty: pre-encoded segments → demuxer concat, no music, intro/outro by * lip-sync scene segments. (`_concat_via_demuxer`, demuxer-first with a * filter fallback.) * - nex: normalized segments → filter concat (drift-free for 8+), optional * background-music bed, optional title-card prepend. (`concat_via_filter` * + `add_background_music`.) * These differences become StitchInput fields: `concatStrategy` * (demuxer | filter | auto), `intro` / `outro` segment paths, and the optional * `music` block (track + volume + tail fade). The demuxer-vs-filter selection in * `auto` mode flips at `FILTER_FALLBACK_SEGMENT_THRESHOLD` segments — the real * sandbox-survival + drift lesson from the Python. * * AV-drift note (stitch_bunty.py ~L471-479): demuxer `-c copy` preserves exact * packet timing and accumulates no re-encode drift, BUT it requires every * segment to share encoding params. The per-segment AV-lock from 3e * (1280×720@24, H.264 libx264 preset-fast crf20, AAC 44100 stereo) guarantees * that, so the demuxer path is the primary one. Filter concat re-encodes (and so * can introduce its own drift) but tolerates mismatched inputs — it is the * fallback. * * IMPORTANT — testing boundary (SAME as 3e): the PURE arg-builders * (`buildConcatDemuxerArgs`, `buildConcatFilterArgs`, `buildMusicMixArgs`) are * the unit-tested surface (arg-shape only). `stitch` actually spawns ffmpeg and * writes the concat list; verifying that the final MP4 looks/sounds right on * real media is a HUMAN integration checkpoint, explicitly OUT OF SCOPE for the * unit tests. Tests use the dry-run path and never run ffmpeg or require media. */ import { writeFile, mkdir, stat } from 'node:fs/promises'; import { dirname, resolve as resolvePath } from 'node:path'; import { runFfmpeg, ffprobeDuration, isValidMp4, type RunFfmpegOptions } from './ffmpeg.js'; import { majoritySize } from './normalize-size.js'; import { probeMedia } from '../final-media.js'; import { VclawError } from '../errors.js'; // Phase 3c: the filter register + stitch constants, the concat/prep arg builders and the multi-layer audio mix live in their own modules; every public name stays importable from here. import { DEFAULT_MUSIC_VOLUME, DEFAULT_MUSIC_FADE_OUT_SEC, STANDARD_AUDIO_BITRATE, resolveGradeFilter } from './stitch-filters.js'; import { type ConcatStrategy, orderedSegments, selectConcatStrategy, buildConcatListContent, buildConcatDemuxerArgs, buildConcatFilterArgs, buildSegmentPrepArgs } from './stitch-concat.js'; import { type AudioLayer, MIX_LIMITER, MIX_LIMITER_DUCKED, MIX_LIMITER_FILTER, buildMultiLayerMixArgs } from './stitch-audio-mix.js'; export { FILTER_FALLBACK_SEGMENT_THRESHOLD, DEFAULT_MUSIC_VOLUME, DEFAULT_MUSIC_FADE_OUT_SEC, STANDARD_AUDIO_BITRATE, GRADE_FILTER_IDS, ON_TWOS_FILTER, DEFAULT_FILM_GRAIN, SHARPEN_FILTER, filmGrainFilter, resolveGradeFilter } from './stitch-filters.js'; export { orderedSegments, selectConcatStrategy, buildConcatListContent, buildConcatDemuxerArgs, buildConcatFilterArgs, buildSegmentPrepArgs } from './stitch-concat.js'; export type { ConcatStrategy, BuildConcatFilterOptions, BuildSegmentPrepOptions } from './stitch-concat.js'; export { buildMultiLayerMixArgs } from './stitch-audio-mix.js'; export type { AudioLayer, BuildMultiLayerMixOptions } from './stitch-audio-mix.js'; export interface MusicMixSettings { /** Path to the background-music track. Looped under the narration. */ trackPath: string; /** Mix level for the music bed (0..1). Defaults to {@link DEFAULT_MUSIC_VOLUME}. */ volume?: number; /** Tail fade-out duration in seconds. Defaults to {@link DEFAULT_MUSIC_FADE_OUT_SEC}. */ fadeOutSec?: number; /** * Voice-forward mix (loudnorm the video's narration + sidechain-duck the music * under it + final limiter). Forwarded to {@link BuildMusicMixOptions.voiceForward}. * Default off → byte-identical legacy bed mix. */ voiceForward?: boolean; } export interface StitchInput { /** * Ordered body segment paths (the slide segments). These sit between the * optional intro and outro segments. From 3e every segment conforms to * 1280×720@24 / H.264 / AAC 44100 stereo, so demuxer concat is valid. */ segments: string[]; /** Optional ordered intro segment paths, prepended before `segments`. */ intro?: string[]; /** Optional ordered outro segment paths, appended after `segments`. */ outro?: string[]; /** Where the final stitched MP4 is written. */ outputPath: string; /** * Path for the concat-demuxer list file. Defaults to `concat.txt` alongside * the output. Only used by the demuxer path. */ concatListPath?: string; /** * Concat strategy. `auto` (default) uses the demuxer up to * {@link FILTER_FALLBACK_SEGMENT_THRESHOLD} segments, then the filter. * bunty maps to `auto`/`demuxer`; nex maps to `filter`. */ concatStrategy?: ConcatStrategy; /** * Whole-cut finishing look, applied in the per-segment prep pass. All three * default off, so omitting them leaves the emitted ffmpeg args byte-identical. * Unlike `segmentGradeIds` these are not per-scene: they are the pass that * binds every generation into one cohesive piece. */ onTwos?: boolean; sharpen?: boolean; filmGrain?: number; /** Optional background-music bed (the nex-brand knob). Omit for bunty. */ music?: MusicMixSettings; /** * Optional EXTRA global audio layers (dialogue / sfx / additional voice) * mixed over the concatenated video at the stitch step, BEYOND the `music` * bed above. Presence-driven and additive: when this is empty/undefined the * stitch is byte-identical to legacy (the music-only {@link buildMusicMixArgs} * path, or the no-audio path). When non-empty, the music bed (if any) is * prepended as a `music` layer and the whole set is mixed via * {@link buildMultiLayerMixArgs}. See assemble.ts Step 6 (dialogue/sfx wiring). */ audioLayers?: AudioLayer[]; /** * Sidechain-duck every `music` layer under the narration/dialogue layers in * the MULTI-LAYER mix (`audioLayers` path). The multi-layer analogue of * `music.voiceForward`: without it a soundtrack bed sits at full level over * a narration layer. Only consulted when `audioLayers` is non-empty; unset → * byte-identical legacy multi-layer mix. */ duckMusicUnderVoice?: boolean; /** * Drop the concatenated segments' own audio from the final mix — the * silent-film switch. assemble sets this when every scene is marked * `silent`, because on `veo-useapi` there is no way to ask the provider for * a silent clip: Veo 3.x is audio-native and the Flow API has no audio-off * parameter, so `generateAudio: false` is a promise the route cannot keep. * Stripping here is the only place the intent can actually be enforced. * Unset → byte-identical legacy mix. */ dropSegmentAudio?: boolean; /** * Segment indices whose audio is silenced, leaving the rest of the film's * audio intact. Indices are into the ORDERED segment list (intro/outro * included), matching `segmentGradeIds`. Prefer this over * {@link dropSegmentAudio} whenever the problem is confined to known shots — * it costs one shot's ambience rather than the whole film's. */ muteSegmentAudio?: number[]; /** * Optional per-clip tail cut (WS9). When set, each ordered segment is * re-encoded with `-t ` before concat, dropping the dead / * freeze frames AI generators append. Omit for byte-identical legacy behavior. */ clipMaxSeconds?: number; /** * Optional letterbox normalization (WS9). When set (e.g. `2.39:1`), each * segment is scaled+padded onto the {@link letterboxCanvas} (default * 1280×720) before concat, producing cinematic bars. Omit to disable. */ letterboxRatio?: string; /** * Canvas for {@link letterboxRatio}. Defaults to the 3e segment standard * (1280×720) so normalized segments stay uniform for demuxer concat. */ letterboxCanvas?: { width: number; height: number }; /** * Path to a `.cube`/`.3dl` LUT applied to every segment via `lut3d` (e.g. a * Kodak Vision3 500T film-stock LUT). Omit to skip. Triggers the prep pass. */ gradeLut?: string; /** * Named color grade applied to every segment (see {@link GRADE_FILTER_IDS}). * A real post transform on the footage, not a prompt hint. Omit to skip. */ gradeId?: string; /** * Per-segment grade overrides, indexed to {@link orderedSegments} (intro + * body + outro). A defined entry overrides {@link gradeId} for that segment — * this is how the narrative color language is realized (e.g. cool-steel for the * calm scenes, crimson-threat for the breach, electric-blue for the resolution). */ segmentGradeIds?: Array; /** * Optional reading-holds (motion-comic). For each listed segment index (into * {@link orderedSegments}: intro+body+outro), a short held, gently-drifting * still of that segment's readable FIRST frame is prepended BEFORE its motion * plays — so viewers can read text-dense info panels (a cast roster, a strategy * map) before the animation takes over. Omit / empty `segments` → byte-identical * legacy (no holds). When any hold is present the concat uses the FILTER path * (re-encode) so the freshly-encoded hold clips concat cleanly with the segments. */ readingHold?: ReadingHoldSettings; } export interface ReadingHoldSettings { /** Hold duration in seconds before each designated segment. Default 3.5. */ holdSec?: number; /** Segment indices (into orderedSegments) that get a reading hold. Empty ⇒ no holds. */ segments: number[]; } export interface BuildReadingHoldOptions { /** Hold duration in seconds. Default 3.5. */ holdSec?: number; /** Output width. Default 1280. */ width?: number; /** Output height. Default 720. */ height?: number; /** Output fps. Default 24. */ fps?: number; } /** * Build ffmpeg args (PURE) that turn a video segment's FIRST frame into a short * held, gently-drifting still clip (a "reading hold"), with a silent stereo audio * track so it concats cleanly. `select=eq(n,0)` grabs frame 0; `zoompan` holds it * for `holdSec` with a faint slow zoom (alive, not frozen); encoded to the segment * standard (libx264 crf20 / AAC 44100 stereo @ fps). `-y` is prepended by runFfmpeg. */ export function buildReadingHoldArgs( segmentPath: string, outputPath: string, opts: BuildReadingHoldOptions = {}, ): string[] { const holdSec = opts.holdSec ?? 3.5; const width = opts.width ?? 1280; const height = opts.height ?? 720; const fps = opts.fps ?? 24; const frames = Math.round(holdSec * fps); // Oversample 2× so the slow zoom stays sharp, then output at the canvas size. const vf = `select=eq(n\\,0),scale=${width * 2}:${height * 2}:force_original_aspect_ratio=increase,` + `crop=${width * 2}:${height * 2},` + `zoompan=z='min(zoom+0.0002,1.03)':x='iw/2-(iw/zoom/2)':y='ih/2-(ih/zoom/2)':d=${frames}:s=${width}x${height}:fps=${fps},setsar=1`; return [ '-i', segmentPath, '-f', 'lavfi', '-t', String(holdSec), '-i', 'anullsrc=r=44100:cl=stereo', '-filter_complex', `[0:v]${vf}[v]`, '-map', '[v]', '-map', '1:a', '-t', String(holdSec), '-r', String(fps), '-c:v', 'libx264', '-preset', 'fast', '-crf', '20', '-pix_fmt', 'yuv420p', '-c:a', 'aac', '-b:a', STANDARD_AUDIO_BITRATE, '-ar', '44100', '-ac', '2', '-movflags', '+faststart', outputPath, ]; } export interface StitchPlannedStep { /** What this step does. */ kind: 'segment-prep' | 'reading-hold' | 'concat-demuxer' | 'concat-filter' | 'music-mix' | 'multi-layer-mix'; /** The ffmpeg args (everything after the binary and the auto-prepended `-y`). */ args: string[]; /** The output this step writes. */ outputPath: string; } export interface StitchResult { status: 'complete' | 'dry-run'; /** Final MP4 path. */ outputPath: string; /** Ordered list of every segment that went into the concat. */ orderedSegments: string[]; /** Which concat path was actually used. */ concatStrategy: 'demuxer' | 'filter'; /** Whether a music bed was mixed. */ music: boolean; /** The planned ffmpeg command sequence (always populated, incl. dry-run). */ plan: StitchPlannedStep[]; /** Final video duration in milliseconds (0 on dry-run — no probe). */ durationMs: number; } export interface BuildMusicMixOptions { /** Mix level for the music bed (0..1). Default {@link DEFAULT_MUSIC_VOLUME}. */ volume?: number; /** Tail fade-out seconds. Default {@link DEFAULT_MUSIC_FADE_OUT_SEC}. */ fadeOutSec?: number; /** * Total video duration in seconds — used to compute the fade-out start * (`total - fadeOut`) and the `-t` cap. The caller probes this via * `ffprobeDuration`. Defaults to 0 (fade starts at 0 / `-t 0`), only used by * the dry-run / pure-builder path where the duration is not yet known. */ totalDurationSec?: number; /** * Voice-forward mix (default false → byte-identical legacy). When true, the * video's own audio (`[0:a]`, i.e. the baked per-scene narration) is loudnorm'd * to a broadcast target, the music bed is sidechain-ducked under it, and the * final mix is limited — so narration is never buried beneath the bed. This is * the film-proven chain (loudnorm I=-15:TP=-1.5 + sidechaincompress + alimiter). * assemble enables it by default whenever per-scene narration was produced. */ voiceForward?: boolean; /** * Drop the video's own audio (`[0:a]`) so the bed plays over a genuinely * silent picture. See {@link BuildMultiLayerMixOptions.dropVideoAudio} — same * reason, the music-only path. Mutually exclusive with `voiceForward`, which * exists to keep the clip's voice audible; when both are set `dropVideoAudio` * wins, because "this film has no voice" is the stronger statement. * Default `false` → byte-identical legacy graph. */ dropVideoAudio?: boolean; /** * Whether the video input actually carries an audio stream. Default `true` * → byte-identical legacy graph. `stitch()` probes the concat on a real run * and passes `false` for a silent picture, in which case the bed is the whole * mix: referencing `[0:a]` on an audioless input is an ffmpeg failure. The * existing `-t` cap already bounds the looped bed. See * {@link BuildMultiLayerMixOptions.videoHasAudio}. */ videoHasAudio?: boolean; } /** Format a number the way the Python f-strings do (`:.2f` / `:.3f`). */ function fixed(value: number, digits: number): string { return value.toFixed(digits); } /** * Build the background-music mix ffmpeg args (PURE). Loops the music under the * narration at a low volume with a tail fade-out, then mixes via amix. * * Ported VERBATIM from assembly_utils.add_background_music (L408-435): * -i video -stream_loop -1 -i music * -filter_complex "[0:a]volume=1.0[v]; * [1:a]volume={vol},afade=t=out:st={fade_start}:d={fade_out}[m]; * [v][m]amix=inputs=2:duration=first:dropout_transition=600:normalize=0[a]" * -map 0:v -map [a] * -c:v copy -c:a aac -b:a 192k * -movflags +faststart * -t {total} output * * fade_start = max(0, total - fade_out). `dropout_transition=600` prevents early * audio cutoff on silent sections; `normalize=0` keeps the narration at full * level while the music stays at `volume`. */ export function buildMusicMixArgs( videoPath: string, musicPath: string, outputPath: string, opts: BuildMusicMixOptions = {}, ): string[] { const volume = opts.volume ?? DEFAULT_MUSIC_VOLUME; const fadeOut = opts.fadeOutSec ?? DEFAULT_MUSIC_FADE_OUT_SEC; const totalDur = opts.totalDurationSec ?? 0; const fadeStart = Math.max(0, totalDur - fadeOut); // Voice-forward (opt-in) loudnorms the video's own audio (the narration in // [0:a]), ducks the music under it via sidechaincompress, and limits the final // mix — the film-proven chain that keeps narration above the bed. Default // (voiceForward unset/false) is the VERBATIM legacy add_background_music graph. // Silent-mode wins over voice-forward: there is no voice to stay forward of. // The clip track is MUTED, not dropped — the music input carries // alimiter's `level` defaults to ENABLED, which is auto-makeup: it does not // only shave peaks above the limit, it also lifts anything QUIETER up to the // limit. Measured on a 3s tone at -12.1 dBFS: `alimiter=limit=0.5` returned // -6.0 dBFS (boosted to the ceiling), `alimiter=limit=0.5:level=disabled` // returned -12.1 dBFS (untouched). A brick wall that silently normalises every // mix to ~-1 dBFS is not what either branch wanted — `MULTI_LAYER_LIMITER` // below already gets this right. Keep the ceiling, drop the makeup. // `-stream_loop -1`, so it must never become the `duration=first` reference // or the mix would run forever. Keeping a zeroed [0:a] anchors the length to // the picture, which is the same shape the legacy graph relies on. const filterComplex = opts.videoHasAudio === false // No amix in this graph, so the bare limiter; the bed is looped, so no pad. ? `[1:a]volume=${volume},afade=t=out:st=${fixed(fadeStart, 2)}:d=${fadeOut},${MIX_LIMITER_FILTER}[a]` : opts.dropVideoAudio ? `[0:a]volume=0[v];` + `[1:a]volume=${volume},afade=t=out:st=${fixed(fadeStart, 2)}:d=${fadeOut}[m];` + `[v][m]amix=inputs=2:duration=first:dropout_transition=600:normalize=0[a]` : opts.voiceForward ? `[0:a]loudnorm=I=-15:TP=-1.5:LRA=11,asplit=2[v0][vsc];` + `[1:a]volume=${volume},afade=t=out:st=${fixed(fadeStart, 2)}:d=${fadeOut}[m];` + `[m][vsc]sidechaincompress=threshold=0.05:ratio=8:attack=5:release=250[mduck];` + `[v0][mduck]amix=inputs=2:duration=first:dropout_transition=600:${MIX_LIMITER_DUCKED}[a]` : `[0:a]volume=1.0[v];` + `[1:a]volume=${volume},afade=t=out:st=${fixed(fadeStart, 2)}:d=${fadeOut}[m];` + // `normalize=0` deliberately disables amix's own gain reduction, so two // near-full-scale inputs sum ABOVE 0 dBFS and the encoder clips. The // legacy graph had no limiter at all and shipped a +4.7 dBFS master. // A brick wall at 0.89 (~-1 dBFS) is inaudible on already-conformant // audio and is the only thing standing between a hot bed and distortion. `[v][m]amix=inputs=2:duration=first:dropout_transition=600:${MIX_LIMITER}[a]`; return [ '-i', videoPath, '-stream_loop', '-1', '-i', musicPath, '-filter_complex', filterComplex, '-map', '0:v', '-map', '[a]', '-c:v', 'copy', '-c:a', 'aac', '-b:a', STANDARD_AUDIO_BITRATE, '-movflags', '+faststart', '-t', fixed(totalDur, 3), outputPath, ]; } // --------------------------------------------------------------------------- // Multi-layer audio mix // --------------------------------------------------------------------------- /** * Compose the full ordered layer list for the multi-layer mix step: the * `music` bed (if present) first as a looped `music` layer, then the extra * `audioLayers` (dialogue/sfx/voice) in order. Pure. Only called when * `input.audioLayers` is non-empty, so the result always has ≥1 layer. */ function combinedAudioLayers(input: StitchInput): AudioLayer[] { const musicLayer: AudioLayer[] = input.music ? [ { trackPath: input.music.trackPath, role: 'music', loop: true, ...(input.music.volume !== undefined ? { volume: input.music.volume } : {}), }, ] : []; return [...musicLayer, ...(input.audioLayers ?? [])]; } export interface StitchOptions extends RunFfmpegOptions { /** Override the ffprobe binary (forwarded to `ffprobeDuration`). */ ffprobeBin?: string; /** * Skip the pre-concat MP4-validity guard. Default OFF (validation ON) for * real runs. The guard probes each input segment with {@link isValidMp4} and * fails fast on a truncated/no-moov MP4 instead of letting ffmpeg concat die * with a cryptic "Invalid data found when processing input". Always skipped on * dry-run (no real files to probe). */ skipSegmentValidation?: boolean; } /** * Orchestrate the stitch: write the concat list (demuxer path), pick the * demuxer-vs-filter path, run via `runFfmpeg`, optionally mix a music bed, and * return the final MP4 path + a plan of the executed command sequence. * * On `dryRun`, returns the planned command sequence WITHOUT writing the concat * list, spawning ffmpeg, or probing durations (music fade-start is computed * from 0). This is the path unit tests use. * * NOTE: the real-spawn path. The final-MP4 quality check is a HUMAN integration * checkpoint — out of scope here. */ export async function stitch( input: StitchInput, opts: StitchOptions = {}, ): Promise { const segs = orderedSegments(input); if (segs.length === 0) { // Defensive: nothing to concat. Caller is expected to pass >=1 segment. throw new Error('stitch: no segments to concatenate'); } const strategy = selectConcatStrategy(input.concatStrategy ?? 'auto', segs.length); const hasMusic = input.music !== undefined; // Extra (dialogue/sfx/voice) global layers beyond the music bed. Presence // here flips the mix step from the legacy single-music buildMusicMixArgs path // to the multi-layer buildMultiLayerMixArgs path. Empty/undefined ⇒ legacy. const hasExtraLayers = (input.audioLayers?.length ?? 0) > 0; // Any audio mix at all (music bed and/or extra layers) requires the // concat→intermediate→mix two-step. When neither is present, concat targets // the final output directly (byte-identical legacy no-audio path). const hasAudioMix = hasMusic || hasExtraLayers; const concatListPath = input.concatListPath ?? resolvePath(dirname(input.outputPath), 'concat.txt'); // When audio is mixed, concat writes an intermediate file and the mix step // produces the final output (mirrors nex_assemble's concat_no_music.mp4). const concatOutput = hasAudioMix ? resolvePath(dirname(input.outputPath), 'concat_no_music.mp4') : input.outputPath; const plan: StitchPlannedStep[] = []; // --- Optional per-segment normalization pre-pass (WS9) --- // When a tail-cut or letterbox is requested, each ordered segment is // re-encoded to the standard (dropping dead tail frames / adding cinematic // bars) before concat; the concat then consumes the normalized clips. Default // (neither field set) emits no prep steps and is byte-identical to legacy. // MIXED-SIZE DETECTION. The demuxer path concats with `-c copy`, which assumes // every segment shares stream parameters. When they do not, ffmpeg does not // error — it emits a master with a wrong duration and frame rate. A 3x15s // mixed 1080p/720p film came out as 108s at ~10fps, with rc=0 and a // plausible-looking file. Nothing downstream noticed. // // The prep pass already re-encodes, so normalising costs nothing extra. // // Target the MAJORITY size, breaking a tie toward the larger. This used to // take the smallest on the reasoning that nothing should be upscaled into // softness — which is right for genuinely mixed sources, and wrong the moment // clips are finished individually. A Flow render now upscales each clip to // 1080p for free, and that step is deliberately non-fatal, so a run where 14 // clips reached 1080p and one stayed at 720p would have dragged the entire // master back down to 720p to accommodate the one that failed. Majority-wins // keeps the old behaviour whenever most segments really are the smaller size, // and stops a single straggler from undoing everything else. let normalizeSize: { width: number; height: number } | undefined; if (!opts.dryRun && segs.length > 1) { const sizes: Array<{ width: number; height: number }> = []; for (const seg of segs) { try { const probe = await probeMedia(seg, { ffprobeBin: opts.ffprobeBin }); if (probe.width && probe.height) sizes.push({ width: probe.width, height: probe.height }); } catch { /* unprobeable: fall through, treated as no-normalize */ } } const distinct = new Set(sizes.map((z) => `${z.width}x${z.height}`)); if (sizes.length === segs.length && distinct.size > 1) { normalizeSize = majoritySize(sizes); process.stderr.write( `[stitch] segments have ${distinct.size} different frame sizes (${[...distinct].join(', ')}); ` + `normalising all to ${normalizeSize.width}x${normalizeSize.height} (the most common size) so the concat stays valid.\n`, ); } } const hasGrade = (typeof input.gradeLut === 'string' && input.gradeLut !== '') || (typeof input.gradeId === 'string' && input.gradeId !== '') || (input.segmentGradeIds?.some((g) => typeof g === 'string' && g !== '') ?? false); // The whole-cut finishing look ALSO requires the prep pass — it is where the // -vf chain is built. Omitting these three from this condition is what made // `assemble --on-twos --sharpen --film-grain 8` a silent no-op: the options // threaded correctly all the way down to buildSegmentPrepArgs, but with no // tail-cut, letterbox or grade also requested, `needsPrep` stayed false and // the prep pass never ran, so those args were never built. // // Caught on real footage, not by the unit tests: they call // buildSegmentPrepArgs DIRECTLY and so proved the filters were correct while // nothing proved they were ever REACHED. Measured on a finished master with // mpdecimate, the near-duplicate frame count was identical to the source. const hasFinishingLook = input.onTwos === true || input.sharpen === true || (input.filmGrain !== undefined && input.filmGrain > 0); // Muting is done in the per-segment prep pass, so asking for it must TURN THAT // PASS ON. A previous bug in this exact condition (see the comment above) // threaded an option all the way down and then never ran the pass. const mutedSegments = new Set( (input.muteSegmentAudio ?? []).filter((i) => Number.isInteger(i) && i >= 0), ); const needsPrep = (input.clipMaxSeconds !== undefined && input.clipMaxSeconds > 0) || (typeof input.letterboxRatio === 'string' && input.letterboxRatio !== '') || hasGrade || hasFinishingLook || normalizeSize !== undefined || mutedSegments.size > 0; const prepDir = resolvePath(dirname(input.outputPath), '.prep'); const concatSegs = needsPrep ? segs.map((_, i) => resolvePath(prepDir, `seg-${String(i).padStart(3, '0')}.mp4`)) : segs; if (needsPrep) { for (let i = 0; i < segs.length; i += 1) { const seg = segs[i]; // Per-segment grade override realizes the narrative color language; falls // back to the uniform gradeId. const gradeFilter = resolveGradeFilter(input.segmentGradeIds?.[i] ?? input.gradeId); // The audio-pad (apad) keeps the prepped segment A/V-aligned so the demuxer // `-c copy` concat cannot drift — but apad needs an audio stream, and the // demuxer path (unlike the filter path) happily concats video-only segments // (a silent intro/outro sting, or a raw provider clip). Probe on real runs // and disable the pad when a segment has no audio; dry-run can't probe (and // never executes ffmpeg) so it keeps the default. let padAudioToVideoDuration = true; // Cap the prepped segment at its own video length. // // `apad` + `-shortest` is supposed to end the output when the video ends, // but it stops doing so once a filter changes frame timing: with // `fps=12,fps=24` in the chain, a 15.0s clip came out with 15.0s of video // and 38.6s of AUDIO, and the container duration followed the audio. Three // such segments concatenated to 121s of "45s" film — with rc=0, correct // frame counts, and a plausible file. Passing an explicit `-t` (via the // existing trimTailArgs path) bounds it deterministically instead of // trusting `-shortest` to notice. let segmentCap = input.clipMaxSeconds; if (!opts.dryRun) { try { const probe = await probeMedia(seg, { ffprobeBin: opts.ffprobeBin }); padAudioToVideoDuration = probe.audioPresent; const videoSeconds = probe.durationSeconds; if (typeof videoSeconds === 'number' && videoSeconds > 0) { segmentCap = segmentCap !== undefined ? Math.min(segmentCap, videoSeconds) : videoSeconds; } } catch { padAudioToVideoDuration = false; } } plan.push({ kind: 'segment-prep', args: buildSegmentPrepArgs(seg, concatSegs[i], { clipMaxSeconds: segmentCap, letterboxRatio: input.letterboxRatio, width: input.letterboxCanvas?.width, height: input.letterboxCanvas?.height, gradeLut: input.gradeLut, gradeFilter, padAudioToVideoDuration, ...(mutedSegments.has(i) ? { muteAudio: true } : {}), ...(input.onTwos ? { onTwos: true } : {}), ...(input.sharpen ? { sharpen: true } : {}), ...(input.filmGrain !== undefined ? { filmGrain: input.filmGrain } : {}), ...(normalizeSize ? { normalizeSize } : {}), }), outputPath: concatSegs[i], }); } } // --- Optional reading-holds (motion-comic) --- // For each designated segment index, generate a held readable still of its // first frame and INSERT it before the segment in the concat. Any hold forces // the FILTER concat (re-encode) so the freshly-encoded holds concat cleanly // with the (possibly differently-encoded) segments. Empty ⇒ unchanged. const holdIndices = new Set( (input.readingHold?.segments ?? []).filter((i) => Number.isInteger(i) && i >= 0 && i < segs.length), ); const holdDir = resolvePath(dirname(input.outputPath), '.holds'); let finalConcatSegs = concatSegs; let effectiveStrategy = strategy; if (holdIndices.size > 0) { const holdSec = input.readingHold?.holdSec ?? 3.5; const expanded: string[] = []; for (let i = 0; i < concatSegs.length; i += 1) { if (holdIndices.has(i)) { // The hold must match the size of the clip it sits beside, or the forced // FILTER concat rejects it (it does not rescale inputs). With a letterbox // prep the prepped segments are the canvas size; otherwise the // concatenated clip keeps its native size — so probe it (real runs) and // match. Dry-run can't probe → canvas / 1280×720 default (args unexecuted). let holdW = input.letterboxCanvas?.width ?? 1280; let holdH = input.letterboxCanvas?.height ?? 720; if (!opts.dryRun && !input.letterboxRatio) { try { const dim = await probeMedia(segs[i], { ffprobeBin: opts.ffprobeBin }); if (dim.width && dim.height) { holdW = dim.width; holdH = dim.height; } } catch { /* unprobeable → fall back to canvas / default */ } } const holdPath = resolvePath(holdDir, `hold-${String(i).padStart(3, '0')}.mp4`); plan.push({ kind: 'reading-hold', args: buildReadingHoldArgs(concatSegs[i], holdPath, { holdSec, width: holdW, height: holdH }), outputPath: holdPath, }); expanded.push(holdPath); } expanded.push(concatSegs[i]); } finalConcatSegs = expanded; effectiveStrategy = 'filter'; } // --- Concat step --- const concatArgs = effectiveStrategy === 'demuxer' ? buildConcatDemuxerArgs(finalConcatSegs, concatListPath, concatOutput) : buildConcatFilterArgs(finalConcatSegs, concatOutput); plan.push({ kind: effectiveStrategy === 'demuxer' ? 'concat-demuxer' : 'concat-filter', args: concatArgs, outputPath: concatOutput, }); // --- Audio-mix step (optional) --- // Two mutually-exclusive paths: // * hasExtraLayers → multi-layer mix (music bed prepended + dialogue/sfx). // * music-only → the EXACT legacy buildMusicMixArgs path (byte-identical). // Neither ⇒ no mix step (legacy no-audio path). if (hasExtraLayers) { const layers = combinedAudioLayers(input); const mixArgs = buildMultiLayerMixArgs(concatOutput, layers, input.outputPath, { ...(input.duckMusicUnderVoice ? { duckMusicUnderVoice: true } : {}), ...(input.dropSegmentAudio ? { dropVideoAudio: true } : {}), }); plan.push({ kind: 'multi-layer-mix', args: mixArgs, outputPath: input.outputPath }); } else if (hasMusic && input.music) { // The fade-start / -t cap need the concat duration; on dry-run we leave it 0. const musicArgs = buildMusicMixArgs(concatOutput, input.music.trackPath, input.outputPath, { volume: input.music.volume, fadeOutSec: input.music.fadeOutSec, totalDurationSec: 0, voiceForward: input.music.voiceForward, ...(input.dropSegmentAudio ? { dropVideoAudio: true } : {}), }); plan.push({ kind: 'music-mix', args: musicArgs, outputPath: input.outputPath }); } if (opts.dryRun) { return { status: 'dry-run', outputPath: input.outputPath, orderedSegments: segs, concatStrategy: effectiveStrategy, music: hasMusic, plan, durationMs: 0, }; } // --- Pre-concat corruption guard --- // useapi sometimes reports a generation 'complete' while the downloaded mp4 // is truncated (no moov atom): it passes existence/size checks but ffmpeg // concat later dies with "Invalid data found when processing input". Probe // each real input segment up front so the corrupt one is named clearly. // Only the actual segment inputs are probed (not derived intermediates). if (!opts.skipSegmentValidation) { for (const seg of segs) { const ok = await isValidMp4(seg, { ffprobeBin: opts.ffprobeBin }); if (!ok) { throw new VclawError( 'ffmpeg_failed', `stitch input segment is a corrupt or truncated MP4 (ffprobe could not read a valid duration): ${seg}`, { segment: seg }, ); } } } // --- Real execution --- await mkdir(dirname(input.outputPath), { recursive: true }); // Per-segment normalization pre-pass: concat then reads the normalized clips. if (needsPrep) { await mkdir(prepDir, { recursive: true }); for (const step of plan) { if (step.kind === 'segment-prep') await runFfmpeg(step.args, opts); } } // Reading-hold clips (motion-comic): built from each designated segment's first // frame AFTER any prep pass, then inserted before it in the concat. if (holdIndices.size > 0) { await mkdir(holdDir, { recursive: true }); for (const step of plan) { if (step.kind === 'reading-hold') await runFfmpeg(step.args, opts); } } if (effectiveStrategy === 'demuxer') { await mkdir(dirname(concatListPath), { recursive: true }); await writeFile(concatListPath, buildConcatListContent(finalConcatSegs), 'utf8'); } await runFfmpeg(concatArgs, opts); let finalPath = concatOutput; // A mix step references the concat's own track (`[0:a]`), and a concat of // silent segments (raw provider clips with no track) has none: ffmpeg then // fails with "matches no streams" AFTER the concat has rendered. Probe once // and hand the builders the truth; unprobeable → assume audio, the legacy // shape, so the failure mode stays the loud one rather than a wrong graph. const concatProbe = (hasExtraLayers || (hasMusic && input.music)) ? await probeMedia(concatOutput, { ffprobeBin: opts.ffprobeBin }).catch(() => undefined) : undefined; const concatSilent = concatProbe?.audioPresent === false; // The same probe carries the duration; only spawn ffprobe again if it did not. const probedConcatMs = async (): Promise => typeof concatProbe?.durationSeconds === 'number' && concatProbe.durationSeconds > 0 ? Math.round(concatProbe.durationSeconds * 1000) : ffprobeDuration(concatOutput, { ffprobeBin: opts.ffprobeBin }); if (hasExtraLayers) { // Multi-layer mix: dialogue/sfx (+ optional music bed) over the concat. // buildMultiLayerMixArgs uses duration=first (driven by the video input), // so it needs no probed total — the args are already final from planning, // unless the picture turned out silent, in which case the mix is rebuilt // without the anchor and capped at the probed picture length. const mixStep = plan.find((s) => s.kind === 'multi-layer-mix'); let mixArgs = mixStep?.args ?? buildMultiLayerMixArgs(concatOutput, combinedAudioLayers(input), input.outputPath, { ...(input.duckMusicUnderVoice ? { duckMusicUnderVoice: true } : {}), ...(input.dropSegmentAudio ? { dropVideoAudio: true } : {}), }); if (concatSilent) { const concatMs = await probedConcatMs(); mixArgs = buildMultiLayerMixArgs(concatOutput, combinedAudioLayers(input), input.outputPath, { ...(input.duckMusicUnderVoice ? { duckMusicUnderVoice: true } : {}), ...(input.dropSegmentAudio ? { dropVideoAudio: true } : {}), videoHasAudio: false, totalDurationSec: concatMs / 1000, }); if (mixStep) mixStep.args = mixArgs; } await runFfmpeg(mixArgs, opts); finalPath = input.outputPath; } else if (hasMusic && input.music) { // The concat duration drives the real fade-start + -t cap. const concatMs = await probedConcatMs(); const musicArgs = buildMusicMixArgs(concatOutput, input.music.trackPath, input.outputPath, { volume: input.music.volume, fadeOutSec: input.music.fadeOutSec, totalDurationSec: concatMs / 1000, voiceForward: input.music.voiceForward, ...(input.dropSegmentAudio ? { dropVideoAudio: true } : {}), ...(concatSilent ? { videoHasAudio: false } : {}), }); // Refresh the plan's music step with the duration-resolved args. const musicStep = plan.find((s) => s.kind === 'music-mix'); if (musicStep) musicStep.args = musicArgs; await runFfmpeg(musicArgs, opts); finalPath = input.outputPath; } const durationMs = await ffprobeDuration(finalPath, { ffprobeBin: opts.ffprobeBin }); // Touch stat so a 0-byte output surfaces as a runtime failure path-side. await stat(finalPath); return { status: 'complete', outputPath: finalPath, orderedSegments: segs, concatStrategy: strategy, music: hasMusic, plan, durationMs, }; }