/** * Concat strategy and the ffmpeg argument builders for the stitch keystone: * segment ordering, demuxer vs filter selection, the concat list, and the * per-segment prep pass (clip, letterbox, grade, on-twos, grain, sharpen, * normalise). Pure: every function returns an argv array. Extracted verbatim * from `stitch.ts` (roadmap Phase 3c), which re-exports every public name. */ import { resolve as resolvePath } from 'node:path'; import { trimTailArgs, letterboxFilter } from './ffmpeg.js'; import { FILTER_FALLBACK_SEGMENT_THRESHOLD, STANDARD_AUDIO_BITRATE, ON_TWOS_FILTER, SHARPEN_FILTER, filmGrainFilter } from './stitch-filters.js'; import type { StitchInput } from './stitch.js'; /** Which concat path to use. `auto` picks demuxer/filter by segment count. */ export type ConcatStrategy = 'demuxer' | 'filter' | 'auto'; /** Assemble the full ordered segment list: intro + body + outro. */ export function orderedSegments(input: StitchInput): string[] { return [...(input.intro ?? []), ...input.segments, ...(input.outro ?? [])]; } /** * Choose the effective concat path. `demuxer`/`filter` are honored directly; * `auto` flips to the filter at {@link FILTER_FALLBACK_SEGMENT_THRESHOLD} * segments (the drift lesson from the Python). */ export function selectConcatStrategy( strategy: ConcatStrategy, segmentCount: number, ): 'demuxer' | 'filter' { if (strategy === 'demuxer' || strategy === 'filter') return strategy; return segmentCount >= FILTER_FALLBACK_SEGMENT_THRESHOLD ? 'filter' : 'demuxer'; } /** Render the concat-demuxer list file body: one `file ''` line per segment. */ export function buildConcatListContent(segments: string[]): string { // Mirrors stitch_bunty._concat_via_demuxer: file ''. // DEVIATION from the Python (which had this bug): single quotes in the path // are escaped ('\'') — an apostrophe anywhere in a parent directory name // would otherwise corrupt the concat list for every segment. return segments.map((seg) => `file '${resolvePath(seg).replace(/'/g, `'\\''`)}'\n`).join(''); } /** * Build the concat-DEMUXER ffmpeg args (PURE). The primary path. * * Ported VERBATIM from stitch_bunty._concat_via_demuxer (L240): * ffmpeg -y -f concat -safe 0 -i concat.txt -c copy output * (`-y` is prepended by `runFfmpeg`, so it is NOT included here.) * * A single ffmpeg invocation regardless of segment count — survives the * sandbox's per-session FFmpeg limit. Requires all segments to share encoding * params (true for 3e segments). */ export function buildConcatDemuxerArgs( _segments: string[], concatListPath: string, outputPath: string, ): string[] { return ['-f', 'concat', '-safe', '0', '-i', concatListPath, '-c', 'copy', outputPath]; } export interface BuildConcatFilterOptions { /** H.264 CRF (Python default 20). */ crf?: number; /** AAC bitrate (Python default "192k"). */ audioBitrate?: string; /** Audio sample rate Hz (Python default 44100). */ sampleRate?: number; /** Audio channel count (Python default 2). */ channels?: number; } /** * Build the concat-FILTER fallback ffmpeg args (PURE). For 8+ segments where * demuxer drift accumulates, or when a segment has incompatible codec params. * * Ported VERBATIM from ffmpeg_wrapper.concat_via_filter (L296-326): * -i f0 -i f1 ... -i f{n-1} * -filter_complex "[0:v][0:a][1:v][1:a]...concat=n=N:v=1:a=1[outv][outa]" * -map [outv] -map [outa] * -c:v libx264 -preset fast -crf 20 * -c:a aac -b:a 192k -ar 44100 -ac 2 * -movflags +faststart output */ export function buildConcatFilterArgs( segments: string[], outputPath: string, opts: BuildConcatFilterOptions = {}, ): string[] { const crf = opts.crf ?? 20; const audioBitrate = opts.audioBitrate ?? STANDARD_AUDIO_BITRATE; const sampleRate = opts.sampleRate ?? 44100; const channels = opts.channels ?? 2; const inputArgs: string[] = []; for (const seg of segments) { inputArgs.push('-i', seg); } // [0:v][0:a][1:v][1:a]...[n-1:v][n-1:a]concat=n=N:v=1:a=1[outv][outa] let filterInputs = ''; for (let i = 0; i < segments.length; i += 1) { filterInputs += `[${i}:v][${i}:a]`; } const filterComplex = `${filterInputs}concat=n=${segments.length}:v=1:a=1[outv][outa]`; return [ ...inputArgs, '-filter_complex', filterComplex, '-map', '[outv]', '-map', '[outa]', '-c:v', 'libx264', '-preset', 'fast', '-crf', String(crf), '-c:a', 'aac', '-b:a', audioBitrate, '-ar', String(sampleRate), '-ac', String(channels), '-movflags', '+faststart', outputPath, ]; } export interface BuildSegmentPrepOptions { /** Per-clip tail cut in seconds (WS9 trimTailArgs). Omit/<=0 disables the trim. */ clipMaxSeconds?: number; /** Letterbox target ratio label (WS9 letterboxFilter). Omit/'' disables the filter. */ letterboxRatio?: string; /** Path to a `.cube`/`.3dl` LUT applied via `lut3d` (e.g. a Kodak Vision3 500T LUT). Omit to skip. */ gradeLut?: string; /** Resolved FFmpeg grade filter chain (from {@link resolveGradeFilter}). Omit/'' to skip. */ gradeFilter?: string; /** Letterbox canvas width. Default 1280 (3e segment standard). */ width?: number; /** Letterbox canvas height. Default 720. */ height?: number; /** Output frame rate. Default 24. */ fps?: number; /** H.264 CRF. Default 20. */ crf?: number; /** AAC bitrate. Default {@link STANDARD_AUDIO_BITRATE}. */ audioBitrate?: string; /** Audio sample rate Hz. Default 44100. */ sampleRate?: number; /** Audio channels. Default 2. */ channels?: number; /** * Pad the audio track with trailing silence to exactly the video duration * (`apad` + `-shortest`), so the prepped segment is A/V-aligned and a demuxer * `-c copy` concat of such segments cannot accumulate drift (the "echo" / * overlapping-narration bug from the film builds). It also TRIMS audio that * runs past the video, preventing one scene's narration bleeding into the next. * Default true; `stitch()` auto-disables it for a segment with no audio stream * (probed on real runs — `apad` needs an audio source on stricter ffmpeg * builds). The re-encode already emits AAC, so this only adds the pad/shortest, * not a second pass. */ padAudioToVideoDuration?: boolean; /** * Silence THIS segment's audio (`volume=0`) while leaving every other segment * untouched. The surgical form of {@link StitchInput.dropSegmentAudio}: when * an audio-native route invents dialogue into one shot of a no-dialogue film, * this loses that shot's ambience instead of the whole film's. */ muteAudio?: boolean; /** * Hold every second frame so motion steps at 12fps on a 24fps timebase — the * hand-drawn "on twos" cadence. See {@link ON_TWOS_FILTER}. */ onTwos?: boolean; /** Mild luma-only sharpen ({@link SHARPEN_FILTER}). */ sharpen?: boolean; /** Film-grain strength 0..100; 0 emits no filter. See {@link filmGrainFilter}. */ filmGrain?: number; /** * Force this segment onto an exact canvas. Required when the ordered segments * do NOT already share one size: the demuxer path concats with `-c copy`, * which assumes identical stream parameters, so mixed sizes yield a master * with wrong duration and frame rate rather than an error. */ normalizeSize?: { width: number; height: number }; } /** * Build the per-segment normalization ffmpeg args (PURE, WS9). Re-encodes one * clip to the 3e segment standard (1280×720@24 / libx264 preset-fast crf20 / AAC * 44100 stereo) while optionally trimming its tail ({@link trimTailArgs}) and * letterboxing it onto a canvas ({@link letterboxFilter}). Re-encoding to the * standard keeps the normalized segments uniform, so the demuxer `-c copy` concat * stays valid. By default the audio is also padded/trimmed to the video length * ({@link BuildSegmentPrepOptions.padAudioToVideoDuration}) so segments are * A/V-aligned and a `-c copy` concat cannot accumulate drift. `-t` is placed just * before the output path so it caps the OUTPUT. */ export function buildSegmentPrepArgs( segment: string, outputPath: string, opts: BuildSegmentPrepOptions = {}, ): string[] { const width = opts.width ?? 1280; const height = opts.height ?? 720; // Compose the vf chain in deterministic order: letterbox (geometry) → LUT // (film-stock emulation) → named grade (state color) → on-twos (motion // cadence) → sharpen → grain. Each is optional; empties are dropped, so // default behavior is no `-vf` at all. // // The last three are ordered, not arbitrary. Sharpen comes BEFORE grain so // the grain itself is not sharpened into speckle, and grain comes AFTER the // fps step so it keeps shimmering at 24fps while the motion holds at 12 — // grain belongs to the stock, not to the drawing. const vf = [ opts.normalizeSize ? `scale=${opts.normalizeSize.width}:${opts.normalizeSize.height}:force_original_aspect_ratio=decrease,` + `pad=${opts.normalizeSize.width}:${opts.normalizeSize.height}:(ow-iw)/2:(oh-ih)/2:black,setsar=1` : '', letterboxFilter(opts.letterboxRatio, width, height), opts.gradeLut ? `lut3d=${opts.gradeLut}` : '', opts.gradeFilter ?? '', opts.onTwos ? ON_TWOS_FILTER : '', opts.sharpen ? SHARPEN_FILTER : '', opts.filmGrain !== undefined ? filmGrainFilter(opts.filmGrain) : '', ] .filter((f) => f !== '') .join(','); const args: string[] = ['-i', segment]; if (vf) args.push('-vf', vf); args.push( '-r', String(opts.fps ?? 24), '-c:v', 'libx264', '-preset', 'fast', '-crf', String(opts.crf ?? 20), '-c:a', 'aac', '-b:a', opts.audioBitrate ?? STANDARD_AUDIO_BITRATE, '-ar', String(opts.sampleRate ?? 44100), '-ac', String(opts.channels ?? 2), '-movflags', '+faststart', ); // Pad/trim the audio to exactly the video length so the prepped segment is // A/V-aligned: apad makes the audio effectively infinite, -shortest then ends // the OUTPUT at the (finite) video stream. Keeps a demuxer `-c copy` concat // drift-free. Default on; opt out for a known audioless segment. // `volume=0` silences THIS segment only, so a film can lose the audio of the // one shot a route misbehaved on and keep the diegetic ambience of the rest. // It runs before apad so the padding is silence too. const af = [opts.muteAudio ? 'volume=0' : '', opts.padAudioToVideoDuration !== false ? 'apad' : ''] .filter((f) => f !== '') .join(','); if (af) { args.push('-af', af); if (opts.padAudioToVideoDuration !== false) args.push('-shortest'); } // -t (when set) just before the output path → caps the output duration. args.push(...trimTailArgs(opts.clipMaxSeconds), outputPath); return args; }