/** * Assemble-stage orchestrator (sub-slice 3i) — the capstone that ties the * post-execution assembly pipeline together behind `vclaw video assemble`. * * Runs the building blocks shipped in 3b–3h IN ORDER and collects an * `AssembleManifestEntry[]`: * 1. (optional) extractPdfSlides — PDF deck -> slide images (3c) * 2. (optional) generateTitleCard — branded title card (3d) * 3. animateSlide per slide — per-slide video segments (3e) * 4. generateTts per scene — per-scene narration audio (3b) * 5. (optional) generateMusic — background music bed (3f) * 6. stitch — final MP4 (3h) * 7. (advisory) qa-* checks — collected into warnings (3g) * * DRY-RUN is the tested surface. `assembleProject({ dryRun: true })` PLANS the * whole pipeline — every FFmpeg command + provider call is recorded into the * manifest/events WITHOUT executing anything or needing API keys. Real * execution (ffmpeg spawns + provider keys) is a HUMAN integration checkpoint, * explicitly out of scope for the unit tests (same boundary as 3e/3h). */ import { readFile } from 'node:fs/promises'; import { existsSync, readdirSync } from 'node:fs'; import { join, resolve as resolvePath } from 'node:path'; import type { VideoProjectWorkspace } from '../workspace.js'; import { artifactPathFor, writeArtifact } from '../artifact-store.js'; import { createAssembleReportArtifact, type AssembleReportArtifact } from '../artifacts.js'; import type { AssembleInput, AssembleManifestEntry, AssembleResult, } from './types.js'; import { extractPdfSlides } from './pdf.js'; import { generateTitleCard } from './title-card.js'; import { animateSlide, alignDurationToFrame } from './animate-slides.js'; import { generateTts } from './tts.js'; import { generateMusic } from './music.js'; import { stitch, type StitchInput, type ConcatStrategy } from './stitch.js'; import { buildAudioLayers, type AudioLayer } from './audio-mix-plan.js'; import { readProjectManifest } from '../workspace.js'; import { dialogueArtifactPathFor, type DialogueArtifact } from '../dialogue.js'; import { readSfxArtifact } from '../sfx.js'; import { runAssembleMediaQc } from './media-qc.js'; import { measureDeliveredDurationMs } from '../audio-platform/measure.js'; import { VclawError } from '../errors.js'; import { lintDialogue } from './qa-dialogue-lint.js'; import { checkNarration } from './qa-narration.js'; import { checkImageFilter } from './qa-image-filter.js'; /** Default per-scene narration duration (sec) used when no probe is available. */ const DEFAULT_SCENE_DURATION_SEC = 5; /** * Music-bed volume for a SELECTED soundtrack auto-attached in clip-stitch mode. * Deliberately far above the 0.05 dialogue-bed default: here the soundtrack IS * the score (sidechain-ducked under the narration layer when one exists). 0.55 * is the value shipped on the hermes-do-launch film (2026-07-04), whose finish * had to be hand-rolled in ffmpeg for want of exactly this wiring. */ const CLIP_STITCH_SOUNDTRACK_VOLUME = 0.55; /** Storyboard scene shape (subset we consume here). */ interface StoryboardScene { sceneIndex: number; description: string; dialogue?: string; durationSeconds?: number; scenePrompt?: { imagePrompt?: string }; /** Narrative color-language state = the post grade id applied to this scene's segment. */ colorState?: string; /** * True when the scene carries NO spoken narration/dialogue by design — its * `description` is a visual action block, not lines to be spoken (e.g. a * mograph motion-graphics clip). Silent scenes are excluded from the * dialogue/narration word-count QA, which would otherwise mis-read the action * text as over-limit lip-sync speech. */ silent?: boolean; } interface StoryboardArtifactShape { projectSlug?: string; scenes?: StoryboardScene[]; } /** * The assemble-relevant knobs read from the brand-profile.json. The shipped * brand-profile.schema.json is presenter-routing focused (presenterName, * characterId, voiceId, intro/outro assets …); the assemble stage reads what it * needs from it loosely and falls back to defaults for anything absent. Extra * assemble fields (deck/music/concat) are honored if present without failing * brand-profile validation, since they live alongside the required routing * fields. */ interface BrandProfileForAssemble { presenterName?: string; voiceId?: string; introAsset?: string; outroAsset?: string; /** Optional PDF deck to rasterize into slides. */ deckPdf?: string; /** Optional title-card config. */ titleCard?: { title: string; subtitle?: string; background?: string }; /** Optional background-music config (the nex-brand knob). */ music?: { enabled?: boolean; prompt?: string; durationSec?: number; volume?: number }; /** Concat strategy override (bunty -> demuxer/auto, nex -> filter). */ concatStrategy?: ConcatStrategy; /** Intro/outro pre-encoded segment paths, if the brand provides them. */ introSegments?: string[]; outroSegments?: string[]; } async function loadStoryboard( workspace: VideoProjectWorkspace, ): Promise { const storyboardPath = artifactPathFor(workspace, 'storyboard'); if (!existsSync(storyboardPath)) return []; const parsed = JSON.parse(await readFile(storyboardPath, 'utf-8')) as StoryboardArtifactShape; const scenes = parsed.scenes ?? []; return [...scenes].sort((a, b) => a.sceneIndex - b.sceneIndex); } async function loadBrandProfile( brandProfilePath?: string, ): Promise { if (!brandProfilePath) return undefined; if (!existsSync(brandProfilePath)) return undefined; return JSON.parse(await readFile(brandProfilePath, 'utf-8')) as BrandProfileForAssemble; } /** Narration text for a scene: explicit dialogue, else the description. */ function narrationFor(scene: StoryboardScene): string { return (scene.dialogue ?? scene.description ?? '').trim(); } /** * Discover the project's dialogue + sfx audio clips (project-relative paths in * the dialogue.json / sfx.json artifacts) and resolve them to absolute, * on-disk-verified paths for the stitch audio mix. * * Presence-driven and additive: a project with NO dialogue.json and NO * sfx.json (or whose clip files are missing) yields empty arrays, so the * caller leaves `stitchInput.audioLayers` unset → byte-identical legacy stitch. * Missing clip files are skipped (existsSync-guarded) rather than failing. */ async function discoverDialogueSfxClips( workspaceRoot: string, slug: string, projectDir: string, ): Promise<{ dialoguePaths: string[]; sfxPaths: string[] }> { const dialoguePaths: string[] = []; const sfxPaths: string[] = []; // Dialogue: dialogue.json → turns[].path (project-relative). const dialoguePath = dialogueArtifactPathFor(workspaceRoot, slug); if (existsSync(dialoguePath)) { try { const artifact = JSON.parse(await readFile(dialoguePath, 'utf-8')) as DialogueArtifact; for (const turn of artifact.turns ?? []) { if (!turn.path) continue; const abs = resolvePath(projectDir, turn.path); if (existsSync(abs)) dialoguePaths.push(abs); } } catch { // Malformed dialogue.json → treat as absent (no layers). Non-fatal. } } // SFX: sfx.json → clips[].path (project-relative). try { const sfx = await readSfxArtifact(workspaceRoot, slug); for (const clip of sfx?.clips ?? []) { if (!clip.path) continue; const abs = resolvePath(projectDir, clip.path); if (existsSync(abs)) sfxPaths.push(abs); } } catch { // Malformed sfx.json → treat as absent. Non-fatal. } return { dialoguePaths, sfxPaths }; } /** * Discover the per-scene rendered clips `vclaw video execute` downloads to * `outputs/scene-.mp4`. Used by clip-stitch mode * (`fromRenderedClips`). * * Enumeration is DRIVEN BY THE `outputs/` DIRECTORY, not the storyboard scene * count: every present `scene-.mp4` is stitched, in ascending numeric order, * joined to the storyboard scene of the same index for its metadata (color * grade, duration) where one exists — otherwise a minimal synthetic scene. This * decouples the master's clip count from the storyboard so that e.g. a 6-clip * project whose storyboard was later rebuilt to 2 scenes still stitches all 6. * `missing` = storyboard scene indices with NO clip (still surfaced as a * warning); `extra` = clip indices with no matching storyboard scene (a clip * present beyond the current storyboard — surfaced informationally). */ function discoverRenderedClips( projectDir: string, scenes: StoryboardScene[], ): { found: Array<{ scene: StoryboardScene; path: string }>; missing: number[]; extra: number[]; } { const outputsDir = join(projectDir, 'outputs'); const sceneByIndex = new Map(); for (const scene of scenes) sceneByIndex.set(scene.sceneIndex, scene); // Glob outputs/scene-.mp4 (N = non-negative integer), collect present indices. const presentIndices: number[] = []; if (existsSync(outputsDir)) { for (const entry of readdirSync(outputsDir)) { const m = /^scene-(\d+)\.mp4$/.exec(entry); if (m) presentIndices.push(Number(m[1])); } } presentIndices.sort((a, b) => a - b); const found: Array<{ scene: StoryboardScene; path: string }> = []; const extra: number[] = []; for (const index of presentIndices) { const scene = sceneByIndex.get(index) ?? { sceneIndex: index, description: '' }; if (!sceneByIndex.has(index)) extra.push(index); found.push({ scene, path: join(outputsDir, `scene-${index}.mp4`) }); } const presentSet = new Set(presentIndices); const missing = scenes.map((s) => s.sceneIndex).filter((i) => !presentSet.has(i)); return { found, missing, extra }; } /** * Orchestrate the assemble pipeline. Returns an `AssembleResult` whose * `manifest` records each produced (or planned, on dry-run) asset in pipeline * order, `events` is a human-readable step log, and `warnings` collects the * advisory QA findings. * * On `dryRun`, every step is PLANNED (manifest entries + events) but nothing is * generated and no API key is required. */ export async function assembleProject(input: AssembleInput): Promise { const { workspace, brandProfilePath, ffmpegBin } = input; const dryRun = input.dryRun ?? false; const fromClips = input.fromRenderedClips ?? false; const manifest: AssembleManifestEntry[] = []; const events: string[] = []; const warnings: string[] = []; const scenes = await loadStoryboard(workspace); const brand = await loadBrandProfile(brandProfilePath); const assembleDir = join(workspace.projectDir, 'assemble'); const slidesDir = join(assembleDir, 'slides'); const audioDir = join(assembleDir, 'audio'); const segmentsDir = join(assembleDir, 'segments'); const outputPath = join(workspace.projectDir, 'outputs', 'final.mp4'); // --- Step 1: PDF slide extraction (optional) ------------------------------- // Skipped entirely in clip-stitch mode (segments are rendered clips, not slides). let slidePaths: string[] = []; if (!fromClips) { if (brand?.deckPdf) { const pdfPath = resolvePath(workspace.projectDir, brand.deckPdf); events.push(`pdf: extract slides from ${brand.deckPdf}`); if (dryRun) { // Plan one slide per scene as the dry-run estimate (no PDF parse). slidePaths = scenes.map((s) => join(slidesDir, `slide_${String(s.sceneIndex).padStart(3, '0')}.png`)); } else { const pdf = await extractPdfSlides({ pdfPath, outputDir: slidesDir }); slidePaths = pdf.pages.map((p) => p.path); } } else { // No deck: each scene's slide is its produced image asset (placeholder path). slidePaths = scenes.map((s) => join(slidesDir, `slide_${String(s.sceneIndex).padStart(3, '0')}.png`)); } } // --- Step 2: title card (optional) ----------------------------------------- if (brand?.titleCard) { const titleCardPath = join(assembleDir, 'title-card.png'); events.push(`title-card: "${brand.titleCard.title}"`); const tc = await generateTitleCard({ title: brand.titleCard.title, subtitle: brand.titleCard.subtitle, background: brand.titleCard.background, outputPath: titleCardPath, dryRun, }); manifest.push({ kind: 'title-card', path: dryRun ? titleCardPath : tc.path, durationMs: 0, sizeBytes: 0, generator: 'assemble/title-card.ts', }); } // --- Steps 3+4: body segments ---------------------------------------------- // `segmentScenes` is the subset of scenes that actually produced a body // segment, in order — it aligns the narrative color grade in step 6 (in clip // mode, scenes missing a rendered clip are dropped from both). const segmentPaths: string[] = []; const segmentScenes: StoryboardScene[] = []; if (fromClips) { // Clip-stitch mode: body segments are the per-scene rendered clips from // `vclaw video execute` (outputs/scene-.mp4). Their native audio is kept, // so there is NO TTS narration step. const { found, missing, extra } = discoverRenderedClips(workspace.projectDir, scenes); if (missing.length > 0) { // A missing scene on a REAL run is a hard gate, not a warning: assemble // used to warn-and-stitch, and an 88-second "12-scene" master was staged // and delivered before anyone noticed scene 6 never rendered (ep57, // 2026-07-22). Dry runs still plan-and-warn (the tested planning // surface); --allow-missing-scenes is the explicit operator escape hatch. if (!dryRun && !(input.allowMissingScenes ?? false)) { throw new VclawError( 'assemble_missing_scene_clips', `clip-stitch: ${missing.length} storyboard scene(s) have no rendered clip at outputs/scene-.mp4: [${missing.join(', ')}]. ` + 'Render the missing scene(s) and re-run, or pass --allow-missing-scenes to stitch a deliberately partial master.', { missingScenes: missing }, ); } warnings.push( `clip-stitch: ${missing.length} storyboard scene(s) have no rendered clip at outputs/scene-.mp4: [${missing.join(', ')}]`, ); } if (extra.length > 0) { warnings.push( `clip-stitch: ${extra.length} rendered clip(s) beyond the current storyboard were still stitched (no matching scene metadata): [${extra.join(', ')}]`, ); } if (found.length === 0) { warnings.push( 'clip-stitch: no rendered clips found at outputs/scene-.mp4 — run `vclaw video execute` first.', ); } for (const { scene, path } of found) { events.push(`clip: scene ${scene.sceneIndex} -> ${path}`); segmentPaths.push(path); segmentScenes.push(scene); manifest.push({ kind: 'rendered-clip', path, durationMs: 0, sceneIndex: scene.sceneIndex, sizeBytes: 0, generator: 'vclaw video execute', }); } } else { // --- Step 4 (computed first): per-scene narration (TTS) ------------------ // TTS is needed to drive the per-slide segment durations in step 3, so we // plan it before animation even though the canonical pipeline lists it after. const ttsSegments = scenes.map((s) => ({ sceneIndex: s.sceneIndex, text: narrationFor(s) })); const narrationDurationMsByScene = new Map(); if (ttsSegments.length > 0 && (brand?.voiceId || dryRun)) { events.push(`tts: ${ttsSegments.length} scene narration(s)`); const tts = await generateTts({ segments: ttsSegments, voiceId: brand?.voiceId ?? 'dry-run-voice', outputDir: audioDir, dryRun, }); for (const scene of tts.scenes) { if (scene.durationMs > 0) { narrationDurationMsByScene.set(scene.sceneIndex, scene.durationMs); } manifest.push({ kind: 'narration', path: scene.path, durationMs: scene.durationMs, sceneIndex: scene.sceneIndex, sizeBytes: scene.sizeBytes, generator: 'assemble/tts.ts', }); } // Real runs also surface tts.manifest entries (already shaped); merge any // not already represented (defensive — dry-run returns an empty manifest). } else if (ttsSegments.length > 0) { warnings.push('tts skipped: no voiceId in brand profile (real run requires one).'); } // --- Step 3: per-slide animation -> segments ---------------------------- for (const scene of scenes) { const slidePath = slidePaths[scene.sceneIndex] ?? slidePaths[scenes.indexOf(scene)] ?? ''; const ttsPath = join(audioDir, `scene_${String(scene.sceneIndex).padStart(3, '0')}.mp3`); const segmentPath = join(segmentsDir, `seg_slide_${String(scene.sceneIndex).padStart(3, '0')}.mp4`); const narrationDurationMs = narrationDurationMsByScene.get(scene.sceneIndex); const durationSec = narrationDurationMs && narrationDurationMs > 0 ? narrationDurationMs / 1000 : scene.durationSeconds ?? DEFAULT_SCENE_DURATION_SEC; events.push(`animate: scene ${scene.sceneIndex} -> ${segmentPath}`); if (dryRun) { // Plan the segment without spawning ffmpeg. segmentPaths.push(segmentPath); segmentScenes.push(scene); manifest.push({ kind: 'slide-animation', path: segmentPath, durationMs: Math.round(alignDurationToFrame(durationSec) * 1000), sceneIndex: scene.sceneIndex, sizeBytes: 0, generator: 'assemble/animate-slides.ts', }); } else { const seg = await animateSlide( { slidePath, ttsPath, outputPath: segmentPath, durationSec, slideNum: scenes.indexOf(scene) + 1, numSlides: scenes.length, }, { ffmpegBin }, ); segmentPaths.push(seg.path); segmentScenes.push(scene); manifest.push({ kind: 'slide-animation', path: seg.path, durationMs: seg.durationMs, sceneIndex: scene.sceneIndex, sizeBytes: 0, generator: 'assemble/animate-slides.ts', }); } } } // --- Step 5: background music (optional) ----------------------------------- let musicPath: string | undefined; /** * The music bed actually attached to the mix, read by the post-stitch loop * check. Declared out here because the check needs the master duration, which * only exists after the stitch block closes. */ let attachedBedPath: string | undefined; if (brand?.music?.enabled) { musicPath = join(assembleDir, 'music.mp3'); const prompt = brand.music.prompt ?? 'Soft ambient background bed, instrumental.'; events.push('music: generate background bed'); const music = await generateMusic({ prompt, durationSec: brand.music.durationSec, outputPath: musicPath, dryRun, }); manifest.push({ kind: 'music', path: music.path, durationMs: music.durationMs, sizeBytes: 0, generator: 'assemble/music.ts', }); } // --- Step 6: stitch -> final MP4 ------------------------------------------- let finalOutputPath = outputPath; if (segmentPaths.length > 0) { // Narrative color language: each scene's colorState becomes the per-segment // grade. Align it to the ordered segments (intro + body + outro): intro/outro // carry no grade. Omitted entirely when no scene is tagged, so default output // is unchanged. segmentScenes is one-per-body-segment in the same (sorted) // order — in clip mode scenes missing a clip are already dropped from both. const bodyGradeIds = segmentScenes.map((s) => s.colorState); const segmentGradeIds = bodyGradeIds.some((g) => typeof g === 'string' && g !== '') ? [ ...new Array(brand?.introSegments?.length ?? 0).fill(undefined), ...bodyGradeIds, ...new Array(brand?.outroSegments?.length ?? 0).fill(undefined), ] : undefined; // Discover dialogue + sfx clips and mix them (with the existing music bed) // as GLOBAL audio layers at the stitch step. Presence-driven & additive: // a project without dialogue.json/sfx.json (or with missing clip files) // yields no extra layers, leaving `audioLayers` unset → byte-identical // legacy stitch (music-only buildMusicMixArgs, or no-audio). // // Clip-stitch used to skip these entirely, on the assumption that a clip // always brings its own diegetic sound. That assumption died with // `--drop-clip-audio`: an audio-native route bakes MUSIC into its clips, so // a scored film ends up with two scores fighting, and the only cure is to // drop the clip track — which also removed every sound effect and left no // way to put one back. Discovery is presence-driven, so a project that // never ran `vclaw video sfx` still yields nothing and stitches unchanged. const { dialoguePaths, sfxPaths } = await discoverDialogueSfxClips( workspace.root, workspace.slug, workspace.projectDir, ); // Clip-stitch audio auto-attach (presence-driven, additive): a project that // ran `vclaw video narrate` and/or selected a soundtrack (`soundtrack // --select`, which writes the manifest `soundtrack` field) gets them mixed // at the stitch step — narration as a global voice layer over the kept clip // audio, the selected soundtrack as the score bed sidechain-ducked under // it. Neither artifact present → no layers → byte-identical legacy // clip-stitch. (Production evidence: the hermes-do-launch film, 2026-07-04, // had to hand-roll exactly this mix in raw ffmpeg.) let clipAudioLayers: AudioLayer[] | undefined; let clipNarrationAttached = false; let clipSoundtrackAttached = false; if (fromClips) { const narrationPath = join(workspace.projectDir, 'artifacts', 'audio', 'narration.mp3'); const haveNarration = existsSync(narrationPath); const projectManifest = await readProjectManifest(workspace); const soundtrackRel = projectManifest?.soundtrack ?? undefined; const soundtrackAbs = soundtrackRel ? resolvePath(workspace.projectDir, soundtrackRel) : undefined; const haveSoundtrack = Boolean(soundtrackAbs && existsSync(soundtrackAbs)); // brand.music generates its own bed via `input.music` — never double-bed. const attachSoundtrack = haveSoundtrack && !brand?.music?.enabled; if (haveSoundtrack && brand?.music?.enabled) { warnings.push( 'clip-stitch: both a selected soundtrack (manifest `soundtrack`) and brand.music are present — using the brand music bed; the selected soundtrack was NOT attached.', ); } // Designed SFX/dialogue are attached ONLY when the clips' own audio has // been dropped. A clip-stitch film that keeps its clip audio already has // diegetic sound and must stitch exactly as before — that is the contract // this mode was built on. The dead end being fixed is narrower: once // `--drop-clip-audio` silences the picture (the cure for an audio-native // route baking its own score into every clip), there was no way to put // any sound effect back at all. const attachDesigned = input.dropClipAudio === true; const useDialogue = attachDesigned ? dialoguePaths : []; const useSfx = attachDesigned ? sfxPaths : []; if (haveNarration || attachSoundtrack || useSfx.length > 0 || useDialogue.length > 0) { clipAudioLayers = buildAudioLayers( { ...(attachSoundtrack && soundtrackAbs ? { musicPath: soundtrackAbs } : {}), ...(haveNarration ? { narrationPath } : {}), ...(useDialogue.length > 0 ? { dialoguePaths: useDialogue } : {}), ...(useSfx.length > 0 ? { sfxPaths: useSfx } : {}), }, { musicVolume: input.musicVolume ?? CLIP_STITCH_SOUNDTRACK_VOLUME }, ); clipNarrationAttached = haveNarration; clipSoundtrackAttached = attachSoundtrack; if (attachSoundtrack && soundtrackAbs) attachedBedPath = soundtrackAbs; if (useSfx.length > 0) { events.push(`audio-mix: ${useSfx.length} designed SFX clip(s) attached as layers`); } } } const audioLayers = fromClips ? clipAudioLayers : dialoguePaths.length > 0 || sfxPaths.length > 0 ? buildAudioLayers({ dialoguePaths, sfxPaths }) : undefined; // Voice-forward the music bed by default whenever per-scene narration was // produced (the manifest carries `narration` entries): loudnorm the voice, // sidechain-duck the music under it, and limit the output — so narration is // never buried beneath the bed. No narration → byte-identical legacy bed mix. const hasNarration = manifest.some((m) => m.kind === 'narration'); // Drop the clips' own audio when the operator asks for it. On veo-useapi // the provider cannot be asked for a silent clip — Veo 3.x is audio-native // and the Flow API exposes no audio-off parameter — so // `executionProfile.generateAudio: false` is recorded and then ignored, and // one delivered short came back with a character saying "I am flying // towards the sun!" mixed under the score by the auto-attach path below. // // This is NOT inferred from `scene.silent`. Silent means no DIALOGUE, not // no sound: the scene prompts request diegetic ambience by name, and an // earlier version of this fix stripped a lighthouse short's entire storm // (−25.9 LUFS of wanted atmosphere) to solve a problem that film never had. // The gate refuses a master containing speech; the operator decides whether // to re-render the shot or strip the track. See AssembleInput.dropClipAudio. const dropClipAudio = input.dropClipAudio === true; // Per-scene muting: silence only the shots a route misbehaved on, keeping // every other shot's ambience. Scene indices are translated to ORDERED // segment indices here (brand intro segments come first), because the // caller thinks in scenes and stitch thinks in segments. const muteScenes = new Set(input.muteClipAudioScenes ?? []); const introCount = brand?.introSegments?.length ?? 0; const muteSegmentAudio: number[] = []; if (muteScenes.size > 0) { const matched = new Set(); segmentScenes.forEach((scene, i) => { if (scene && muteScenes.has(scene.sceneIndex)) { muteSegmentAudio.push(i + introCount); matched.add(scene.sceneIndex); } }); // A scene index that matches nothing would otherwise be a silent no-op — // the operator would believe a shot was muted and ship it unmuted. const unmatched = [...muteScenes].filter((i) => !matched.has(i)); if (unmatched.length > 0) { warnings.push( `assemble.mute-clip-audio: no stitched clip for scene(s) [${unmatched.join(', ')}] — nothing was muted for them.`, ); } if (muteSegmentAudio.length > 0) { events.push( `audio-mix: clip audio muted for scene(s) [${[...matched].sort((a, b) => a - b).join(', ')}]; every other scene keeps its ambience`, ); } } const stitchInput: StitchInput = { segments: segmentPaths, intro: brand?.introSegments, outro: brand?.outroSegments, outputPath, concatStrategy: brand?.concatStrategy, ...(segmentGradeIds ? { segmentGradeIds } : {}), // Whole-cut finishing look. Presence-driven so an assemble without these // flags emits byte-identical ffmpeg args to before. ...(input.onTwos ? { onTwos: true } : {}), ...(input.sharpen ? { sharpen: true } : {}), ...(input.filmGrain !== undefined ? { filmGrain: input.filmGrain } : {}), ...(musicPath ? { music: { trackPath: musicPath, volume: brand?.music?.volume, ...(hasNarration ? { voiceForward: true } : {}) } } : {}), ...(audioLayers ? { audioLayers } : {}), // Duck the soundtrack under the narration in the multi-layer mix — the // clip-stitch analogue of the slide path's voiceForward. Only meaningful // when both a voice layer and a music layer are present. ...(clipNarrationAttached && (clipSoundtrackAttached || Boolean(musicPath)) ? { duckMusicUnderVoice: true } : {}), ...(dropClipAudio ? { dropSegmentAudio: true } : {}), ...(muteSegmentAudio.length > 0 ? { muteSegmentAudio } : {}), }; if (audioLayers) { events.push( fromClips ? `audio-mix: clip-stitch auto-attach — ${clipNarrationAttached ? 'narration' : ''}${clipNarrationAttached && clipSoundtrackAttached ? ' + ' : ''}${clipSoundtrackAttached ? 'selected soundtrack (ducked under voice)' : ''} mixed ${dropClipAudio ? 'over a SILENCED picture' : 'over clip audio'}` : `audio-mix: ${dialoguePaths.length} dialogue + ${sfxPaths.length} sfx clip(s) mixed as global layers` + (musicPath ? ' (+music bed)' : ''), ); } // Say it out loud. This discards audio the provider generated and the // prompts asked for, so an operator who wanted it needs to know where it // went — and it is a warning, not just an event, because it is a loss. if (dropClipAudio) { events.push( "audio-mix: --drop-clip-audio — the clips' own audio (including any diegetic ambience) is DROPPED; only the score and explicit layers remain", ); warnings.push( 'assemble.drop-clip-audio: the rendered clips\' audio was discarded on request. Diegetic ambience the scene prompts asked for went with it.', ); } events.push( `stitch: ${segmentPaths.length} segment(s) -> ${outputPath}` + (musicPath ? ' (+music)' : '') + (audioLayers ? ' (+dialogue/sfx)' : ''), ); const stitched = await stitch(stitchInput, { dryRun, ffmpegBin }); finalOutputPath = stitched.outputPath; for (const step of stitched.plan) { events.push(`stitch.plan: ${step.kind} -> ${step.outputPath}`); } manifest.push({ kind: 'final-video', path: stitched.outputPath, durationMs: stitched.durationMs, sizeBytes: 0, generator: 'assemble/stitch.ts', }); } else if (!fromClips) { // Clip mode already pushed a precise clip-stitch warning above. warnings.push('stitch skipped: no slide segments (empty storyboard).'); } const qc = dryRun ? undefined : await runAssembleMediaQc({ manifest, outputPath: finalOutputPath }); if (qc) { for (const issue of qc.issues) { warnings.push(`qc.${issue.code}[${issue.scope}]: ${issue.message}`); } } // A music bed shorter than the picture is not an error — `-stream_loop -1` // repeats it and the film is scored end to end. What is wrong is that this // happens SILENTLY, and the loop seam lands wherever the arithmetic puts it. // On one delivered short a 30.7s bed under a 64.1s cut seamed 6.7s into the // climax shot, and nothing in the pipeline mentioned it: `soundtrack.json` // recorded the 64s that was REQUESTED, so every duration in view agreed. // // Now that the artifact records the delivered length, say where the seam is. // Advisory, never blocking — a looped bed is a legitimate choice, and the // operator decides whether to commission a full-length cue or move the seam. const bedPath = attachedBedPath ?? musicPath; const masterMs = qc?.master?.durationMs; if (!dryRun && bedPath && masterMs && masterMs > 0) { const bedMs = await measureDeliveredDurationMs(bedPath, 0); // 1s of slack: a bed within a second of the picture has no audible seam. if (bedMs > 0 && bedMs + 1000 < masterMs) { const seams: string[] = []; for (let at = bedMs; at < masterMs && seams.length < 4; at += bedMs) { seams.push(`${(at / 1000).toFixed(1)}s`); } warnings.push( `assemble.music-bed-loops: the score is ${(bedMs / 1000).toFixed(1)}s under a ` + `${(masterMs / 1000).toFixed(1)}s cut, so it repeats. Loop seam(s) at ${seams.join(', ')}` + `${seams.length >= 4 ? ' (and on)' : ''} — check what lands there, or commission a full-length cue.`, ); } } // --- Step 7: advisory QA --------------------------------------------------- // Silent scenes (visual action blocks with no spoken lines — e.g. mograph // clips) are excluded from the dialogue/narration word-count checks, which // would otherwise mis-read the action text as over-limit lip-sync speech. const spokenScenes = scenes.filter((s) => !s.silent); if (spokenScenes.length > 0) { const dialogueResult = lintDialogue({ segments: spokenScenes.map((s) => ({ sceneIndex: s.sceneIndex, text: narrationFor(s) })), }); for (const w of dialogueResult.warnings) { warnings.push(`qa.dialogue[scene ${w.sceneIndex}/${w.rule}]: ${w.message}`); } const narrationResult = checkNarration({ scenes: spokenScenes.map((s) => ({ sceneIndex: s.sceneIndex, narration: narrationFor(s) })), slideCount: slidePaths.length || undefined, }); for (const w of narrationResult.warnings) { warnings.push(`qa.narration[scene ${w.sceneIndex}/${w.rule}]: ${w.message}`); } const imageFilterResult = checkImageFilter({ candidates: scenes .filter((s) => s.scenePrompt?.imagePrompt) .map((s) => ({ sceneIndex: s.sceneIndex, prompt: s.scenePrompt!.imagePrompt! })), }); for (const w of imageFilterResult.warnings) { warnings.push(`qa.image-filter[scene ${w.sceneIndex}/${w.verdict}]: ${w.message}`); } } const status: AssembleResult['status'] = dryRun ? 'dry-run' : warnings.length > 0 ? 'partial' : 'complete'; return { status, outputPath: finalOutputPath, manifest, events, warnings, ...(qc ? { qc } : {}), }; } /** * Persist an `assemble-report.json` artifact for a completed (or dry-run) * assemble pass via the TYPED `writeArtifact` helper (so the artifact is no * longer an "alternate writer" — it's schema-covered). Validates against * schemas/video/artifacts/assemble-report.schema.json. */ export async function writeAssembleReport( workspace: VideoProjectWorkspace, result: AssembleResult, brandProfilePath?: string, ): Promise<{ artifactPath: string; report: AssembleReportArtifact }> { const report = createAssembleReportArtifact({ projectSlug: workspace.slug, status: result.status, brandProfile: brandProfilePath ?? null, outputPath: result.outputPath, manifest: result.manifest.map((entry) => ({ kind: entry.kind, path: entry.path, durationMs: entry.durationMs, ...(entry.sceneIndex !== undefined ? { sceneIndex: entry.sceneIndex } : {}), sizeBytes: entry.sizeBytes, generator: entry.generator, })), warnings: result.warnings, events: result.events, ...(result.qc ? { qc: result.qc } : {}), }); const artifactPath = await writeArtifact(workspace, 'assemble-report', report); return { artifactPath, report }; }