import { readFile } from 'node:fs/promises'; import { rewriteEmotionAsPhysical } from './emotion-cues.js'; import { CAMERA_MOVE_VOCABULARY, SHOT_TYPE_VOCABULARY, SHOT_SIZE_VOCABULARY, LENS_VOCABULARY, ANGLE_VOCABULARY, matchVocabularyTerm, } from './prompt-quality.js'; import type { CategoryDescriptor } from './category-registry.js'; import { soundDesign, beats } from './cinematography.js'; import { STYLE_BY_ID } from './style-register.js'; export interface MultiShotPreset { name: string; totalSeconds: number; minShotSeconds: number; maxShotSeconds: number; minShots: number; maxShots: number; maxChars: number; styleLine: string; audioLine: string; } export const CINEMATIC_15S_PRESET: MultiShotPreset = { name: 'cinematic-15s', totalSeconds: 15, minShotSeconds: 2, maxShotSeconds: 5, minShots: 3, maxShots: 7, maxChars: 1500, styleLine: 'Cool shadows, natural skin tones. IMAX-scale composition, deep focus, practical lighting. High contrast, grounded realism. In the style of a Christopher Nolan movie.', audioLine: 'Diegetic sound only — natural ambience, environmental foley, and subject-driven sound.', }; export const SEEDANCE_10S_PRESET: MultiShotPreset = { name: 'seedance-10s', totalSeconds: 10, minShotSeconds: 2, maxShotSeconds: 5, minShots: 2, maxShots: 5, maxChars: 1500, styleLine: CINEMATIC_15S_PRESET.styleLine, audioLine: CINEMATIC_15S_PRESET.audioLine, }; export const VEO_8S_PRESET: MultiShotPreset = { name: 'veo-8s', totalSeconds: 8, minShotSeconds: 2, maxShotSeconds: 4, minShots: 2, maxShots: 4, maxChars: 1500, styleLine: CINEMATIC_15S_PRESET.styleLine, audioLine: CINEMATIC_15S_PRESET.audioLine, }; export const RUNWAY_10S_PRESET: MultiShotPreset = { name: 'runway-10s', totalSeconds: 10, minShotSeconds: 2, maxShotSeconds: 5, minShots: 2, maxShots: 5, maxChars: 1000, styleLine: CINEMATIC_15S_PRESET.styleLine, audioLine: CINEMATIC_15S_PRESET.audioLine, }; export const MUSIC_VIDEO_PRESET: MultiShotPreset = { name: 'music-video-15s', totalSeconds: 15, minShotSeconds: 1, maxShotSeconds: 4, minShots: 4, maxShots: 12, maxChars: 1500, styleLine: 'Saturated stage color, rhythmic lighting, performance energy. Bold contrast, expressive grade, beat-driven cutting.', audioLine: 'Music-driven — the track sets the tempo and cuts land on the beat, with diegetic accents layered under the mix.', }; export const SOCIAL_HOOK_PRESET: MultiShotPreset = { name: 'social-hook-8s', totalSeconds: 8, minShotSeconds: 1, maxShotSeconds: 3, minShots: 3, maxShots: 8, maxChars: 1000, styleLine: 'Bright phone-native vertical look, punchy saturation, fast pattern-interrupt cutting, captiony hook energy.', audioLine: 'Punchy social mix — a hook SFX up front, trending-audio energy, and captiony beats driving fast cuts.', }; const PRESET_REGISTRY: ReadonlyMap = new Map([ [CINEMATIC_15S_PRESET.name, CINEMATIC_15S_PRESET], [SEEDANCE_10S_PRESET.name, SEEDANCE_10S_PRESET], [VEO_8S_PRESET.name, VEO_8S_PRESET], [RUNWAY_10S_PRESET.name, RUNWAY_10S_PRESET], [MUSIC_VIDEO_PRESET.name, MUSIC_VIDEO_PRESET], [SOCIAL_HOOK_PRESET.name, SOCIAL_HOOK_PRESET], ]); /** * Resolve a preset style line for a genre. Falls back to the cinematic Nolan * line (today's hardcoded default) for unknown/absent genres. * * The style lines live in {@link VISUAL_STYLES} (`style-register.ts`) so this * and the character-sheet / storyboard-grid emitters in `filmmaking-prompts.ts` * read ONE source — a style applied to the clip but not the sheet drifts away * from the sheet that locked the character's identity. * * Two behaviours are deliberately preserved from the map this replaced: * NO alias resolution (`resolveGenreStyle` aliases, this does not — aliasing * here would change what `vlog` and `cinematic` emit), and NO trimming (so * `' noir '` still falls through to the fallback). * * A registered style with NO `videoStyleLine` — `live-action` — must also fall * through. The check is truthiness rather than `??` so an empty string behaves * the same as an absent one; `'' ?? fallback` would wrongly yield `''`. */ export function resolveStyleLine(genre?: string): string { if (!genre) return CINEMATIC_15S_PRESET.styleLine; const line = STYLE_BY_ID.get(genre.toLowerCase())?.videoStyleLine; return line ? line : CINEMATIC_15S_PRESET.styleLine; } export function knownPresetNames(): readonly string[] { return Array.from(PRESET_REGISTRY.keys()); } // Provider/route hint → preset, keyed on the provider FAMILY (the first token of // the hint, e.g. `seedance-direct` → `seedance`, `veo-useapi` → `veo`, // `google-flow` → `google`). Matching the family token rather than a substring // avoids misfires like `veo-via-runway-proxy` resolving to runway, and keeps the // mapping next to the preset registry it points at — the single source of truth. const PROVIDER_FAMILY_PRESET: ReadonlyMap = new Map([ ['seedance', SEEDANCE_10S_PRESET.name], ['veo', VEO_8S_PRESET.name], ['google', VEO_8S_PRESET.name], ['flow', VEO_8S_PRESET.name], ['runway', RUNWAY_10S_PRESET.name], ]); export function presetNameForProvider(hint: string | undefined): string | undefined { if (!hint) return undefined; const family = hint.trim().toLowerCase().split(/[-_:/\s]/)[0]; return PROVIDER_FAMILY_PRESET.get(family); } export function listMultiShotPresets(): readonly MultiShotPreset[] { return Array.from(PRESET_REGISTRY.values()); } export function resolvePreset(name?: string): MultiShotPreset { if (name === undefined) return CINEMATIC_15S_PRESET; const preset = PRESET_REGISTRY.get(name); if (!preset) { throw new Error( `unknown preset "${name}" (known: ${knownPresetNames().join(', ')})`, ); } return preset; } // Suggested camera-grid vocabularies. Shot sizes/angles/lenses are local to the // framework; prompt-quality's SHOT_TYPE_VOCABULARY is only re-exported for // consumers (it is not used when building the plan — SHOT_SIZES is). const SHOT_SIZES = SHOT_SIZE_VOCABULARY; const LENSES = LENS_VOCABULARY; const ANGLES = ANGLE_VOCABULARY; const MOVEMENTS = CAMERA_MOVE_VOCABULARY; export interface ShotSlot { index: number; start: number; end: number; timecode: string; shotSize: string; lens: string; angle: string; movement: string; } export interface ParsedMultiShotShot extends ShotSlot { description: string; } export interface ShotPlan { preset: MultiShotPreset; shots: ShotSlot[]; /** * The PRNG seed this plan was generated from. Always populated — when the * caller does not pass `--seed`, buildShotPlan picks one and records it here, * so the exact plan is reproducible from the emitted JSON: passing this value * back as `--seed` regenerates the identical shot sequence. This is the * determinism anchor for "exactly what you know you're going to get". */ seed: number; } export interface BuildShotPlanOptions { shots?: number; seed?: number; } // Deterministic, seedable PRNG so plans vary across calls but are reproducible in tests. function mulberry32(seed: number): () => number { let a = seed >>> 0; return () => { a |= 0; a = (a + 0x6d2b79f5) | 0; let t = Math.imul(a ^ (a >>> 15), 1 | a); t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; return ((t ^ (t >>> 14)) >>> 0) / 4294967296; }; } export function formatTimecode(seconds: number): string { const mm = Math.floor(seconds / 60); const ss = seconds % 60; return `${String(mm).padStart(2, '0')}:${String(ss).padStart(2, '0')}`; } // Partition totalSeconds into `count` integer durations, each within [min, max]. function partitionDurations( total: number, count: number, min: number, max: number, rand: () => number, ): number[] { if (count * min > total || count * max < total) { throw new Error( `cannot partition ${total}s into ${count} shots within [${min}, ${max}]`, ); } const durations = new Array(count).fill(min); let remaining = total - count * min; const open = durations.map((_, i) => i); // indices still below max while (remaining > 0) { const pick = Math.floor(rand() * open.length); const i = open[pick]; durations[i] += 1; remaining -= 1; if (durations[i] >= max) open.splice(pick, 1); } return durations; } function pickNonRepeating(pool: readonly T[], prev: T | undefined, rand: () => number): T { if (pool.length === 1) return pool[0]; const candidates = prev === undefined ? pool : pool.filter((v) => v !== prev); return candidates[Math.floor(rand() * candidates.length)]; } export function buildShotPlan( preset: MultiShotPreset, options: BuildShotPlanOptions = {}, ): ShotPlan { // Resolve the seed once and record it on the returned plan so the exact // sequence is reproducible from disk (pass it back via --seed). const resolvedSeed = options.seed ?? Math.floor(Math.random() * 1e9); const rand = mulberry32(resolvedSeed); const arithMin = Math.ceil(preset.totalSeconds / preset.maxShotSeconds); const arithMax = Math.floor(preset.totalSeconds / preset.minShotSeconds); const minCount = Math.max(preset.minShots, arithMin); const maxCount = Math.min(preset.maxShots, arithMax); if (minCount > maxCount) { throw new Error( `preset "${preset.name}": shot-count window [${preset.minShots}, ${preset.maxShots}] cannot satisfy duration partition [${arithMin}, ${arithMax}]`, ); } let count = options.shots ?? minCount + Math.floor(rand() * (maxCount - minCount + 1)); // Clamp to [minCount, maxCount] so an explicit --shots stays feasible. if (count < minCount) count = minCount; if (count > maxCount) count = maxCount; const durations = partitionDurations( preset.totalSeconds, count, preset.minShotSeconds, preset.maxShotSeconds, rand, ); const shots: ShotSlot[] = []; let cursor = 0; let prevSize: string | undefined; let prevLens: string | undefined; let prevAngle: string | undefined; let prevMove: string | undefined; for (let i = 0; i < count; i += 1) { const start = cursor; const end = cursor + durations[i]; cursor = end; const shotSize = pickNonRepeating(SHOT_SIZES, prevSize, rand); const lens = pickNonRepeating(LENSES, prevLens, rand); const angle = pickNonRepeating(ANGLES, prevAngle, rand); const movement = pickNonRepeating(MOVEMENTS, prevMove, rand); prevSize = shotSize; prevLens = lens; prevAngle = angle; prevMove = movement; shots.push({ index: i, start, end, timecode: `[${formatTimecode(start)} - ${formatTimecode(end)}]`, shotSize, lens, angle, movement, }); } return { preset, shots, seed: resolvedSeed }; } export function assembleMetadataBlock( preset: MultiShotPreset, location: string, timeOfDay: string, genre?: string, constraintsLine?: string, ): string { const loc = timeOfDay ? `${location}, ${timeOfDay}` : location; const styleLine = genre !== undefined ? resolveStyleLine(genre) : preset.styleLine; const lines = [ `Location: ${loc}`, `Style: ${styleLine}`, `Audio: ${preset.audioLine}`, ]; // Opt-in 4th line. It stays opt-in because it costs ~68 chars and // `runway-10s` has barely any headroom left after the scaffold; an // unconditional line would push that preset over budget. The validator's // metadata test is three independent presence regexes, so an extra labeled // line is invisible to it — this is additive, not a contract change. if (constraintsLine) lines.push(`Constraints: ${constraintsLine}`); return lines.join('\n'); } // Compose a full prompt body from a plan whose shots already carry `description`. export function composePromptText( plan: Array & { line: string }>, metadataBlock: string, ): string { const body = plan.map((s) => `${s.timecode} ${s.line}`).join('\n\n'); return `${body}\n\n${metadataBlock}`; } // Render a ShotPlan in Seedance's native prompt format: one flowing paragraph // with inline labeled segments (Style & Mood / Dynamic Description / Static // Description), a camera block in the existing per-shot emitter phrasing, and an // Audio footer. Pure and deterministic — it only reads `plan`/`descriptor`. export function composeSeedanceParagraph( plan: ShotPlan, descriptor: CategoryDescriptor, ): string { const { preset, shots } = plan; // Camera block: reuse the per-shot "shotSize, lens, angle, movement" phrasing, // collapsed onto one line so the paragraph stays a single block. const cameraBlock = shots .map((s) => `${s.shotSize}, ${s.lens}, ${s.angle}, ${s.movement}`) .join('; '); const motion = shots.map((s) => s.movement).join(', '); // Style & Mood is driven by the category's genre so a category actually looks // like its genre (e.g. cgi/cartoon/anime), not the generic Nolan preset line. // live-action genres have no GENRE_STYLE_LINES entry, so resolveStyleLine // returns the same Nolan line as the preset — those categories stay byte-stable. const styleMood = `${descriptor.label} — ${resolveStyleLine(descriptor.genre)}`; const dynamic = `the ${descriptor.subjectType} carries the action across ${shots.length} continuous beats (${motion})`; const staticScene = `${descriptor.label} scene, ${descriptor.genre} look, beat structure ${descriptor.beatTemplate}`; const segments = [ `Style & Mood: ${styleMood}`, `Dynamic Description: ${dynamic}`, `Static Description: ${staticScene}`, `Camera: ${cameraBlock}`, `Audio: ${preset.audioLine}`, ]; // Single space joins keep this one flowing paragraph (no "\n\n" block breaks). return segments.join(' '); } // Render a ShotPlan as a structured per-shot video-prompt layout: one block per // shot headed `SHOT — `, followed by labeled lines (Framing / Scene / // Dialogue / SFX / Camera), closed by a single Audio footer from the preset. // Blocks are separated by blank lines. Pure and deterministic — it only reads // `plan`/`descriptor`. ShotSlot carries no per-shot name/scene/dialogue/sfx // fields, so those are derived deterministically (NAME falls back to the shot // size, then `Shot `; Dialogue/SFX render an em-dash placeholder when absent). export function composePerShotFormat( plan: ShotPlan, descriptor: CategoryDescriptor, ): string { const { preset, shots } = plan; const blocks = shots.map((s, i) => { const n = i + 1; const name = s.shotSize || `Shot ${n}`; const framing = `${s.shotSize}, ${s.angle}, ${s.movement}, ${s.lens}`; const scene = `${descriptor.label} — ${descriptor.subjectType} carries beat ${n} of ${shots.length} (${s.movement}); ${descriptor.genre} look, ${descriptor.beatTemplate} structure`; return [ `SHOT ${n} — ${name}`, `Framing: ${framing}`, `Scene: ${scene}`, `Dialogue: —`, `SFX: ${soundDesign(descriptor.audioProfile)}`, `Camera: ${s.movement}`, ].join('\n'); }); return `${blocks.join('\n\n')}\n\nAudio: ${preset.audioLine}`; } /** The standing no-baked-text rule, used as the default `Constraints:` line. */ export const DEFAULT_CONSTRAINTS_LINE = 'no text, logos, or readable writing anywhere in frame.'; export interface TimecodedContext { location: string; timeOfDay: string; genre?: string; /** Emit the opt-in 4th tail line. */ constraintsLine?: string; /** Budget the CALLER will consume after this returns, reserved up front. */ reservedChars?: number; /** Per-shot prose overrides, keyed by 0-based shot index. */ shotLines?: ReadonlyMap; } /** Render `, , , ` — the part that is never trimmed. */ function shotSpec(slot: ShotSlot): string { const size = slot.shotSize.charAt(0).toUpperCase() + slot.shotSize.slice(1); return `${size}, ${slot.lens}, ${slot.angle}, ${slot.movement}`; } /** * Does `line` still read as the slot's own camera parameters? Beat prose can * SHADOW the spec — the `turntable` template's "continue the orbit through the * rear profile" makes a line read as `orbit` no matter what the slot says, * because matching is longest-first. When that happens the validator sees two * consecutive shots sharing a movement and reports a repeat that is not real. */ function specSurvives(line: string, slot: ShotSlot): boolean { return ( matchVocabularyTerm(line, [...SHOT_SIZE_VOCABULARY]) === slot.shotSize && matchVocabularyTerm(line, [...LENS_VOCABULARY]) === slot.lens && matchVocabularyTerm(line, [...ANGLE_VOCABULARY]) === slot.angle && matchVocabularyTerm(line, [...CAMERA_MOVE_VOCABULARY]) === slot.movement ); } /** * Map each shot to the beat its MIDPOINT falls inside, labelling with an * ordinal when one beat spans several shots (`Rising (1/2)`, `Rising (2/2)`). * The ordinal is what keeps consecutive lines from being byte-identical, which * would otherwise look like a copy-paste mistake in the emitted prompt. */ function beatLabels(plan: ShotPlan, descriptor: CategoryDescriptor): Array<{ label: string; direction: string }> { const timeline = beats(descriptor.beatTemplate, plan.preset.totalSeconds, descriptor.hookSeconds); const chosen = plan.shots.map((slot) => { const mid = (slot.start + slot.end) / 2; return timeline.find((b) => mid >= b.start && mid < b.end) ?? timeline[timeline.length - 1]; }); const totals = new Map(); for (const beat of chosen) totals.set(beat.label, (totals.get(beat.label) ?? 0) + 1); const seen = new Map(); return chosen.map((beat) => { const total = totals.get(beat.label) ?? 1; if (total === 1) return { label: beat.label, direction: beat.direction }; const n = (seen.get(beat.label) ?? 0) + 1; seen.set(beat.label, n); return { label: `${beat.label} (${n}/${total})`, direction: beat.direction }; }); } /** * Render a ShotPlan in the canonical timecoded format: one `[MM:SS - MM:SS]` * paragraph per shot, then the `Location:` / `Style:` / `Audio:` tail. This is * the format `parseMultiShotPrompt` reads and `runMultiShotChecks` validates — * the round trip already existed, only the emit side was missing. * * Prose is scaffolding, not finished writing: it comes from the category's beat * template so the command emits a complete, VALIDATING prompt with no model in * the loop, which an operator then rewrites shot by shot. The `Label:` prefix * is deliberate — it makes each line unmistakably a placeholder and greps for * slots nobody has filled in yet. * * Three trim rungs, taking the first that fits the preset budget: * 1. ` —