/** * The Seedance packet composer: the 13-block prompt body, its block-order * check and the frame-map / dialogue / subject-lock helpers it composes from * the primitives in `seedance-blocks.ts`. Extracted verbatim from * `filmmaking-prompts.ts` (roadmap Phase 3b), which re-exports the public * names, so importers are unchanged. `seedance-blocks.ts` stays the * dependency-light primitive layer that `prompt-lint` also reads. */ import { performanceDirection, type FilmmakingCharacterContext } from './character-performance.js'; import { STARTING_POSE_GUIDANCE } from './cinema-motion-review.js'; import { captureRealismBlock, cameraSpec, musicSyncLine, phoneCaptureBlock, type DetailLevel, dynamicRegisterClause, strobeBlock } from './cinematography.js'; import type { BriefArtifact, StoryboardArtifact } from './artifacts.js'; import { crossFrameBlock, frameMapBlock, lastFrameBlock, subjectLockBlock, buildPositionalDescriptorLine, buildIdentityLockLine, noOnScreenTextBlock, captureCadenceBlock, atmosphereBlock, SINGLE_FULL_FRAME_GUARD, POSITIONAL_BINDING, type FrameMapEntry, type SubjectLockEntry } from './seedance-blocks.js'; import { withDialogue, type DialogueLine } from './multi-shot-prompt.js'; import type { ResolvedCinemaProfile } from './cinema-profile.js'; import type { FilmmakingReferenceSlot, FilmmakingSeedancePacket, FilmmakingPromptIssue } from './filmmaking-prompts.js'; import { RICH_CAMERA_MOVE, richCinematographySuffix, richAudioSuffix, type GenreStyle, resolveGenreStyle, richSuffixOptsFromProfile, cleanSentence, formatSeconds } from './filmmaking-shared.js'; // Canonical 13-block order for the text-driven Seedance master-prompt (WS6, // reshaped for Joey 3.0: two directive blocks lead, ATMOSPHERE is its own block). // Source of truth shared by seedancePromptText (which emits them) and // checkSeedanceBlockOrder (which validates them) so they cannot drift. const SEEDANCE_BLOCK_ORDER = [ // Two directive blocks lead: overlay text and shutter cadence are both // decided early in the frame, so the instruction has to sit early in the // prompt. Text suppression used to ride at the bottom inside LAST FRAME, // which is where it was competing with every block above it. 'NO ON-SCREEN TEXT', 'CAPTURE CADENCE', 'SCENE & MOOD', 'FRAME MAP', 'SUBJECT LOCK', 'CROSS-FRAME', 'MOVEMENT', 'ATMOSPHERE', 'LAST FRAME', 'WORLD PLATE', 'SOUND BED', 'CAPTURE REALISM', 'CAMERA CAPTURE', ] as const; /** * QA validator (WS6): confirm a text-driven Seedance packet carries all ten * Joey master-prompt blocks in the canonical order. Returns a warning-level * {@link FilmmakingPromptIssue} when a block is missing or out of sequence, or * `null` when the packet is well-formed. Pure/deterministic — operates on the * rendered prompt text only. */ export function checkSeedanceBlockOrder(promptText: string): FilmmakingPromptIssue | null { let last = -1; for (const block of SEEDANCE_BLOCK_ORDER) { const idx = promptText.indexOf(block); if (idx === -1) { return { code: 'seedance-block-order', severity: 'warning', message: `text-driven Seedance packet is missing the "${block}" block (expected the 13-block order: ${SEEDANCE_BLOCK_ORDER.join(' → ')}).`, }; } if (idx < last) { return { code: 'seedance-block-order', severity: 'warning', message: `text-driven Seedance packet has the "${block}" block out of order (expected the 13-block order: ${SEEDANCE_BLOCK_ORDER.join(' → ')}).`, }; } last = idx; } return null; } export function seedancePromptText(input: Parameters[0]): string { const canon = (input.scene.characters ?? []).map(name => performanceDirection(input.characterContext?.get(name)?.slot ?? input.characterContext?.get(name)?.description ?? name, input.characterContext?.get(name) ?? {})).filter(Boolean); return seedancePromptBody(input) + (canon.length ? `\nCHARACTER PERFORMANCE: ${canon.join(' ')}` : ''); } /** @internal — exported for testing the text-driven AUDIO / MOOD branch only. */ function seedancePromptBody(input: { scene: StoryboardArtifact['scenes'][number]; brief: BriefArtifact | undefined; references: FilmmakingReferenceSlot[]; variant: FilmmakingSeedancePacket['variant']; durationSeconds: number; noFaces?: boolean; genreStyle?: GenreStyle; aspectRatio?: string; characterContext?: Map; detail?: DetailLevel; /** * Resolved cinema profile. Drives the prose-vs-numeric register, whether the * CAPTURE REALISM block fires (realism), its haze/wet knobs, and the cinema-vs- * phone capture register. Omitted → the legacy default (numeric register, * realism on with the legacy no-arg capture block) so existing callers/tests * stay byte-stable. */ profile?: ResolvedCinemaProfile; /** * Opt-in (WS-C) canonical multi-reference text discipline. Default false → * output byte-identical to today. See {@link GenerateFilmmakingPromptsOptions.textDiscipline}. */ textDiscipline?: boolean; /** Scene audio flag, only consulted when {@link textDiscipline} is set. */ generateAudio?: boolean; /** * The active blueprint forbids handheld camera. When set, the * `storyboard-grid-reference` packet drops its default "Handheld camera moments * may be used" boilerplate so it does not contradict the blueprint (and the * forbidden-move checker no longer flags its own boilerplate). Default false → * the clause stays, byte-identical to legacy. */ forbidHandheld?: boolean; /** * Spoken dialogue (ai-filmmaking "Dialog scenes"): the text-driven packet * carries it on the opening FRAME MAP beat; the grid variants carry it on the * Storyline line. Speakers matching {@link characterContext} render as their * visual descriptor. Absent (default) → output byte-identical to today. */ dialogue?: DialogueLine; /** Rewrite named dialogue emotions as physical cues; only with `dialogue`. */ emotionCues?: boolean; }): string { const genreStyle = input.genreStyle ?? resolveGenreStyle(); const aspectRatio = input.aspectRatio ?? '16:9'; const detail = input.detail ?? 'standard'; const textDiscipline = input.textDiscipline ?? false; const profile = input.profile; const characterRefs = input.references.filter((reference) => reference.role === 'character-sheet'); const gridRef = input.references.find((reference) => reference.role === 'storyboard-grid'); const startFrame = input.references.find((reference) => reference.role === 'start-frame'); const duration = `${input.durationSeconds} seconds`; const action = cleanSentence(input.scene.scenePrompt?.animationPrompt ?? input.scene.description); const noFaceLine = input.noFaces ? 'Keep all figures as backlit silhouettes, backs, or distance with faces obscured (content-filter safe).' : ''; // Spoken-dialogue clause (ai-filmmaking Dialog scenes). Speakers that match a // stored character render as their visual descriptor, never a proper name — // the same character-lock discipline as SUBJECT LOCK. Absent → `speak` is the // identity function and every variant stays byte-identical to today. const dialogue = input.dialogue ? dialogueWithDescriptors(input.dialogue, input.characterContext) : undefined; const speak = (line: string): string => ( dialogue ? withDialogue(line, dialogue, { emotionCues: input.emotionCues ?? false }) : line ); // Opt-in WS-C discipline lines, computed once and spliced into each variant. // All-empty (default) → byte-identical output. The positional descriptor reuses // the same scene.characters → stored-descriptor mapping as SUBJECT LOCK, so the // emitted text is a visual descriptor per subject, never a proper name. const positionalLine = textDiscipline ? buildPositionalDescriptorLine(subjectLockEntriesFromContext(input)) : ''; const identityLockLine = textDiscipline ? buildIdentityLockLine() : ''; // Music suffix on the grid variants. Default keeps the hardcoded "NO MUSIC"; with // text-discipline on it follows the scene's generateAudio flag (audio → diegetic). const gridMusicClause = textDiscipline && input.generateAudio ? 'NO TEXT ON SCREEN. Diegetic soundscape — natural ambience, environmental foley, and subject-driven sound, no added music.' : 'NO TEXT ON SCREEN, NO MUSIC.'; // Invariant: ANY packet that references a storyboard grid OR a character // reference sheet must carry the single-frame guard, or the sheet/grid leaks // into the video as an animated split-screen / reproduced reference card. // Character sheets are themselves multi-panel layouts, so a character-sheets // packet needs the guard even when no storyboard grid is attached. With // text-discipline on, the guard is unconditional. Enforce it once at the exit // so a new variant can't silently omit it. `withGridGuard` is a no-op when the // body already includes the guard text (the explicit branches keep it inline). const withGridGuard = (body: string): string => ( (gridRef || textDiscipline || characterRefs.length > 0) && !body.includes(SINGLE_FULL_FRAME_GUARD) ? `${body}\n${SINGLE_FULL_FRAME_GUARD}` : body ); if (input.variant === 'character-sheets-plus-storyboard-grid' && gridRef) { // The per-character header lines carry the sheet label (which may include a // proper name). With text-discipline on they are suppressed in favour of the // positional visual-descriptor line (visual descriptors only, never names). const characterHeaders = textDiscipline ? [] : characterRefs.map((reference, index) => `Character ${index + 1}: ${reference.slot} (${reference.label})`); return withGridGuard([ ...characterHeaders, positionalLine, '', `Use the provided character sheets and cinematic storyboard grid ${gridRef.slot} as visual and motion reference. Create a ${duration} ${genreStyle.formatTone} sequence at ${aspectRatio}. ${SINGLE_FULL_FRAME_GUARD} Follow the panel order, camera logic, motion, and framing consistently and temporally. ${gridMusicClause}`, identityLockLine, noFaceLine, startFrame ? `Use ${startFrame.slot} as the scene start-frame continuity anchor.` : '', // The camera-body clause belongs on BOTH bodies. Every real project emits // a grid packet, so a clause added only to the text-driven variant ships // to nobody. profile?.dynamicRegister ? dynamicRegisterClause(profile.dynamicRegister) : '', '', `Storyline: ${speak(action)}`, ].filter(Boolean).join('\n')); } if (input.variant === 'storyboard-grid-reference' && gridRef) { // Drop the handheld-realism boilerplate when the blueprint forbids handheld // (otherwise the clause contradicts the bible and trips its own forbidden-move // checker). No blueprint / handheld allowed → clause stays, byte-identical. const handheldClause = input.forbidHandheld ? '' : 'Handheld camera moments may be used to boost realism. '; return withGridGuard([ // The two anti-artifact directive blocks lead here too. The grid variant // carries its identity discipline inline rather than as labelled blocks, // but these two are not identity — they suppress artifacts that apply to // any video render, and the grid packet is the WORST case for the text // one: the storyboard grid is a sheet of CAM/MOVE/MOOD annotation strips, // so it hands the model rendered text as a reference image. noOnScreenTextBlock(), captureCadenceBlock(profile?.captureRegister ?? 'cinema', detail, profile?.strobeBpm ? { strobeQuarantine: true } : {}), profile?.strobeBpm ? strobeBlock(profile.strobeBpm, detail) : '', positionalLine, `Use the provided cinematic storyboard grid ${gridRef.slot} as visual and motion reference. Create a ${duration} ${genreStyle.formatTone} sequence at ${aspectRatio}. ${SINGLE_FULL_FRAME_GUARD} Follow the panel order, camera logic, motion and camera framing consistently. ${handheldClause}${gridMusicClause}`, identityLockLine, noFaceLine, startFrame ? `Use ${startFrame.slot} as the scene start-frame continuity anchor.` : '', // The camera-body clause belongs on BOTH bodies. Every real project emits // a grid packet, so a clause added only to the text-driven variant ships // to nobody. profile?.dynamicRegister ? dynamicRegisterClause(profile.dynamicRegister) : '', '', `Storyline: ${speak(action)}`, ].filter(Boolean).join('\n')); } // Variant A (text-driven) — the Joey 13-block Seedance master-prompt (WS6, now // the default packet shape). The blocks are assembled in a fixed contract order // (SCENE & MOOD → FRAME MAP → SUBJECT LOCK → CROSS-FRAME → MOVEMENT → LAST FRAME // → WORLD PLATE → SOUND BED → CAPTURE REALISM → CAMERA CAPTURE) using the // shared seedance-blocks.ts / cinematography.ts emitters. `terse`/`standard` // detail emit no quantified cinematography tokens; `rich` appends them. // SOUND BED. With text-discipline on, the diegetic-vs-no-music choice is keyed // on the scene's generateAudio flag (audio → diegetic soundscape; otherwise the // existing no-music line). Default (text-discipline off) keeps today's wording. const soundBed = genreStyle.genre === 'music-video' ? `SOUND BED: Music-driven — ${musicSyncLine(undefined, detail)}.` : textDiscipline && input.generateAudio ? `SOUND BED: Diegetic soundscape — natural ambience, environmental foley, and subject-driven sound, no added music.${detail === 'rich' ? ` ${richAudioSuffix()}` : ''}` : `SOUND BED: No music. Natural ambience and subject-driven sound only.${detail === 'rich' ? ` ${richAudioSuffix()}` : ''}`; // FRAME MAP rows; with dialogue, the spoken line is baked into the opening // beat (the guide's "bake it into the TIMELINE in quoted brackets" rule). // Same weave as the machine-readable `timeline` field (shared helper). const frameMap = weaveDialogueIntoFrameMap( threeBeatFrameMap(input.durationSeconds, action, aspectRatio), input.dialogue, input.characterContext, input.emotionCues ?? false, ); return withGridGuard([ // Two directive blocks lead the packet. Overlay text and shutter cadence are // both decided early in the frame, so the instructions have to sit above the // descriptive body to carry weight (Joey 3.0 Cinema Director block order). noOnScreenTextBlock(), '', captureCadenceBlock(profile?.captureRegister ?? 'cinema', detail, profile?.strobeBpm ? { strobeQuarantine: true } : {}), '', // THE STROBE sits with the other directive blocks, above the body. Setting // a BPM is what turns the quarantine on above — the two are coupled here so // a strobe can never ship without it (that combination returns genuinely // broken frames, and it is not something to rely on remembering). profile?.strobeBpm ? strobeBlock(profile.strobeBpm, detail) : '', profile?.strobeBpm ? '' : '', `SCENE & MOOD: ${duration} ${genreStyle.formatTone} at ${aspectRatio}. ${cleanSentence(input.brief?.intent ?? input.scene.description)}.`, '', STARTING_POSE_GUIDANCE, frameMapBlock(frameMap), '', subjectLockBlock(subjectLockEntriesFromContext(input)), positionalLine, identityLockLine, '', crossFrameBlock(), '', timecodedMovementBlock(input.durationSeconds, action, aspectRatio, detail), profile?.dynamicRegister ? dynamicRegisterClause(profile.dynamicRegister) : '', '', // Haze gets its own block and is suppressed inside CAPTURE REALISM below — // it is the most drift-prone element in the grammar and needs the full // negation battery, which only reads at weight when it stands alone. atmosphereBlock(profile?.haze ?? 'light', [], detail), '', lastFrameBlock('resolved final beat, clean composition'), '', `WORLD PLATE: ${input.brief?.title ?? input.scene.description}.`, '', soundBed, '', captureRealismLine(profile, detail), '', `CAMERA CAPTURE: ${genreStyle.gridStyleDescriptors}, ${aspectRatio} held across every shot.${detail === 'rich' ? ` ${richCinematographySuffix(richSuffixOptsFromProfile(profile))}` : ''}`, noFaceLine, ].filter(Boolean).join('\n')); } /** * Render the CAPTURE REALISM block from the resolved cinema profile. * - no profile (legacy callers) → today's `captureRealismBlock({}, detail)`. * - realism off → a single line stating capture realism is dialled off. * - phone capture register → the Joey 2.0 phone-capture (UGC) block. * - cinema capture register → the anti-plastic capture-realism block with the * profile's haze/wet knobs. */ function captureRealismLine(profile: ResolvedCinemaProfile | undefined, detail: DetailLevel): string { if (!profile) { // omitHaze for the same reason as the profiled branch below: the packet's // ATMOSPHERE block carries the haze, and it is stated in exactly one place. return `CAPTURE REALISM: ${captureRealismBlock({ omitHaze: true }, detail)}`; } if (!profile.realism) { return 'CAPTURE REALISM: realism dialled off — render the scene without the anti-plastic capture-realism treatment.'; } if (profile.captureRegister === 'phone') { return `CAPTURE REALISM: ${phoneCaptureBlock({}, detail)}`; } // omitHaze: the ATMOSPHERE block above owns the haze for this packet. return `CAPTURE REALISM: ${captureRealismBlock({ haze: profile.haze, omitHaze: true, ...(profile.wet ? { wet: true } : {}) }, detail)}`; } // Above this runtime (seconds) a single Seedance packet is treated as a genuine // multi-cut sequence, matching cinema-worldbuilder-pro-2.0's shot-complexity // guidance: "4–8 seconds — one strong character action, single locked // composition" (one main idea per shot) vs "12–15 seconds — 2–3 simple beats // with hard cuts inside the prompt". A 9–10s scene stays a single flowing shot; // 11s+ earns the per-shot hard-cut Movement form. const MULTI_CUT_DURATION_THRESHOLD_SECONDS = 10; // TIMECODED MOVEMENT block (Joey discipline): the Movement block carries per-beat // timestamps inline. The beats are sourced from the SAME three-beat timeline split // as the FRAME MAP (threeBeatFrameMap → 0:00 / d÷3 / 2d÷3 / d) so the two blocks // stay aligned, and the camera move per beat reuses cameraSpec(RICH_CAMERA_MOVE, // detail) so detail (terse/standard/rich) still governs quantified-token emission. // // The FORM depends on genuine multi-shot intent so the packet does not contradict // itself (cinema-worldbuilder-pro-2.0 Universal Rule #6 reserves the inline // per-shot "Hard cut to" form "for any multi-cut sequence"; Rule #21 is "one main // idea per shot"): // - SINGLE-SHOT (≤ MULTI_CUT_DURATION_THRESHOLD_SECONDS): one flowing paragraph // with the per-beat timestamps inline and NO "Shot N" / "Hard cut to" labels — // matching the skill's single-shot Movement example, which is one continuous // evolving take. This is the FRAME MAP's "wide establish → medium develop → // resolved close" rendered as one shot, not three cuts. // - MULTI-CUT (above the threshold): "Shot 1 (…): … Hard cut to Shot 2 (…): …" // per the skill's multi-shot Movement example and cut-trigger rule. // Pure/deterministic — no Date, no Math.random, no I/O. function timecodedMovementBlock( durationSeconds: number, action: string, aspectRatio: string, detail: DetailLevel, ): string { const beats = threeBeatFrameMap(durationSeconds, action, aspectRatio); const move = cameraSpec(RICH_CAMERA_MOVE, detail); if (durationSeconds > MULTI_CUT_DURATION_THRESHOLD_SECONDS) { const shots = beats.map((beat, index) => { const lead = index === 0 ? `Shot ${index + 1}` : `Hard cut to Shot ${index + 1}`; return `${lead} (${beat.t}): ${move} — ${beat.beat}.`; }); return `MOVEMENT: ${shots.join(' ')}`; } // Single-shot: one continuous take. Lead the paragraph with the camera move once, // then evolve the same locked composition across the inline-timestamped beats. const flow = beats.map((beat) => `${beat.beat} (${beat.t}).`).join(' '); return `MOVEMENT: ${move} held across one continuous take — ${flow}`; } // FRAME MAP rows for the text-driven packet — the original three-beat timeline // (wide establish → medium develop → resolved close), now emitted as ordered // FrameMapEntry rows keyed on the same 0:00 / d÷3 / 2d÷3 / d split. export function threeBeatFrameMap(durationSeconds: number, action: string, aspectRatio: string): FrameMapEntry[] { const third = Math.floor(durationSeconds / 3); const twoThirds = Math.floor((durationSeconds * 2) / 3); return [ { t: `0:00-${formatSeconds(third)}`, beat: `Wide establishing shot, ${aspectRatio} — ${action}` }, { t: `${formatSeconds(third)}-${formatSeconds(twoThirds)}`, beat: `Medium shot, ${aspectRatio} — preserve subject identity and geography while the action develops` }, { t: `${formatSeconds(twoThirds)}-${formatSeconds(durationSeconds)}`, beat: `Close-up or resolved final frame, ${aspectRatio} — complete the scene beat cleanly` }, ]; } // SUBJECT LOCK entries from the carried character context. Per WS0/WS5, @imageN // slots are emitted as hard binding labels ONLY when POSITIONAL_BINDING is true; // otherwise each character's locked identity description is carried as a // descriptor-only guidance label (visual descriptor, never a proper name as the // binding key). Falls back to a single primary-subject entry when no characters // are present on the scene. function subjectLockEntriesFromContext(input: { scene: StoryboardArtifact['scenes'][number]; characterContext?: Map; }): SubjectLockEntry[] { const names = input.scene.characters ?? []; if (names.length === 0) return []; return names.map((name, index) => { const ctx = input.characterContext?.get(name); const label = ctx?.description ?? name; const slot = POSITIONAL_BINDING && ctx?.slot ? ctx.slot : `subject ${index + 1}`; return { label, slot }; }); } // Render dialogue speakers through the character-lock discipline: a speaker // whose name matches a stored character is replaced by that character's locked // visual descriptor (exact-name match, same as the scene.characters lookup), // so the packet never carries a proper name. Unmatched speakers (e.g. "She", // "He") pass through unchanged. function dialogueWithDescriptors( dialogue: DialogueLine, characterContext?: Map, ): DialogueLine { if (!characterContext) return dialogue; const resolve = (speaker: string): string => characterContext.get(speaker)?.description ?? speaker; return { ...dialogue, speaker: resolve(dialogue.speaker), ...(dialogue.secondSpeaker ? { secondSpeaker: { ...dialogue.secondSpeaker, speaker: resolve(dialogue.secondSpeaker.speaker) } } : {}), }; } // Bake spoken dialogue onto the FIRST beat of a frame map — the ai-filmmaking // "bake dialogue into the opening beat" rule. Resolves each speaker to its // locked visual descriptor first (never a proper name), then applies the // says:/replies: notation. Pure; returns `beats` unchanged when there is no // dialogue or no beats, so the default path stays byte-identical. Shared by the // prose FRAME MAP and the machine-readable `timeline` field so the two can't // drift (the timeline previously omitted the dialogue the prose carried). export function weaveDialogueIntoFrameMap( beats: FrameMapEntry[], dialogue: DialogueLine | undefined, characterContext: Map | undefined, emotionCues: boolean, ): FrameMapEntry[] { const opening = beats[0]; if (!dialogue || !opening) return beats; const resolved = dialogueWithDescriptors(dialogue, characterContext); const next = beats.slice(); next[0] = { ...opening, beat: withDialogue(opening.beat, resolved, { emotionCues }) }; return next; }