/** * Joey-style Seedance block emitters: Subject Lock / Frame Map / Cross-Frame / * Last Frame. Pure/deterministic — no I/O. * * WS0 binding probe (docs/design/notes/ark-reference-order-result.md) is * RESOLVED positional: per UseAPI docs + production ARK payloads (2026-05-30), * all three gateways honor reference ORDER as the `@imageN` mapping, so `@imageN` * slots are emitted as a HARD positional binding contract — the slot label is the * reference-order key, not a descriptive hint. */ import { volumetricHaze, type DetailLevel, type HazeDensity } from './cinematography.js'; import { noFaceMorphTag } from './prompt-rules.js'; /** * WS0 reference-order binding mode. Resolved to `true`: positional binding is * confirmed (UseAPI docs + production ARK payloads, 2026-05-30 — all three * gateways honor reference ORDER as the `@imageN` mapping). `@imageN` is a hard * positional reference-order contract, not guidance only. */ export const POSITIONAL_BINDING = true; export interface SubjectLockEntry { label: string; // visual descriptor, NOT a proper name slot: string; // e.g. '@image1' } export interface FrameMapEntry { t: string; // timecode range beat: string; } /** FRAME MAP — ordered beats with timecodes. */ export function frameMapBlock(entries: FrameMapEntry[]): string { const lines = entries.map((e) => ` ${e.t}: ${e.beat}`).join('\n'); return `FRAME MAP:\n${lines}`; } /** SUBJECT LOCK — per-character identity binding to @imageN slots. */ export function subjectLockBlock(entries: SubjectLockEntry[]): string { if (entries.length === 0) { return 'SUBJECT LOCK: preserve the primary subject identical across every frame.'; } const lines = entries.map((e) => ` ${e.slot}: ${e.label} — lock this identity, do not alter.`).join('\n'); return `SUBJECT LOCK:\n${lines}`; } /** CROSS-FRAME RULES — identity/geography stability across cuts. */ export function crossFrameBlock(): string { return 'CROSS-FRAME RULES: face, hair, wardrobe, silhouette, palette, and geography stay identical across every cut; lighting logic and lens language established once and held.'; } // --------------------------------------------------------------------------- // Text-discipline emitters (WS-C). The canonical multi-reference discipline // proven by the user's ARK payloads, factored as pure helpers so the same // wording is reused everywhere (and never drifts). All are opt-in: callers gate // them behind a flag so default output stays byte-stable. // --------------------------------------------------------------------------- /** * Positional placements for {@link buildPositionalDescriptorLine}. Index 0 is * the hero/centre, then the frame is read outward L/R. Matches the ARK habit of * naming where each subject sits ("Center: …. Left: …. Right: …."). */ const POSITIONAL_PLACEMENTS = [ 'Center', 'Left', 'Right', 'Far left', 'Far right', 'Background left', 'Background right', 'Background center', 'Foreground', ] as const; const POSITIONAL_FALLBACK_DESCRIPTOR = 'as established in the reference image'; /** * Per-character POSITIONAL visual-descriptor line. Each subject is placed at a * frame position (Center / Left / Right / …) carrying its VISUAL DESCRIPTOR * (`entry.label`) — never a proper name. A missing descriptor falls back to a * stable "as established in the reference image" so the slot still reads. Returns * `''` for an empty list so single-subject / character-free scenes add nothing. */ export function buildPositionalDescriptorLine(entries: SubjectLockEntry[]): string { if (entries.length === 0) return ''; return entries .map((entry, index) => { const placement = POSITIONAL_PLACEMENTS[index] ?? `Subject ${index + 1}`; const descriptor = entry.label.trim() || POSITIONAL_FALLBACK_DESCRIPTOR; return `${placement}: ${descriptor}.`; }) .join(' '); } /** * Explicit identity-lock / no-face-morph discipline line. Reuses the standing * {@link noFaceMorphTag} rule (prompt-rules) so the "no face morphing" wording * stays canonical across surfaces. */ export function buildIdentityLockLine(): string { return `Keep each character identical to her reference image, ${noFaceMorphTag().split(',')[0]}.`; } /** * Single-full-frame guard: forces the model to perform grid panels over time * instead of reproducing a 3x3 collage as a moving split-screen. Shared between * the grid-bearing packet variants and the opt-in text-discipline path so every * multi-reference packet can carry the same guard text. * See multi-shot-framework Anti-patterns ("Grid leakage"). */ export const SINGLE_FULL_FRAME_GUARD = 'Output a single full-frame cinematic shot that fills the entire frame edge to edge — no 3x3 grid, no split-screen, no panel borders, no collage, no multi-panel montage, no contact sheet, no reference-card layout. The character reference sheets and storyboard grid are reference ONLY — never reproduce the multi-panel sheet, the character-sheet grid, or any panel layout in the video; read identity and motion from them and render one continuous live scene. Perform any grid panels as consecutive moments over time, never as one image.'; /** * LAST FRAME — closing composition ONLY. * * Text suppression deliberately does NOT live here. Overlay text is decided * early in the frame, so the instruction has to sit early in the prompt: * {@link noOnScreenTextBlock} is emitted as the first directive block instead. * Restating it at the bottom bought length without buying weight, and left the * only copy of the rule below every block competing for attention. */ export function lastFrameBlock(closing: string): string { return `LAST FRAME: ${closing}.`; } // --------------------------------------------------------------------------- // Directive blocks (Joey 3.0 Cinema Director). These sit ABOVE the descriptive // body because these models weight early text more heavily, and each one // suppresses a failure mode that positive phrasing alone does not fix. // --------------------------------------------------------------------------- /** * NO ON-SCREEN TEXT — the mandatory first directive block. * * Deliberately carries NO exception clause. An "other than…" carve-out for * in-world signage reopens the door and the model renders captions again; * physical text that genuinely exists in a scene is described elsewhere as an * object (shape, colour, placement), never as an exception here. Weighs * heaviest on phone / selfie / talking-head packets, which pull captions * straight from social training data. */ export function noOnScreenTextBlock(): string { return ( 'NO ON-SCREEN TEXT: the frame carries no text of any kind, anywhere, from the first frame to the last — ' + 'no captions, no subtitles, no burned-in dialogue, no auto-captions, no karaoke text, no lower thirds, ' + 'no titles, no title cards, no credits, no watermarks, no logos, no timecode, no interface elements, ' + 'and no social-media overlays. Every frame is clean of overlay graphics.' ); } /** Capture family for {@link captureCadenceBlock}. */ export type CaptureRegisterKind = 'cinema' | 'phone'; export interface CaptureCadenceOpts { /** * Move the stepped/stuttering quality onto the LIGHT and off the footage. * Required whenever the scene strobes: without it the model reads "stepped" * as an instruction about the capture and returns genuinely broken frames. */ strobeQuarantine?: boolean; } /** * CAPTURE CADENCE (cinema) / CAPTURE FORMAT (phone) — the anti-artifact block. * * Emitted SECOND, immediately after the text block, before anything else * competes for attention. The negation run is held at every detail level on * purpose: it is the part that suppresses interpolation judder, and dropping * items brings the corresponding artifact straight back, so only the * descriptive prose scales with `d`. */ export function captureCadenceBlock( register: CaptureRegisterKind, d: DetailLevel, opts: CaptureCadenceOpts = {}, ): string { const quarantine = opts.strobeQuarantine ? ' Any stepped or stuttering quality here comes entirely from the strobe lighting described below — from bodies ' + 'being revealed only in discrete flashes — never from broken or choppy footage. Camera motion between flashes ' + 'stays continuous and smooth even while the bodies appear to jump between positions.' : ''; if (register === 'phone') { const body = d === 'terse' ? 'captured on a handheld phone at 30 frames per second with a fast electronic shutter — motion crisp and ' + 'slightly clipped rather than softly blurred.' : 'captured on a handheld phone at 30 frames per second with a fast electronic shutter — motion crisp and ' + 'slightly clipped rather than softly blurred. Digitally sharp with heavy edge sharpening and high ' + 'micro-contrast, deep phone depth of field, visible rolling-shutter skew on fast pans, automatic exposure ' + 'that visibly hunts and pumps, automatic white balance drifting between zones, and phone HDR tone-mapping ' + 'with lifted milky shadows and compressed highlights. Fine digital luminance noise, not film grain.'; // The film-grammar kill is mandatory on phone: without it the model splits // the difference and returns footage that reads as neither format. // Both registers share the CAPTURE CADENCE label — it is this contract's // block-order slot name, and the body says which family it is. return ( `CAPTURE CADENCE: ${body} No anamorphic character, no oval bokeh, no horizontal streak flares, no 35mm grain, ` + 'no colour-negative rendition, no cinema-camera look, no 24fps cadence, no 180-degree shutter blur, ' + `no shallow cinema focus, and no cinematic grade.${quarantine}` ); } const body = d === 'terse' ? 'captured natively at 24 frames per second on a true 180-degree shutter, a real 1/48-second exposure on every frame.' : 'captured natively at 24 frames per second on a true 180-degree shutter, a real 1/48-second exposure on every ' + 'frame. Every frame carries genuine photographic motion blur and blends smoothly into the next, so motion ' + 'reads fluid, filmic, and continuous rather than choppy, stuttering, or stepping between positions.'; return ( `CAPTURE CADENCE: ${body} No frame interpolation, no frame blending, no digital smoothing, no ghosting, ` + `no double-imaging, no dropped frames, and no high-shutter video crispness.${quarantine}` ); } /** Haze density for {@link atmosphereBlock}; `'none'` emits the clean-air lock. */ export type AtmosphereDensity = HazeDensity | 'none'; /** * ATMOSPHERE — depth separation, never mood. * * Haze is the most drift-prone element in the grammar, so it gets its own * block with a complete negation battery rather than a clause inside CAPTURE * REALISM. Like the cadence run, the battery is held at every detail level: * each dropped item is one artifact allowed back in. `planes` should name the * actual planes of the actual shot, nearest to furthest — a generic haze line * reads as mood and drifts toward fog. */ export function atmosphereBlock( density: AtmosphereDensity, planes: string[], d: DetailLevel, ): string { if (density === 'none') { return ( 'ATMOSPHERE: the air is clean — no haze, no fog, no smoke, no atmospheric density, ' + 'no visible light beams, and no suspended particulate.' ); } const planeClause = planes.length > 0 ? ` It sits between every plane so depth reads in clearly separated layers — ${planes.join(', then ')}.` : ''; const stillness = d === 'terse' ? '' : ' Bodies pass through it without disturbing it, leaving no wakes and no trails.'; return ( `ATMOSPHERE: ${volumetricHaze(density, d)}, hanging still and motionless for the full shot.` + `${planeClause}${stillness}` + ' It reads as thickened still air only — no drifting, no currents, no plumes, no banks, no wisps, ' + 'no tendrils, no swirls, no rolling, no fog-machine texture, no smoke shapes, and nothing that ever ' + 'reads as fog or smoke.' ); }