/** * lip-sync.ts — prove a clip is synced to THE BAR IT SITS ON, by measurement. * * On routes that return the SUPPLIED track as the clip's audio, lip sync stops * being a judgement call: correlate the clip's own speech-band envelope against * the song at that segment's timecode. Calibrated on real films: * * genuinely synced to this bar : 0.96 - 0.99 at 0 ms offset * rendered against another bar : 0.13 - 0.35 * * That gap is the whole point, and it is why this exists in the first place. A * picture-only review — a contact strip, or an agent reading frames — can only * tell you a mouth is OPEN. It cannot tell you the mouth is open to the RIGHT * WORDS. On one film a multi-agent visual pass picked six "performance" clips for * six lip-synced windows and every one measured 0.13-0.26; installing them would * have put a man mouthing the wrong lyrics into every hook of a film that had * already been rebuilt once to remove exactly that defect. * * Ported from `skills/rap-avatar-mv/scripts/check_sync.py`, which has gated real * films. The constants are its constants — they are calibration, not taste. * * OFFSET IS NOT A RE-RENDER. A window scoring well but late is a CUT POINT * problem: move the boundary between it and the previous segment by the offset, * and the previous shot simply holds longer. Re-rendering those would have burnt * 9 of 24 windows on the film this came from. * * The maths here is pure and offline-testable; {@link measureSegmentSync} is the * only part that shells out to ffmpeg, and it is injectable. */ /** 8 kHz mono is ample for a speech-band envelope and decodes fast. */ export const ANALYSIS_SAMPLE_RATE = 8000; /** 25 ms analysis frames. */ export const ANALYSIS_FRAME_HZ = 40; /** * Below this the clip's audio is not the song at this segment — its mouth is * moving to something else. Genuine matches sit at 0.96-0.99. */ export const SYNC_MIN = 0.6; /** +/- 2 s offset search, in analysis frames. */ export const OFFSET_WINDOW_FRAMES = 80; /** Beyond this a picture/vocal offset is visible rather than merely present. */ export const OFFSET_TOLERANCE_MS = 80; /** Pearson correlation over the common length. 0 when there is nothing to compare. */ export function correlate(a: readonly number[], b: readonly number[]): number { const n = Math.min(a.length, b.length); // Fewer than 20 frames is half a second; a correlation over that is noise. if (n < 20) return 0; let sumA = 0; let sumB = 0; for (let i = 0; i < n; i++) { sumA += a[i] as number; sumB += b[i] as number; } const meanA = sumA / n; const meanB = sumB / n; let varA = 0; let varB = 0; let cov = 0; for (let i = 0; i < n; i++) { const da = (a[i] as number) - meanA; const db = (b[i] as number) - meanB; varA += da * da; varB += db * db; cov += da * db; } const denom = Math.sqrt(varA) * Math.sqrt(varB); // A silent or perfectly flat channel has no shape to match. Reporting 0 rather // than NaN keeps "no evidence" and "no match" the same answer downstream. return denom === 0 ? 0 : cov / denom; } export interface SyncMatch { /** Best correlation found across the offset search. */ score: number; /** Where it was found, in ms. Positive = the clip runs AHEAD of the song. */ offsetMs: number; } /** * The shortest overlap an offset may be scored on, as a fraction of the shorter * envelope. * * Sliding far enough leaves only a sliver in common, and a short sliver of two * envelopes matches almost anything — two unrelated bars scored 0.996 that way in * testing, at an offset that had shifted nearly the whole signal off the end. The * absolute 20-frame floor inside {@link correlate} is not enough on its own, * because it is absolute: it protects a long signal and not a short one. */ export const MIN_OVERLAP_FRACTION = 0.5; /** * Best correlation and its offset. A real match survives the search; a clip cut * against a different bar does not improve at any offset. */ export function bestOffset( songEnvelope: readonly number[], clipEnvelope: readonly number[], windowFrames: number = OFFSET_WINDOW_FRAMES, ): SyncMatch { let score = 0; let atFrame = 0; const minOverlap = Math.floor( Math.min(songEnvelope.length, clipEnvelope.length) * MIN_OVERLAP_FRACTION, ); for (let off = -windowFrames; off <= windowFrames; off++) { const song = off >= 0 ? songEnvelope.slice(off) : songEnvelope; const clip = off >= 0 ? clipEnvelope : clipEnvelope.slice(-off); // An offset that leaves too little in common cannot be evidence of anything. if (Math.min(song.length, clip.length) < minOverlap) continue; const value = correlate(song, clip); if (value > score) { score = value; atFrame = off; } } return { score, offsetMs: Math.round((atFrame * 1000) / ANALYSIS_FRAME_HZ) }; } /** Speech-band RMS envelope, one value per analysis frame. */ export function rmsEnvelope(samples: Int16Array | readonly number[]): number[] { const step = Math.floor(ANALYSIS_SAMPLE_RATE / ANALYSIS_FRAME_HZ); const out: number[] = []; for (let i = 0; i + step < samples.length; i += step) { let sum = 0; for (let k = i; k < i + step; k++) { const v = (samples[k] as number) / 32768; sum += v * v; } out.push(Math.sqrt(sum / step)); } return out; } export interface SyncSegment { /** Where this segment sits on the song's timeline, in seconds. */ startSec: number; seconds: number; clipPath: string; /** * Seek point INSIDE the clip, in seconds. The assembler cuts from here, so the * audio that actually plays starts here — reading the clip from zero measures a * part of it the film never uses and reports a mismatch that is not there. */ clipStartSec?: number; /** * Whether this segment is CLAIMED to be lip-synced. A segment that carries no * vocal — B-roll, or a window measured shot-led — is reported and never failed: * there is nothing for it to be out of sync WITH. */ synced: boolean; label?: string; } export interface SegmentSyncResult extends SyncMatch { segment: SyncSegment; /** A `synced` segment that measured below SYNC_MIN, or beyond the offset tolerance. */ pass: boolean; /** * How to fix a failure — the two causes need OPPOSITE responses, and conflating * them is expensive in both directions: * * 're-render' : the mouth is on different words. Only a new render fixes it. * 'move-cut' : the mouth is on the RIGHT words, just early or late. Move the * boundary with the previous segment; that shot holds longer. * Re-rendering these would have burnt 9 of 24 windows on the * film this came from. * 'unmeasurable' : nothing could be read. Not a verdict on the picture at all * — the clip or the song gave no audio over this span, so there * is no evidence either way. Distinct because the other two both * assert something about the take, and this asserts nothing. */ remedy?: 're-render' | 'move-cut' | 'unmeasurable'; /** Human-readable statement of the defect and its remedy. */ issue?: string; } /** Decode one span to a speech-band envelope. Injected so tests never spawn ffmpeg. */ export type EnvelopeReader = ( path: string, startSec: number, seconds: number, ) => Promise; /** * Measure every segment against the song at its own timecode. * * Pure apart from `readEnvelope`. Segments that do not claim sync are measured * and reported but never failed. */ export async function measureSegmentSync( songPath: string, segments: readonly SyncSegment[], readEnvelope: EnvelopeReader, syncMin: number = SYNC_MIN, ): Promise { const results: SegmentSyncResult[] = []; for (const segment of segments) { const [song, clip] = await Promise.all([ readEnvelope(songPath, segment.startSec, segment.seconds), readEnvelope(segment.clipPath, segment.clipStartSec ?? 0, segment.seconds), ]); const match = bestOffset(song, clip); const result: SegmentSyncResult = { segment, ...match, pass: true }; // A segment that never claimed sync has nothing to be out of sync WITH: it is // measured and reported, never failed. Failing B-roll makes the gate cry wolf. if (segment.synced) { // "Could not measure" is NOT "measured and wrong". An empty envelope means // ffmpeg returned no audio for that span — no audio track, an unreadable // file, or a seek past the end — and correlate() reports 0 for it, which is // indistinguishable from a genuine mismatch. Saying "its mouth is moving to // different words" about a clip nobody could read sends the operator to // re-render footage that may be perfect. if (song.length === 0 || clip.length === 0) { result.pass = false; result.remedy = 'unmeasurable'; const missing = song.length === 0 ? 'the SONG' : 'the CLIP'; result.issue = `could not measure — no readable audio from ${missing} over this span. ` + 'Lip sync is only measurable on routes that return the supplied track ' + 'as the clip audio; on a silent-clip lane there is nothing to correlate.'; } else if (match.score < syncMin) { result.pass = false; result.remedy = 're-render'; result.issue = `not synced to this bar (${match.score.toFixed(2)} < ${syncMin}) — ` + 'its mouth is moving to different words'; } else if (Math.abs(match.offsetMs) > OFFSET_TOLERANCE_MS) { // Blocks, because a slip this size is visible and has survived two visual // reviews before now. But the remedy is the cut point, not a new render. result.pass = false; result.remedy = 'move-cut'; result.issue = `synced but ${match.offsetMs > 0 ? '+' : ''}${match.offsetMs}ms out — move ` + 'the cut point before it rather than re-rendering or freezing a frame'; } } results.push(result); } return results; } /** * Speech-band envelope of one span, via ffmpeg. The only impure part of this file. * * Band-limited to 300-3400 Hz for the same reason the reference implementation is: * it makes the measurement robust to codec, level and bitrate differences between * a provider's render and the master track. */ export const defaultEnvelopeReader: EnvelopeReader = async (mediaPath, startSec, seconds) => { const { spawn } = await import('node:child_process'); const { resolveFfmpegBin } = await import('./assemble/ffmpeg.js'); const args = [ '-v', 'error', // -ss BEFORE -i is the fast seek, and it is correct here: we want the span // itself, not frame-accurate trimming of a re-encode. ...(startSec > 0 ? ['-ss', startSec.toFixed(3)] : []), '-i', mediaPath, ...(seconds > 0 ? ['-t', seconds.toFixed(3)] : []), '-vn', '-ac', '1', '-ar', String(ANALYSIS_SAMPLE_RATE), '-af', 'highpass=f=300,lowpass=f=3400', '-f', 's16le', 'pipe:1', ]; const chunks: Buffer[] = await new Promise((resolve, reject) => { const child = spawn(resolveFfmpegBin(), args, { stdio: ['ignore', 'pipe', 'ignore'] }); const acc: Buffer[] = []; child.stdout.on('data', (c: Buffer) => acc.push(c)); child.on('error', reject); // A file with no audio track exits non-zero. That is "no evidence", which // correlates to 0 — the same answer as "no match", and not a crash. child.on('close', () => resolve(acc)); }); const buffer = Buffer.concat(chunks); if (buffer.length < 2) return []; const usable = buffer.length - (buffer.length % 2); const samples = new Int16Array(usable / 2); for (let i = 0; i < samples.length; i++) samples[i] = buffer.readInt16LE(i * 2); return rmsEnvelope(samples); }; /** * Turn an assembled plan into the segments to measure. * * Segments are laid end to end, so a segment's position on the SONG timeline is * the running total of the durations before it. `performer` becomes `synced`: * B-roll is measured and reported but never failed. */ export function syncSegmentsFromPlan( segments: readonly { clip: string; inSec: number; durationSec: number; performer: boolean }[], clipPaths: Readonly>, ): SyncSegment[] { let cursor = 0; return segments.map((seg, i) => { // `seg.clip` is a clip ID, not a path — the planner is pure and works entirely // in IDs (see its injected `clipDuration: (clipId) => number`). Passing the ID // straight to ffmpeg, as the first version of this did, opens nothing: every // envelope comes back empty, every correlation is 0.00, and the gate reports // "its mouth is moving to different words" for all 25 cuts of a fine film. const path = clipPaths[seg.clip]; if (!path) { throw new Error( `syncSegmentsFromPlan: clip "${seg.clip}" is not in the registry. ` + 'Verification silently measures nothing without a path, so this throws ' + 'rather than reporting an unsynced cut that was never read.', ); } const out: SyncSegment = { startSec: cursor, seconds: seg.durationSec, clipPath: path, clipStartSec: seg.inSec, synced: seg.performer, label: `cut ${i} (${seg.clip})`, }; cursor += seg.durationSec; return out; }); }