/** * lip-sync.ts — prove a clip is synced to THE BAR IT SITS ON, by measurement. * * On routes that return the SUPPLIED track as the clip's audio, lip sync stops * being a judgement call: correlate the clip's own speech-band envelope against * the song at that segment's timecode. Calibrated on real films: * * genuinely synced to this bar : 0.96 - 0.99 at 0 ms offset * rendered against another bar : 0.13 - 0.35 * * That gap is the whole point, and it is why this exists in the first place. A * picture-only review — a contact strip, or an agent reading frames — can only * tell you a mouth is OPEN. It cannot tell you the mouth is open to the RIGHT * WORDS. On one film a multi-agent visual pass picked six "performance" clips for * six lip-synced windows and every one measured 0.13-0.26; installing them would * have put a man mouthing the wrong lyrics into every hook of a film that had * already been rebuilt once to remove exactly that defect. * * Ported from `skills/rap-avatar-mv/scripts/check_sync.py`, which has gated real * films. The constants are its constants — they are calibration, not taste. * * OFFSET IS NOT A RE-RENDER. A window scoring well but late is a CUT POINT * problem: move the boundary between it and the previous segment by the offset, * and the previous shot simply holds longer. Re-rendering those would have burnt * 9 of 24 windows on the film this came from. * * The maths here is pure and offline-testable; {@link measureSegmentSync} is the * only part that shells out to ffmpeg, and it is injectable. */ /** 8 kHz mono is ample for a speech-band envelope and decodes fast. */ export declare const ANALYSIS_SAMPLE_RATE = 8000; /** 25 ms analysis frames. */ export declare const ANALYSIS_FRAME_HZ = 40; /** * Below this the clip's audio is not the song at this segment — its mouth is * moving to something else. Genuine matches sit at 0.96-0.99. */ export declare const SYNC_MIN = 0.6; /** +/- 2 s offset search, in analysis frames. */ export declare const OFFSET_WINDOW_FRAMES = 80; /** Beyond this a picture/vocal offset is visible rather than merely present. */ export declare const OFFSET_TOLERANCE_MS = 80; /** Pearson correlation over the common length. 0 when there is nothing to compare. */ export declare function correlate(a: readonly number[], b: readonly number[]): number; export interface SyncMatch { /** Best correlation found across the offset search. */ score: number; /** Where it was found, in ms. Positive = the clip runs AHEAD of the song. */ offsetMs: number; } /** * The shortest overlap an offset may be scored on, as a fraction of the shorter * envelope. * * Sliding far enough leaves only a sliver in common, and a short sliver of two * envelopes matches almost anything — two unrelated bars scored 0.996 that way in * testing, at an offset that had shifted nearly the whole signal off the end. The * absolute 20-frame floor inside {@link correlate} is not enough on its own, * because it is absolute: it protects a long signal and not a short one. */ export declare const MIN_OVERLAP_FRACTION = 0.5; /** * Best correlation and its offset. A real match survives the search; a clip cut * against a different bar does not improve at any offset. */ export declare function bestOffset(songEnvelope: readonly number[], clipEnvelope: readonly number[], windowFrames?: number): SyncMatch; /** Speech-band RMS envelope, one value per analysis frame. */ export declare function rmsEnvelope(samples: Int16Array | readonly number[]): number[]; export interface SyncSegment { /** Where this segment sits on the song's timeline, in seconds. */ startSec: number; seconds: number; clipPath: string; /** * Seek point INSIDE the clip, in seconds. The assembler cuts from here, so the * audio that actually plays starts here — reading the clip from zero measures a * part of it the film never uses and reports a mismatch that is not there. */ clipStartSec?: number; /** * Whether this segment is CLAIMED to be lip-synced. A segment that carries no * vocal — B-roll, or a window measured shot-led — is reported and never failed: * there is nothing for it to be out of sync WITH. */ synced: boolean; label?: string; } export interface SegmentSyncResult extends SyncMatch { segment: SyncSegment; /** A `synced` segment that measured below SYNC_MIN, or beyond the offset tolerance. */ pass: boolean; /** * How to fix a failure — the two causes need OPPOSITE responses, and conflating * them is expensive in both directions: * * 're-render' : the mouth is on different words. Only a new render fixes it. * 'move-cut' : the mouth is on the RIGHT words, just early or late. Move the * boundary with the previous segment; that shot holds longer. * Re-rendering these would have burnt 9 of 24 windows on the * film this came from. * 'unmeasurable' : nothing could be read. Not a verdict on the picture at all * — the clip or the song gave no audio over this span, so there * is no evidence either way. Distinct because the other two both * assert something about the take, and this asserts nothing. */ remedy?: 're-render' | 'move-cut' | 'unmeasurable'; /** Human-readable statement of the defect and its remedy. */ issue?: string; } /** Decode one span to a speech-band envelope. Injected so tests never spawn ffmpeg. */ export type EnvelopeReader = (path: string, startSec: number, seconds: number) => Promise; /** * Measure every segment against the song at its own timecode. * * Pure apart from `readEnvelope`. Segments that do not claim sync are measured * and reported but never failed. */ export declare function measureSegmentSync(songPath: string, segments: readonly SyncSegment[], readEnvelope: EnvelopeReader, syncMin?: number): Promise; /** * Speech-band envelope of one span, via ffmpeg. The only impure part of this file. * * Band-limited to 300-3400 Hz for the same reason the reference implementation is: * it makes the measurement robust to codec, level and bitrate differences between * a provider's render and the master track. */ export declare const defaultEnvelopeReader: EnvelopeReader; /** * Turn an assembled plan into the segments to measure. * * Segments are laid end to end, so a segment's position on the SONG timeline is * the running total of the durations before it. `performer` becomes `synced`: * B-roll is measured and reported but never failed. */ export declare function syncSegmentsFromPlan(segments: readonly { clip: string; inSec: number; durationSec: number; performer: boolean; }[], clipPaths: Readonly>): SyncSegment[]; //# sourceMappingURL=lip-sync.d.ts.map