/** * Clip QC — per-second headcount over a rendered clip. * * The defect this exists for: a video model handed an unbounded plural invents * and duplicates people, and walks extra figures in from the frame edges * PARTWAY THROUGH a clip. Eight clips of a 37-shot film shipped that way. Every * existing check missed it: * * - `consistency-audit` samples ONE mid-frame for identity, so a figure that * enters at t≈4s is invisible. * - `motion-qc` samples nine frames but looks for morph/vanish artefacts, not * how many people are in shot. * - the operator's own frame-grab QC sampled one frame per clip and passed * all eight. * * What caught it was a filmstrip — every second of the clip, side by side. This * productizes that: sample N frames, count people in each, and flag a clip whose * count CHANGES across the frames (someone entered or left) or exceeds the cast * the scene actually names. * * The analysis is pure and offline-testable; the frame extractor and the vision * client are injected. */ export type ClipQcCode = 'headcount-variance' | 'headcount-exceeds-cast' | 'frame-extraction-failed'; export interface ClipQcFinding { sceneIndex: number; code: ClipQcCode; severity: 'error' | 'warning'; message: string; } /** One sampled frame's observation. */ export interface ClipQcFrameObservation { /** Fraction (0–1) of the clip at which the frame was sampled. */ fraction: number; /** People visible in this frame. */ people: number; } export interface ClipQcSceneResult { sceneIndex: number; clipPath: string; framesSampled: number; /** People counted per sampled frame, in time order. */ counts: number[]; /** Cast the storyboard names for this scene, when known. */ expectedCast?: number; /** Contact sheet written for human review, when one was rendered. */ filmstripPath?: string; findings: ClipQcFinding[]; ok: boolean; } export interface ClipQcReport { schemaVersion: 1; projectSlug: string; generatedAt: string; scenes: ClipQcSceneResult[]; findings: ClipQcFinding[]; ok: boolean; } /** Runtime boundary for persisted clip-QC JSON consumed by other subsystems. */ export function isClipQcReport(value: unknown): value is ClipQcReport { if (!value || typeof value !== 'object') return false; const report = value as Partial; if (report.schemaVersion !== 1 || typeof report.projectSlug !== 'string' || !report.projectSlug || typeof report.generatedAt !== 'string' || Number.isNaN(Date.parse(report.generatedAt)) || typeof report.ok !== 'boolean' || !Array.isArray(report.scenes) || !Array.isArray(report.findings)) return false; return report.scenes.every((scene) => ( scene && Number.isInteger(scene.sceneIndex) && scene.sceneIndex >= 0 && typeof scene.clipPath === 'string' && scene.clipPath.length > 0 && Number.isInteger(scene.framesSampled) && scene.framesSampled >= 0 && Array.isArray(scene.counts) && scene.counts.every((count) => Number.isInteger(count) && count >= 0) && (scene.filmstripPath === undefined || typeof scene.filmstripPath === 'string') && Array.isArray(scene.findings) && typeof scene.ok === 'boolean' )); } /** Input passed to the vision client for one sampled frame. */ export interface ClipQcFrameInput { framePath: string; sceneIndex: number; fraction: number; } /** Injectable per-frame client. The default is Gemini-backed; tests inject a fake. */ export interface ClipQcFrameClient { countPeople(input: ClipQcFrameInput): Promise; } export interface AnalyzeClipHeadcountInput { sceneIndex: number; clipPath: string; observations: ClipQcFrameObservation[]; /** How many distinct people the scene is supposed to contain. */ expectedCast?: number; filmstripPath?: string; /** * True when the headcount lane actually ran. With no vision key the command * still renders filmstrips, and reporting "no frames sampled" for every clip * in that mode is noise, not signal. */ countingAttempted?: boolean; /** * Ignore a spread of at most this many people (default 1). * * Calibrated against a live 37-clip run: a ±1 spread is boundary noise — a * subject half-out of frame in the first or last sample, or the model missing * a small/occluded figure. Six clips flagged at ±1 and every one was clean or * intentional (two-shots where a subject deliberately walks out). The real * defect was 2 -> 6, a spread of 4. Flagging ±1 would train the operator to * ignore the check, which is worse than not having it. */ varianceTolerance?: number; } /** * Decide whether a clip's per-frame headcounts describe a defect. * * PURE. Two independent signals: * * - **variance** — the count is not the same in every frame. Someone entered or * left mid-clip. This is the signal that the single-frame checks structurally * cannot see, and the one that shipped. * - **over cast** — the maximum count exceeds the number of people the scene * names. The model invented or duplicated someone. * * A clip can trip both; each is reported once with the evidence inline. */ export function analyzeClipHeadcount(input: AnalyzeClipHeadcountInput): ClipQcSceneResult { const counts = input.observations .slice() .sort((a, b) => a.fraction - b.fraction) .map((observation) => observation.people); const findings: ClipQcFinding[] = []; if (counts.length === 0 && input.countingAttempted) { findings.push({ sceneIndex: input.sceneIndex, code: 'frame-extraction-failed', severity: 'warning', message: `scene ${input.sceneIndex}: no frames could be sampled from ${input.clipPath}.`, }); } const tolerance = input.varianceTolerance ?? 1; if (counts.length > 0) { const min = Math.min(...counts); const max = Math.max(...counts); if (max - min > tolerance) { findings.push({ sceneIndex: input.sceneIndex, code: 'headcount-variance', // WARNING, not error: vision headcounts are approximate. A verified // false positive (one woman in a colonnade counted as 2-3, confirmed // against the filmstrip) is why this triages rather than gates — it // points a human at N filmstrips instead of all 37. severity: 'warning', message: `scene ${input.sceneIndex}: people in frame changes from ${min} to ${max} across the clip ` + `(per-second counts: ${counts.join(', ')}). Someone enters or leaves mid-clip — the defect a ` + 'single-frame grab cannot see.', }); } } if (input.expectedCast !== undefined && counts.length > 0) { const max = Math.max(...counts); if (max > input.expectedCast) { findings.push({ sceneIndex: input.sceneIndex, code: 'headcount-exceeds-cast', severity: 'warning', message: `scene ${input.sceneIndex}: up to ${max} people in frame but the scene names ` + `${input.expectedCast}. The model invented or duplicated cast — state an exact count in the ` + 'prompt and describe the background as empty.', }); } } return { sceneIndex: input.sceneIndex, clipPath: input.clipPath, framesSampled: counts.length, counts, ...(input.expectedCast !== undefined ? { expectedCast: input.expectedCast } : {}), ...(input.filmstripPath ? { filmstripPath: input.filmstripPath } : {}), findings, ok: findings.every((finding) => finding.severity !== 'error'), }; } /** Roll per-scene results into a report. Pure. */ export function buildClipQcReport( projectSlug: string, generatedAt: string, scenes: ClipQcSceneResult[], ): ClipQcReport { const findings = scenes.flatMap((scene) => scene.findings); return { schemaVersion: 1, projectSlug, generatedAt, scenes, findings, ok: findings.every((finding) => finding.severity !== 'error'), }; } export interface AuditClipHeadcountOptions { /** Frames sampled per clip (default 8 — one per second of a typical 8s clip). */ sampleCount?: number; /** * Vision client. Omit to run FILMSTRIP-ONLY: the contact sheets are still * rendered, because they are the artefact a human reads and the one that * actually exposed the original defect. Unlike motion-qc, a missing vision * key degrades this command rather than failing it. */ frameClient?: ClipQcFrameClient; /** Injectable ffmpeg runner (offline tests). */ runFfmpeg?: (args: string[]) => Promise; /** Injectable clock so reports are deterministic in tests. */ now?: () => string; } export interface ClipQcSceneSource { sceneIndex: number; clipPath: string; /** Where to write this clip's contact sheet. */ filmstripPath: string; /** Distinct people the storyboard names for this scene, when known. */ expectedCast?: number; /** Absolute paths of frames already extracted for this clip, in time order. */ framePaths: string[]; } /** * Run the headcount audit over pre-extracted frames. The caller owns frame * extraction and temp-dir lifecycle (mirroring how the other QC lanes split * I/O from analysis), so this stays offline-testable end to end. */ export async function auditClipHeadcount( projectSlug: string, sources: ClipQcSceneSource[], options: AuditClipHeadcountOptions = {}, ): Promise { const now = options.now ?? ((): string => new Date().toISOString()); const scenes: ClipQcSceneResult[] = []; for (const source of sources) { if (options.runFfmpeg) { await options.runFfmpeg( buildFilmstripArgs(source.clipPath, source.filmstripPath, source.framePaths.length || 8), ); } const observations: ClipQcFrameObservation[] = []; if (options.frameClient) { for (const [index, framePath] of source.framePaths.entries()) { const fraction = (index + 1) / (source.framePaths.length + 1); const people = await options.frameClient.countPeople({ framePath, sceneIndex: source.sceneIndex, fraction, }); observations.push({ fraction, people }); } } scenes.push( analyzeClipHeadcount({ sceneIndex: source.sceneIndex, clipPath: source.clipPath, observations, countingAttempted: Boolean(options.frameClient), ...(source.expectedCast !== undefined ? { expectedCast: source.expectedCast } : {}), filmstripPath: source.filmstripPath, }), ); } return buildClipQcReport(projectSlug, now(), scenes); } /** Prompt for the per-frame headcount. Deliberately narrow: one number. */ export function buildHeadcountPrompt(input: ClipQcFrameInput): string { return [ `Count the people visible in this single frame, sampled at ${Math.round(input.fraction * 100)}% of an AI-generated video clip (scene ${input.sceneIndex}).`, 'Count every human figure that is at least partly visible, including people at the edges of frame, in the background, out of focus, or turned away from camera.', 'Do NOT count reflections, statues, paintings, posters, mannequins, or people printed on fabric.', 'If you are unsure whether a shape is a person, do not count it.', '', 'Reply in this EXACT format (no other text):', 'verdict: ', 'reason: count=', ].join('\n'); } /** Parse `count=` out of the reason line. Null when absent/unparseable. */ export function parseHeadcountReason(reason: string): number | null { const match = reason.match(/count\s*=\s*(\d+)/i); if (!match?.[1]) return null; const value = Number(match[1]); return Number.isInteger(value) && value >= 0 ? value : null; } export interface DefaultClipQcClientOptions { endpoint?: string; keyOverride?: string; fetcher?: typeof fetch; } /** * Gemini-backed headcount client, mirroring `createDefaultMotionFrameClient`. * A frame the model will not commit on counts as 0 rather than guessing — an * invented count would produce false variance findings, which is worse than a * quiet miss for a check whose whole value is that it can be trusted. */ export function createDefaultClipQcFrameClient( options: DefaultClipQcClientOptions = {}, ): ClipQcFrameClient { return { async countPeople(input: ClipQcFrameInput): Promise { const { classifyImageWithGemini, resolveVisionQaEndpoint } = await import( './assemble/gemini-vision-classify.js' ); const classified = await classifyImageWithGemini({ imagePath: input.framePath, prompt: buildHeadcountPrompt(input), allowedVerdicts: ['people'] as const, endpoint: resolveVisionQaEndpoint(options.endpoint), ...(options.keyOverride ? { keyOverride: options.keyOverride } : {}), ...(options.fetcher ? { fetcher: options.fetcher } : {}), }); return parseHeadcountReason(classified.reason) ?? 0; }, }; } /** * ffmpeg args for a one-pass filmstrip contact sheet: every sampled second * tiled left-to-right. This is the artefact a human actually reads — it is what * exposed the original defect — so it is rendered whether or not a vision key * is present. Pure arg builder; the caller spawns. */ export function buildFilmstripArgs( clipPath: string, outPath: string, frameCount: number, tileWidth = 200, ): string[] { const n = Math.max(1, Math.floor(frameCount)); return [ '-i', clipPath, '-vf', `fps=1,scale=${tileWidth}:-1,tile=${n}x1`, '-frames:v', '1', outPath, ]; }