/** * `vclaw video match-highlights` — the PURE core. * * Cuts a long fixed-camera sports recording down to only the events (by default * every cricket delivery, run-up → dead ball) using Gemini agentic video * understanding, one synchronous Interactions call per time segment, then * stitches the kept windows into reels. * * This file holds everything that is a deterministic function of its inputs: * the segment plan, the `-f segment` argv, the segment-list CSV parser, the * response parser, the offset merge, the usage roll-up and the on-disk cache * shapes. Every side effect (ffmpeg, uploads, Interactions calls, file writes) * lives in `src/cli/handlers/match-highlights.ts`, and the cutting half lives in * `./match-highlights-cut.js` (re-exported at the bottom of this file). * * PRODUCTISED FROM A LIVE RUN (2026-09-16): a 91-minute amateur club-cricket * recording → 7 segments → 120 deliveries, timestamps accurate to ~1 s, two * reels. The facts the design is pinned to, all verified on that run and in * `docs/audits/2026-09-16-gemini-agentic-contract.md`: * * - `background: true` plus a Files API `uri` DIES server-side ("Unsupported * file uri: blobstore://…"), so the calls are synchronous — which is why the * handler passes `longFetch` (see `./gemini-long-fetch.js`): undici aborts a * synchronous call at ~300 s and a segment call can take 270 s. * - Segment boundaries land on keyframes, so a segment's real offset comes from * ffmpeg's `-segment_list` CSV. NEVER assume `index * segmentSeconds`. * - A Files API upload is visible ONLY to the key that uploaded it, so the * upload and the Interactions call that references its `uri` share ONE pinned * key — which is why the upload cache stores a key FINGERPRINT and is * discarded when the pinned key is no longer in the pool. * - `status: "incomplete"` means the output cap was hit (thinking tokens count * against it). It is surfaced as a failed segment and NEVER retried: a retry * re-spends a paid call that will fail the same way. */ import { createHash } from 'node:crypto'; import { basename } from 'node:path'; import type { InteractionUsage } from './gemini-interactions.js'; import { DEFAULT_SPORT, SPORT_PROMPTS, type JudgeVerdict } from './match-highlights-verify.js'; import type { EventPlacement } from './match-highlights-place.js'; import { DEFAULT_LONG_FETCH_TIMEOUT_MS } from './gemini-long-fetch.js'; export const MATCH_HIGHLIGHTS_SCHEMA_VERSION = 1 as const; /** Segment length bounds, in seconds. Below 60 s the per-call overhead dominates; above 1800 s the call outruns any sane wait. */ export const MIN_SEGMENT_SECONDS = 60; export const MAX_SEGMENT_SECONDS = 1800; export const DEFAULT_SEGMENT_SECONDS = 900; /** * Parallelism bounds. FOUR concurrent 175 MB upload buffers got the live run * killed for memory on a 16 GB machine, which is why the DEFAULT is 2; 4 stays * the accepted ceiling for a machine with the headroom, so `--parallel 4` is the * operator taking that risk knowingly rather than a value we hand out. */ export const MIN_PARALLEL = 1; export const MAX_PARALLEL = 4; export const DEFAULT_PARALLEL = 2; /** Output cap for each segment call. Thinking tokens count against it, per model invocation. */ export const MATCH_HIGHLIGHTS_MAX_OUTPUT_TOKENS = 32768; /** Low temperature: this is a timestamp-extraction task, not a creative one. */ export const MATCH_HIGHLIGHTS_TEMPERATURE = 0.1; /** Files API objects expire after 48 h; re-upload before a cached `uri` can go stale mid-run. */ export const UPLOAD_CACHE_TTL_HOURS = 47; /** Total per-call cap, in seconds. Matches the reference harness's `curl --max-time 2400`. */ export const DEFAULT_TIMEOUT_SECONDS = DEFAULT_LONG_FETCH_TIMEOUT_MS / 1000; export const MIN_TIMEOUT_SECONDS = 60; export const MAX_TIMEOUT_SECONDS = 7200; /** * The built-in listing prompt — the EXACT wording that produced 120 usable * deliveries on the live run. It now lives in the sport table * (`SPORT_PROMPTS.cricket.listing`) so another sport is a data addition; * `--prompt-file` still overrides it wholesale. */ export const DEFAULT_MATCH_PROMPT: string = SPORT_PROMPTS[DEFAULT_SPORT].listing; /** Event types the built-in prompt emits; `--types` is validated against nothing (a custom prompt may use its own vocabulary). */ export const DEFAULT_HIGHLIGHT_TYPES: readonly string[] = ['four', 'six', 'wicket']; export interface MatchSegment { index: number; /** Basename of the segment file, as ffmpeg wrote it into the segment list. */ file: string; /** Absolute offset of this segment inside the source, in seconds. */ start: number; end: number; } export type MatchSegmentStatus = 'planned' | 'analyzed' | 'cached' | 'failed'; export interface MatchSegmentReport extends MatchSegment { status: MatchSegmentStatus; uploadSeconds?: number; analysisSeconds?: number; /** Events parsed out of this segment's response (0 on failure). */ events: number; usage?: InteractionUsage; error?: string; } export interface MatchEvent { /** Absolute start second inside the source. */ s: number; /** Absolute end second inside the source. */ e: number; /** Event type, e.g. `four` / `wicket` / `dot`. */ t?: string; /** Short operator note. */ n?: string; /** The segment file this event came from. */ segment: string; /** * Set only by the `--place` pass: `model` when it timed this window, `unplaced` * when it was asked and could not. Absent everywhere else. */ placement?: EventPlacement; /** * Seconds of lead the placement pass gave this event's candidate window, chosen by * its type. Recorded so a run can be reproduced and so an `unplaced` event says how * much room it was actually given. */ lead?: number; } export interface MatchReel { name: string; path: string; windows: number; seconds: number; types?: string[]; /** Set when the reel was skipped (e.g. `--types` matched nothing). */ skipped?: string; } /** One verification pass's cost, reported separately so the passes can be compared. */ export interface MatchPassReport { /** Interactions calls this pass made (a failed call still counts: it was billed). */ calls: number; /** Wall-clock seconds the pass took, uploads included. */ seconds: number; usage?: InteractionUsage; } export interface MatchPassReports { listing: MatchPassReport; gap?: MatchPassReport; judge?: MatchPassReport; judge2?: MatchPassReport; /** The `--place` pass: the only pass that spends in `--events-file` mode. */ place?: MatchPassReport; } export interface MatchHighlightsArtifact { schemaVersion: typeof MATCH_HIGHLIGHTS_SCHEMA_VERSION; source: string; generatedAt: string; model: string; sport: string; segmentSeconds: number; segments: MatchSegmentReport[]; events: MatchEvent[]; reels: MatchReel[]; /** Gap-pass stretches cut and scanned. */ gapsScanned?: number; /** Deliveries the gap pass found that the listing had missed. */ recovered?: number; /** Duplicate rows collapsed at the 4-second rule. */ deduped?: number; /** Every judge verdict, kept whether or not it survived into a reel. */ judged?: JudgeVerdict[]; /** `--place`: candidate events the model timed. */ placed?: number; /** `--place`: candidate events it was asked about and could not time; their original window stands. */ unplaced?: number; /** Per-pass cost and wall time. */ passes?: MatchPassReports; usage?: InteractionUsage; providerCalls: number; /** Non-fatal observations: dropped rows, skipped reels, failed segments. */ notes?: string[]; } /* ------------------------------------------------------------------ * * Segment planning * ------------------------------------------------------------------ */ /** * Nominal segment plan for `--dry-run`, where nothing has been cut yet so the * real `-segment_list` CSV does not exist. Boundaries are `i * segmentSeconds` * and are explicitly NOT measured — the real run reads the CSV because ffmpeg * lands each cut on the next keyframe. */ export function planSegments(input: { durationSeconds: number; segmentSeconds: number; prefix?: string }): MatchSegment[] { const { durationSeconds, segmentSeconds } = input; if (!(durationSeconds > 0)) return []; const prefix = input.prefix ?? 'seg'; const count = Math.max(1, Math.ceil(durationSeconds / segmentSeconds)); const segments: MatchSegment[] = []; for (let index = 0; index < count; index += 1) { const start = index * segmentSeconds; segments.push({ index, file: `${prefix}-${String(index).padStart(2, '0')}.mp4`, start: round1(start), end: round1(Math.min(durationSeconds, start + segmentSeconds)), }); } return segments; } /** * ffmpeg argv (everything AFTER `ffmpeg -y`, matching {@link runFfmpeg}) that * splits `source` into stream-copied segments plus a CSV segment list. * * `-c copy` keeps this near-instant and lossless; `-reset_timestamps 1` makes * every segment start at 0 so the model's timestamps are segment-relative. Audio * is mapped only when the source actually has an audio stream — `-map 0:a:0` on * a silent source fails with "matches no streams". */ export function buildSegmentArgs(input: { source: string; outDir: string; segmentSeconds: number; prefix?: string; hasAudio?: boolean; }): string[] { const prefix = input.prefix ?? 'seg'; const hasAudio = input.hasAudio ?? true; return [ '-v', 'error', '-i', input.source, '-c', 'copy', '-map', '0:v:0', ...(hasAudio ? ['-map', '0:a:0'] : []), '-f', 'segment', '-segment_time', String(input.segmentSeconds), '-reset_timestamps', '1', '-segment_list', `${input.outDir}/segments.csv`, '-segment_list_type', 'csv', '-movflags', '+faststart', `${input.outDir}/${prefix}-%02d.mp4`, ]; } /** * Parse ffmpeg's `-segment_list_type csv` output: `,,` per * row. These are the REAL offsets (ffmpeg cuts on keyframes, so they drift from * the nominal `i * segmentSeconds` by a fraction of a second and, on a sparse * GOP, by much more). */ export function parseSegmentListCsv(text: string): MatchSegment[] { const segments: MatchSegment[] = []; for (const line of text.split(/\r?\n/)) { const row = line.trim(); if (row === '') continue; const parts = row.split(','); if (parts.length < 3) continue; const file = parts[0].trim(); const start = Number.parseFloat(parts[1]); const end = Number.parseFloat(parts[2]); if (file === '' || !Number.isFinite(start) || !Number.isFinite(end)) continue; segments.push({ index: segments.length, file: basename(file), start, end }); } return segments; } /** * The segment length an existing cut was made with, read back from its CSV. * * A `--out` produced before the plan sidecar existed has a `segments.csv` and no * record of the `--segment-seconds` that made it. Consecutive starts are that * length apart (give or take the keyframe drift that moved each cut), so the * first interval recovers it. Returns undefined for a single-segment cut, where * the source was shorter than one segment and nothing can be inferred. */ export function inferSegmentSeconds(segments: readonly MatchSegment[]): number | undefined { if (segments.length < 2) return undefined; const interval = segments[1].start - segments[0].start; return interval > 0 ? interval : undefined; } /** True when a flag value disagrees with an existing cut by more than keyframe drift. */ export function segmentSecondsConflict( requested: number, existing: number | undefined, toleranceSeconds = 2, ): boolean { return existing !== undefined && Math.abs(requested - existing) > toleranceSeconds; } /* ------------------------------------------------------------------ * * Response parsing * ------------------------------------------------------------------ */ /** Strip a ```json fence (or bare backticks) from a model answer. */ export function stripJsonFences(text: string): string { let out = text.trim(); const fenced = /^```[A-Za-z0-9]*[ \t]*\r?\n([\s\S]*?)\r?\n?```$/.exec(out); if (fenced) return fenced[1].trim(); // The looser shape the live run produced: stray backticks plus a `json` tag. out = out.replace(/^`+/, '').replace(/`+$/, '').trim(); if (/^json\b/i.test(out)) out = out.slice(4).trim(); return out; } export interface ParsedEventRows { rows: Array<{ s: number; e: number; t?: string; n?: string }>; /** Rows the model emitted that had a non-numeric `s`/`e` and were dropped. */ dropped: number; } /** * Parse one segment's answer into event rows. A row whose `s`/`e` is not a * finite number is DROPPED and COUNTED rather than silently coerced — the count * surfaces in the artifact notes so a degraded segment is visible. * * Throws when the answer is not a JSON array at all; the caller turns that into * a failed segment (never a retry). */ /** * A timestamp the model emitted, as a number or a numeric string. * * Bare `Number()` is not enough: `Number(null)`, `Number('')`, `Number(false)` * and `Number([])` are all 0, so a row with a missing timestamp would silently * become an event at second zero instead of being dropped and counted. */ function finiteSecond(value: unknown): number | undefined { if (typeof value === 'number') return Number.isFinite(value) ? value : undefined; if (typeof value === 'string' && value.trim() !== '') { const parsed = Number(value); return Number.isFinite(parsed) ? parsed : undefined; } return undefined; } export function parseEventRows(text: string): ParsedEventRows { const payload: unknown = JSON.parse(stripJsonFences(text)); if (!Array.isArray(payload)) { throw new Error(`expected a JSON array of {s,e,t,n} objects, got ${payload === null ? 'null' : typeof payload}`); } const rows: ParsedEventRows['rows'] = []; let dropped = 0; for (const raw of payload) { if (!raw || typeof raw !== 'object') { dropped += 1; continue; } const row = raw as { s?: unknown; e?: unknown; t?: unknown; n?: unknown }; const s = finiteSecond(row.s); const e = finiteSecond(row.e); if (s === undefined || e === undefined) { dropped += 1; continue; } rows.push({ s, e, ...(typeof row.t === 'string' && row.t !== '' ? { t: row.t } : {}), ...(typeof row.n === 'string' && row.n !== '' ? { n: row.n } : {}), }); } return { rows, dropped }; } /** One decimal place, matching the live run's merged artifact. */ export function round1(value: number): number { return Math.round(value * 10) / 10; } /** * Shift one segment's segment-relative rows onto the source timeline. * This is the step that makes per-segment analysis add up to a whole match. */ export function offsetEventRows( rows: ParsedEventRows['rows'], segment: Pick, ): MatchEvent[] { return rows.map((row) => ({ s: round1(row.s + segment.start), e: round1(row.e + segment.start), ...(row.t !== undefined ? { t: row.t } : {}), ...(row.n !== undefined ? { n: row.n } : {}), segment: segment.file, })); } /** Merge every segment's events onto one timeline, ordered by start then end. */ export function mergeSegmentEvents(perSegment: readonly MatchEvent[][]): MatchEvent[] { return perSegment.flat().sort((a, b) => (a.s === b.s ? a.e - b.e : a.s - b.s)); } /* ------------------------------------------------------------------ * * Usage roll-up * ------------------------------------------------------------------ */ const USAGE_FIELDS = [ 'inputTokens', 'outputTokens', 'totalTokens', 'thoughtTokens', 'toolUseTokens', 'cachedTokens', ] as const; /** Sum the numeric usage fields across segments. `byModality` is per-call detail and is not summed. */ export function sumInteractionUsage(usages: ReadonlyArray): InteractionUsage | undefined { const total: Record = {}; for (const usage of usages) { if (!usage) continue; for (const field of USAGE_FIELDS) { const value = usage[field]; if (typeof value === 'number' && Number.isFinite(value)) { total[field] = (total[field] ?? 0) + value; } } } return Object.keys(total).length > 0 ? (total as InteractionUsage) : undefined; } /* ------------------------------------------------------------------ * * Flag resolution * ------------------------------------------------------------------ */ /** `--types four,six,wicket` → `['four','six','wicket']`; empty/absent → the default trio. */ export function resolveHighlightTypes(flag: string | undefined): string[] { if (flag === undefined) return [...DEFAULT_HIGHLIGHT_TYPES]; const types = flag.split(',').map((t) => t.trim()).filter((t) => t !== ''); return types; } /* ------------------------------------------------------------------ * * On-disk caches (a re-run after a failure must not re-upload or re-spend) * ------------------------------------------------------------------ */ /** * Fingerprint of the pinned Gemini key. The raw key is NEVER written to disk — * the sidecar only has to prove that a cached `uri` belongs to a key still in * the pool, because a Files API object is visible only to its uploader. */ export function fingerprintKey(key: string): string { return createHash('sha256').update(key).digest('hex').slice(0, 16); } /** Identity of the bytes a cached upload was made from. */ export interface UploadedFileIdentity { sizeBytes: number; mtimeMs: number; } export interface MatchUploadCache { schemaVersion: typeof MATCH_HIGHLIGHTS_SCHEMA_VERSION; uri: string; mimeType: string; sizeBytes: number; /** * Modification time of the file that was uploaded. * * The cache is keyed by PATH, and the pass reels reuse fixed paths * (`gaps-0.mp4`, `candidates.mp4`, `disputed.mp4`). A second run whose gap * stretches or candidate set changed rebuilds those paths with DIFFERENT * content, so without this the stale `uri` would be handed to a new window * table and every recovered timestamp would be silently wrong — after paid * calls. Segment files are written once and never change, so they stay cached. */ mtimeMs?: number; /** sha256 prefix of the uploading key — resolved back against the pool on re-use. */ keyFingerprint: string; uploadedAt: string; uploadSeconds: number; } /** * True when a cached upload can still be used: the pinned key is one of the * pool's current keys AND the object has not aged past the Files API lifetime. * A stale/foreign cache must be discarded and re-uploaded — reusing it pairs the * `uri` with the wrong key, and every segment then fails with no retry. */ export function isUploadCacheUsable( cache: MatchUploadCache, input: { poolFingerprints: ReadonlySet; now?: Date; ttlHours?: number; /** Current bytes at the cached path. A mismatch means the file was rebuilt. */ file?: UploadedFileIdentity; }, ): boolean { if (cache.schemaVersion !== MATCH_HIGHLIGHTS_SCHEMA_VERSION) return false; if (!cache.uri || !cache.keyFingerprint) return false; if (!input.poolFingerprints.has(cache.keyFingerprint)) return false; if (input.file) { // A cache written before mtime was recorded cannot prove the bytes match. if (cache.mtimeMs === undefined) return false; if (cache.sizeBytes !== input.file.sizeBytes) return false; if (Math.abs(cache.mtimeMs - input.file.mtimeMs) > 1) return false; } const uploadedAt = Date.parse(cache.uploadedAt); if (!Number.isFinite(uploadedAt)) return false; const ageHours = ((input.now ?? new Date()).getTime() - uploadedAt) / 3_600_000; return ageHours >= 0 && ageHours < (input.ttlHours ?? UPLOAD_CACHE_TTL_HOURS); } /** * A completed segment's answer, cached beside the segment so a re-run after one * failed segment does not re-spend the calls that already succeeded. Only the * parsed result is kept: the raw payload's `signature` blobs are opaque and * must never be written into an artifact. */ export interface MatchResultCache { schemaVersion: typeof MATCH_HIGHLIGHTS_SCHEMA_VERSION; model: string; status: string; text: string; usage?: InteractionUsage; analysisSeconds: number; completedAt: string; /** * The CACHE KEY: the hash of the prompt the caller asked under. A changed * prompt invalidates the entry. * * On a tool-overflow retry this is deliberately the ORIGINAL prompt's hash * even though `text` came from the `[retry]` variant, so the next run's first * attempt finds the answer instead of paying, overflowing and only then * hitting the cache. The two prompts ask the same question of the same reel * and the same windows; the retry only adds a "inspect fewer, larger windows" * preamble, so the answer is interchangeable. */ promptHash: string; } export function hashPrompt(prompt: string): string { return createHash('sha256').update(prompt).digest('hex').slice(0, 16); } /** A cached result is reusable only when it completed AND came from the same prompt and model. */ export function isResultCacheUsable( cache: MatchResultCache, input: { promptHash: string; model: string }, ): boolean { return ( cache.schemaVersion === MATCH_HIGHLIGHTS_SCHEMA_VERSION && cache.status === 'completed' && typeof cache.text === 'string' && cache.text.trim() !== '' && cache.promptHash === input.promptHash && cache.model === input.model ); } /* ------------------------------------------------------------------ * * Artifact assembly * ------------------------------------------------------------------ */ export function buildMatchHighlightsArtifact(input: { source: string; generatedAt: string; model: string; sport?: string; segmentSeconds: number; segments: MatchSegmentReport[]; events: MatchEvent[]; reels: MatchReel[]; gapsScanned?: number; recovered?: number; deduped?: number; judged?: JudgeVerdict[]; placed?: number; unplaced?: number; passes?: MatchPassReports; providerCalls: number; notes?: string[]; }): MatchHighlightsArtifact { // Usage is summed across every pass, not just the listing: the gap and judge // reels are full agentic calls and are the larger share on a long recording. const passUsage = input.passes ? [input.passes.gap?.usage, input.passes.judge?.usage, input.passes.judge2?.usage, input.passes.place?.usage] : []; const usage = sumInteractionUsage([...input.segments.map((segment) => segment.usage), ...passUsage]); const notes = (input.notes ?? []).filter((note) => note.trim() !== ''); return { schemaVersion: MATCH_HIGHLIGHTS_SCHEMA_VERSION, source: input.source, generatedAt: input.generatedAt, model: input.model, sport: input.sport ?? DEFAULT_SPORT, segmentSeconds: input.segmentSeconds, segments: input.segments, events: input.events, reels: input.reels, ...(input.gapsScanned !== undefined ? { gapsScanned: input.gapsScanned } : {}), ...(input.recovered !== undefined ? { recovered: input.recovered } : {}), ...(input.deduped !== undefined ? { deduped: input.deduped } : {}), ...(input.judged && input.judged.length > 0 ? { judged: input.judged } : {}), ...(input.placed !== undefined ? { placed: input.placed } : {}), ...(input.unplaced !== undefined ? { unplaced: input.unplaced } : {}), ...(input.passes ? { passes: input.passes } : {}), ...(usage ? { usage } : {}), providerCalls: input.providerCalls, ...(notes.length > 0 ? { notes } : {}), }; } export * from './match-highlights-cut.js'; export * from './match-highlights-verify.js'; export * from './match-highlights-place.js';