/** * Transcript beats — turn a timed transcript (SRT / WebVTT / whisper JSON) * into ~5s, pause-aware beats, and seed a coverage-only motion-pack skeleton * from them. The VO is the timeline: every beat eventually gets a decision * (generated / v2v / talking-head / screen-rec / reuse). * * Pure text-in, data-out — no transcription here. Media without a transcript * goes through existing STT tooling first. */ import type { MotionPackArtifact, MotionPackCoverageRow, TranscriptBeat } from './types.js'; export interface TranscriptCue { t0Sec: number; t1Sec: number; text: string; } /** A gap this long between cues is a hard pause — a natural clip boundary. */ export const DEFAULT_PAUSE_GAP_SEC = 0.8; export const DEFAULT_BEAT_TARGET_SEC = 5; /** Coverage skeleton rows aim for topic-chunk length, not beat length. */ const SEGMENT_TARGET_SEC = 60; function parseClockTime(raw: string): number { const cleaned = raw.trim().replace(',', '.'); const parts = cleaned.split(':'); if (parts.length < 2 || parts.length > 3) throw new Error(`invalid cue timestamp "${raw}"`); const nums = parts.map((p) => Number(p)); if (nums.some((n) => Number.isNaN(n) || n < 0)) throw new Error(`invalid cue timestamp "${raw}"`); return parts.length === 3 ? nums[0] * 3600 + nums[1] * 60 + nums[2] : nums[0] * 60 + nums[1]; } export function parseSrt(text: string): TranscriptCue[] { const cues: TranscriptCue[] = []; const chunks = text.replace(/\r/g, '').split(/\n\n+/); for (const chunk of chunks) { const lines = chunk.split('\n').filter((l) => l.trim() !== ''); const timeLineIdx = lines.findIndex((l) => l.includes('-->')); if (timeLineIdx === -1) continue; const [from, to] = lines[timeLineIdx].split('-->'); const body = lines.slice(timeLineIdx + 1).join(' ').trim(); if (!body) continue; cues.push({ t0Sec: parseClockTime(from), t1Sec: parseClockTime(to.split(' ')[1] ?? to), text: body }); } return cues; } export function parseVtt(text: string): TranscriptCue[] { // WebVTT is SRT-shaped once the header and cue settings are stripped. const withoutHeader = text.replace(/^?WEBVTT[^\n]*\n/, ''); const noteStripped = withoutHeader.replace(/^NOTE[^\n]*(\n(?!\n)[^\n]*)*/gm, ''); return parseSrt( noteStripped .split('\n') .map((line) => (line.includes('-->') ? line.replace(/ --> ([^ ]+).*$/, ' --> $1') : line)) .join('\n'), ); } interface WhisperSegment { start?: number; end?: number; text?: string; } export function parseWhisperJson(text: string): TranscriptCue[] { const parsed = JSON.parse(text) as { segments?: WhisperSegment[] }; const segments = Array.isArray(parsed.segments) ? parsed.segments : []; return segments .filter((s) => typeof s.start === 'number' && typeof s.end === 'number' && s.text?.trim()) .map((s) => ({ t0Sec: s.start as number, t1Sec: s.end as number, text: (s.text as string).trim() })); } export function parseTranscriptCues(content: string, fileName: string): TranscriptCue[] { const lower = fileName.toLowerCase(); if (lower.endsWith('.srt')) return parseSrt(content); if (lower.endsWith('.vtt')) return parseVtt(content); if (lower.endsWith('.json')) return parseWhisperJson(content); throw new Error(`unsupported transcript format for "${fileName}" — expected .srt, .vtt, or whisper .json`); } /** * Merge cues into ~targetSec beats, never straddling a hard pause. Beats are * the unit blocks map onto; a pause after a beat marks a natural boundary a * block should not cross. */ export function buildBeats( cues: TranscriptCue[], opts: { targetSec?: number; pauseGapSec?: number } = {}, ): TranscriptBeat[] { const targetSec = opts.targetSec ?? DEFAULT_BEAT_TARGET_SEC; const pauseGapSec = opts.pauseGapSec ?? DEFAULT_PAUSE_GAP_SEC; const beats: TranscriptBeat[] = []; let current: { t0Sec: number; t1Sec: number; texts: string[] } | null = null; const flush = (pauseAfter: boolean) => { if (!current) return; beats.push({ index: beats.length, t0Sec: current.t0Sec, t1Sec: current.t1Sec, text: current.texts.join(' '), ...(pauseAfter ? { pauseAfter: true } : {}), }); current = null; }; for (const cue of [...cues].sort((a, b) => a.t0Sec - b.t0Sec)) { const gapFromPrev = current ? cue.t0Sec - current.t1Sec : 0; if (current && gapFromPrev >= pauseGapSec) { flush(true); } if (!current) { current = { t0Sec: cue.t0Sec, t1Sec: cue.t1Sec, texts: [cue.text] }; } else { current.t1Sec = Math.max(current.t1Sec, cue.t1Sec); current.texts.push(cue.text); } if (current.t1Sec - current.t0Sec >= targetSec) { flush(false); } } flush(false); return beats; } function gistOf(text: string, maxWords = 8): string { const words = text.trim().split(/\s+/); return words.length <= maxWords ? text.trim() : words.slice(0, maxWords).join(' ') + '…'; } /** * Seed a coverage-only pack skeleton from beats: rows are topic chunks split * at hard pauses (capped near SEGMENT_TARGET_SEC), every row defaulting to * `talking-head` — the authoring pass upgrades rows to generated/v2v and adds * blocks. Blocks start empty on purpose: choreography is a creative decision. */ export function initPackFromBeats( beats: TranscriptBeat[], opts: { projectSlug: string; video: string; sheetId: string; generatedAt: string; aspect?: string; durationSec?: number; }, ): MotionPackArtifact { const coverage: MotionPackCoverageRow[] = []; let seg: { t0Sec: number; t1Sec: number; texts: string[] } | null = null; let segNo = 1; const flush = () => { if (!seg) return; coverage.push({ segment: `S${String(segNo).padStart(2, '0')}`, t0Sec: seg.t0Sec, t1Sec: seg.t1Sec, voGist: gistOf(seg.texts.join(' ')), plan: 'talking-head', }); segNo += 1; seg = null; }; for (const beat of beats) { if (!seg) seg = { t0Sec: beat.t0Sec, t1Sec: beat.t1Sec, texts: [beat.text] }; else { seg.t1Sec = Math.max(seg.t1Sec, beat.t1Sec); seg.texts.push(beat.text); } if (beat.pauseAfter || seg.t1Sec - seg.t0Sec >= SEGMENT_TARGET_SEC) flush(); } flush(); return { schemaVersion: 1, projectSlug: opts.projectSlug, video: opts.video, sheetId: opts.sheetId, generatedAt: opts.generatedAt, aspect: opts.aspect ?? '16:9', durationSec: opts.durationSec ?? 5, timing: 'draft', coverage, blocks: [], }; }