import type { ProviderRouteId } from './types.js'; /** * Declarative, single-sourced capability envelope for one provider route. * * The numeric/enum values here are NOT re-derived heuristics — they mirror the * limits the transports already enforce: * - dreamina-useapi: `src/video/providers/dreamina-useapi.ts` * (DREAMINA_MAX_IMAGE_REFS / _VIDEO / _AUDIO, resolutionFor(), clampDuration()). * - runway-useapi: `src/video/providers/runway-useapi.ts` * (RUNWAY_MAX_IMAGE_REFS / _VIDEO, clampDuration(), default mode 'explore'). * - seedance-direct: `src/video/native-seedance.ts` * (REFERENCE_BUDGET images/videos/audios via assertReferenceBudget). * - veo-useapi: `src/video/native-veo.ts` (veoRatio landscape/portrait/square; * conservative defaults — UseAPI enforces the rest server-side). * * This matrix is the read-only contract used by `assertRouteRequestValid` to * fail fast (or warn under VCLAW_ALLOW_UNSAFE_MODELS) BEFORE a provider submit. * Keep it in sync with the transport limits above — do not let it drift. */ export interface RouteCapabilities { /** Submission modes the route understands (e.g. 'explore' | 'credits'). */ modes: string[]; /** Output resolutions the route can deliver (e.g. '720p', '1080p'). */ resolutions: string[]; /** * Discrete clip durations (seconds) the route accepts/clamps to. An empty * array means the transport enforces no client-side duration set (the * provider applies model-specific limits server-side), so duration is not * checked for this route. */ durationsSec: number[]; /** Maximum number of image reference inputs per submission. */ maxImageRefs: number; /** Maximum number of video reference inputs per submission. */ maxVideoRefs: number; /** Maximum number of audio reference inputs per submission. */ maxAudioRefs: number; /** Whether the route can attach/generate audio. */ audioCapable: boolean; /** * Max jobs this lane can have IN FLIGHT at once, across ALL processes. * * `undefined` = unenforced: the lane behaves exactly as it always has. That is * deliberate — an over-high guess reintroduces the contention this exists to * remove, and an over-low one needlessly serialises a lane that works today. * Only declare a number you have MEASURED. * * Note this is a different scope from `execute-pool.ts`'s `maxConcurrent`, * which caps concurrency inside a single process and cannot see other agents. */ maxConcurrentJobs?: number; /** When a capability is gated to a region (e.g. '1080p is CA-only'). */ regionGated?: string; /** Free-form operator notes carried into warnings/diagnostics. */ notes?: string[]; } /** * Capability matrix for the four live provider routes. `veo-direct` is * intentionally absent — it was an adapter-only phantom route with no native * transport and has been removed from `ProviderRouteId`. */ export const ROUTE_CAPABILITIES: Record = { 'veo-useapi': { // These describe the ROUTE's capability envelope, not the current // native-veo.ts transport's completeness. The Bun/Flow transport today // forwards only referencePaths[0] and silently ignores extra references // (it never rejects them), so the matrix must NOT hard-fail payloads the // transport tolerates — doing so would change existing behavior: // - first-frame + last-frame image-to-video and ingredients-to-video // legitimately carry up to ~3 image references (cf. the former veo // descriptor's ingredients-to-video maxReferenceImages: 3). // - scene chaining carries the previous scene's video output as a single // seed input → maxVideoRefs: 1. // - no audio reference path exists, and no flow submits audio refs to this // route → audioCapable:false, maxAudioRefs:0. // - durationSeconds is passed straight through; UseAPI enforces // model-specific limits server-side → durationsSec left empty (= // unconstrained) so valid durations are never rejected here. // veoRatio maps aspect → landscape/portrait/square (no resolution clamp). modes: ['default'], resolutions: ['720p', '1080p'], durationsSec: [], maxImageRefs: 3, maxVideoRefs: 1, maxAudioRefs: 0, audioCapable: false, notes: [ 'Aggregator Veo route (native-veo.ts, Google Flow via useapi.net). Image refs cover first-frame/last-frame/ingredients (≤3); a single video ref is the scene-chaining seed. The transport currently forwards only the first image and ignores extras, so these caps describe route capability, not transport completeness. UseAPI enforces model-specific duration limits server-side.', 'Real human faces are rejected by content moderation on image-to-video. MEASURED 4 Sep 2026: four i2v scenes from real photographs of a dental practice; the one keyframe with no person in it landed on the first attempt, and all three with a person were refused with the byte-identical masked failure `Generation failed: All operations failed`. isIpProhibitedFailure was false, so this is the moderation class and not the intellectual-property one. Clinical imagery is NOT the variable — the refused set includes a receptionist holding a telephone with nothing anatomical in frame. BRACKETED 5 Sep 2026 on the lite row at 0 credits (no-person PASS -> real-face REFUSE -> no-person PASS), so the refusal is pinned to the image and not to session/account state; quality and lite rows both refuse, so no cheaper tier clears it. Refusals are not charged (balance unchanged across all of them). Note the R2V character lane is a different path and is recorded as clearing photoreal faces (see `video flow-characters`), but it SYNTHESISES a scene from a registered identity rather than animating the supplied photograph — not a substitute where the photograph itself is the deliverable.', ], }, 'runway-useapi': { // runway-useapi.ts: RUNWAY_MAX_IMAGE_REFS=11, RUNWAY_MAX_VIDEO_REFS=3, // default mode 'explore' (free, low-res). modes: ['explore', 'credits'], resolutions: ['720p', '1080p'], // Left empty (= unconstrained) for the same reason as veo-useapi above: the // valid set is MODEL-specific — seedance-2.0 takes any integer 4-15 while the // gen-4 family is 5/8/10 — so a single flat list here can only be wrong. It // used to read [5,8,10,15] and would have rejected a legitimate 7 s request. // native-runway.ts clampDuration owns this per model, and warns on a change. durationsSec: [], maxImageRefs: 11, maxVideoRefs: 3, maxAudioRefs: 0, audioCapable: false, notes: [ "Default mode 'explore' is free, queued, and low-res; 'credits' is the paid faster path.", ], }, 'dreamina-useapi': { // dreamina-useapi.ts: 9 image / 3 video / 3 audio refs; resolutionFor -> // 1080p (CA-only) else 720p; clampDuration -> 4/5/8/10/12/15. modes: ['default'], resolutions: ['720p', '1080p'], durationsSec: [4, 5, 8, 10, 12, 15], maxImageRefs: 9, maxVideoRefs: 3, maxAudioRefs: 3, audioCapable: true, regionGated: '1080p is CA-only; 720p works on both US and CA regions.', notes: [ 'Seedance 2.0 via Dreamina (useapi.net). Real human faces are rejected by content moderation.', ], }, 'seedance-direct': { // MEASURED 2026-08-10: the Higgsfield free lane runs ONE job at a time. Two // drivers on it produced 88 `free slot stayed busy` events and 0 NSFW // rejections over ~7 hours; scene 13 burned 3 attempts across 70 minutes and // then landed in 45 SECONDS once the second driver was killed. maxConcurrentJobs: 1, // The authenticated Higgsfield Unlimited Seedance 2.5 route accepts integer // durations through 20s for the production policy used here. Keep every // integer selectable: duration is a directing decision, not an 8/15s preset. modes: ['fast', 'quality'], resolutions: ['720p', '1080p'], durationsSec: [4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20], maxImageRefs: 9, maxVideoRefs: 3, maxAudioRefs: 3, audioCapable: true, notes: [ 'Higgsfield Unlimited uses Seedance 2.5 when the in-tree engine is bootstrapped; production duration ceiling is 20s. Allocate duration by dramatic/action density rather than defaulting every shot to the ceiling.', ], }, 'magnific-rest': { // Magnific/Freepik proxied video catalog (native-magnific.ts). Image-to-video // only; durations enforced server-side. Cost-aware: defaults to the cheapest // catalog model that fits the operation; the premium model (kling-o1-pro) is opt-in. modes: ['default'], resolutions: ['720p', '1080p'], durationsSec: [], maxImageRefs: 1, maxVideoRefs: 0, maxAudioRefs: 0, audioCapable: false, notes: [ 'Magnific REST video generation, image-to-video only. Cheap-by-default: the default is the cheapest catalog model that fits the requested operation. The premium model (kling-o1-pro) is opt-in via VCLAW_MAGNIFIC_MODEL; unknown ids throw. Seedance is NOT on this REST catalog — seedance-pro-1080p 404s, so Seedance generation stays on dreamina-useapi/seedance-direct. Requires MAGNIFIC_API_KEY.', ], }, }; /** * A request to validate against a route's capability envelope. Every field is * optional: an absent field is simply not checked (so callers that only know a * subset of the request — e.g. just resolution + aspect ratio at payload-build * time — never trip a false violation for a value they did not supply). */ export interface RouteRequest { resolution?: string; durationSec?: number; aspectRatio?: string; imageRefs?: number; videoRefs?: number; audioRefs?: number; mode?: string; /** * Resolved Veo model for a `veo-useapi` request (`fast | quality | omni-flash`). * Only consulted by the veo-useapi-specific gate below. Other routes ignore it. */ veoModel?: string; /** omni-flash GENERATION tier ('360p' | '720p'); undefined = field not sent. */ flowResolution?: string; /** * True when this request carries a voice-narration preset (`flow.ts --voice`). * Provider-side this is `omni-flash`-only on the veo-useapi route. */ hasVoice?: boolean; /** * True when this request carries a V2V edit video reference * (`flow.ts --ref-video`). NOTE: this is NOT the scene-chaining seed * (`inputKind: 'video'`), which the route supports on any model — it is the * dedicated omni-flash V2V edit source. omni-flash-only on the veo-useapi route. */ hasVideoRef?: boolean; /** * True when this request carries an omni-flash First-Frame (startImage / I2V) * seed — the gated build-ahead identity-lock path (see VideoExecutionTask. * firstFrame). A start image satisfies the omni-flash "voice needs a visual * anchor" precondition just like an R2V image or V2V video. Only ever true on * omni-flash once the `VCLAW_OMNI_FIRST_FRAME` gate is enabled; absent → * byte-identical legacy behavior on the veo-useapi route. */ hasStartImage?: boolean; } /** Options for {@link assertRouteRequestValid}. */ export interface AssertRouteRequestValidOptions { /** Defaults to `process.env`. Pure callers can inject a fixed map. */ env?: NodeJS.ProcessEnv; } /** Result of a capability check. `ok` is always true when no Error is thrown. */ export interface RouteRequestValidation { ok: boolean; warnings: string[]; } function isTruthyEnv(value: string | undefined): boolean { if (!value) return false; const v = value.trim().toLowerCase(); return v !== '' && v !== '0' && v !== 'false' && v !== 'no' && v !== 'off'; } /** * Gate for the omni-flash First-Frame (startImage / I2V) path, shared by the * execution-runtime payload builder and the veo-useapi transport so a single env * flag controls both layers. * * As of 2026-06-06 omni-flash startImage is LIVE on useapi.net (live-verified: * the request routes to abra_i2v_8s / IMAGE_TO_VIDEO), so this is now ENABLED BY * DEFAULT. `VCLAW_OMNI_FIRST_FRAME` is a kill-switch — only 0/false/no/off * disables it (e.g. if the provider regresses). See VideoExecutionTask.firstFrame. */ export function isOmniFirstFrameEnabled(env: NodeJS.ProcessEnv = process.env): boolean { const v = (env.VCLAW_OMNI_FIRST_FRAME ?? '').trim().toLowerCase(); return !(v === '0' || v === 'false' || v === 'no' || v === 'off'); } /** * Validate a {@link RouteRequest} against {@link ROUTE_CAPABILITIES} for the * given route. Pure and deterministic (no fs/network); the only side input is * `env` (default `process.env`), read once for the `VCLAW_ALLOW_UNSAFE_MODELS` * escape hatch. * * Behavior: * - On a hard violation (resolution/duration/mode not supported, or refs over * budget), throws a clear {@link Error} — UNLESS `VCLAW_ALLOW_UNSAFE_MODELS` * is truthy, in which case each violation is downgraded to a `warnings[]` * entry and `{ ok: true }` is returned. * - A fully-valid (or empty) request returns `{ ok: true, warnings: [] }`. * - veo-useapi only: a `hasVoice` or `hasVideoRef` request whose `veoModel` * is not `omni-flash` is a violation (voice/V2V are omni-flash-only), * pushed through the same downgradable path. * * Only fields that are present on the request are checked, so partial requests * never produce spurious violations for fields the caller did not supply. */ export function assertRouteRequestValid( routeId: ProviderRouteId, req: RouteRequest, opts: AssertRouteRequestValidOptions = {}, ): RouteRequestValidation { const env = opts.env ?? process.env; const allowUnsafe = isTruthyEnv(env.VCLAW_ALLOW_UNSAFE_MODELS); const caps = ROUTE_CAPABILITIES[routeId]; const violations: string[] = []; const warnings: string[] = []; if (!caps) { // Unknown route — treat as a hard violation (caller passed a non-live id). violations.push(`Unknown provider route: ${routeId}`); } else { if (req.resolution !== undefined && !caps.resolutions.includes(req.resolution)) { violations.push( `Route ${routeId} does not support resolution ${req.resolution} (supported: ${caps.resolutions.join(', ')}).`, ); } if ( req.durationSec !== undefined && caps.durationsSec.length > 0 && !caps.durationsSec.includes(req.durationSec) ) { violations.push( `Route ${routeId} does not support duration ${req.durationSec}s (supported: ${caps.durationsSec.join(', ')}s).`, ); } // `seedance-direct` is a compatibility route name for two explicit // transports. The bootstrapped Higgsfield Unlimited path accepts the 20s // production ceiling above; forcing the paid native/xskill path restores // Seedance 2.0's measured 15s cap. Never let the compatibility alias blur // those provider contracts. if ( routeId === 'seedance-direct' && isTruthyEnv(env.VCLAW_SEEDANCE_DIRECT_NATIVE) && req.durationSec !== undefined && req.durationSec > 15 ) { violations.push( `Route seedance-direct paid native transport supports at most 15s (got ${req.durationSec}s); 16–20s requires the Higgsfield Unlimited transport.`, ); } if (req.mode !== undefined && !caps.modes.includes(req.mode)) { violations.push( `Route ${routeId} does not support mode '${req.mode}' (supported: ${caps.modes.join(', ')}).`, ); } if (req.imageRefs !== undefined && req.imageRefs > caps.maxImageRefs) { violations.push( `Route ${routeId} accepts at most ${caps.maxImageRefs} image references (got ${req.imageRefs}).`, ); } if (req.videoRefs !== undefined && req.videoRefs > caps.maxVideoRefs) { violations.push( `Route ${routeId} accepts at most ${caps.maxVideoRefs} video references (got ${req.videoRefs}).`, ); } if (req.audioRefs !== undefined && req.audioRefs > caps.maxAudioRefs) { violations.push( `Route ${routeId} accepts at most ${caps.maxAudioRefs} audio references (got ${req.audioRefs}).`, ); } if ( req.audioRefs !== undefined && req.audioRefs > 0 && caps.maxAudioRefs > 0 && !caps.audioCapable ) { violations.push(`Route ${routeId} is not audio-capable but ${req.audioRefs} audio reference(s) were supplied.`); } // veo-useapi model gate: voice narration and V2V edits are unlocked only by // the omni-flash Flow v1 model (mirrors the Bun sidecar's validateFlowVideo // throws). Fires only when the carrier field is actually present, so every // legacy payload (voice/V2V absent) is a no-op and stays byte-identical. if (routeId === 'veo-useapi') { if (req.hasVideoRef && req.veoModel !== 'omni-flash') { violations.push( "Route veo-useapi: video-to-video (referenceVideoMediaId) requires veoModel 'omni-flash'.", ); } if (req.hasVoice && req.veoModel !== 'omni-flash') { violations.push( "Route veo-useapi: voice narration (voicePreset) requires veoModel 'omni-flash'.", ); } // The 360p/720p GENERATION tier exists only on omni-flash. Veo publishes // no 360p variant, so the API rejects the pairing rather than quietly // rendering 720p at full price — catch it here instead of at the provider. if (req.flowResolution !== undefined && req.veoModel !== 'omni-flash') { violations.push( `Route veo-useapi: generation resolution '${req.flowResolution}' requires veoModel 'omni-flash' — Veo publishes no 360p variant.`, ); } // Provider rule: referenceAudio requires an image (R2V), a video (V2V), // or a first-frame startImage (I2V) reference — pure text-to-video + voice // is rejected by Google. Fail fast in-process rather than burning a submit // on a guaranteed rejection. (`hasStartImage` only goes true on omni-flash // once the VCLAW_OMNI_FIRST_FRAME gate is enabled, so legacy stays exact.) if ( req.hasVoice && req.veoModel === 'omni-flash' && !(req.imageRefs && req.imageRefs > 0) && !req.hasVideoRef ) { violations.push( "Route veo-useapi: voice narration requires a reference image (R2V) or referenceVideoMediaId (V2V); omni-flash text-to-video + voice is rejected by the provider.", ); } // A first-frame startImage does NOT satisfy that anchor — it excludes it. // The spec's reference table marks omni-flash I2V and I2V-FL "the frame // only — no referenceImage_*, no character_*, no referenceAudio_*", and // POST /videos says those "are rejected alongside them". We previously // treated a startImage as satisfying the voice anchor, which composed a // request the provider was always going to refuse. if (req.hasVoice && req.veoModel === 'omni-flash' && req.hasStartImage) { violations.push( 'Route veo-useapi: omni-flash image-to-video takes the frames only — a voice preset (referenceAudio_*) is rejected alongside a first-frame startImage.', ); } } } if (violations.length === 0) { return { ok: true, warnings }; } if (allowUnsafe) { for (const v of violations) { warnings.push(`VCLAW_ALLOW_UNSAFE_MODELS: ${v}`); } return { ok: true, warnings }; } throw new Error( `Route capability check failed for ${routeId}: ${violations.join(' ')} ` + `Set VCLAW_ALLOW_UNSAFE_MODELS=1 to downgrade these to warnings.`, ); }