/** * modelark.ts — the BytePlus ModelArk video-generation client behind the * `seedance-modelark` route (the international Volcengine Ark: the official * Dreamina Seedance 2.5 / 2.0 API). * * This is a PURE protocol module: it turns a `VideoExecutionTask` into the * exact `content[]` request ModelArk accepts, and wraps the three HTTP verbs. * It owns no job state and touches no disk — `native-modelark.ts` does that. * Nothing here is shared with `seedance-direct` (the xskill aggregator, a * different host, key, body shape and biller — ADR 0007): the two are kept * apart so one key can never silently mean two billers. * * Contract (docs.byteplus.com ModelArk 1520757 / 1521309 / 1521720 / * 2607688 / 2298881, read 2026-09-21): * POST /contents/generations/tasks → { id } * GET /contents/generations/tasks/{id} → { status, content.video_url, error, usage, … } * DELETE /contents/generations/tasks/{id} → {} (queued → cancelled; running cannot be cancelled; * a finished task's RECORD is deleted — never call it * on a task whose result has not been downloaded) * A first/last-frame task and an omni reference task are mutually exclusive in * one request; a first/last-frame task must use `ratio: 'adaptive'`. * `content.video_url` is valid for 24 hours and at most 100 downloads. */ import type { VideoExecutionPayload, VideoExecutionTask } from '../types.js'; import { safeErrorBody } from '../http-error-safety.js'; import { ROUTE_CAPABILITIES } from '../provider-platform/route-capabilities.js'; import { withRetry, fetchTransientRetry, type WithRetryOptions } from '../with-retry.js'; export const MODELARK_DEFAULT_BASE_URL = 'https://ark.ap-southeast.bytepluses.com/api/v3'; export const MODELARK_TASKS_PATH = '/contents/generations/tasks'; export const MODELARK_API_KEY_ENV = 'ARK_API_KEY'; export const MODELARK_MODEL_ENV = 'VCLAW_MODELARK_MODEL'; export const MODELARK_BASE_URL_ENV = 'VCLAW_MODELARK_BASE_URL'; /** * How a chained scene continues the previous one. `first-frame` (default): the * previous clip's last frame becomes this scene's `first_frame`. `extend`: the * previous clip itself rides as `@Video 1` and the task is an explicit video * extension (`omni_reference_task_type: extend`) — the vendor bills the input * clip's seconds on top of the output, and documents the field for 2.5 only. */ export const MODELARK_CHAIN_MODE_ENV = 'VCLAW_MODELARK_CHAIN_MODE'; export type ModelArkChainMode = 'first-frame' | 'extend'; /** * The planner options for the chain mode. An unknown value is carried as an * error and refused only for CHAINED scenes — a scene that does not chain (a * batch job, scene 0) is not blocked by a setting it never reads. */ export function modelArkChainOptions(env: NodeJS.ProcessEnv): { chainMode?: ModelArkChainMode; chainModeError?: string } { try { return { chainMode: resolveModelArkChainMode(env) }; } catch (error) { return { chainModeError: error instanceof Error ? error.message : String(error) }; } } /** The chain mode from the environment; unset means `first-frame`, anything unknown is refused, never guessed. */ export function resolveModelArkChainMode(env: NodeJS.ProcessEnv = process.env): ModelArkChainMode { const raw = env[MODELARK_CHAIN_MODE_ENV]?.trim(); if (!raw || raw === 'first-frame') return 'first-frame'; if (raw === 'extend') return 'extend'; throw new Error(`${MODELARK_CHAIN_MODE_ENV}=${raw} is not a chain mode this route knows (first-frame, extend).`); } export type ModelArkModelId = | 'dreamina-seedance-2-5-260628' | 'dreamina-seedance-2-0-fast-260128' | 'dreamina-seedance-2-0-mini-260615'; export const DEFAULT_MODELARK_MODEL: ModelArkModelId = 'dreamina-seedance-2-5-260628'; export interface ModelArkModelLimits { family: '2.5' | '2.0'; /** Inclusive integer range of `duration` seconds the model accepts. */ durationRange: readonly [number, number]; maxImageRefs: number; maxVideoRefs: number; maxAudioRefs: number; /** Combined length of every reference video and audio clip, in seconds. */ maxReferenceSeconds: number; /** Each reference video must be at least this long (seconds). */ minReferenceVideoSeconds: number; /** Whether a request may carry audio references and nothing else. */ audioOnlyReferences: boolean; /** Output resolutions the model sells (the 2.0 fast/mini models stop at 720p). */ resolutions: readonly ModelArkResolution[]; /** * List price in USD per million output tokens, by resolution, for a request * WITHOUT and WITH a video input (the vendor bills the two differently). * From the ModelArk pricing page, read 2026-09-21; the 1080p launch discount * ended 2026-09-17, so these are the prices that apply. A resolution the * model does not sell has no row. The vendor's minimum-token floors for * video input are NOT modelled: this is the list rate × the tokens reported. */ usdPerMillionTokens: Readonly>>; } export type ModelArkResolution = '480p' | '720p' | '1080p'; /** * Per-model limits, copied from the vendor's create-task reference. These are * what the preflight enforces BEFORE the first paid submit; a model missing * here is refused rather than guessed at. */ export const MODELARK_MODELS: Record = { 'dreamina-seedance-2-5-260628': { family: '2.5', durationRange: [4, 30], maxImageRefs: 30, maxVideoRefs: 10, maxAudioRefs: 10, maxReferenceSeconds: 30, minReferenceVideoSeconds: 2, audioOnlyReferences: true, resolutions: ['480p', '720p', '1080p'], usdPerMillionTokens: { '480p': { noVideoInput: 10.70, videoInput: 6.40 }, '720p': { noVideoInput: 10.70, videoInput: 6.40 }, '1080p': { noVideoInput: 11.70, videoInput: 7.00 }, }, }, 'dreamina-seedance-2-0-fast-260128': { family: '2.0', durationRange: [4, 15], maxImageRefs: 9, maxVideoRefs: 3, maxAudioRefs: 3, maxReferenceSeconds: 15, minReferenceVideoSeconds: 2, audioOnlyReferences: false, resolutions: ['480p', '720p'], usdPerMillionTokens: { '480p': { noVideoInput: 5.60, videoInput: 3.30 }, '720p': { noVideoInput: 5.60, videoInput: 3.30 }, }, }, 'dreamina-seedance-2-0-mini-260615': { family: '2.0', durationRange: [4, 15], maxImageRefs: 9, maxVideoRefs: 3, maxAudioRefs: 3, maxReferenceSeconds: 15, minReferenceVideoSeconds: 2, audioOnlyReferences: false, resolutions: ['480p', '720p'], usdPerMillionTokens: { '480p': { noVideoInput: 3.50, videoInput: 2.10 }, '720p': { noVideoInput: 3.50, videoInput: 2.10 }, }, }, }; /** * The list-price cost of one task, from the `usage` the vendor returns on a * succeeded task and the rate for (model, resolution, video input). The rate * is per OUTPUT token, so `completion_tokens` is read first and `total_tokens` * only when that is absent (on the one measured job both were 87,300 — a 4 s * 720p text-to-video clip on 2.5, ≈ 21,800 tokens per second, USD 0.934 — so * the vendor's precedence is unverified; if a prompt component ever appears, * this order under-reports rather than over-reports). Returns null when the * usage is not a finite token count, the resolution is not one the model * sells, or the fields are missing (a job-state file from before cost * accounting) — a cost is reported or absent, never guessed. Unrounded: the * job total is rounded once, by the caller. */ export function modelArkCostUsd( model: ModelArkModelId, resolution: ModelArkResolution | undefined, videoInput: boolean | undefined, usage: unknown, ): { tokens: number; usd: number } | null { if (resolution === undefined || typeof videoInput !== 'boolean') return null; // Own keys only: a hand-edited model of "constructor" must price as unknown, not throw. const row = Object.hasOwn(MODELARK_MODELS, model) ? MODELARK_MODELS[model].usdPerMillionTokens[resolution] : undefined; if (!row) return null; const rate = videoInput ? row.videoInput : row.noVideoInput; const node = usage && typeof usage === 'object' ? usage as { total_tokens?: unknown; completion_tokens?: unknown } : {}; const tokens = typeof node.completion_tokens === 'number' ? node.completion_tokens : typeof node.total_tokens === 'number' ? node.total_tokens : Number.NaN; if (!Number.isFinite(tokens) || tokens < 0) return null; const usd = tokens * rate / 1_000_000; // An absurd count can overflow to Infinity, which JSON writes as null: a // cost the job-state reader then refuses. return Number.isFinite(usd) ? { tokens, usd } : null; } export function isModelArkModelId(value: unknown): value is ModelArkModelId { return typeof value === 'string' && Object.prototype.hasOwnProperty.call(MODELARK_MODELS, value); } /** `VCLAW_MODELARK_MODEL` or the 2.5 default; an unknown id throws (never a silent fallback). */ export function resolveModelArkModel(env: NodeJS.ProcessEnv): ModelArkModelId { const raw = env[MODELARK_MODEL_ENV]?.trim(); if (!raw) return DEFAULT_MODELARK_MODEL; if (isModelArkModelId(raw)) return raw; throw new Error( `${MODELARK_MODEL_ENV}=${raw} is not a ModelArk model this route knows. Known: ${Object.keys(MODELARK_MODELS).join(', ')}.`, ); } export function resolveModelArkApiKey(env: NodeJS.ProcessEnv): string { const key = env[MODELARK_API_KEY_ENV]?.trim(); if (!key) { throw new Error(`seedance-modelark requires ${MODELARK_API_KEY_ENV} (BytePlus ModelArk API key).`); } return key; } export function modelArkBaseUrl(env: NodeJS.ProcessEnv): string { return (env[MODELARK_BASE_URL_ENV]?.trim() || MODELARK_DEFAULT_BASE_URL).replace(/\/+$/, ''); } export type ModelArkReferenceKind = 'image' | 'video' | 'audio'; export type ModelArkReferenceRole = 'first_frame' | 'last_frame' | 'reference_image' | 'reference_video' | 'reference_audio'; export interface ModelArkReferencePlan { kind: ModelArkReferenceKind; role: ModelArkReferenceRole; /** Local path, http(s) URL, or ModelArk `asset://` URI, as the task carried it. */ source: string; } export interface ModelArkTaskPlan { sceneIndex: number; model: ModelArkModelId; mode: 'text' | 'first-frame' | 'omni'; /** The prompt as it will be submitted (markers / frame directives already applied). */ prompt: string; ratio: '16:9' | '9:16' | '1:1' | 'adaptive'; resolution: '480p' | '720p' | '1080p'; duration: number; generateAudio: boolean; refs: ModelArkReferencePlan[]; omniReferenceTaskType?: 'auto' | 'extend'; /** Advisory notes the operator should see (never blocking). */ issues: string[]; } export interface ModelArkContentItem { type: 'text' | 'image_url' | 'video_url' | 'audio_url'; text?: string; image_url?: { url: string }; video_url?: { url: string }; audio_url?: { url: string }; role?: ModelArkReferenceRole; } export interface ModelArkCreateBody { model: ModelArkModelId; content: ModelArkContentItem[]; resolution: ModelArkTaskPlan['resolution']; ratio: ModelArkTaskPlan['ratio']; duration: number; generate_audio: boolean; watermark: false; omni_reference_task_type?: 'auto' | 'extend'; } export type ModelArkTaskStatus = 'queued' | 'running' | 'succeeded' | 'failed' | 'cancelled' | 'expired'; export interface ModelArkTask { id: string; status: ModelArkTaskStatus | string; content?: { video_url?: string; last_frame_url?: string }; error?: { code?: string; message?: string } | null; usage?: unknown; [key: string]: unknown; } const VIDEO_EXTENSIONS = new Set(['.mp4', '.mov', '.webm', '.avi', '.mkv']); const AUDIO_EXTENSIONS = new Set(['.mp3', '.wav', '.m4a', '.aac', '.ogg', '.flac']); function referenceExtension(reference: string): string { let pathname = reference; try { pathname = new URL(reference).pathname; } catch { pathname = reference.split('?')[0] ?? reference; } const dot = pathname.lastIndexOf('.'); return dot >= 0 ? pathname.slice(dot).toLowerCase() : ''; } /** * Bucket references by kind, using the same extension rule the other transports * use so a voice-clone `.mp4` lands in the video bucket everywhere. A ModelArk * `asset://` digital character counts as an image. The xskill Asset Library's * `Asset://` URIs (capital A) belong to a different registry and are refused: * ModelArk has never heard of them, and sending one would fail after the * request was already billed a queue slot. */ export function classifyModelArkReferences(referencePaths: readonly string[]): { images: string[]; videos: string[]; audios: string[]; } { const out = { images: [] as string[], videos: [] as string[], audios: [] as string[] }; for (const reference of referencePaths) { if (reference.startsWith('Asset://')) { throw new Error( `seedance-modelark cannot use ${reference}: that is an xskill Asset Library avatar, which ModelArk does not know. ` + 'Use a hosted image, a local file, or a ModelArk digital character (asset://…) on this route.', ); } if (reference.startsWith('asset://')) { out.images.push(reference); continue; } const extension = referenceExtension(reference); if (VIDEO_EXTENSIONS.has(extension)) out.videos.push(reference); else if (AUDIO_EXTENSIONS.has(extension)) out.audios.push(reference); else out.images.push(reference); } return out; } /** * ModelArk's omni mode wants each reference named in the prompt as * `@Image 1` / `@Video 2` / `@Audio 1` (1-based, in `content[]` order). The * vendor's own examples write both `@Image 1` and `@Image1`, so either spelling * counts as present. Missing markers are prepended; existing ones are left in * place and never duplicated. Pure. */ export function ensureModelArkMarkers( prompt: string, counts: { images: number; videos: number; audios: number }, ): string { const wanted: Array<{ label: 'Image' | 'Video' | 'Audio'; n: number }> = []; for (let i = 1; i <= counts.images; i += 1) wanted.push({ label: 'Image', n: i }); for (let i = 1; i <= counts.videos; i += 1) wanted.push({ label: 'Video', n: i }); for (let i = 1; i <= counts.audios; i += 1) wanted.push({ label: 'Audio', n: i }); const missing = wanted.filter(({ label, n }) => !new RegExp(`@${label}\\s?${n}(?!\\d)`, 'i').test(prompt)); if (missing.length === 0) return prompt; return `${missing.map(({ label, n }) => `@${label} ${n}`).join(' ')} ${prompt.trimStart()}`; } function planError(sceneIndex: number, message: string): Error { return new Error(`seedance-modelark scene ${sceneIndex}: ${message}`); } function resolvePlanResolution( sceneIndex: number, task: VideoExecutionTask, profile: VideoExecutionPayload['executionProfile'], model: ModelArkModelId, ): ModelArkTaskPlan['resolution'] { // 1080p only when the profile — or the scene's packet (`filmmaking-prompts // --resolution`, the one per-task override any transport honours) — asks for // it. The per-task value bypasses the profile gate in execution-runtime, so it // is checked here against the route's own capability list; 4k is refused // because Seedance 2.5 does not offer it and 2.0's 4k would be a price change. const requested = (task.resolution ?? profile.resolution).toLowerCase(); const allowed = ROUTE_CAPABILITIES['seedance-modelark'].resolutions; if (allowed.includes(requested) && (requested === '480p' || requested === '720p' || requested === '1080p')) { // The 2.0 fast/mini models stop at 720p: sending 1080p would be refused // by the vendor after queueing, or worse, quietly served at another price. if (!MODELARK_MODELS[model].resolutions.includes(requested)) { throw planError(sceneIndex, `${model} does not sell ${requested} (${MODELARK_MODELS[model].resolutions.join(', ')}); pick the 2.5 model for 1080p.`); } return requested; } throw planError(sceneIndex, `resolution ${requested} is not available on this route (${allowed.join(', ')}).`); } function resolvePlanDuration(sceneIndex: number, task: VideoExecutionTask, limits: ModelArkModelLimits, model: string): number { const seconds = task.durationSeconds; // ModelArk bills per second and also accepts `-1` ("the model picks"). Neither // an unbounded duration nor a silently substituted default belongs in a paid // contract, so the scene has to say how long it is. if (seconds === undefined) { throw planError(sceneIndex, 'durationSeconds is not set; this route bills per second and never picks a length for you.'); } const [min, max] = limits.durationRange; if (!Number.isInteger(seconds) || seconds < min || seconds > max) { throw planError(sceneIndex, `durationSeconds ${seconds} is outside ${model}'s range (${min}–${max} whole seconds).`); } return seconds; } /** * Decide, without touching the network, exactly what one scene will ask * ModelArk for. This is the contract the operator reviews in a dry run and * the preflight that runs over EVERY scene before the first paid submit. * * Mode selection: * - no references → text-to-video, the profile's aspect ratio. * - one image, no video/audio, and the reference is a keyframe * (`referenceRole` !== 'character') → strict first-frame i2v * (`first_frame`, plus `last_frame` from `endKeyframePath`); the vendor * requires `ratio: 'adaptive'` here, so the profile aspect is not sent. * - anything else → omni reference-to-video. A keyframe that arrives with a * voice clip cannot use `first_frame` (the two modes cannot mix), so it * becomes `@Image 1` and the prompt is told it is the first frame — a SOFT * lock the vendor documents as the way to combine the two. */ export function planModelArkTask( task: VideoExecutionTask, profile: VideoExecutionPayload['executionProfile'], model: ModelArkModelId, options: { chainMode?: ModelArkChainMode; chainModeError?: string } = {}, ): ModelArkTaskPlan { const limits = MODELARK_MODELS[model]; const sceneIndex = task.sceneIndex; const exact = task.promptPolicy === 'exact'; const classified = classifyModelArkReferences(task.referencePaths); const issues: string[] = []; const resolution = resolvePlanResolution(sceneIndex, task, profile, model); const duration = resolvePlanDuration(sceneIndex, task, limits, model); if (options.chainModeError && task.chainedFromCandidateId) throw planError(sceneIndex, options.chainModeError); if (options.chainMode === 'extend' && task.chainedFromCandidateId) { return planModelArkExtension(task, profile, model, { classified, resolution, duration }); } const keyframeShaped = task.referenceRole !== 'character' && classified.images.length === 1; if (task.endKeyframePath && !keyframeShaped) { // Deliberately narrow: omni mode could carry an end frame as another // reference image, but without a lone first frame there is nothing for it // to be the end OF, and a silent reinterpretation is worse than a refusal. throw planError(sceneIndex, 'endKeyframePath needs exactly one keyframe image as the first frame; character references cannot carry an end frame.'); } if (task.endKeyframePath && classifyModelArkReferences([task.endKeyframePath]).images.length !== 1) { throw planError(sceneIndex, `endKeyframePath ${task.endKeyframePath} is not an image; a last frame must be a still.`); } if (classified.images.length + classified.videos.length + classified.audios.length === 0) { return { sceneIndex, model, mode: 'text', prompt: task.prompt, ratio: profile.aspectRatio, resolution, duration, generateAudio: profile.generateAudio, refs: [], issues, }; } const strictFrames = keyframeShaped && classified.videos.length === 0 && classified.audios.length === 0; if (strictFrames) { const refs: ModelArkReferencePlan[] = [{ kind: 'image', role: 'first_frame', source: classified.images[0] }]; if (task.endKeyframePath) refs.push({ kind: 'image', role: 'last_frame', source: task.endKeyframePath }); issues.push( `scene ${sceneIndex}: first-frame task, so ratio is 'adaptive' (the clip takes the keyframe's shape); the profile aspect ${profile.aspectRatio} is not enforced.`, ); return { sceneIndex, model, mode: 'first-frame', prompt: task.prompt, ratio: 'adaptive', resolution, duration, generateAudio: profile.generateAudio, refs, issues, }; } // Omni reference-to-video. const images = [...classified.images]; if (task.endKeyframePath) images.push(task.endKeyframePath); if (images.length === 0 && classified.videos.length === 0 && !limits.audioOnlyReferences) { throw planError(sceneIndex, `${model} cannot take audio-only references; add at least one image or video, or use the 2.5 model.`); } if (images.length > limits.maxImageRefs) { throw planError(sceneIndex, `${images.length} image references exceed ${model}'s limit of ${limits.maxImageRefs}.`); } if (classified.videos.length > limits.maxVideoRefs) { throw planError(sceneIndex, `${classified.videos.length} video references exceed ${model}'s limit of ${limits.maxVideoRefs}.`); } if (classified.audios.length > limits.maxAudioRefs) { throw planError(sceneIndex, `${classified.audios.length} audio references exceed ${model}'s limit of ${limits.maxAudioRefs}.`); } let prompt = task.prompt; if (keyframeShaped) { // The frame directive leads and names every image itself (the keyframe, and // the end keyframe when present, are the ONLY images in this shape), so the // marker pass below only has to cover the video/audio references. const directive = task.endKeyframePath ? 'The first frame is @Image 1 and the last frame is @Image 2.' : 'The first frame is @Image 1.'; if (exact) { throw planError(sceneIndex, 'this keyframe scene also carries video/audio references, which needs the prompt rewritten to name the first frame, but promptPolicy is exact.'); } const marked = ensureModelArkMarkers(prompt, { images: 0, videos: classified.videos.length, audios: classified.audios.length }); prompt = `${directive} ${marked.trimStart()}`; issues.push( `scene ${sceneIndex}: the keyframe rides as @Image 1 in omni mode (video/audio references present); the first-frame lock is soft, not exact.`, ); } else { const marked = ensureModelArkMarkers(prompt, { images: images.length, videos: classified.videos.length, audios: classified.audios.length, }); if (marked !== prompt) { if (exact) { throw planError(sceneIndex, 'the prompt does not name every reference (@Image N / @Video N / @Audio N) and promptPolicy is exact, so it cannot be amended.'); } prompt = marked; } } const refs: ModelArkReferencePlan[] = [ ...images.map((source): ModelArkReferencePlan => ({ kind: 'image', role: 'reference_image', source })), ...classified.videos.map((source): ModelArkReferencePlan => ({ kind: 'video', role: 'reference_video', source })), ...classified.audios.map((source): ModelArkReferencePlan => ({ kind: 'audio', role: 'reference_audio', source })), ]; return { sceneIndex, model, mode: 'omni', prompt, ratio: profile.aspectRatio, resolution, duration, generateAudio: profile.generateAudio, refs, omniReferenceTaskType: 'auto', issues, }; } /** * A chained scene in `extend` mode: the previous scene's clip (the chain seed, * always the FIRST reference path, so the first video) rides as `@Video 1` and * the task is an explicit extension. The vendor's rules for it: at least one * `reference_video`, `ratio: adaptive` (the clip keeps the extended clip's * shape), and a prompt that SAYS extend/continue — so `Continue @Video 1:` is * prepended unless the prompt already carries that intent. Refused, never * guessed: a 2.0 model (the explicit field is documented for 2.5 only), a * chained scene with no video (the seed became an image — the chain mode * differs between where the payload was built and here), an end keyframe * (frame roles cannot mix with references), and an `exact` prompt that would * need rewording. */ function planModelArkExtension( task: VideoExecutionTask, profile: VideoExecutionPayload['executionProfile'], model: ModelArkModelId, known: { classified: ReturnType; resolution: ModelArkTaskPlan['resolution']; duration: number }, ): ModelArkTaskPlan { const limits = MODELARK_MODELS[model]; const sceneIndex = task.sceneIndex; const { images, videos, audios } = known.classified; if (limits.family !== '2.5') { throw planError(sceneIndex, `${MODELARK_CHAIN_MODE_ENV}=extend asks for an explicit video extension (omni_reference_task_type: extend), which ModelArk documents only for Seedance 2.5; ${model} is a 2.0 model. Use dreamina-seedance-2-5-260628 or ${MODELARK_CHAIN_MODE_ENV}=first-frame.`); } // The chain seed is always the FIRST reference; ANOTHER video (a voice clip in // a video container, a scene asset) must never be what gets extended. if (classifyModelArkReferences(task.referencePaths.slice(0, 1)).videos.length !== 1) { throw planError(sceneIndex, `this chained scene's first reference is not the previous clip as a video (it arrived as an image), so ${MODELARK_CHAIN_MODE_ENV}=extend cannot apply; set it where the payload is built too (the workspace .env.local).`); } if (known.resolution === '1080p') { // The vendor's pages disagree (the Seedance 2.5 tutorial: reference videos at // 480p or 720p; the API reference: up to 4k). A 1080p OMNI reference was // measured accepted on 2026-09-23 (task cgt-20260923052923-r5f10), but an // extension is its own task type, so the stricter reading stands here until // a 1080p extension has been tested. throw planError(sceneIndex, 'an extension at 1080p is untested (the Seedance 2.5 tutorial takes reference videos at 480p or 720p only; the API reference lists up to 4k, and a 1080p reference in OMNI mode was measured accepted on 2026-09-23 — but an extension is a different task type), and the previous clip of a 1080p scene is a 1080p video; render the chain at 720p or use first-frame mode.'); } if (task.endKeyframePath) { throw planError(sceneIndex, 'an extension cannot also pin a last frame: frame roles and reference roles cannot share one request.'); } if (images.length > limits.maxImageRefs) throw planError(sceneIndex, `${images.length} image references exceed ${model}'s limit of ${limits.maxImageRefs}.`); if (videos.length > limits.maxVideoRefs) throw planError(sceneIndex, `${videos.length} video references exceed ${model}'s limit of ${limits.maxVideoRefs}.`); if (audios.length > limits.maxAudioRefs) throw planError(sceneIndex, `${audios.length} audio references exceed ${model}'s limit of ${limits.maxAudioRefs}.`); // The lead is kept only when the prompt ALREADY ties the intent to @Video 1 // ("Continue @Video 1…", "Extend @Video 1 backward…"): ordinary words ("the // rain continues", "an extended close-up", "continue the orbit") must not // switch it off, or the vendor may not read the task as an extension. It goes // on first, so the marker pass sees its @Video 1 and does not name it twice. const namesIntent = /\b(?:extend|continue)\b[^.@]{0,40}@Video\s?1(?!\d)/i.test(task.prompt); const withIntent = namesIntent ? task.prompt : `Continue @Video 1: ${task.prompt.trimStart()}`; const prompt = ensureModelArkMarkers(withIntent, { images: images.length, videos: videos.length, audios: audios.length }); if (prompt !== task.prompt && task.promptPolicy === 'exact') { throw planError(sceneIndex, 'an extension needs the prompt to name @Video 1 and say continue/extend, and promptPolicy is exact, so it cannot be amended.'); } const refs: ModelArkReferencePlan[] = [ ...images.map((source): ModelArkReferencePlan => ({ kind: 'image', role: 'reference_image', source })), ...videos.map((source): ModelArkReferencePlan => ({ kind: 'video', role: 'reference_video', source })), ...audios.map((source): ModelArkReferencePlan => ({ kind: 'audio', role: 'reference_audio', source })), ]; return { sceneIndex, model, mode: 'omni', prompt, ratio: 'adaptive', resolution: known.resolution, duration: known.duration, generateAudio: profile.generateAudio, refs, omniReferenceTaskType: 'extend', issues: [ `scene ${sceneIndex}: extends the previous clip (@Video 1), so ratio is 'adaptive' (the clip keeps the previous clip's shape; the profile aspect ${profile.aspectRatio} is not enforced) and the previous clip's seconds are billed as input on top of the output.`, ], }; } /** * The wire body. `resolvedUrls[i]` is what `plan.refs[i]` became after hosting * or inlining (see `modelark-references.ts`); order is preserved because the * `@Image N` markers count in `content[]` order. */ export function buildModelArkCreateBody(plan: ModelArkTaskPlan, resolvedUrls: readonly string[]): ModelArkCreateBody { if (resolvedUrls.length !== plan.refs.length) { throw new Error(`seedance-modelark scene ${plan.sceneIndex}: ${plan.refs.length} references planned but ${resolvedUrls.length} resolved.`); } const content: ModelArkContentItem[] = [{ type: 'text', text: plan.prompt }]; plan.refs.forEach((ref, index) => { const url = resolvedUrls[index]; if (ref.kind === 'image') content.push({ type: 'image_url', image_url: { url }, role: ref.role }); else if (ref.kind === 'video') content.push({ type: 'video_url', video_url: { url }, role: ref.role }); else content.push({ type: 'audio_url', audio_url: { url }, role: ref.role }); }); return { model: plan.model, content, resolution: plan.resolution, ratio: plan.ratio, duration: plan.duration, generate_audio: plan.generateAudio, watermark: false, ...(plan.omniReferenceTaskType ? { omni_reference_task_type: plan.omniReferenceTaskType } : {}), }; } export type ModelArkErrorClass = 'content-violation' | 'task-constraint' | 'other'; /** Group a vendor error so the transport can word its remedy; never used to retry a paid submit. */ export function classifyModelArkError(code: string | undefined, message: string | undefined): ModelArkErrorClass { const haystack = `${code ?? ''} ${message ?? ''}`; if (/TaskTypeConstraint|TaskTypeMismatch/i.test(haystack)) return 'task-constraint'; if (/sensitive|moderation|content.?(policy|filter|violation)/i.test(haystack)) return 'content-violation'; return 'other'; } export class ModelArkHttpError extends Error { constructor( readonly status: number, readonly code: string | undefined, readonly vendorMessage: string | undefined, readonly rawBody: string, ) { super(`ModelArk request failed with HTTP ${status}${code ? ` (${code})` : ''}: ${safeErrorBody(vendorMessage ?? rawBody)}`); this.name = 'ModelArkHttpError'; } } export interface ModelArkFetchResponse { ok: boolean; status: number; text(): Promise; json(): Promise; arrayBuffer(): Promise; } export type ModelArkFetchLike = (input: string, init?: { method?: string; headers?: Record; body?: string; }) => Promise; export interface ModelArkClientContext { fetchImpl: ModelArkFetchLike; baseUrl: string; apiKey: string; /** Retry policy for the idempotent verbs (GET / DELETE); tests inject `{ retries: 0 }`. */ retry?: Pick; } /** A thrown transient (network error, or a 5xx after the retries ran out) as the caller can act on it. */ function transientAsResult(error: unknown): { ok: false; status: number; body: string } { const status = error && typeof error === 'object' && typeof (error as { status?: unknown }).status === 'number' ? (error as { status: number }).status : 0; return { ok: false, status, body: error instanceof Error ? error.message : String(error) }; } function authHeaders(apiKey: string): Record { return { 'Content-Type': 'application/json', Authorization: `Bearer ${apiKey}` }; } function parseVendorError(raw: string): { code?: string; message?: string } { try { const parsed = JSON.parse(raw) as { error?: { code?: unknown; message?: unknown }; code?: unknown; message?: unknown }; const node = parsed.error ?? parsed; return { code: typeof node.code === 'string' ? node.code : undefined, message: typeof node.message === 'string' ? node.message : undefined, }; } catch { return {}; } } async function failedResponse(response: ModelArkFetchResponse): Promise { const raw = await response.text().catch(() => ''); const { code, message } = parseVendorError(raw); return new ModelArkHttpError(response.status, code, message, raw); } /** * ONE paid POST, never retried: the create endpoint bills and is not * idempotent, so replaying it on a gateway error could queue two charged * tasks. A transient failure surfaces to the caller as-is. */ export async function modelArkCreateTask(ctx: ModelArkClientContext, body: ModelArkCreateBody): Promise<{ id: string; raw: unknown }> { const response = await ctx.fetchImpl(`${ctx.baseUrl}${MODELARK_TASKS_PATH}`, { method: 'POST', headers: authHeaders(ctx.apiKey), body: JSON.stringify(body), }); if (!response.ok) throw await failedResponse(response); const raw = await response.json(); const id = raw && typeof raw === 'object' ? (raw as { id?: unknown }).id : undefined; if (typeof id !== 'string' || !id.trim()) { throw new Error(`ModelArk create returned no task id: ${safeErrorBody(JSON.stringify(raw))}`); } return { id, raw }; } /** * Idempotent GET with transient retry (network errors and 5xx). Throws a * `ModelArkHttpError` on a 4xx and the transient error when the retries run * out — the poll isolates that per scene so one task's error never hides * another's progress. */ export async function modelArkGetTask(ctx: ModelArkClientContext, taskId: string): Promise { const response = await withRetry(() => fetchTransientRetry(ctx.fetchImpl, `${ctx.baseUrl}${MODELARK_TASKS_PATH}/${encodeURIComponent(taskId)}`, { method: 'GET', headers: authHeaders(ctx.apiKey), }), ctx.retry); if (!response.ok) throw await failedResponse(response); const raw = await response.json(); if (!raw || typeof raw !== 'object') { throw new Error(`ModelArk task ${taskId} returned a non-object body: ${safeErrorBody(JSON.stringify(raw))}`); } return raw as ModelArkTask; } /** * A task's `created_at` as unix seconds. It is documented as unix seconds; a * numeric string or an ISO string is read too, because a row silently dropped * from the list would let the resolver claim a task was never created. */ export function modelArkCreatedAtSeconds(value: unknown): number | undefined { return typeof value === 'number' ? value : typeof value === 'string' && /^\d+$/.test(value) ? Number(value) : typeof value === 'string' && Number.isFinite(Date.parse(value)) ? Date.parse(value) / 1000 : undefined; } export interface ModelArkTaskSummary { id: string; status: string; model?: string; /** Unix seconds. */ created_at?: number; } /** * `GET /contents/generations/tasks?filter.model=…` — the vendor's own list of * the key's tasks (last 7 days, newest first, up to 500 a page). It echoes no * client reference, so the only way to find a task whose create response was * lost is by model and creation time; the transport does that with a narrow * window and refuses to bind when the window holds more than one task. */ export async function modelArkListTasks( ctx: ModelArkClientContext, filter: { model: string; status?: string; pageSize?: number }, ): Promise { const query = new URLSearchParams({ page_num: '1', page_size: String(filter.pageSize ?? 500), 'filter.model': filter.model }); if (filter.status) query.set('filter.status', filter.status); const response = await withRetry(() => fetchTransientRetry(ctx.fetchImpl, `${ctx.baseUrl}${MODELARK_TASKS_PATH}?${query.toString()}`, { method: 'GET', headers: authHeaders(ctx.apiKey), }), ctx.retry); if (!response.ok) throw await failedResponse(response); const raw = await response.json(); const items = raw && typeof raw === 'object' && Array.isArray((raw as { items?: unknown }).items) ? (raw as { items: unknown[] }).items : null; if (!items) throw new Error(`ModelArk task list returned no items array: ${safeErrorBody(JSON.stringify(raw))}`); return items.flatMap((item) => { if (!item || typeof item !== 'object') return []; const node = item as { id?: unknown; status?: unknown; model?: unknown; created_at?: unknown }; if (typeof node.id !== 'string') return []; const createdAt = modelArkCreatedAtSeconds(node.created_at); return [{ id: node.id, status: typeof node.status === 'string' ? node.status : '', ...(typeof node.model === 'string' ? { model: node.model } : {}), ...(createdAt !== undefined ? { created_at: createdAt } : {}), }]; }); } /** * Whether a failed create might still have created (and billed) a task. A 4xx * is the vendor saying no; anything else — a network error, a timeout, a 5xx * that outlived nothing (the create is never retried), a 2xx with no id — is * a lost answer, and the task may exist. */ export function isAmbiguousCreateError(error: unknown): boolean { // 408 (the request timed out server-side) and 499 (the client went away) // are the two 4xx that say nothing about whether the body was processed. if (error instanceof ModelArkHttpError) return !(error.status >= 400 && error.status < 500) || error.status === 408 || error.status === 499; return true; } /** * DELETE cancels a QUEUED task. It cannot stop a running one (the vendor * returns an error, which is reported here rather than thrown so the caller * can say "still billing"), and on a finished task it deletes the record — * which is why the transport only ever calls it on a scene it is still waiting for. * NEVER throws: a network error or a 5xx that outlives the retries comes back * as `{ ok: false, status, body }` too, because "we could not cancel it" is * exactly the answer the caller has to relay, not swallow. */ export async function modelArkDeleteTask(ctx: ModelArkClientContext, taskId: string): Promise<{ ok: boolean; status: number; body: string }> { let response: ModelArkFetchResponse; try { response = await withRetry(() => fetchTransientRetry(ctx.fetchImpl, `${ctx.baseUrl}${MODELARK_TASKS_PATH}/${encodeURIComponent(taskId)}`, { method: 'DELETE', headers: authHeaders(ctx.apiKey), }), ctx.retry); } catch (error) { return transientAsResult(error); } const body = await response.text().catch(() => ''); return { ok: response.ok, status: response.status, body }; }