import type { ResolvedDevice } from "../../device.js"; /** * The four ONNX sessions a Chatterbox model is split into, named by their * *file* names: Transformers.js resolves per-session `dtype` entries against * the ONNX file name (`language_model`), not the session key (`model`) — a * mismatched key is silently ignored and the device default (fp32 on WebGPU, * 2 GB for the language model) loads instead. */ export type ChatterboxSession = "embed_tokens" | "speech_encoder" | "language_model" | "conditional_decoder"; /** Per-session quantization, as accepted by `from_pretrained({ dtype })`. */ export type DtypeConfig = Record; /** One attempt at loading the model: a device plus a quantization choice. */ export interface LoadPlan { readonly device: ResolvedDevice; readonly dtype: DtypeConfig; } /** * Build the ordered list of load attempts for a device. * * fp16 support varies by GPU, so a WebGPU request degrades to integer-only * weights before giving up on the GPU entirely and landing on WASM. A WASM * request has nothing to fall back to and yields a single plan. * * Pass `fp16: false` when the adapter lacks `shader-f16`: an f16 plan on such * a device *loads* fine and only fails at the first inference, which the * load-time fallback can no longer catch. */ export declare function buildLoadPlans(device: ResolvedDevice, overrides?: Partial, fp16?: boolean): LoadPlan[]; //# sourceMappingURL=dtype-plan.d.ts.map