/** * evermind_lm.ts — EvermindLM: a small but complete generative language model. * * This is what turns a trained checkpoint into an *AI that generates text* (the * thing a marketplace buyer actually runs). Architecture (Mamba-flavoured, the * minimal exact-gradient CPU reference): * * x_t = Embed[token_t] * per layer: * x_t += DepthwiseCausalConv(x)_t // temporal mixing (short conv) * x_t += SharedExpertMoE(x_t) // per-position channel mixing (sparse) * logits_t = x_t · Embedᵀ // tied output head * * The token mixer is a depthwise causal convolution (each channel sees a short * window of its own past — Mamba's pre-conv) and the channel mixer is the * shared-expert MoE, so the model is genuinely sparse. Embeddings are tied * (input lookup == output head), which the gradient code accounts for. * * Pure CPU, exact forward + backward (finite-difference checked), reusing the * engine's MoE, cross-entropy, and AdamW. The WGSL/WebGPU path is a future * acceleration with the same shapes. */ import { type AdamWOptions } from "../optim/adamw.js"; import { DynamicLossScaler, type LossScalerOptions } from "../training/mixed_precision.js"; export interface EvermindLMConfig { /** Vocabulary size (the only required field; everything else has a default). */ vocabSize: number; /** Model (channel) dimension. Default 64. */ dModel?: number; /** Number of (conv + MoE) blocks. Default 2. */ numLayers?: number; /** Causal conv kernel width. Default 3. */ convKernel?: number; /** Hidden width of each MoE expert FFN. Default 2·dModel. */ hiddenDim?: number; /** Routed experts per MoE layer. Default 4. */ numExperts?: number; /** Experts activated per token. Default 2. */ topK?: number; /** Deterministic init seed. */ seed?: number; } export declare const DEFAULT_LM_CONFIG: Required>; export declare const DEFAULT_LM_SEED = 1163283533; interface MoECacheLike { x: Float32Array; route: { experts: number[]; gates: number[]; probs: Float32Array; }; sharedPre: Float32Array; sharedH: Float32Array; expertOut: Float32Array[]; expertPre: Float32Array[]; expertH: Float32Array[]; } interface LayerCache { layerIn: Float32Array[]; normedConv: Float32Array[]; rmsConv: number[]; afterConv: Float32Array[]; rmsMoe: number[]; moeCache: MoECacheLike[]; } interface ForwardCache { tokens: number[]; layers: LayerCache[]; finalX: Float32Array[]; } /** A tokenizer the LM can read/write text through (the engine's `BPETokenizer` fits). */ export interface TextCodec { encode(text: string): number[]; decode(ids: number[]): string; } export interface LMGenerateOptions { maxNewTokens: number; /** Sampling temperature; ≤0 ⇒ greedy argmax. Default 0 (greedy). */ temperature?: number; /** Deterministic sampler seed (only used when temperature > 0). */ seed?: number; /** Stop generating when this token id is produced. */ stopToken?: number; } export declare class EvermindLM { readonly config: Required>; /** Tied token embedding / output head: vocabSize × dModel (row-major). */ emb: Float32Array; private gEmb; /** Per-layer depthwise causal conv kernels: dModel × convKernel. */ private readonly conv; private readonly gConv; /** Per-layer pre-conv / pre-MoE RMSNorm gains (dModel each). */ private readonly nConv; private readonly gNConv; private readonly nMoe; private readonly gNMoe; /** Per-layer channel mixer. */ private readonly moe; constructor(config: EvermindLMConfig); /** Embed a token sequence into per-position channel vectors. */ private _embed; /** * One (conv + MoE) block: pre-norm → depthwise causal conv → residual, then * pre-norm → MoE channel mixer → residual. Returns the block output and the * activation cache its backward needs. Isolating this is what lets * {@link lossAndBackwardCheckpointed} recompute a layer's activations on demand * instead of retaining every layer's cache at once. */ private _forwardLayer; /** Tied output head: logits_t[v] = x_t · emb[v]. */ private _head; /** Run the model over a token sequence; returns per-position logits + a cache. */ forward(tokens: number[]): { logits: Float32Array[]; cache: ForwardCache; }; /** * Head + tied-embedding gradient. Accumulates dL/d(head→emb) into gEmb and * returns the mean next-token loss plus dL/d(finalX). Shared by the full and * checkpointed backward paths so the head maths lives in one place. */ private _headBackward; /** * Backward through one (conv + MoE) block given dL/d(block output) and the * block's activation cache. Accumulates conv/norm/MoE gradients and returns * dL/d(block input). The inverse of {@link _forwardLayer}. */ private _backwardLayer; /** Embedding lookup: dL/d(layer-0 input) flows into the row for each token. */ private _embedBackward; /** * Next-token cross-entropy over the sequence (predict tokens[t+1] from * position t), accumulating exact gradients. Returns the mean loss. Call * {@link zeroGrad} before and an optimiser step after. */ lossAndBackward(tokens: number[]): number; /** * Activation-checkpointed backward — numerically identical gradients to * {@link lossAndBackward}, but retains only the per-LAYER inputs during the * forward instead of every layer's full activation cache. Each layer's * activations are RECOMPUTED (a cheap extra forward) when its backward runs, * so peak activation memory is one layer's cache, not all of them. This is the * memory-for-compute trade the cookbook pairs with FSDP to fit longer * sequences / bigger models on a constrained device. */ lossAndBackwardCheckpointed(tokens: number[]): number; /** * Text-level generation: encode the prompt, generate, decode. `codec` is any * tokenizer exposing encode/decode (the engine's `BPETokenizer` satisfies it), * so the LM consumes and emits real text rather than raw token ids. The model's * `vocabSize` must match the codec's vocabulary. */ generateText(prompt: string, codec: TextCodec, opts: LMGenerateOptions): string; /** Greedy / temperature-sampled autoregressive generation. Returns NEW token ids. */ generate(prompt: number[], opts: LMGenerateOptions): number[]; /** All trainable parameters as {data} (AdamW-compatible), canonical order. */ parameters(): { data: Float32Array; }[]; /** Gradient buffers, index-aligned with {@link parameters}. */ gradients(): { data: Float32Array; }[]; zeroGrad(): void; /** Serialise to an "EVL0" binary (fp16 or f32), params in {@link parameters} order. */ exportWeights(opts?: { fp16?: boolean; }): ArrayBuffer; /** Load weights from an "EVL0" binary. Validates CRC (when present), magic + dims. */ loadWeights(buffer: ArrayBuffer): void; /** Parse + validate an EVL0 checkpoint to a flat f32 param vector (no mutation). */ private _readFlat; /** Flat concatenation of all params, canonical order (matches checkpoint layout). */ private _flat; /** Distribute a flat param vector back into the model's parameters. */ private _setFlat; /** * Export a SPARSE DELTA of the current weights against a base EVL0 checkpoint * (EVM-6). Online WSLA updates only a few rows, so a delta persists kilobytes * instead of rewriting the whole model. Reconstruct with {@link loadDelta}. */ exportDelta(baseCheckpoint: ArrayBuffer, opts?: { eps?: number; }): ArrayBuffer; /** Reconstruct weights from a base EVL0 checkpoint + a delta (EVM-6). */ loadDelta(baseCheckpoint: ArrayBuffer, delta: ArrayBuffer): void; } export interface EvermindLMTrainOptions extends AdamWOptions { epochs?: number; /** * Gradient accumulation: average gradients over this many sequences (micro- * batches) before each optimiser step, for a larger effective batch on a * memory-constrained device. Default 1. */ accumSteps?: number; /** * Use activation checkpointing (recompute layer activations in backward) to * cap peak activation memory. Identical gradients, a little extra compute. * Default false. */ checkpoint?: boolean; /** * Mixed-precision training: fp16-rounded gradients with dynamic loss scaling * over fp32 master weights (the model params). Overflowing steps are skipped * and the scale backs off. Default false. */ mixedPrecision?: boolean | LossScalerOptions; } /** Minimal sequence trainer: AdamW over next-token cross-entropy. */ export declare class EvermindLMTrainer { private readonly model; private readonly opts; private readonly adam; private readonly scaler; private readonly grads; constructor(model: EvermindLM, opts?: EvermindLMTrainOptions); /** The dynamic loss scaler (mixed-precision mode only) — exposes scale/overflow stats. */ get lossScaler(): DynamicLossScaler | null; private _backward; /** Divide accumulated gradients by `n` (accumulation averaging), in place. */ private _scaleGrads; /** Round accumulated gradients to fp16 precision (mixed-precision simulation), in place. */ private _fp16Grads; /** Multiply accumulated gradients by `s`, in place. */ private _mulGrads; /** Train on a set of token sequences; returns per-epoch mean loss. */ fit(sequences: number[][]): number[]; } export {}; //# sourceMappingURL=evermind_lm.d.ts.map