import type { RateLimitConfig } from '../types/config.js'; /** * The parts of a request the limiter buckets on. * * Structural rather than tied to `CompletionRequest`, so every operation family — completions, * embeddings, and whatever comes next — shares one limiter instance and one budget. */ export interface RateLimitedRequest { /** Model requested, for per-model limits. */ model?: string; /** Caller, for per-user limits. */ userId?: string; } /** Raised when a caller exceeds its rate limit. */ export declare class NexusRateLimitError extends Error { /** The bucket that was full. */ key: string; /** Epoch milliseconds when the window resets, when the store reported one. */ readonly resetAt?: number | undefined; constructor( /** The bucket that was full. */ key: string, /** Epoch milliseconds when the window resets, when the store reported one. */ resetAt?: number | undefined); /** Seconds a caller should wait, suitable for a `Retry-After` header. */ get retryAfterSeconds(): number | undefined; } /** * Fixed-window rate limiting, per user, per model, or globally, in memory or through a shared * store. */ export declare class RateLimiter { private buckets; /** * Counts one call against a distributed store. * * Separate from `check()` rather than replacing it: a store is asynchronous, and making the * common in-memory path await a promise would add a microtask to every request that does not use * one. Callers pick the path by whether `config.store` is set. */ checkAsync(request: RateLimitedRequest, config?: RateLimitConfig): Promise; /** Counts one call in memory. Throws `NexusRateLimitError` when the bucket is full. */ check(request: RateLimitedRequest, config?: RateLimitConfig): void; private getKey; }