import type { HeapSample } from "./control-server.js"; import { launchInstrumented } from "./launcher.js"; import { runAbandonPhase, type AbandonOrigin } from "./abandon-load.js"; import { runLoadPhase } from "./load.js"; import { type TrendResult } from "./trend.js"; export type RitualOptions = { /** Absolute path to the standalone server.js. */ serverPath: string; /** Concrete request path (dynamic params already resolved), e.g. "/products/42". */ route: string; /** Directory for this route's snapshots and control file. */ workDir: string; /** Built bootstrap module for `--import`. */ bootstrapPath: string; appPort: number; warmupRequests?: number; loadRequests?: number; connections?: number; cycles?: number; idleMs?: number; /** Old-space cap for the measured process (MB). Default: 512. */ maxOldSpaceMb?: number; /** How long the process gets to start listening (ms). Default: 60_000. */ readyTimeoutMs?: number; /** Headers sent with every request during warm-up and load. */ headers?: Record; /** * The load is driving a cached route with keys it has not served before, so * the store fills up as a side effect of measuring. Recorded on the verdict * for disclosure; it does not change how the verdict is reached. */ cacheDriven?: boolean; /** Emulate clients that disconnect before the response arrives. */ abandonAfterMs?: number; /** Where that deadline starts. Defaults to `first-byte`. */ abandonFrom?: AbandonOrigin; }; export type PhaseTiming = { phase: string; seconds: number; }; /** What each load phase actually did — auditable after the fact. */ export type LoadOutcome = { phase: string; sent: number; ok2xx?: number; non2xx?: number; errors?: number; timeouts?: number; abandoned?: number; /** Abandonments where the response had already started — the mid-stream path. */ abandonedMidStream?: number; /** Abandonments where the first-byte budget expired in silence. */ abandonedBeforeResponse?: number; }; /** * Whether the heap actually held still before each sample was taken. * * `unknown` is not a softer `moving`: with fewer than two GC readings there is * nothing to compare, so the run never learned whether the heap was steady. * Conflating the two made every short-idle run look like a moving heap. */ export type SettleStatus = "settled" | "moving" | "unknown"; export type SettleOutcome = { phase: string; status: SettleStatus; /** GC polls taken before converging or giving up. */ polls: number; }; /** * Highest memory observed *during* a load cycle, per class. * * Every other number in a run is taken after idle and a forced GC, which is * what a verdict about retention needs. It is also blind to the process that * climbs to 3.5 GB under load and hands it all back: `stable`, and dead in a * 1 GB container. A peak is a lower bound — a spike shorter than the poll * interval is never seen. */ export type PeakSample = { phase: string; heapUsed: number; external: number; arrayBuffers: number; rss: number; /** Readings taken; 0 means the poller never got one. */ polls: number; }; /** What survived a process that ran out of heap partway through a run. */ export type HeapExhaustedEvidence = { /** Post-GC readings taken before the death: baseline first, then per cycle. */ memorySamples: HeapSample[]; peaks: PeakSample[]; unreclaimedSamples: HeapSample[]; /** Cycles that finished. Zero means it died inside the first one. */ cyclesCompleted: number; cyclesRequested: number; requestsPerCycle: number; baselineSnapshot: string; }; /** * The measured process ran out of heap mid-run. * * Thrown rather than returned because there is no `RitualResult` to build: no * after-snapshot was taken and the trend has nothing complete to classify. It * is still a finding, not a failure — see `explainExit` in `launcher.ts` — and * the evidence it carries is what the report shows instead of a curve. */ export declare class HeapExhaustedError extends Error { readonly evidence: HeapExhaustedEvidence; constructor(message: string, evidence: HeapExhaustedEvidence); } export type RitualResult = { route: string; /** Wall-clock per phase, so slow runs can be explained instead of guessed. */ timings: PhaseTiming[]; /** Per-phase request outcomes; without these a run cannot be audited. */ loadOutcomes: LoadOutcome[]; /** Per-cycle settle results: a sample taken while the heap moved is suspect. */ settleOutcomes: SettleOutcome[]; /** Post-GC heapUsed per phase: baseline first, then one per cycle. */ samples: number[]; /** Full memory samples in the same order. */ memorySamples: HeapSample[]; /** Highest memory seen during each load cycle, sampled without collecting. */ peaks: PeakSample[]; /** * One reading per cycle taken before any collection is forced — what the * process is holding on its own. * * Every other number here is post-GC, which is what makes a verdict mean * something and is also blind to a whole class of leak: memory a full GC * would reclaim that a production process never runs one often enough to * (vercel/next.js#96533). Absolute values overstate retention — this is * taken seconds after load, not hours, so it includes garbage not yet * collected — but a roughly constant offset across identical cycles cancels * out of the deltas the trend is read from. */ unreclaimedSamples: HeapSample[]; /** Trend over `unreclaimedSamples`. Reported, never the verdict. */ unreclaimedTrend: TrendResult; baselineSnapshot: string; /** * Empty when the final snapshot could not be taken. The curve is already * complete by then, so the run still has a verdict — only the attribution * is lost. `snapshotFailure` says why. */ afterSnapshot: string; /** Why `afterSnapshot` is empty, when it is. */ snapshotFailure?: string; trend: TrendResult; requestsPerCycle: number; /** * The gate (bytes per cycle) this verdict was judged against. Travels with * the result so the confidence audit grades against the same number the * verdict used, and so the report can print what it measured against. */ minGrowthPerCycle: number; }; /** Injectable seams for unit tests; production uses the real implementations. */ export type RitualDeps = { launch: typeof launchInstrumented; load: typeof runLoadPhase; /** Early-disconnect load; a seam like `load`, for the same reasons. */ abandon: typeof runAbandonPhase; sleep: (ms: number) => Promise; /** GC-free read, polled while the app is under load. */ readMemory: (port: number) => Promise; }; /** * The pre-collection wait, taken *out of* the idle rather than added to it. * * Capped at a quarter of the idle so short profiles keep their settle budget: * `--quick` idles for 8 s, and a flat 3 s would be nearly half of it. */ export declare function unreclaimedSettleFor(idleMs: number): number; /** * Single source of truth for ritual defaults — reports must echo them. * * `cycles` is 4 rather than the minimum 3 because of how much the verdict * actually sees: the sample array is `[baseline, ...cycles]` and * `classifyTrend` drops the warm-up delta, leaving `cycles - 1` numbers. At 3 * cycles that is **two** deltas, and `anyFlatOrDown` needs only one of the two * to be non-positive to call a route stable — one noisy cycle flips the * verdict. Four cycles give three deltas, which is what the real-app * validation ran with. The 3-cycle minimum stays available for users who want * a faster, weaker read. */ export declare const RITUAL_DEFAULTS: { readonly warmupRequests: 200; readonly loadRequests: 5000; readonly connections: 100; readonly cycles: 4; readonly idleMs: 30000; readonly maxOldSpaceMb: 512; }; /** * Runs the validated phase-0 ritual against one route in a fresh process: * * warm-up → GC → baseline snapshot * → [load → idle → GC → sample] × cycles (last cycle snapshots) * * Warm-up before the baseline and idle+GC before every sample are what * separate a real measurement from the classic false positive. */ export declare function runRitual(options: RitualOptions, deps?: RitualDeps): Promise;