import { type WindowsNpmShimDeps, type WindowsNpmShimResolution } from "./windows-npm-shim.js"; import { type JobJournal, type JournalRow, type JournalStartingTree } from "./job-journal.js"; import type { LaneExecutionSnapshot } from "../lane-execution-broker.js"; import { type JobArchive } from "./job-archive.js"; import type { TreeSnapshot } from "./tree-delta.js"; import type { DispatchLaneStatus } from "../dispatch-lane-stats.js"; /** Depth marker written into every child's environment, read back to bound recursion. */ export declare const DEPTH_ENV = "LLM_RELAY_DISPATCH_DEPTH"; /** * How deep a chain of dispatches may go. A lane is itself an agent that can reach this same * server, so without a bound a loop is reachable — `agent-dispatch` bounds its own at 3 and this * matches it. The value is a ceiling on nesting, not on concurrency. */ export declare const DEFAULT_MAX_DEPTH = 3; /** Hard ceiling on captured lane output, so one runaway lane cannot exhaust memory. */ export declare const MAX_OUTPUT_BYTES: number; /** Ceiling on one lane run. agy's own `--print-timeout` convention here is 30 minutes. */ export declare const DEFAULT_LANE_TIMEOUT_MS: number; /** * `"timed_out"` is its own terminal status, DISTINCT from `"failed"` (2026-09-03, * C:\Code\docs\backlog.md "A bounded llm-relay design dispatch can consume its full 1,200-second * wait and return no result"). A killed-by-timeout run used to fall into the same bucket as an * ordinary nonzero exit, so a caller reading `status` could not tell "the lane ran and failed" * from "the lane never finished" — the two call for different next actions (retry a different * lane vs. maybe poll a little longer next time). `complete()` sets it whenever the run result * says `timedOut`, ahead of the exit-code/semantic-failure check. */ export type JobStatus = "running" | "completed" | "failed" | "cancelled" | "timed_out" | "killed"; /** * The four states `LaneJobStore.complete()`/`cancel()` can put a job into — `"running"` is the * only non-terminal member of `JobStatus`. Exported so a rendering test can iterate every * terminal status the store actually knows, rather than hand-copying the list a second time. */ export declare const TERMINAL_JOB_STATUSES: readonly ["completed", "failed", "cancelled", "timed_out", "killed"]; export type TerminalJobStatus = (typeof TERMINAL_JOB_STATUSES)[number]; /** * Terminal statuses worth reporting as lane telemetry: every terminal status but `cancelled` and * `killed` — a caller cancellation is not lane evidence, so it is discarded, never reported. Keyed as a `Record` over `Exclude<…, "cancelled">` so a NEW terminal status is * a compile error HERE, at the classifier, rather than a silent drop at the forwarder (the * closed-union gotcha in CLAUDE.md); the forwarder ranges over this list, never a hand copy. */ declare const REPORTABLE_JOB_STATUS_MAP: Record, true>; export type ReportableJobStatus = keyof typeof REPORTABLE_JOB_STATUS_MAP; export declare const REPORTABLE_JOB_STATUSES: readonly ("completed" | "failed" | "timed_out")[]; /** * Narrowing guard over `REPORTABLE_JOB_STATUSES` — the forwarder ranges over the list * through this, so the statuses stay narrowed past the check (a bare `.includes` would * not narrow, and the report below needs the `cancelled`/`running` members gone). */ export declare function isReportableJobStatus(status: JobStatus): status is ReportableJobStatus; /** * The relay's own response headers that announce what happened during an answer-mode HTTP call — * the direct-fetch sibling of the provenance every spawned-lane answer already carries (lane id, * spec, elapsed). Only these five, allow-list style: never the whole `Headers` object, and NEVER * read at all on a non-2xx response (a failure carries status + a bounded body excerpt, not * headers — see `runAnswerFetch`). */ export interface RelayAnnouncements { servedBy?: string; poolAttempts?: string; hedged?: string; latencyDemoted?: string; degraded?: string; } /** * "skipped" — a rung the walk never started at all, because it was already at its configured * `maxConcurrent` cap when its turn in the ladder came (backlog item "a per-lane CONCURRENCY cap * on cli dispatch rungs", 2026-09-09). * * ⚠ Deliberately NOT a member of `DispatchLaneStatus`. That type is what a settled RUN is reported * to the daemon as (`DispatchedTelemetryReport.status`, validated by `routes/admin.ts`'s pin/demote * and failure-kind total tables, and folded into `dispatch-lane-stats.json`'s wall-clock window by * `recordLaneRun`), and a skipped rung never ran — "spawn nothing … report no telemetry for it, * nothing ran" is the brief's whole point. Widening the REPORTED vocabulary to include a state * nothing ever reports would be exactly the closed-union defect CLAUDE.md warns about: a member * reachable only in theory, decided nowhere real. */ export declare const SKIPPED_LANE_STATUS: "skipped"; /** * Every state one WALK ATTEMPT can end in — `LaneAttempt.status`'s own closed union, strictly wider * than `DispatchLaneStatus` by the one member above. The two unions describe different questions * (this one is "what happened to this attempt in THIS job's record"; `DispatchLaneStatus` is "what * does the daemon's persisted lane history say"), so they are two types on purpose rather than one * reused for both. */ export type LaneAttemptStatus = DispatchLaneStatus | typeof SKIPPED_LANE_STATUS; /** * Did this attempt actually run a lane, or was it skipped before any process started? Total over * `LaneAttemptStatus` (`const _never: never` below), so a future member — whether added here or * inherited from a `DispatchLaneStatus` that grows — is a compile error at this switch rather than * a silent "counts as tried" default (the closed-union gotcha in CLAUDE.md). `mcp/server.ts` uses * it to keep a walk's terminal "N tried" message honest when every remaining lane was capped rather * than actually attempted — a skipped rung "counts as NOT TRIED" per the backlog item's own wording. */ export declare function attemptWasTried(status: LaneAttemptStatus): boolean; /** * One lane a dispatch WALK tried, in the order it tried them. * * ⚠ Its `status` is a `LaneAttemptStatus`, IMPORTED rather than restated — see that type's own * doc comment for why it is not simply `DispatchLaneStatus`. Two hand-written copies of one closed * set is the most-repeated defect in this repository's history, and here the type additionally buys * a guarantee: that union has no `cancelled` member, so "a caller cancellation is never reported as * lane evidence" becomes a property the compiler enforces at every call site rather than a runtime * check somebody can forget. */ export interface LaneAttempt { laneId: string; spec: string | undefined; status: LaneAttemptStatus; /** Wall clock for THIS attempt, not for the walk. Always 0 for a `"skipped"` attempt — nothing * ran, so there is no duration to report, the same rule an `"abandoned"` run's OWN duration * is withheld from `dispatch-lane-stats.ts`'s persisted window for. */ elapsedMs: number; /** Why the attempt ended this way, when it did not succeed. */ reason?: string; } /** * The status one lane attempt earns. Shares its PRIORITY ORDER with `LaneJobStore.complete()` * below, and `test/mcp-server.test.ts` pins that the two agree — they cannot share one * implementation because `complete()` maps onto `JobStatus`, which describes the WALK, while this * maps onto `DispatchLaneStatus`, which describes one lane. * * ⚠ `abandoned` is tested FIRST, ahead of `timedOut`. A lane the walk kills at its budget usually * reports `timedOut` from the killed child as well, and between the two the walk's own decision is * the MORE SPECIFIC claim: the relay knows it stopped this lane after N seconds, whereas the * child's timeout flag would report it as having exhausted a ceiling it never reached. */ export declare function classifyLaneAttempt(run: LaneRunResult, opts?: { abandoned?: boolean | undefined; semanticFailure?: string | undefined; }): DispatchLaneStatus; export type LaneActivityState = "starting" | "active" | "quiet" | "advancing" | "unmonitored"; export type LaneWalkVerdict = "keep-running" | "no-idle-stop"; /** * The walk's OWN liveness decision for the attempt running now. * * This is a published snapshot, not a second activity probe. `mcp/server.ts` computes it while * making the same keep/stop decision the walk already has to make, then `dispatch_status` merely * renders it. A status read must never call the relay-traffic, process-CPU or working-tree readers: * those readers carry baselines/state and consuming them from a poll could change routing. */ export interface LaneLiveness { activity: LaneActivityState; verdict: LaneWalkVerdict; /** When the walk last evaluated/published this snapshot. */ checkedAt: number; /** * Timestamp the idle timer is anchored to. Starts at attempt launch so a new lane gets a full * idle window even before it emits observable activity; moves forward when real activity arrives. */ idleBaselineAt: number; /** Newest REAL activity timestamp the walk accepted for this attempt, or null before any. */ lastActivityAt: number | null; /** What established the current idle baseline: attempt-start or a first-party activity signal. */ source: string; /** Idle cutoff applied to this attempt; null means the walk will not idle-stop it. */ idleMs: number | null; } export interface LaneJob { id: string; status: JobStatus; /** * Ladder rung this job is running NOW. A walk repoints it as it advances, so a `dispatch_status` * poll always names the lane currently doing the work; `attempts` holds the ones already tried. */ laneId: string; /** What the rung addresses (`pool/high`, `agy`, …), when it names one. */ spec: string | undefined; /** * Every lane this job tried, oldest first. Empty for a job that never advanced past its first * lane and is still running; one entry for an ordinary single-lane dispatch that settled. * * ⚠ The JOB is the WALK, not one lane, and this field is what makes that legible. The walk * cannot finish inside one blocking call — the measured client tool-call ceiling on this machine * is between 45 s and 100 s, and above it the call fails AND destroys the job handle — so it * continues in the background behind ONE handle. Re-pointing the handle at each new lane instead * would break polling outright. */ attempts: LaneAttempt[]; /** * Selectable lanes the walk's `maxLanes` bound kept it from trying. Absent when it tried * everything the ladder offered — see `LaneJobStore.noteWalkScope`. */ lanesNotTried?: number; /** * Whether this job ran as a WALK at all, or as the single pre-walk lane * (`routing.dispatchWalk: false`). * * ⚠ It exists because a renderer cannot tell those apart from `attempts` alone: a walk that * legitimately exhausted a one-rung ladder and a walk-disabled dispatch that tried its one lane * produce the same record. Saying "every lane has been tried" for the second is false, and it * also breaks the documented promise that `dispatchWalk: false` restores the pre-walk behaviour * exactly — the pre-walk answer carried no such advice at all. */ walkEnabled?: boolean; /** * The caller NAMED what to run — a `lane` override, or a `model` run as its own one-lane view — * so the walk ran exactly one lane on purpose. `jobAnswer` then says that only that lane ran: * "every dispatch lane has now been tried" would be false there, and it tells an autonomous * caller to stop delegating altogether (measured 2026-09-10 on jobs 0023 and 0024). */ forcedLane?: boolean; /** * How long the lane now running usually takes to ANSWER in this dispatch's mode, from that * lane's own completed runs. Diagnostic only: it gives duration context, but the caller follows * `liveness.verdict` rather than deciding "stuck" from history. 23 of the 182 unanswered * dispatches in the 2026-09-10 transcript sweep ended with the caller simply no longer polling. * Absent when the lane has no completed run on record. */ expected?: { medianMs: number | null; p80Ms: number | null; samples: number; }; startedAt: number; endedAt: number | undefined; exitCode: number | null; stdout: string; stderr: string; timedOut: boolean; cwd: string; /** Set only when the run failed before or during the spawn. */ error: string | undefined; /** Set only for an answer-mode job whose relay response reached the 2xx branch. */ relay?: RelayAnnouncements; /** Set when the job's dispatch view came from a fallback rather than the live daemon. */ dispatchSource?: "daemon" | "fallback"; /** * Who owns the currently-running agent process tree. Absent preserves the pre-D1/local shape for * embeds and answer-mode jobs. `local-fallback` is explicit because it does NOT survive MCP exit. */ executionOwner?: "relay-daemon" | "local-fallback"; /** * What this dispatcher started for this job, and what became of it — written once, when the job * reaches a terminal state. See `LaneProcessReport`. */ process?: LaneProcessReport; /** * What the lane now running has produced so far, and when it last produced anything — so a poll * can say how long a lane has been SILENT (`docs/backlog.md`: a Muse Spark lane logged a stream * error in its first second, produced nothing more, and read `running` for nine minutes with * nothing on the status to distinguish it from a lane still thinking). Present only while a * SPAWNED attempt runs; an answer-mode call has no output stream and carries none. */ activity?: LaneActivity; /** * This record was read back from the archive after an MCP server restart: the job ended in a * previous process, and its report is exactly what that process wrote (`job-archive.ts`). */ restored?: boolean; /** The first line of the task, cut short (`taskLabel`), so a job can be recognised in a list. */ label?: string; /** * The read-only TOOL binding applied to the lane now running (`readonly-boundary.ts` * `readOnlyInvoke`). Replaced per lane, like `expected`; absent for an ordinary dispatch. */ readOnly?: { laneId: string; binding: string; }; /** * What the launcher changed before the lane now running started (`prepareLaneLaunch`): an * environment value expanded or removed, or the working directory given to AGY. Names only, never * a value. Replaced per lane, like `readOnly`; absent when the launcher changed nothing. */ launch?: string[]; /** * What the job changed in its git working tree, rendered (`tree-delta.ts`): set once, when the job * ends, for an agent-mode job whose cwd is in a git work tree. Report only. */ treeDelta?: string; /** * The walk's authoritative liveness snapshot for the lane now running. Replaced per lane and * removed when the job becomes terminal; terminal attempt records already say how the lane ended. */ liveness?: LaneLiveness; } /** * Output progress of the attempt now running. Byte counts and timestamps only — never the bytes. * * ⚠ Silence is REPORTED here, never acted on. `claude -p` — the transposed `cliLane` form every * `relay` rung takes in agent mode, i.e. the free pool itself — buffers its whole answer until * exit (module header), so "zero bytes after N seconds" is the ordinary shape of a healthy run on * the most-used lane. A threshold that killed on it would manufacture the false failure the backlog * item names as worse than a slow honest status. This figure is diagnostic only; the caller follows * the walk's published liveness verdict instead of interpreting silence. */ export interface LaneActivity { /** When the running attempt was spawned. */ attemptStartedAt: number; /** When the lane last wrote anything to either stream; null while it has written nothing. */ lastOutputAt: number | null; stdoutBytes: number; stderrBytes: number; } /** * The record of one job's OWNED processes, written at the moment the job goes terminal. * * ⚠ It exists because ownership is the only reliable signal. A stale lane and a slow lane are * indistinguishable from the outside (`docs/backlog.md`: one legitimately ran 29 minutes), so a * rule based on age would eventually kill a live lane — which is why the dispatcher terminates what * IT started, and why what it started is enumerable by job id rather than by hand. * * `survivors` is the honest half: a termination that did not take is REPORTED, never assumed away. * `terminated: false` means no process was ever registered for this job (a pre-spawn failure, or an * answer-mode job whose direct HTTP call owns no OS process), and is distinct from `pids: []` with * `terminated: true` — the first says "nothing of ours ran", the second says "ours ran and is gone". */ export interface LaneProcessReport { /** Root pids this dispatcher started for the job. */ pids: number[]; /** Pids still alive after termination was attempted. */ survivors: number[]; /** Whether a termination was attempted at all. */ terminated: boolean; } /** What a completed run looks like to a caller. */ export interface LaneRunResult { code: number | null; stdout: string; stderr: string; timedOut: boolean; } /** * A lane can exit 0 (agent mode) or answer HTTP 200 (answer mode) with text that is * syntactically nonempty but carries nothing usable — a lone `#`, a bare `---`, a block of * `***`. The pre-existing check caught only the LITERALLY empty string, which a scaffold-only * fragment slips past (C:\Code\docs\backlog.md: a 652-second review returned only `#`). * * Deliberately STRUCTURAL, not semantic: it strips whitespace, punctuation and Markdown * scaffolding characters and asks only whether anything ALPHANUMERIC remains — it does not judge * whether the content actually answers the task, which would cross this repo's own repair * boundary ("routing comes from config and deterministic classification, never from an LLM's * opinion inserted into the request path"). `isContentEmpty("Here is")` is `false`: a generic * lead-in with nothing after it is a real judgement call this predicate refuses to make. `"OK"` * and `"42"` must both read as content, and do. */ export declare function isContentEmpty(text: string): boolean; /** * The distinct failure reason both dispatch modes report for a content-empty result — a single * string constant so `describeJob`'s `error:` line and any caller-side matching agree on the * literal token, rather than each spelling it out separately. */ export declare const EMPTY_OUTPUT_REASON = "empty-output: the lane's output has no usable content after stripping formatting"; /** * Expand the `%NAME%` references a Windows environment value still holds (`docs/backlog.md`: a lane * inherited `HOME=%USERPROFILE%` literally, and a child that honours `HOME` wrote into a directory * named `%USERPROFILE%`). A reference expands from the SAME environment, looked up without case as * Windows does, in one pass. A value that is ONE unresolved reference is removed, because the literal * is never a usable value. An unresolved reference inside a longer value (a `PATH` entry) stays: removing * the whole value would lose the parts that are real. Windows only — `%` has no meaning to a POSIX * shell, and a POSIX value that holds one is data. * * Returns the notes `LaneJobStore.noteLaunch` records: variable and reference NAMES, never a value. */ export declare function expandEnvReferences(env: NodeJS.ProcessEnv, platform?: NodeJS.Platform): { env: NodeJS.ProcessEnv; notes: string[]; }; type LaneInvoke = { command: string; args: string[]; env?: Record; }; /** * Give an AGY lane the caller's working directory (`docs/backlog.md`: AGY works in its own scratch * directory whatever `cwd` its process gets, so a lane asked to edit a worktree saw none of its * files). AGY reads a directory only through `--add-dir`, so the directory goes on the command line, * and the task text names it. Null for every other lane. * * The prompt is the argument after AGY's own `-p`; a rung with no `-p` gets `--add-dir` alone. An * `--add-dir` the rung already declares for the same directory is not repeated. */ export declare function agyWorkingDirInvoke(invoke: T, cwd: string): { invoke: T; note: string; } | null; export interface DispatchedQuotaReport { laneId: string; tier: string | undefined; outcome: "rate_limited" | "quota_exhausted"; retryAfterMs?: number; } /** * Classify only positive quota evidence from a dispatched lane. Nonzero results use the same * fail-safe classifier as probes. Exit-zero AGY JSON is special-cased because AGY can encode an * error in its envelope; answer prose is never searched. */ export declare function classifyDispatchedResult(input: { result: LaneRunResult; laneId: string; tier: string | undefined; command: string; args: readonly string[]; }): DispatchedQuotaReport | undefined; export interface LaneSpawnOptions { env: NodeJS.ProcessEnv; cwd: string; timeoutMs: number; /** * Called for every chunk the child writes to either stream, with the chunk's SIZE — the bytes * themselves stay in the exec buffer that becomes `LaneRunResult`. Optional so every existing * caller and test double is unchanged; the store's silence figure is fed from it. */ onOutput?: ((chunk: { stream: "stdout" | "stderr"; bytes: number; }) => void) | undefined; } /** * The spawn seam. Injected so the suite can exercise every path without spending real lane quota — * the same discipline `lane-quota-probe.ts` applies, and the reason its default spawner refuses to * run under vitest. */ /** * What the store needs in order to REAP what a job owns and to report it: a way to stop it, and a * way to name what it started. * * ⚠ Narrower than a spawn handle on purpose. `startLane` wraps the spawner's promise in its own * outcome promise, so the value the walk registers is NOT a `LaneSpawnHandle` — typing this seam as * one would have forced the walk to register something other than what it actually holds. */ export interface OwnedProcess { /** * Terminate the process TREE this spawn started. Idempotent, and safe to call after the child has * already exited — the descendants are the whole reason it exists. */ kill: () => void; /** * The root pids this spawn started, read LAZILY. * * ⚠ A thunk, not a number: the Windows shell fallback REPLACES the root process (an npm `.cmd` * shim answers ENOENT and the real child is spawned through `exec`), so the pid is only knowable * after the fact. Optional, so a hand-written test double that owns no OS process can omit it. */ pids?: () => number[]; } export interface LaneSpawnHandle extends OwnedProcess { result: Promise; } export type LaneSpawner = (command: string, args: readonly string[], opts: LaneSpawnOptions) => LaneSpawnHandle; type LaneExecError = Error & { killed?: boolean | undefined; code?: unknown; }; /** * Terminate a full process tree (Windows: taskkill /T /F; POSIX: the process group created by * `createLaneSpawner`). The positive-pid fallback is defence for an injected/legacy child that was * not started as a group leader; the real POSIX spawner always owns group `pid`. */ export declare function terminateProcessTree(pid: number, platform?: NodeJS.Platform): void; /** A readable the spawner can watch for progress; the `data` event is all it subscribes to. */ export interface LaneOutputStream { on: (event: "data", listener: (chunk: string | Buffer) => void) => unknown; } /** The small child-process surface the lane spawner needs. */ export interface LaneChildProcess { pid?: number | undefined; stdin: { end: () => void; } | null | undefined; /** * Optional, so a test double that models no streams is unchanged. When present, the spawner * subscribes for `onOutput` ALONGSIDE `execFile`'s own buffering listener — a second `data` * listener never consumes what the first collects, so `LaneRunResult` is byte-identical. */ stdout?: LaneOutputStream | null | undefined; stderr?: LaneOutputStream | null | undefined; kill: () => boolean; } /** Spawn options for the POSIX process-group path. */ export interface LaneSpawnProcessOptions { cwd: string; env: NodeJS.ProcessEnv; windowsHide: true; detached: true; } /** Child events the POSIX buffered spawn path consumes. */ export interface LaneSpawnedProcess extends LaneChildProcess { on(event: "error", listener: (err: Error) => void): unknown; on(event: "close", listener: (code: number | null, signal: NodeJS.Signals | null) => void): unknown; } /** Spawn options common to the direct and Windows shell-fallback paths. */ export interface LaneExecOptions { encoding: "utf8"; maxBuffer: number; timeout: number; windowsHide: true; env: NodeJS.ProcessEnv; cwd: string; } export type LaneExecCallback = (err: LaneExecError | null, stdout: string, stderr: string) => void; /** Injectable process boundary: tests prove the real spawn contract without launching a lane. */ export interface LaneProcessApi { platform: NodeJS.Platform; execFile: (command: string, args: string[], opts: LaneExecOptions, callback: LaneExecCallback) => LaneChildProcess; exec: (command: string, opts: LaneExecOptions, callback: LaneExecCallback) => LaneChildProcess; /** POSIX-only real spawn seam: `detached` is supported by spawn, not execFile. */ spawn?: (command: string, args: string[], opts: LaneSpawnProcessOptions) => LaneSpawnedProcess; } /** Injectable npm-shim resolver so unit tests never inspect the host filesystem. */ export type WindowsNpmShimResolver = (command: string, args: readonly string[], deps: WindowsNpmShimDeps) => WindowsNpmShimResolution; /** * The real spawner. * * ⚠ Every guard here answers a measured failure; read the module header before removing one. * It never rejects — a failure becomes a result the caller can report, because a lane that died * is information, not an exception. * * ⚠ Refuses to spawn under vitest unless the suite injects its own seam. Same rule as * `winenv.ts`, `os-keyring.ts` and `lane-quota-probe.ts`: a test run must never spend real quota * or touch the operator's live agent sessions. */ export declare function createLaneSpawner(processApi: LaneProcessApi, hostEnv?: NodeJS.ProcessEnv, resolveNpmShim?: WindowsNpmShimResolver): LaneSpawner; export declare const defaultLaneSpawner: LaneSpawner; /** * The answer-mode HTTP seam — a direct POST to the running relay's own `/v1/messages`, * bypassing a spawned harness entirely for a `relay`-kind rung. Typed as `typeof fetch` (this * repo's established convention — `backend.ts`, `catalog.ts`, `key-checker.ts`, `reshaper.ts` all * inject the real global `fetch` this same way), so a test can hand it a fake built from the * real `Response` constructor exactly as those modules' tests already do. * * Injected for the same reason `LaneSpawner` is, and guarded the same way: this process is * launched by a host that may be running on a machine whose OWN `llm-relay` daemon is live on * its default port, so an accidentally-unmocked call here would not spend a lane's OWN quota (the * spawner's concern) but would reach a REAL locally-running relay and spend a REAL provider's. * `createAnswerFetch` copies the `LaneSpawner` guard pattern exactly: under vitest, refuse unless * the caller injects its own seam. */ export type AnswerFetch = typeof fetch; export declare function createAnswerFetch(hostEnv?: NodeJS.ProcessEnv): AnswerFetch; export declare const defaultAnswerFetch: AnswerFetch; /** * Where this relay itself is listening, for the answer-mode HTTP call. Deliberately a tiny local * copy of `cli.ts`'s `proxyUrl` rather than an import of it: `cli.ts` already imports * `McpDispatchServer` from this module's sibling, and importing back would create the first * import cycle between `cli.ts` and `mcp/`, for two lines neither side is likely to drift on. */ export declare function relayLoopbackUrl(config: { host: string; port: number; }, path: string): string; export declare function readRelayAnnouncements(headers: Headers): RelayAnnouncements; /** One line of `LaneJobStore.recent`. `elsewhere` marks a job another live process runs. */ export interface RecentJob { id: string; status: JobStatus; laneId: string; startedAt: number; endedAt: number | undefined; elsewhere: boolean; label?: string; } /** * Build one process-local job-id allocator. * * The old allocator read a shared max sequence and then incremented a process-local counter. Two * MCP processes could both read N before either wrote N+1, so both handed out the same handle. * * New ids keep the established `job-` wire shape but replace the shared sequence with a * 128-bit random PROCESS instance plus a local monotonic suffix. Independent processes need no * coordination. The leading 9 plus fixed-width 39-digit instance guarantees every new id is * outside JavaScript's safe-integer range, so `jobSeqOf` can continue recognizing only legacy * sequential ids in old archive files. */ export declare function createJobIdFactory(entropy?: Uint8Array): () => string; /** * The job store. * * IN MEMORY for what RUNS: this process is stdio-attached to one host session, so its children die * with it, and a durable record of a running job would describe a process that no longer exists. * Two things ARE durable, each in its own file: the set of jobs a restart killed (`job-journal.ts`, * a row per running job, cleared when it ends) and every job's FINAL record once it is terminal * (`job-archive.ts`, written eagerly at the terminal transition). `cancelAll` is wired to process * exit so nothing is orphaned. */ export declare class LaneJobStore { private readonly jobs; private readonly kills; private readonly owned; /** * Is this pid still alive? Injected so the suite can prove the survivor path without racing a * real termination, and so a platform with no `process.kill(pid, 0)` has one place to change. * * ⚠ `process.kill(pid, 0)` is an EXISTENCE probe, not a signal: it throws ESRCH when the pid is * gone and EPERM when it exists but belongs to another user — both of which mean the process the * dispatcher started is no longer a usable child. */ isAlive: (pid: number) => boolean; private readonly journal; private readonly archive; /** Starting trees carried off ordinary killed rows before the journal drops them on its next write. */ private readonly adoptedStartingTrees; /** Dead-MCP rows whose actual process owner is the daemon, atomically claimed for reconciliation. */ private readonly brokerOrphans; /** Running/recovered jobs whose process tree is owned by the daemon rather than this MCP process. */ private readonly brokerExecutions; private readonly nextJobId; constructor(journal?: JobJournal, archive?: JobArchive, nextJobId?: () => string); /** * The finished jobs a previous process archived, read back so `dispatch_status`/`dispatch_result` * answer for them after a restart instead of `unknown jobId`. Marked `restored` so the rendering * can say the report predates this process. */ private restoreArchive; /** Run the archive's pending write now — the shutdown seam, called from `McpDispatchServer.shutdown`. */ flush(): void; /** * The jobs a PREVIOUS process died holding. They are adopted as terminal `"killed"` rows rather * than dropped, because the measured cost of dropping them was ninety lane-minutes lost behind a * bare `unknown jobId: job-0051` on a routine poll (2026-09-06). * * ⚠ No process is terminated here and none could be: this process holds no handle for those pids, * and a pid from a previous process may already belong to something else. `process.terminated` is * therefore false, and `ownedProcesses()` reports the job as un-reaped rather than claiming a * termination that never happened. */ private adoptOrphans; create(laneId: string, spec: string | undefined, cwd: string, dispatchSource?: "daemon" | "fallback", label?: string): LaneJob; /** * Record the process-handle this dispatcher started for `id`, so the job can be REAPED when it * reaches a terminal state and so what it owned is reportable afterwards. */ registerProcess(id: string, handle: OwnedProcess): void; /** Root pids currently owned by a still-running job, through the same handle the reaper uses. */ runningPids(id: string): number[]; /** * Register a bare kill callback — the pre-existing seam, kept because a caller that owns no OS * process (an answer-mode job's direct HTTP call, or a test double) still has something to stop. */ registerKill(id: string, kill: () => void): void; /** * Terminate everything this dispatcher started for `id`, and record what happened. Called from * EVERY terminal transition — `complete`, `fail` and `cancel` alike — because "the job ended" and * "the job's processes ended" were two different facts on HEAD, and only the second one is true. * * ⚠ Never throws. A termination failure is data (`survivors`), not an exception: the job is * already terminal, and throwing here would replace a reported outcome with a crashed handler. */ private reap; /** * Every process this dispatcher started for a TERMINAL job, keyed by job id — so a stale lane can * be found without enumerating the machine's processes by hand, which is the only way the * measured case was ever found (four `opencode` processes, ~530 MB, burning CPU hours after their * jobs had returned, appearing in no `dispatch_status` listing). * * A still-running job is deliberately absent: it owns its processes on purpose. */ ownedProcesses(): Array<{ jobId: string; laneId: string; spec: string | undefined; } & LaneProcessReport>; /** * Point a still-RUNNING walk at the lane it has just moved to, so a `dispatch_status` poll * names the lane currently doing the work rather than the one already abandoned. * * ⚠ The `running` guard is what keeps a cancelled walk cancelled. `cancel()` sets the status * synchronously and the walk loop checks it, but the two are not atomic with respect to each * other; without the guard a walk that advanced in the same tick would repoint a job the * operator had already stopped. */ setCurrentLane(id: string, laneId: string, spec: string | undefined, expected?: LaneJob["expected"]): void; /** * A spawned attempt has started for `id`: from now until it settles, `noteOutput` counts what it * writes and a poll can say how long it has been silent. Not called for an answer-mode HTTP call, * which has no output stream — its absence is what keeps the figure honest there. */ beginAttemptActivity(id: string, now: number): void; /** The running attempt wrote `bytes` to `stream` at `now`. Ignored once the attempt is over. */ noteOutput(id: string, stream: "stdout" | "stderr", bytes: number, now: number): void; /** Persist the job-wide starting tree while the job is still running. */ noteStartingTree(id: string, tree: TreeSnapshot, scope: readonly string[] | undefined): void; /** * Return, once, daemon-backed orphan rows this process atomically claimed. The server owns the * broker client, so reconciliation is asynchronous and deliberately outside this synchronous store. */ takeBrokerOrphans(): JournalRow[]; /** Broker execution id for a job this process is collecting, if any. */ brokerExecution(id: string): string | undefined; /** Record which execution boundary owns the current attempt. */ noteExecutionOwner(id: string, owner: LaneJob["executionOwner"]): void; /** * Clear an attempt-scoped daemon execution after it is definitively terminal and the walk will * continue. The job itself stays running; the journal row is rewritten without broker metadata. */ clearBrokerExecution(id: string): void; /** * Bind a live job to the daemon execution that now owns its process tree, and persist the opaque * reference before the broker start request is sent. Used by the later ownership switchover. */ noteBrokerExecution(id: string, executionId: string): void; /** * Materialize a claimed broker row as a running job while the daemon is queried/retried. * The daemon, not this process, owns its process tree; no local kill handle is registered. */ adoptBrokerRunning(row: JournalRow, snapshot?: LaneExecutionSnapshot): LaneJob | undefined; /** * Apply one identity-checked daemon snapshot to an adopted broker job. * Returns false on mismatch/inconsistent terminal data; callers then retain the row and retry * rather than turn malformed transport data into a death claim. */ applyBrokerSnapshot(id: string, snapshot: LaneExecutionSnapshot): boolean; /** Definitive reachable-daemon 404: the broker no longer knows the execution. */ markBrokerKilled(id: string, message: string): boolean; /** * Return, once, the starting trees carried by jobs adopted as killed after a restart. The server * owns the git reader, so the store cannot render their deltas synchronously during construction. */ takeAdoptedStartingTrees(): Array<{ jobId: string; cwd: string; startingTree: JournalStartingTree; }>; /** Record the read-only tool binding the lane now running was given, so the reply can state it. */ noteReadOnly(id: string, laneId: string, binding: string): void; /** Record what the launcher changed for the lane now running, so the reply can state it. */ /** * Record the job's tree delta. Accepted on a TERMINAL job too — a cancellation ends the job before * the second `git status` returns — and the archive row is then written again so it carries it. */ noteTreeDelta(id: string, text: string): void; /** Publish the walk's liveness decision for the attempt now running. Pure status data only. */ noteLiveness(id: string, liveness: LaneLiveness): void; noteLaunch(id: string, notes: readonly string[]): void; /** * Append one tried lane to the walk's record. Appended for EVERY attempt, the winner included, * so the answer can say what it cost to get there — and so an abandoned lane leaves a trace, * which is the case `docs/backlog.md` says matters most ("an operator who gives up on a slow * lane leaves no trace"). */ recordAttempt(id: string, attempt: LaneAttempt): void; /** * Record what this dispatch was allowed to reach: whether it ran as a WALK, and how many * selectable lanes its own `maxLanes` bound kept it from trying. * * ⚠ No silent caps, and no false claim of exhaustion. Both halves feed the same decision — a * dispatch may only tell the caller "every lane has been tried" when it actually walked and * nothing was left over. A walk that stopped at four of nine lanes, or a dispatch with the walk * turned off entirely, saying that would be false on the one surface the caller acts on. Zero is * not stored, so an uncapped walk renders exactly as it did before this field existed. */ noteWalkScope(id: string, scope: { enabled: boolean; lanesNotTried: number; forced?: boolean; }): void; /** * A rung the walk SKIPPED for its `maxConcurrent` cap "counts as NOT TRIED" (the backlog item's * own wording), so it grows the same counter `noteWalkScope`'s `maxLanes` bound uses — one caller * before the walk starts (a fixed bound known in advance), this one during it (a skip discovered * lane by lane) — so `jobAnswer` cannot tell them apart and always prefers the PARTIAL advice over * the EXHAUSTED one whenever anything was left untried, whichever reason left it untried. */ noteSkippedLane(id: string): void; /** * How many jobs THIS MCP server process currently has a spawned process running for `laneId` — * the job's CURRENT attempt (`job.laneId`, repointed by `setCurrentLane` as the walk advances * from lane to lane), never its history. A job counts only while `status === "running"`: the * moment an attempt settles (or the whole job ends), it stops occupying a slot. * * ⚠ `excludeJobId` matters, and omitting it is a real bug, not a cosmetic nicety: `create()` sets * a fresh job's `laneId` to its FIRST candidate lane at CREATION, before the walk has attempted * anything — so a walk asking "is my own first lane already at its cap?" would count ITSELF and * skip its own opening attempt, on every dispatch, the moment any `maxConcurrent` is configured. * Passing the asking job's own id excludes it, so this answers "how many OTHER jobs are running * this lane right now" — the question a skip check actually needs. * * ⚠ Per MCP SERVER PROCESS by design. D1 moves the physical child into the daemon, but this * admission policy still counts the jobs THIS MCP store owns or recovered. Another simultaneously * live MCP process is not folded into the number, so two host sessions can together exceed a * rung's `maxConcurrent`. This preserves the documented per-host cap rather than silently turning * it into a machine-wide semaphore. */ inFlight(laneId: string, excludeJobId?: string): number; /** A job THIS store holds. The walk reads this; it never needs another process's job. */ get(id: string): LaneJob | undefined; /** * A job for a CALLER: one this store holds, else one another `llm-relay mcp` process finished and * archived after this store started (kept from then on, marked `restored`), else undefined. * * ⚠ Why a caller needs more than `get`. Claude Desktop and Codex Desktop each start their own * server, and a session can poll through a different connection than the one that dispatched — * observed 2026-09-16 as `unknown jobId` for a job that had finished. The archive is shared on * disk, so the answer exists; it was read only once, at start. */ find(id: string): LaneJob | undefined; /** * The newest `limit` jobs this machine knows: this store's, every finished job on disk, and every * job another live process is running. Newest start first. */ recent(limit: number): RecentJob[]; /** * The journal row for a job another LIVE `llm-relay mcp` process is running now, or undefined. * Nothing about it can be read here except that it runs; its answer reaches `find` once it ends. */ runningElsewhere(id: string): { laneId: string; spec?: string; startedAt: number; pid: number; } | undefined; list(): LaneJob[]; /** * `relay` carries an answer-mode job's captured provenance headers — absent for every * agent-mode job, since those never touch HTTP directly. * * ⚠ `run.timedOut` is checked BEFORE the exit-code/semantic-failure branch and wins outright: * a killed-by-timeout run gets `status: "timed_out"`, never `"failed"`, regardless of what * `semanticFailure` a caller also passed (agent-mode quota classification already refuses to * run on a timed-out result, so in practice this only ever arbitrates against the empty-output * check — and a timeout is the more informative, more specific claim of the two). */ complete(id: string, run: LaneRunResult, semanticFailure?: string, relay?: RelayAnnouncements): void; fail(id: string, error: string): void; cancel(id: string): boolean; cancelAll(): void; } export interface CwdCheck { ok: boolean; reason?: string; } /** * Validate a caller-supplied working directory. * * The design question the prior-art survey raised: an executing tool needs a directory, and a * caller-supplied filesystem path is a strictly larger version of the hazard `dispatch.ts` already * refuses (request content becoming process configuration). * * The answer taken here, and why: the directory must EXIST and be a directory, and when the * operator declares `allowedRoots` it must sit under one of them. With no `allowedRoots` declared * the check is existence only — the caller is already a trusted agent on the operator's own * machine, and refusing by default would make the tool useless for its stated purpose. The bound * is offered, not imposed; that is the operator's call to make in config, not this file's. * * ⚠ **The containment test resolves BOTH sides with `path.resolve` before comparing** (closed * 2026-09-03, docs/history/audit-findings-2026-09-03.md finding 1 / DR-002). Without it a literal `..` * segment in `cwd` — a raw string a caller sends verbatim, never normalized — satisfied a bare * `startsWith` prefix test while `existsSync`/`statSync` above had already resolved `..` at the * OS level against the REAL, escaped directory: `allowedRoots: ["C:/allowed"]` admitted * `C:/allowed/../other`. `path.resolve` collapses `..`/`.` the same way the OS already does for * the existence check, so the two agree; it is a no-op on an already-clean absolute path, so the * common case is unaffected. `normalizePath`'s trailing-separator LOOP is untouched — resolving * first does not remove the need for it (`path.resolve` does not fold case on win32). */ export declare function checkCwd(cwd: string, allowedRoots: readonly string[] | undefined): CwdCheck; /** * Current dispatch depth, read from the environment this process was launched with. * Absent or unparseable ⇒ 0, so a hand-launched server behaves as the top of the chain. */ export declare function currentDepth(env?: NodeJS.ProcessEnv): number; export {};