/** * Thin HTTP client for the supervisor's process broker (plan §4.1, Phase 2). * * Liveness, identity and kill move OFF the Node event loop and OUT of child * processes: the supervisor answers from one native NtQuerySystemInformation * snapshot and kills via native TerminateProcess/TerminateJobObject. Nothing * here ever spawns a process — the EDR-gated tasklist/powershell/taskkill * machinery this replaces was itself the leading event-loop stall generator. * * Degradation contract (the audited-safe direction, same as the old breakers): * every function returns `null` when the broker cannot answer (unreachable, * auth-rejected, timeout) and the caller MUST treat null as "cannot confirm * → assume alive / kill not performed". Communication is strictly one-way * MCP → supervisor. */ /** * How a failed broker call failed — the distinction the spawn contract pivots on: * "no-connection" the TCP connection was never established (refused, host * unreachable, connect timeout). The request CANNOT have * reached the broker, so no process can have been created — * a legacy local spawn is safe. * "uncertain" the request may have been received and processed (read * timeout, connection dropped mid-flight, unknown error). * A process MAY exist broker-side — a blind local spawn * could create a duplicate and is forbidden. */ export type BrokerFetchFailure = "no-connection" | "uncertain"; export interface BrokerAliveResult { alive: boolean; pidFound: boolean; identityChecked: boolean; identityMismatch: boolean; startTimeMs?: number; /** Exit code from the broker's ledger once the process is gone (spawn-side * handle-wait recorded it). Undefined while alive / never managed. */ exitCode?: number | null; /** Ledger state of the pid's latest row ("running"/"exited"/"killed"/…). */ registryState?: string | null; } /** * Ask the broker whether `pid` is alive AND (when `startedAtMs` is given) still * the same process we spawned (creation-time identity, ±60s tolerance — the * same rule the deleted tasklist/Get-Process snapshots enforced). * `startTimeTicks` (native creation FILETIME from a broker spawn) upgrades the * check to exact identity. Returns null when the broker cannot answer. */ export declare function brokerCheckAlive(pid: number, startedAtMs?: number, startTimeTicks?: number): Promise; export interface BrokerThreadAliveResult { /** False when the broker has never managed a process for this thread (HTTP 404). */ known: boolean; /** Alive iff the thread's job has >=1 live member (kernel truth), or a live * assign-failed process is still registered for it. */ alive: boolean; /** How the verdict was reached ("job" | "registry"). */ source?: string; /** Recorded exit code once the worker is truly gone (job drained). */ exitCode?: number | null; /** Ledger state of the thread's latest row ("running"/"exited"/"killed"/…). */ registryState?: string | null; } /** * Job-based thread liveness — the single source of truth for a broker-spawned * worker. A launcher CLI (claude.exe/copilot.exe forks the real agent then exits) * makes the originally-spawned pid a lie; the thread's JOB is not. Returns * `known:false` when the broker has no record of the thread, and null when the * broker cannot answer (unreachable → "cannot confirm → treat alive"). */ export declare function brokerCheckAliveByThread(threadId: number): Promise; export interface BrokerKillResult { killed: boolean; alive: boolean; alreadyDead: boolean; identityMismatch: boolean; refused: boolean; error?: string; } /** * Ask the broker to kill the process tree rooted at `pid` (native, identity- * checked when `startedAtMs` is given; the broker records the outcome in its * on-disk registry). Returns null when the broker cannot answer — the caller * must then fall back / re-verify, never assume the kill happened. * * Pass `pid` as undefined to route the kill by THREAD instead: the broker then * takes the KillByThread path (job kill of every live member, or the registry- * identity survivor when the job assign failed) rather than KillByPid. This is * the correct call for a launcher-forked worker whose recorded pid is the * already-exited launcher — KillByPid(launcher) hits the broker's AlreadyDead * fast-path and never touches the live forked child. */ export declare function brokerKillTree(pid: number | undefined, opts?: { threadId?: number; killedBy?: string; reason?: string; startedAtMs?: number; }): Promise; export interface BrokerSpawnSpec { threadId: number; role: string; exe: string; args: string[]; env: Record; cwd?: string; logFilePath: string; /** Prompt-sized stdin payload for CLIs that read their prompt from stdin * (Codex). Omitted → the child gets NUL stdin (Node's "ignore"). */ stdinData?: string; } export interface BrokerSpawnSuccess { pid: number; /** Native creation FILETIME — exact process identity for later checks/kills. */ startTimeTicks: number; startTimeMs: number; jobName: string | null; /** Launched but job containment failed (Avecto downgrade §4.2) — managed via * registry identity only. Already logged loudly broker-side. */ jobAssignFailed: boolean; /** Idempotent hit: the thread ALREADY had a live managed process and pid * describes that process — nothing new was spawned. */ existing: boolean; } /** * The four spawn outcomes the caller MUST keep apart: * success → a process is running and registered (existing=true when it * was already there — the idempotency contract); * { error } → the broker answered and the spawn FAILED. No process was * created (the broker's idempotency check also found none); * { unavailable } → the request DEFINITIVELY never spawned anything. Two * sub-cases, told apart by `endpointAbsent`: * endpointAbsent:true the broker answered 404 ONLY — the * /proc/spawn endpoint is genuinely ABSENT (an OLD * supervisor binary that predates it). The caller may * grant a one-release local-spawn grace; * endpointAbsent:false the endpoint is PRESENT but the * spawn did not happen: either the supervisor is DOWN * (no connection established: refused / host * unreachable / connect timeout) or a LIVE supervisor * REJECTED the request (401 — bad bearer secret). A * local spawn would reintroduce the uncontained-zombie * hole, so the caller must fail loudly instead — NO * local spawn. * Either way no process was created, so nothing can be * duplicated; * { uncertain } → timeout or connection lost mid-flight: the broker MAY * have created a process. A blind local spawn could * duplicate the agent — the caller must reconcile via * brokerListThread / an idempotent spawn retry instead. */ export type BrokerSpawnResult = BrokerSpawnSuccess | { error: string; } | { unavailable: string; endpointAbsent: boolean; } | { uncertain: string; }; /** * Ask the supervisor to spawn an agent process: CreateProcess(CREATE_SUSPENDED) * → assign to the thread's named job → resume. The supervisor is the process * owner; children (hooks, grandchildren) land in the job automatically. * Spawn is idempotent by threadId server-side (create-or-return), which is what * makes retrying after an uncertain outcome safe. */ export declare function brokerSpawn(spec: BrokerSpawnSpec): Promise; export interface BrokerProcEntry { pid: number; startTimeTicks: number; startTimeMs: number; name: string; } /** * The broker's live process list for a thread (job members, falling back to the * registry root's tree). This is the reconcile source of truth after an * uncertain spawn: a live entry here means the thread HAS its process — adopt * it, never spawn another. `known:false` → the broker answered and has no * record of the thread. Null → broker could not answer. */ export declare function brokerListThread(threadId: number): Promise<{ known: boolean; processes: BrokerProcEntry[]; } | null>; /** * Ask the broker to adopt an already-running process (spawned before the broker * owned spawning, or during a broker outage) into the thread's job + registry. * Idempotent server-side. Best-effort: null when the broker cannot answer. */ export declare function brokerAdopt(threadId: number, pid: number, startedAtMs?: number, role?: string): Promise<{ adopted: boolean; alreadyManaged: boolean; } | null>; export type BrokerRestartVerdict = "accepted" | "in-progress"; /** * Ask the supervisor — the single process-lifecycle owner — to tear down and * respawn THIS server (self-update path). The supervisor answers 202 before * doing anything: the requester is the process about to be killed, so it can * never observe completion. 409 means a restart is already in progress, which * delivers exactly what the caller wants — treat it as success. * * Returns null when the supervisor cannot answer (unreachable, or an old * binary without /restart → 404/401). The caller decides between the flagged * legacy self-spawn fallback and a loud abort — same contract as brokerSpawn. */ export declare function brokerRequestRestart(reason: string): Promise; export interface BrokerKeeperState { threadId: number; retryCount: number; fastExitCount: number; fastExitEscalation: number; cooldownUntilMs: number | null; survivedKillCount: number; lastStartAtMs: number | null; } /** Load a thread's persisted keeper counters. Null → broker unreachable OR no * state saved yet (both mean "start from defaults"). */ export declare function brokerGetKeeperState(threadId: number): Promise; /** Persist a thread's keeper counters (fire-and-forget from the keeper; a lost * write degrades to today's in-memory behavior, already logged at WARN by the * unreachable path). */ export declare function brokerSaveKeeperState(state: BrokerKeeperState): Promise; /** Test hook: reset the unreachable-cooldown state. */ export declare function resetBrokerClientState(): void; export declare function brokerSpawnTimeoutForTest(ms: number | null): void; //# sourceMappingURL=broker-client.d.ts.map