/** * Supervisor positive-liveness probe. * * # Why this exists * * The supervisor loops in `cli.ts` (`runDaemonSupervisor`, `runForegroundSupervisor`) * watch for the worker child process to **exit** and respawn on a configured set * of exit codes. That works for clean crashes, segfaults, and graceful * shutdowns. It does **NOT** catch the beacon-01-class zombie: a worker * whose event loop is wedged after a partial shutdown attempt, with a dead * HTTP listener and no pending exit. The process is alive enough that * `child.exit` never fires; the supervisor waits forever. * * PR-1 (#655) catches the specific "deadlocked-shutdown" shape via a hard * timeout in the shutdown handler. This module catches the **generic** zombie * shape: anything that leaves the HTTP listener dead while the process * remains alive. Defense-in-depth. * * # HTTP response liveness * * The kernel can complete a TCP handshake while the worker's JavaScript * event loop is stuck. Require an HTTP response to a cheap, unknown API * HEAD route. Any status (including 401/404/503) proves request handling * progressed, without credentials or a database/chain health dependency. * A single absolute deadline bounds connecting and receiving the status * line, and the socket is destroyed on every outcome. * * # Env gate * * `DKG_SUPERVISOR_LIVENESS_PROBE` controls behaviour: * - unset / `"on"` / `"1"` / `"true"`: probe is active. * - `"off"` / `"0"` / `"false"`: probe disabled. * * Why an env gate: tests + the worker-without-HTTP scenario (e.g. * benchmark scripts that spawn a worker headlessly without the API * server) shouldn't get SIGKILLed because they have no listener to * probe. Defaults to on so production benefits; tests opt out. */ /** Default total HTTP liveness deadline — 5s. */ export declare const LIVENESS_PROBE_TIMEOUT_MS = 5000; export declare const DEFAULT_LIVENESS_SHUTDOWN_GRACE_MS = 30000; /** Default tick — 30s. Picked to be longer than typical request handling but short enough that a 5-failure quorum triggers within ~2.5 min. */ export declare const LIVENESS_PROBE_INTERVAL_MS = 30000; /** Default trigger threshold — 5 consecutive failures. With 30s tick → ~2.5 min unresponsive before SIGKILL. */ export declare const LIVENESS_CONSECUTIVE_FAILURES_TO_KILL = 5; /** Derive watchdog grace from the already-validated worker hard timeout. */ export declare function resolveLivenessShutdownGraceMs(hardTimeoutMs: number): number; /** * One-shot probe: require an HTTP status line from `host:port`; a TCP * connection alone is insufficient. Node owns HTTP framing and status * parsing; this policy accepts any parsed response. Returns false on any * error or timeout. * * The socket is force-destroyed on every outcome (success or failure) to * avoid leaking file descriptors when the supervisor probes the worker * thousands of times over a long uptime. */ export declare function probeWorkerAlive(port: number, host?: string, timeoutMs?: number): Promise; /** * Parse the env-gate value. Exported so tests can verify the truth table * without re-implementing the parser. * * - unset / `"on"` / `"1"` / `"true"` (case-insensitive) → `true` (enabled). * - `"off"` / `"0"` / `"false"` (case-insensitive) → `false` (disabled). * - anything else → `true` (fail-safe: unknown value = enable rather than * silently disable an operator's protection). */ export declare function isLivenessProbeEnabled(envValue: string | undefined): boolean; export interface LivenessWatcherOpts { /** Port to probe (the worker's API listener). */ port: number; /** Host to probe. Defaults to IPv4 loopback for the common apiHost case. */ host?: string; /** Called whenever the threshold trips. Wired by the supervisor to `child.kill('SIGKILL')`. */ onUnresponsive: () => void; /** * Called once per failure increment (not per healthy probe) for log/telemetry. * Receives the running consecutive-failure count after the increment. */ onFailure?: (consecutive: number) => void; intervalMs?: number; timeoutMs?: number; consecutiveFailuresToKill?: number; /** Injectable probe — tests pass a stub. Defaults to `probeWorkerAlive`. */ probe?: (port: number, host?: string, timeoutMs?: number) => Promise; /** * Optional graceful-shutdown detector. Called on every failed probe BEFORE * the consecutive-failure counter is incremented. If it returns truthy, the * watcher enters a bounded shutdown-grace window — failed probes during * this window do not count toward the SIGKILL threshold so we don't * SIGKILL the worker mid-teardown (e.g. `agent.stop()` / DB close after * `server.close()`). * * The supervisor wires this to `existsSync(apiPortFile) === false`: the * worker's `shutdown()` removes `api.port` BEFORE the slow awaits, so its * absence is the "graceful shutdown initiated" signal the watcher needs. * * Codex (#664#discussion_r3302432762): the previous implementation * PERMANENTLY disarmed the watcher on the first shutdown observation. * If a later teardown step hung (DB close, network shutdown, …), the * supervisor could no longer SIGKILL or respawn the worker. The fix: * keep probing during shutdown, but only fire SIGKILL after a bounded * grace window (`shutdownGraceMs`) so the worker's own * SHUTDOWN_HARD_TIMEOUT_MS gets first crack at force-exiting itself. */ isShuttingDown?: () => boolean | Promise; /** * Maximum time to wait after the worker enters graceful shutdown before * the watcher resumes counting failures toward `consecutiveFailuresToKill`. * Default remains 30s. The CLI supervisor supplies a value derived from the * worker's resolved hard timeout so an operator override cannot be * preempted. Set to a negative value to disable the bounded fallback * (legacy "disarm forever" behavior). */ shutdownGraceMs?: number; } /** * Start a periodic liveness watcher. Returns a `stop()` handle that the * supervisor's `child.exit` handler must call to cancel the interval — * otherwise the watcher keeps probing into the void after the worker * exits cleanly, eventually generating spurious SIGKILL attempts. * * The watcher self-throttles: if a probe is already in flight when the * tick fires, the second tick is dropped (we don't pile up probes when * the worker is slow but not dead). */ export declare function startLivenessWatcher(opts: LivenessWatcherOpts): { stop(): void; }; //# sourceMappingURL=supervisor-liveness.d.ts.map