/** * daemon/funnel-watchdog.ts — notice when public ingress dies, and only then * reconnect the node. * * Tailscale Funnel can report itself healthy while refusing every connection * from the internet. `tailscale funnel status` prints "Funnel on", the serve * config is intact, the client version is unchanged, the node is Online — and * connections through Tailscale's ingress relays are reset. Inbound webhooks * stop arriving and nothing anywhere says so; the outage is discovered when a * human notices a message never landed, hours later. * * The state clears when the node reconnects. So this watchdog is a probe and a * single lever, with the probe doing nearly all of the work. * * WHY THE PROBE HAS TO GO OUT TO THE INTERNET. The tailnet resolver answers the * funnel hostname with the node's own 100.x address, so any request made on * this machine against that name travels over the tailnet and is answered * happily by the same daemon that would answer a real one. That is a false * green, and it is what hid this failure before: local curl said 405 while the * public path was dead. We therefore resolve the hostname through a PUBLIC * resolver, connect to the ingress address it returns, and set SNI by hand. * Reaching the node from outside is the only fact that means anything. * * WHY IT IS RELUCTANT TO ACT. A reconnect drops every live tailnet connection * for a moment — SSH, file transfers, other services. So a bounce needs proof, * not a hunch: * * - three consecutive failed probes, not one, so a blip is never enough; * - an "unknown" verdict (no public DNS, no route off the machine) resets the * counter instead of incrementing it — when the whole network is down the * node is not the problem and bouncing it fixes nothing; * - a cooldown after each attempt, so a fault this lever cannot fix degrades * into a loud log line rather than a reconnect loop. * * The healthy path costs one TLS handshake every few minutes and touches * nothing. */ export type Verdict = "up" | "down" | "unknown"; export interface WatchdogState { /** Failed probes in a row. Reset by anything that is not a clear failure. */ consecutiveDown: number; /** When the lever was last pulled, so it cannot be pulled again at once. */ lastHealAt?: number; /** Whether the current outage has already been announced, so we say it once. */ announced: boolean; } export declare function initialState(): WatchdogState; export interface ProbeResult { ip: string; /** An HTTP status — any status — means the ingress reached the node. */ status?: number; /** ECONNRESET, ETIMEDOUT, … when it did not. */ error?: string; } /** * One reachable relay is enough. * * Tailscale publishes several ingress addresses and they do not fail together: * during recovery two answered while the third still reset. Requiring all of * them would call a working funnel broken and bounce a healthy node. */ export declare function classify(results: ProbeResult[]): Verdict; export interface Decision { action: "sleep" | "heal"; sleepMs: number; reason: string; } export interface DecideOptions { failuresBeforeHeal?: number; healCooldownMs?: number; healthyIntervalMs?: number; suspectIntervalMs?: number; } /** * What to do about a verdict. Pure, because this is the part that must be * right: everything it can get wrong is either an outage nobody notices or a * reconnect nobody asked for. * * Mutates `state`, returns the decision. */ export declare function decide(state: WatchdogState, verdict: Verdict, now: number, opts?: DecideOptions): Decision; /** The `tailscale` binary, or undefined when it is not installed here. */ export declare function tailscaleBinary(): string | undefined; export interface CliResult { ok: boolean; stdout?: string; error?: string; } /** * Run the client, and report failure rather than swallowing it. * * The macOS app's CLI is not always usable from a background process: it wants * to talk to the GUI app and answers "The Tailscale GUI failed to start" when * it cannot. The first version of this file treated that error as "no funnel * configured", so the watchdog returned quietly and looked exactly like a * watchdog that was working — the precise failure it exists to prevent. */ export declare function runTailscale(bin: string, args: string[]): CliResult; /** * The node's funnel hostname, read from the client rather than configured. * * Hard-coding it would put one machine's name in the source and go stale the * first time the node is renamed. */ export declare function funnelHostname(bin?: string | undefined): CliResult & { hostname?: string; }; /** Is a funnel even configured? Nothing to watch when it is not. */ export declare function funnelConfigured(bin?: string | undefined): CliResult & { configured?: boolean; }; /** Public ingress addresses for the funnel hostname, via a public resolver. */ export declare function resolveIngress(hostname: string): Promise; /** One request to one ingress address, with SNI set to the funnel hostname. */ export declare function probeOne(ip: string, hostname: string, path?: string): Promise; /** Probe every published ingress address. */ export declare function probeIngress(hostname: string): Promise; /** * Did the reconnect take? Ask more than once, and give it time between asks. * * Exported because the retry policy is the whole point of it: a single probe * immediately after `up` measures the coordination server's propagation delay * rather than the health of the node. */ export declare function confirmAfterHeal(probe: (hostname: string) => Promise, hostname: string, attempts?: number, settleMs?: number, sleep?: (ms: number) => Promise): Promise; /** Reconnect the node. The one lever, pulled only on proof. */ export declare function reconnect(bin?: string | undefined): boolean; /** * Watch public ingress for as long as the daemon runs. * * Returns a stop function. Safe to start when Tailscale is absent: it says so * once and does nothing further. */ export declare function startFunnelWatchdog(opts?: { probe?: (hostname: string) => Promise; heal?: () => boolean; hostname?: string; }): () => void; //# sourceMappingURL=funnel-watchdog.d.ts.map