/** * retryPolicy — when a wspace-files-svc call through the workspace gateway is * worth trying again, and how long to wait before doing so. * * ## Why this exists: a scale-to-zero cold start, measured in production * * `wspace-files-svc` runs at `minReplicas: 0`, 0.25 vCPU / 0.5 GiB, so after an * idle period its revision sits at `replicas: 0, state: "ScaledToZero"`. Timed on * the deployed image at that exact resource envelope: * * bunx prisma migrate deploy 26.0s * app boot ~25.5s * ------------------------------------ * >= 51s before the port binds * * The gateway's per-operation budget is `OPERATION_TIMEOUT_MS = 15_000`. The first * request after a quiet period therefore CANNOT succeed: it wakes the container and * is then abandoned ~36s before that container can answer. * * Seven-attempt production reproduction, same 1,305-byte file, idle gaps * 0/5/40/5/40/90/5s: * * #1 16.2s FAIL "Operation timeout" {"code":"OPERATION_TIMEOUT","timeoutMs":15000} * #2 16.2s OK #3 6.1s OK #4 6.1s OK #5 6.1s OK #6 2.1s OK #7 2.1s OK * * Only the first attempt fails; everything after it is warm and gets faster. With a * 300s Azure Container Apps cooldown, a real user's first attachment after any quiet * period lands on exactly this. The proper fix is `minReplicas: 1` — that needs Azure * access, so this module is the client-side half that makes the cold start survivable * meanwhile. Several services in this fleet are scale-to-zero, so it is not a * one-off workaround. * * ## Two retry classes, deliberately different * * - `cold-start` — the gateway gave up waiting (`OPERATION_TIMEOUT` and friends). * The container is probably still booting, so the delays are sized * in TENS of seconds. A single immediate retry is useless here. * - `transient-blip` — a masked `INTERNAL_SERVER_ERROR` or a bare network error, the * pre-existing `completeUpload` case (a brief Azure Managed Redis * slowdown). Sub-second delays, unchanged from before this module. * * ## What is NEVER retried * * Classification is an ALLOWLIST: an unrecognised code is not retried. On top of * that, `NEVER_RETRY_CODES` is a hard stop — if a quota rejection, an auth failure * or any other 4xx-class business error appears anywhere in the error, the whole * error is unretryable even if a timeout also appears. Retrying those is pure harm: * the answer will not change and the user waits a minute to be told the same thing. */ /** How an error should be retried, or `null` for "do not retry this". */ export type FilesRetryKind = 'cold-start' | 'transient-blip'; /** * Attempt ceiling for a cold start. Three attempts, and the third one starts at * roughly t+61s: * * attempt 1 t=0 -> gateway gives up at ~16s * wait 10s * attempt 2 t=26 -> gateway gives up at ~41s * wait 20s * attempt 3 t=61 -> container bound its port at ~51s, so this is the one that lands * * The container's boot clock starts when attempt 1 reaches the ACA ingress, which is * why the ladder is measured from the FIRST attempt and not from each failure. The * numbers above come from the measurement at the top of this file: >=51s to port * bind, 15s gateway budget. Change them if that measurement changes, not before. */ export declare const COLD_START_MAX_ATTEMPTS = 3; /** Delay before cold-start attempt 2, then before attempt 3. See the ladder above. */ export declare const COLD_START_RETRY_DELAYS_MS: readonly number[]; /** * Hard wall-clock ceiling for one operation's cold-start retries, measured from the * first attempt. The ladder above tops out at ~76s of real time; this stops a * pathologically slow gateway from stretching that indefinitely. Past this, we give * up honestly and show the user the real error. */ export declare const COLD_START_RETRY_BUDGET_MS = 75000; /** Attempt ceiling for a transient blip — unchanged from the pre-existing behaviour. */ export declare const BLIP_MAX_ATTEMPTS = 3; /** Delays before blip attempts 2 and 3 — unchanged from the pre-existing behaviour. */ export declare const BLIP_RETRY_DELAYS_MS: readonly number[]; /** * Best-effort message extraction, preferring the first GraphQL error's message * (Apollo v4 `CombinedGraphQLErrors.errors`, or v3 `.graphQLErrors`) so callers can * surface the real reason instead of a generic envelope. * * Lives here rather than in `uploadBlob` because both the upload path and the folder * path need it and both now sit on top of this module. `uploadBlob` re-exports it, so * its public name is unchanged. */ export declare function extractErrorMessage(err: unknown): string | undefined; /** * Decide what KIND of failure this is, or `null` if it must not be retried. * * Pure. This is the whole safety argument in one function, so it is unit-tested * directly rather than only through the upload path. */ export declare function classifyFilesError(err: unknown): FilesRetryKind | null; export interface FilesRetryContext { /** 1-based index of the attempt that just failed. */ attempt: number; /** Milliseconds since the FIRST attempt started (the container's boot clock). */ elapsedMs: number; /** Which classes this call site permits. Defaults to cold-start only. */ allow?: readonly FilesRetryKind[]; } export type FilesRetryDecision = { retry: true; kind: FilesRetryKind; /** How long to wait before the next attempt. */ delayMs: number; /** 1-based number of the attempt that is about to run. */ nextAttempt: number; maxAttempts: number; } | { retry: false; reason: 'not-retryable' | 'kind-not-allowed' | 'attempts-exhausted' | 'budget-exhausted'; }; /** * The retry policy as a pure function: given the error and where we are, should we go * again and after how long? * * Separated from any I/O so the safety properties can be asserted directly — retries * on a gateway timeout, does NOT retry a quota/auth/4xx rejection, respects the * attempt ceiling and the wall-clock budget, and reports honestly why it stopped. */ export declare function decideFilesRetry(err: unknown, ctx: FilesRetryContext): FilesRetryDecision; /** What one attempt reports back to the runner. */ export type FilesAttemptResult = { ok: true; value: T; } | { ok: false; error: unknown; }; /** Handed to `onRetry` just before the runner sleeps. Enough to render honest UI. */ export interface FilesRetryNotice { kind: FilesRetryKind; /** 1-based number of the attempt about to run (2 means "second try"). */ attempt: number; maxAttempts: number; /** How long the runner is about to wait. */ delayMs: number; /** The failure that triggered the retry, for logging — never for the user. */ error: unknown; } export interface RunWithFilesRetryOptions { /** Which classes this call site permits. Defaults to cold-start only. */ allow?: readonly FilesRetryKind[]; /** Fired before each wait, so the UI can say what it is waiting for. */ onRetry?: (notice: FilesRetryNotice) => void; /** Injected by tests. */ now?: () => number; /** Injected by tests. */ sleep?: (ms: number) => Promise; } export interface FilesRetryOutcome { /** Present only on success. */ value?: T; /** The last in-band error, when `value` is absent. */ error?: unknown; /** How many attempts were actually made. */ attempts: number; /** Why we stopped, when `value` is absent. */ gaveUpBecause?: 'not-retryable' | 'kind-not-allowed' | 'attempts-exhausted' | 'budget-exhausted'; } /** * Run `attempt` until it succeeds, the policy says stop, or the budget runs out. * * Two failure channels, on purpose: * * - A THROWN error is classified and, if unretryable, rethrown **as-is**. Identity * and stack are preserved, so callers that let network errors through untouched * keep doing exactly that. * - An IN-BAND error (`{ ok: false, error }` — what `errorPolicy: 'all'` produces on * `result.error`) is returned in the outcome. The runner never invents a message * for it; the caller owns the wording, which is what fe-libs #132 established. */ export declare function runWithFilesRetry(attempt: (attemptNumber: number) => Promise>, options?: RunWithFilesRetryOptions): Promise>; //# sourceMappingURL=retryPolicy.d.ts.map