/** * Shared offload-status lookup helper. * * The receiver stamps `routeState="drop"` on every event it routes to the * customer-owned offload bucket (per `project_offload_loop_handoff.md`). * That stamp is visible on the metric surface * (`all_events_summaryBytes_total{routeState="drop"}`) — so any tool that * has resolved a `pattern_hash` can ask "is this pattern currently being * offloaded?" with a single PromQL instant query. * * This module is the one canonical place that question gets asked. * `retriever_query`, `event_lookup`, and `investigate` each call into * here so the offload-detection logic (and its tolerances — thresholds, * timeout, defaults) lives in exactly one spot. * * Design split (preserved from `metric_surface_owns_overflow_visibility.md`): * the offload lookup is a TSDB query on the metric surface, NOT a * retriever-archive scan. The helper does not touch S3 / the bloom index; * it issues at most two Prometheus instant queries per call, each wrapped * in its OWN 2s timeout. Heavy-cohort tail latency: a slow kept-side scan * does not poison a fast dropped-side answer — see the partial-result * contract on `OffloadStatus`. `ok: false` is reserved for the case where * BOTH cohorts timed out. * * Why not co-located with promql.ts: `promql.ts` is a pure query-builder * module (returns strings, no env / no executor). This helper needs both * the builder (`includeToSelector`) AND the executor * (`queryInstant`), so it sits one layer up. */ import type { EnvConfig } from './environments.js'; import { type LabelNameMap } from './promql.js'; /** * Resolved offload status for a single `pattern_hash`. * * `ok=false` signals neither cohort returned data within the timeout — * callers should treat the other fields as undefined and suppress any * offload-related UI/markdown. * * Partial-result contract (heavy-cohort tail-latency fix): when the * dropped side resolves with bytes > 0 but the kept side times out, the * envelope still surfaces `ok: true, is_offloaded: true, * dropped_bytes_in_window: ` with `kept_bytes_in_window: null`, * `dropped_share_pct: null`, and `kept_timed_out: true`. The agent still * gets the actionable signal (this pattern IS offloaded → use * retriever_query); the share math is omitted. Symmetric flag * `dropped_timed_out` covers the inverse case (kept resolves, dropped * times out) — in that branch we cannot claim offload, so `is_offloaded` * is forced false and `dropped_bytes_in_window` is null. */ export interface OffloadStatus { /** * True when the drop/offload cohort (`routeState="drop"`) has bytes in the * window. NOTE: today `routeState="drop"` does NOT distinguish * offload-to-S3 (fetchable via retriever_query) from hard-drop (gone, * never offloaded). So `is_offloaded` means "in the engine's drop/offload * cohort", NOT "confirmed offloaded/fetchable". Consumers must not promise * fetchability from this alone — a true distinction needs the dedicated * `routeState="offload"` setter (D1b). Until then, gate fetch-back claims * on a found result / retriever-configured, not on this flag. */ is_offloaded: boolean; dropped_bytes_in_window: number | null; dropped_share_pct: number | null; kept_bytes_in_window: number | null; sample_count: number; /** Unix-ms of the latest dropped sample bucket; null when no dropped series exists. */ last_seen_dropped_ts: number | null; /** True when at least one cohort returned data; false only when BOTH timed out. */ ok: boolean; /** True when the kept-cohort PromQL scan timed out (share math suppressed). */ kept_timed_out?: boolean; /** True when the dropped-cohort PromQL scan timed out (is_offloaded forced false). */ dropped_timed_out?: boolean; } export interface OffloadStatusArgs { patternHash: string; metricsEnv: string; /** Window for the increase() over the bytes metric. Default `24h`. */ range?: string; /** Per-query timeout. Default 2000ms (each cohort runs in parallel). */ timeoutMs?: number; /** Label name map. Default `DEFAULT_LABELS`. */ labels?: LabelNameMap; } /** * Lookup the offload status for a single `pattern_hash`. * * Issues two parallel Prometheus instant queries (kept + dropped cohorts) * via `includeToSelector('both')`. Both calls share the * `timeoutMs` budget (parallel, not sequential). If either cohort times * out the result is marked `ok: false`. */ export declare function getOffloadStatus(env: EnvConfig, args: OffloadStatusArgs): Promise; /** * Batch variant. Used by `retriever_query` and `investigate`, which both * already hold a top-N `pattern_hash` list and would otherwise issue N * round trips. Drops the per-hash filter, groups `sum by (hash)`, then * locally joins back to the input list. * * Hashes that returned no data are absent from the returned record. The * caller's `is_offloaded` default for absent entries is `false` — but * the caller decides, not this helper (so "queried and absent" stays * distinguishable from "lookup failed", which is a missing entry plus * an empty record). */ export declare function getOffloadStatusBatch(env: EnvConfig, args: Omit & { patternHashes: string[]; }): Promise>;