/** * Retriever end-to-end probe. * * Fires a synthetic query at the deployed Retriever and asserts every stage of * the chain — offload-bucket freshness → indexer pipeline running → SQS queues * drained → pod ready → query submission → CloudWatch scan match → CloudWatch * stream fetch → S3 qr/*.jsonl write → MCP-side events returned. Each stage is * a named assert with a one-line `observed` summary and a stored remedy keyed * on the assert name. * * Designed as a "60-second post-install verify" and a deep doctor diagnostic. * Catches the silent-failure shapes that took ~2 hours to debug manually: * - Stream pipeline failing to launch (chart 1.0.20 streamer→retriever * rename residue): cw_stream_fetch fails with that remedy. * - Indexer pod up but not booted: indexer_pipeline_running fails. * - Forwarder/Fluentd offload broken: offload_bucket_has_recent_data fails. * - IRSA s3:PutObject misconfigured: s3_qr_jsonl_written fails. * - MCP input_bucket misaligned with engine write location: mcp_events_returned fails. * * The offload sink is not always S3. When the resolved destination type is * `azure_blob`, the two storage asserts list the blob container through * `az storage blob list` and report container facts: what the listing * returned and which blob-data role is missing. Neither one mentions IAM, * because an Azure operator has no IAM to fix. * * Dependencies are injectable via a `ProbeDeps` parameter so the test suite * can replace AWS / kubectl / metric-backend / submitQuery wholesale without * touching the production code paths. The default deps wire up: * - the object store lister for the destination type (aws or az CLI) * - aws CLI (via execFile) * - kubectl (via execFile) * - customer metric backend (resolveBackend → queryInstant) * - executeRetrieverQuery (called as a library — submits via the retriever * URL and polls S3 markers; never round-trips through the MCP IPC layer) */ import { type ObjectStoreKind } from './object-store.js'; export interface ProbeArgs { /** Kubernetes namespace where the retriever pod runs (e.g. 'log10x'). */ namespace: string; /** Bucket or blob container where the receiver offloads data (where the indexer reads from). */ offload_bucket: string; /** Bucket or blob container where the retriever writes qr//*.jsonl result objects. */ input_bucket: string; /** Destination type of both containers. Default `s3`. */ store_kind?: ObjectStoreKind; /** Azure storage account holding them. Read when `store_kind` is `azure_blob`. */ storage_account?: string; /** CloudWatch log group the retriever writes per-query execution events to. */ query_log_group: string; /** Label selector for the retriever pod. Default: 'app=retriever-10x'. */ pod_label_selector?: string; /** Pre-picked hash to query for; skips the metric-backend probe. */ target_hash?: string; /** Query window size in minutes. Default: 5. */ window_minutes?: number; /** Overall probe timeout in ms. Default: 90000. */ timeout_ms?: number; } export interface ProbeAssert { name: string; pass: boolean; observed: string; remedy?: string; } export interface ProbeResult { verdict: 'green' | 'broken' | 'unknown'; picked_hash?: string; query_id?: string; asserts: ProbeAssert[]; first_failed_assert?: string; surfaced_remedy?: string; total_runtime_ms: number; /** Populated when verdict==='unknown'. */ reason?: string; } export declare const REMEDIES: Record; /** * Remedies for the two storage asserts when the sink is a blob container. * The S3 wording sends an Azure operator to an IAM policy they do not have; * these name the blob-data role instead. Every other assert is store-neutral * and keeps its entry in REMEDIES. */ export declare const AZURE_BLOB_REMEDIES: Record; export interface ProbeDeps { /** List objects under prefix; returns the array of objects with LastModified. */ listObjects: (bucket: string, prefix: string) => Promise>; /** Run `kubectl logs -c --since=Ns` and return stdout. */ kubectlLogs: (namespace: string, podLabelSelector: string, sinceSeconds: number) => Promise; /** Get SQS queue depths (ApproximateNumberOfMessages) for the retriever queues by URL. */ sqsDepths: (queueUrls: string[]) => Promise>; /** List SQS queues matching a name prefix. Returns URLs. */ sqsListQueues: (namePrefix: string) => Promise; /** Check pod ready state; returns the pod name and whether all containers are ready. */ kubectlGetPod: (namespace: string, labelSelector: string) => Promise<{ name?: string; ready: boolean; observed: string; }>; /** CloudWatch Logs filter — returns matching events for a substring + filter pattern. */ cwFilterLogEvents: (logGroup: string, filterPattern: string, sinceMs: number) => Promise>; /** Pick the top tenx_hash from the metric backend; null if backend missing or empty. */ pickTopHash: () => Promise<{ status: 'ok'; hash: string; } | { status: 'no_backend'; } | { status: 'no_data'; }>; /** Submit a retriever query and poll until completion. Returns the response shape. */ submitRetrieverQuery: (req: { search: string; from: string; to: string; target: string; limit: number; writeResults: boolean; }) => Promise<{ queryId: string; eventsMatched: number; eventsReturned: number; }>; } /** * Run the e2e probe. Pre-flight asserts (1-4) run in parallel; post-query * asserts (5-8) run sequentially because each depends on the queryId. * * The verdict is: * - 'green' when every assert passed. * - 'broken' when at least one assert failed (with first_failed_assert * populated to the first false entry and surfaced_remedy set * to that entry's remedy). * - 'unknown' when the probe could not reach stage 3 (target hash could * not be picked because the metric backend is not configured * or has no data, AND the caller did not pass target_hash). */ export declare function runRetrieverProbe(args: ProbeArgs, deps?: ProbeDeps): Promise;