/** * Retriever query diagnostics — polls CloudWatch Logs to extract structured * execution metadata (query plan, Bloom filter scan stats, worker stats, * errors) from a query's per-stream CW log. * * Shape this module emits: `RetrieverQueryDiagnostics`. Consumers (the * retriever-query tool, investigate fallback) use this to: * (a) report real execution progress to the LLM (not just "0 events"), and * (b) diagnose WHY a query returned 0 — stale indexer vs. no matching tokens * vs. Bloom false positives vs. mid-execution timeout. * * Polling is best-effort. Failures surface via `diagnostics.pollingError`, * not silent undefined — callers MUST check and degrade explicitly. * * Env: `LOG10X_RETRIEVER_LOG_GROUP` selects the CW log group to poll. When * unset, diagnostics are skipped and `pollingError` is set so the renderer * can surface the fact to the user. */ import type { QueryDiagnosis } from './query-funnel.js'; /** * Execution diagnostics for a retriever query, built from CloudWatch Logs events. * * Field guide (how to diagnose a 0-event result): * - queryPlan missing → query submission failed or CW events not visible yet * - emptyReason present → search expression produced no parseable tokens * - scanStats missing → no sub-queries reported scan completion (likely no * index data for the time range, i.e. stale indexer) * - scanStats.matched === 0 → Bloom filter rejected every index object * - scanStats.matched > 0 but * workerStats.totalResultEvents === 0 * → Bloom false positives OR workers timed out * - workerStats.complete < started * → some workers still running when MCP polled * - pollingError present → CW polling failed; diagnostics are incomplete * - partialResults === true → MCP poll timeout hit; server query may still be running */ export interface RetrieverQueryDiagnostics { /** Coordinator's query plan. Present if the main query log stream was readable. */ queryPlan?: { templateHashes: number; vars: number; timeslice: number; dispatch: 'local' | 'remote' | 'unknown'; /** The bloom pre-filter expression (querySearch), when the engine reports it. */ search?: string; /** The exact in-memory predicate applied to fetched events (queryFilters). */ filter?: string; }; /** Populated when isEmptyQuery() returned true — no parseable search tokens. */ emptyReason?: string; /** * Resolution stage: index blobs the scan mapped for the window, summed across * every `scan range:` event (sub-queries included). This is the ground-truth * empty-window signal: submittedKeys===0 on all slices means the time->blob * mapping returned no blobs (no data indexed for the range) — distinct from * "blobs were scanned but the bloom rejected them all". */ resolution?: { submittedKeys: number; submittedTasks: number; }; /** Aggregated Bloom filter scan stats across all sub-queries. */ scanStats?: { scanned: number; matched: number; skippedSearch: number; skippedTemplate: number; skippedDuplicate: number; }; /** * Coordinator's view of stream dispatch. Represents SQS send() calls, NOT * worker completion. Use workerStats for actual worker execution. */ streamDispatch?: { requests: number; objects: number; blobs: number; }; /** * Stream worker execution. `started` and `complete` are counted from CW events, * so they reflect CW visibility, not necessarily ground truth if CW buffer hasn't * flushed. If complete < started, some workers were still running when polled. */ workerStats?: { started: number; complete: number; /** * The engine's "stream worker complete: fetched N bytes", summed. NOTE: this * is the WRITTEN-event utf8 volume from the q/ byte-count writer, NOT the * bytes S3 returned. The engine emits no separate S3-read byte count, so this * must never be read as a fetch success/failure signal. */ totalFetchedBytes: number; /** Events the results writer wrote after the exact predicate (qr/). */ totalResultEvents: number; /** * Flushes that reached the results writer but were EMPTY because the exact * predicate dropped the event pre-write. The distinguisher for a 0-result: * >0 with totalResultEvents===0 => the filter rejected events that DID arrive * (FILTER_NO_MATCH); ===0 with totalResultEvents===0 => no rows reached the * writer at all, so the fetch or the parse produced nothing. */ totalEmptyFlushes: number; /** Events the results writer dropped because the result cap was hit. */ totalResultsTruncated: number; /** * Bytes actually read from the source objects (S3 GET), summed across workers, * from "stream worker fetch complete: s3BytesRead". The REAL fetch counter (vs * totalFetchedBytes, which is pre-filter event volume). 0 when the engine * predates the counter; > 0 distinguishes a parse miss from a fetch-empty. */ totalS3BytesRead: number; }; /** Main coordinator pipeline elapsed time (NOT full query wall time). */ coordinatorElapsedMs?: number; /** ERROR-level CW events from any pipeline component. */ errors?: string[]; /** * True when the MCP's S3 marker poll timed out before the server query finished. * The server query may still be running and writing more results. */ partialResults?: boolean; /** * If CW polling itself failed (CW SDK error, access denied, log group wrong), * the reason is reported here. Diagnostics are incomplete when set. */ pollingError?: string; /** * Always-available funnel verdict derived from the coordinator's _DONE marker * (set by the query path, not from CloudWatch). Present even when CW * diagnostics are empty — which is the common case when the query-events log * group is unconfigured or its buffer hasn't flushed. */ funnel?: QueryDiagnosis; } interface CWEvent { level: string; message: string; data?: Record; /** CW event timestamp in ms. */ ts: number; } /** * Build diagnostics from CW events. Deterministic: each event contributes * to exactly one field so aggregation is idempotent across re-polls. */ export declare function buildDiagnostics(cwEvents: CWEvent[]): RetrieverQueryDiagnostics; /** * Populate diagnostics into a response-shaped object. Polling errors become * `diagnostics.pollingError`, not silent undefined. */ export declare function attachDiagnostics(resp: T, queryStartTimeMs: number): Promise; /** * Look up current execution diagnostics for a previously-submitted queryId. * Use this after runRetrieverQuery returns `partialResults: true` to check * whether the server query has finished. */ export declare function getRetrieverQueryStatus(queryId: string, queryStartTimeMs: number): Promise; /** * Generate a human-readable explanation for a zero-result query. * Returns null when no specific explanation can be derived — callers should * fall back to a generic message. */ export declare function explainZeroResults(diag: RetrieverQueryDiagnostics): string | null; export {};