/** * product-kb/search — in-memory TF-IDF search over a Page[] corpus. * * Design notes * * - The corpus is small (~135 pages, ~1500 chunks). A full in-memory * inverted index fits in well under 5 MB and rebuilds at startup * in <100ms. No persistence needed. * * - Tokenisation: lowercase, split on whitespace + punctuation, drop * short-stopwords. The corpus is technical English; aggressive * stemming would hurt precision (collapse "patterns" → "pattern" * is fine, but "metric" → "metr" is not). * * - Scoring: classic TF-IDF on the chunk level. Per-document (page) * score is the max over its chunks — the user wants to find the * ONE chunk that answers their question, not an average over * unrelated sections. * * - Boosts: * slug-token superset (every query token is a slug segment) → +50 * slug-token tail subpath (>=2 contiguous trailing tokens) → +25 * per matched slug token → +10 * per matched heading token, focus-scaled → +4 * focus * * - Per-page scores are length-normalized (raw / sqrt(pageTokens)) * before boosts are applied. Without this, long body-heavy pages * win on token-shared queries against focused short pages that * are actually on-topic. * * - Rare-token gate: a query token whose page-df is below 2% of the * corpus is "discriminating"; when ANY query token is rare (or * missing entirely) AND no rare token has a hit, the search * returns [] so the caller can surface a found:false envelope * instead of low-quality residual matches. * * - SearchResult.matched_chunks contains ONLY the top-scoring chunks * per page (max 3), not the whole page. Keeps the envelope small. */ import type { Page, SearchResult } from './types.js'; /** * Tokenise a string for indexing or query parsing. Lowercase, split on * any non-alphanumeric character, drop stopwords and single chars. * * Preserves "pattern_hash" and similar underscore-joined identifiers * because they are user-facing field names in this product. */ export declare function tokenize(text: string): string[]; /** A flattened chunk reference into the corpus, used by the indexer. */ interface ChunkRef { pageIdx: number; chunkIdx: number; } /** The inverted index produced by buildIndex(). */ export interface SearchIndex { pages: Page[]; /** token → list of (page, chunk) refs that contain it, with TF. */ postings: Map>; /** token → number of distinct chunks that contain it (for IDF). */ df: Map; /** Total chunk count across the corpus (for IDF denominator). */ totalChunks: number; /** * Per-page total token count (sum of TF over all chunks). Used as * the document-length denominator in BM25-style length normalization, * which prevents long body-heavy pages from outranking focused * shorter pages on token-shared queries. */ pageTotalTokens: number[]; /** * token → number of distinct PAGES (not chunks) that contain it. * Used as the rare-token denominator: a query token is "discriminating" * when (pageDf / pages.length) is below a small fraction. The chunk- * level df is too noisy for that decision because one big page can * push a token into many chunk postings. */ pageDf: Map; /** * Per-chunk token count (pageIdx → chunkIdx → token count, heading * counted 3x to mirror indexing). Used to length-normalize a result by * the size of the chunks that actually matched, not the whole page — so * a focused chunk on a large reference page is not buried under the * page's total length. */ chunkTotalTokens: number[][]; } /** * Build an inverted index from a loaded corpus. Run once at startup, * pass the resulting SearchIndex to `searchIndex()` for each query. */ export declare function buildIndex(pages: Page[]): SearchIndex; /** * Options for `searchIndex()`. * * query — natural-language query string. * category — when set, restrict results to pages in this category * (faq / apps / engine / api / config / manage). * maxPages — cap the number of pages returned (default 10). * maxChunksPerPage — cap matched_chunks per page (default 3). * minScore — drop results whose final (length-normalized) score is * below this floor. Default 0.5. Without this, queries * whose only matching tokens are common filler words * return junk hits instead of falling through to the * found:false / similar_topics path in product-qa. */ export interface SearchOptions { query: string; category?: string; maxPages?: number; maxChunksPerPage?: number; minScore?: number; } /** * Run a TF-IDF search against an in-memory index. * * Returns the top-N SearchResult[] ordered by score descending. Each * result carries only its top-K matched_chunks to keep responses small. * * Scoring overview: * * 1. Per-chunk TF-IDF + heading-focus boost (see `scoreChunk`). * 2. Per-page score = sum of top-K chunk scores, then divided by * `sqrt(pageTotalTokens)` so long body-heavy pages cannot drown * focused short FAQ pages on token-shared queries. * 3. Slug boosts on the normalized score: * +50 when the slug is a token-superset of the query * +25 when the query tokens match a contiguous tail subpath * +10 per matched slug token otherwise (was +5) * The old "querySlug-as-string" comparison (querySlug retained * whitespace, so it never matched natural-language queries) is gone. * 4. Rare-token gate: when any query token has df/totalChunks below * RARE_TOKEN_PAGE_DF_FRACTION (or is absent entirely), at least * one of those rare tokens must hit. Otherwise [] is returned so * the caller's fallback (found:false + similar_topics) fires. * 5. minScore floor (default DEFAULT_MIN_SCORE) drops any normalized * result below the threshold. */ export declare function searchIndex(index: SearchIndex, opts: SearchOptions): SearchResult[]; /** * Exact slug lookup. Returns the matching page wrapped as a * SearchResult, or null when no page has that topic. * * Used by the `topic` arg path of the product_qa tool. When a `query` is * supplied alongside the topic, the page's chunks are ranked by that query * and the top `maxChunks` are returned, so drilling into a known * multi-section page (e.g. the Receiver deploy guide) lands on the * asked-about section rather than the page intro. With no query, the first * `maxChunks` chunks are returned (immediate context, no follow-up call). */ export declare function lookupTopic(index: SearchIndex, topic: string, query?: string, maxChunks?: number): SearchResult | null; /** * Return the N topic slugs that are textually closest to the given * query (by token overlap on the slug). Used when an exact-topic * lookup misses, so we can suggest "did you mean…" candidates. */ export declare function nearestTopics(index: SearchIndex, query: string, n?: number): string[]; export {};