import type { CorpusAdapter, IndexHealth, StoredRecord, StoredRecordMeta } from './types.js'; /** * Corpus-neutral derived SQLite index. * * This engine stores and ranks generic "records" supplied by a `CorpusAdapter` * (see `./types.ts`). Domain semantics belong to the adapter. The engine only * knows: enumerate the adapter's files, decode each into a * record, keep an SQLite + FTS5 cache of those records in sync with their * source files, and rank a candidate pool by fusing lexical search with * whatever extra ranked signal lists the adapter supplies. * * The adapter's source files are the single source of truth; this index is * fully rebuildable from them at any time. */ export declare const CORPUS_INDEX_SCHEMA_VERSION = 3; export declare const CORPUS_INDEX_DB_FILENAME = "index.db"; export declare class CorpusIndex { private readonly root; private readonly adapter; static open(opts: { root: string; adapter: CorpusAdapter; }): Promise; private db; /** * The `CorpusAdapter.rootMtimeMs()` value observed on the last {@link * ensureFreshRecords} check that actually ran `listFiles()` — `null` * before any check has run. Used to skip that O(corpus) enumeration when * the root's mtime is unchanged. */ private lastKnownRootMtimeMs; /** * Cached prepared statements for the `records`/`records_fts` read path * (the `CorpusAdapter` storage mode's hot per-query operations), so a * query prepares each of these exactly once per store rather than once per * call. Cleared by {@link invalidateRecordStatements} whenever * `records`/`records_fts` is dropped and recreated — a statement prepared * against the OLD table object must not be reused once the table has been * dropped, even though the replacement carries the same name. */ private allRecordsStmt?; private allRecordsMetaStmt?; private recordCountStmt?; /** Cached by `IN` placeholder count for targeted post-ranking body reads. */ private readonly recordsByIdsStmts; /** * Cached by `IN` placeholder count for targeted metadata-only reads (no * `body`) — used for bounded graph-neighbour target lookups * ({@link recordsMetaByIds}), never for a whole-corpus read. */ private readonly recordsMetaByIdsStmts; /** * Cached candidate-query statements ({@link candidateRecordsMeta}), keyed * by `:` — at most four variants (topic * present/absent x history included/excluded). */ private readonly candidateStmts; /** * Cached un-scoped `count(*)` statement backing {@link unscopedMatchCount}/ * {@link narrowedMatchPattern} — a single statement (no topic/status * variants; it never joins to `records`). */ private candidateCountStmt?; private constructor(); private get dbPath(); private openDatabase; /** * Bootstrap the schema ONLY on a genuinely fresh database file (no known * known table exists yet). An EXISTING database — even one * on an outdated schema version — is left untouched here: {@link ensureSchema} * runs `CREATE TABLE IF NOT EXISTS`, a no-op against an already-existing * table that would silently skip a newly added column (e.g. `adapter_meta`), * and then unconditionally stamps `PRAGMA user_version` to the CURRENT * version — which would make {@link schemaIsValid} * report that stale table as valid forever, so an outdated on-disk database * would never actually rebuild. Every caller runs {@link ensureHealthy} * before reading or writing, which detects a genuine version/table mismatch * and performs a full drop + recreate rebuild instead. */ private bootstrapSchema; /** * Create the tables this corpus needs: `records` plus its `records_fts` * full-text index, one record per source file. */ private ensureSchema; private tableNames; private schemaVersion; /** * True when every required table exists, no legacy pre-engine table is * present, and the schema version matches. Table absence/presence is * checked explicitly and BEFORE relying on `user_version` at all: a legacy * database can carry a `user_version` that already equals the current * constant (an old build stamped it, or the constant was bumped without a * corresponding structural check), so the version comparison alone can * never be trusted to detect a stale schema. */ private schemaIsValid; /** * True when the `records` table itself carries the indexed `topic`/`status` * facet columns this schema version requires. An explicit STRUCTURAL check, * independent of `user_version` — the same defense {@link LEGACY_TABLES} * provides for a table that was dropped, applied here for a table whose * COLUMN SET changed instead ({@link CORPUS_INDEX_SCHEMA_VERSION} 2 -> 3, * `topic`/`status` promoted out of `adapter_meta` into real columns). * `ensureSchema` only ever stamps `PRAGMA user_version` — it never runs * `ALTER TABLE` against an existing table — so a database file left over * from schema version 2 has a `records` table with NO `topic`/`status` * columns at all; comparing `user_version` alone would already correctly * force a rebuild on THIS specific bump (2 -> 3 is a version this codebase * never wrote before), but a structural check that does not depend on * every prior build's version-stamping discipline having been perfect is * cheap here and closes that class of trap for good. */ private recordsFacetColumnsPresent; /** * Open + required-table + schema-version check. If the derived cache is * missing tables or on the wrong schema version, drop and rebuild it from * the authoritative source files. Freshness (new/changed files) is the job * of {@link syncIncremental}, not this method. */ ensureHealthy(): Promise; /** * Indexed record count — a single cheap `SELECT count(*)`. * * Private: the freshness comparison in {@link ensureFreshRecords} is its only caller, and * nothing outside this class has ever asked the index how many records it holds. (It was public * for adapters that needed a freshness check which does not load the corpus; the check moved * in here, and the old docstring explaining the public-ness outlived it by one audit pass — * left stacked above its own replacement.) */ private recordCount; /** * Cheap per-query freshness gate. * * The corpus root's mtime is checked first, which is a single `stat`. Only * when that mtime changed does this method pay the O(corpus) `listFiles()` * enumeration, so the overwhelming majority of queries skip it entirely. * * `CorpusAdapter` corpora (journal) keep their existing cheap count * comparison: compare the adapter's current file COUNT to the indexed * record count and only run the full incremental sync when they differ * (add/remove drift) or the schema is invalid. Same-count out-of-band * content edits are NOT detected here — those are the job of an explicit * {@link rebuild} / {@link syncIncremental}. When the adapter supplies * {@link CorpusAdapter.rootMtimeMs}, that count comparison itself is * pre-gated by a single cheap `stat` (see {@link ensureFreshRecords}): * `listFiles()` — an O(corpus) directory enumeration — only runs when the * root's mtime has actually changed since the last check, which is the * common case for the overwhelming majority of queries. */ ensureFresh(): Promise; /** * The `CorpusAdapter` (records) half of {@link ensureFresh}. Pre-gated by * the adapter's optional {@link CorpusAdapter.rootMtimeMs}: when supplied * and UNCHANGED since the last check, this skips `listFiles()` (an * O(corpus) directory enumeration) entirely — nothing could have been * added or removed if the directory's own mtime never moved. When absent, * or when the mtime DID change, falls through to the exact same * `listFiles().length !== recordCount()` comparison this always ran. */ private ensureFreshRecords; /** * Migration cleanup only — never schema creation. Drops every table owned * by the deleted pre-engine journal store, if present. A fresh or already- * current database has none of these and the statements are no-ops. */ private dropLegacyTables; /** * Drop every cached prepared statement bound to `records`/`records_fts`. * Must run before those tables are dropped: a statement prepared against * the table object being dropped must be re-`prepare()`d against its * replacement, not reused. */ private invalidateRecordStatements; /** * Drop `records`/`records_fts` and recreate them empty, then clear any * legacy pre-engine table. Cached statements are invalidated FIRST, because * a statement prepared against the table object being dropped must not be * reused against its replacement. */ private dropAndRecreateSchema; /** * Full rebuild: recreate the schema and re-derive every row from the * adapter's source files, into `records`/`records_fts` (one record per file). */ rebuild(): Promise; /** * Incremental sync keyed by source path + mtime + content hash. Only files * whose mtime AND content hash differ from the stored row are re-decoded and * upserted; files that vanished are dropped. Cheap enough to run before * every retrieval. * * CONCURRENCY. Two phases, and the split is load-bearing rather than stylistic. * * Phase 1 reads, hashes and decodes every candidate file while holding NO database * transaction. Phase 2 opens `BEGIN IMMEDIATE` and applies the already-computed changes, * so the write lock is held for the duration of a few prepared statements instead of for * the duration of reading the corpus. * * The earlier version did all of that inside one deferred `BEGIN`, which produced a real * crash: two concurrent `journal_recall` legs both ran this sync, and the loser died with * `runner_crash`. Two separate faults combined. * * - The write lock was held across every `readFile`, so the contention window grew with * the corpus rather than staying at write time. `PRAGMA busy_timeout = 5000` cannot * absorb a window that scales. * - A DEFERRED transaction takes a read lock first and must UPGRADE to a write lock at * its first write. SQLite fails that upgrade with `SQLITE_BUSY` immediately and does * NOT apply `busy_timeout`, because waiting while already holding a read lock could * deadlock two upgraders against each other. `BEGIN IMMEDIATE` takes the write lock up * front, which is the case `busy_timeout` does retry. * * Callers therefore no longer need to serialise journal access to stay alive. Serialising * remains harmless, and is still worthwhile for write-heavy callers, but concurrent * READERS must not crash — recall is the common path and runs on every flow. */ syncIncremental(): Promise; private upsertRecord; private deleteRecord; private loadFile; /** Every derived-index row, decoded back into structured form. */ allRecords(): StoredRecord[]; /** * Every derived-index row's metadata ONLY — id/path/title/mtime/hash/ * topic/status/adapter-meta, never `body`. Ranking (lexical fusion, tag * overlap, graph-neighbour expansion) needs none of a record's body text; * only a caller that will actually return a record's text should pay to * project it (see {@link recordsByIds}). Same row set as {@link * allRecords}, just a cheaper column list. This is still a WHOLE-CORPUS * read — callers on the per-query candidate path should use {@link * candidateRecordsMeta} instead, which is SQL-bounded; this remains for * callers that genuinely need every record (e.g. the reindex CLI's node * count). */ allRecordsMeta(): StoredRecordMeta[]; /** * Full records (including `body`) for exactly the given ids — the targeted * lookup a caller runs once to materialize text for the records it has * already decided, via {@link allRecordsMeta} + ranking, that it will * actually return. Unmatched ids are silently omitted. Empty `ids` * short-circuits without touching the database. Not cached as a prepared * statement: the `IN (...)` placeholder count varies per call, and this * runs once against a small, already-narrowed id set — not per corpus row. */ recordsByIds(ids: string[]): StoredRecord[]; /** * Metadata-only records (no `body`) for exactly the given ids — the * `recordsByIds` counterpart used for BOUNDED graph-neighbour target * lookups: a handful of specific link targets a candidate pool's seed * documents point to, never a whole-corpus read. Cached by `IN` placeholder * count, same reasoning as {@link recordsByIds}. Empty `ids` short-circuits * without touching the database. */ recordsMetaByIds(ids: string[]): StoredRecordMeta[]; /** * SQL-BOUNDED candidate query: lexical FTS5/BM25 match, joined against * `records` and scoped by an optional exact `topic` equality and an optional `status` * exclusion (the excluded VALUE is the caller's — see `excludeStatus`) — both against the * real, indexed columns added in schema version 3 (see {@link ensureSchema}) — capped at * `limit` rows and returned in bm25 order (best match first). This is the * query that replaces reading the whole corpus (or a whole topic) into JS * and filtering there: every predicate that can run in SQL does, and the * row COUNT returned is bounded by `limit`, never by corpus or topic size. * * Empty `tokens` short-circuits without touching the database: an empty * FTS5 MATCH pattern cannot express "match nothing", and every ranking * signal this engine fuses for a query (lexical, tag overlap, and * neighbour expansion seeded from lexical/tag) is driven off `tokens` — no * tokens can never win any signal regardless of what the candidate pool * contains, so there is nothing a query against the database could add. */ candidateRecordsMeta(opts: { tokens: string[]; topic?: string; /** A `status` value to exclude, compared for equality and never interpreted. Omit to * include every status. This used to be a boolean `includeHistory` with the literal * `'superseded'` written into the SQL — journal vocabulary inside an engine whose own * contract says it "only ever compares [status] for equality, never interprets what they * mean". The adapter owns the word now; the engine owns the comparison. */ excludeStatus?: string; limit: number; }): StoredRecordMeta[]; private candidateSelectStmt; /** * Un-scoped (no topic/status join) document count for one FTS5 MATCH * pattern — a cheap, bounded posting-list-length lookup FTS5 answers from * its own term statistics without materializing a single row or joining to * `records`. Used only to RANK tokens by rarity in {@link * narrowedMatchPattern}; a topic/status-SCOPED count would need the same * per-row join {@link candidateSelectStmt} does, which is exactly the cost * narrowing exists to avoid paying more than once. */ private unscopedMatchCount; /** * Reorder `tokens` rarest-document-frequency-first, then OR-join a PREFIX * of them — the query's actual MATCH pattern — stopping once the prefix's * SUMMED individual match counts (an upper bound on their true union: union * size <= sum of sizes) exceeds `limit x NARROW_FANOUT_MARGIN`. This exists * because `ORDER BY bm25(...)` (needed for a genuinely best-first result) * forces SQLite to fully materialize AND sort every MATCHing row before * applying `LIMIT` — cheap when the match set is small, but proportional to * corpus size when a query's tokens include common words that appear in a * large, roughly-constant FRACTION of the corpus (their match count grows * with corpus size, not with how many documents are actually relevant). * Narrowing the pattern down to its rarest tokens first keeps the match set * — and so the sort — bounded regardless of corpus size, while never * dropping the rarest (most discriminating) tokens a query has. * * Ranking is by UN-SCOPED count ({@link unscopedMatchCount}, no topic/status * join) rather than the topic-scoped count the final query will actually * see: a topic-scoped count is itself only answerable by the same per-row * join {@link candidateSelectStmt} pays for, which would reintroduce * per-token O(matches) work — exactly what narrowing exists to avoid. Using * the un-scoped count as a rarity PROXY is safe: the topic-scoped result is * always a subset of the un-scoped one, so a token rare un-scoped is at * least as rare within any one topic, and `NARROW_FANOUT_MARGIN` leaves * headroom for a topic's share of common tokens to still clear `limit` * matches once genuinely scoped. * * Skipped entirely (returns the full OR of every token) when there is at * most one token, or the full pattern's UN-SCOPED count is already at or * below `limit`: a topic-scoped count can only be smaller, so sorting that * few rows is cheap and narrowing would only add cost for no benefit. */ private narrowedMatchPattern; /** Reflected schema table list (for health/diagnostics tests). */ inspectSchema(): Promise<{ tables: string[]; schemaVersion: number; }>; close(): void; } //# sourceMappingURL=index-store.d.ts.map