import type { CorpusAdapter, CorpusRecord, IndexHealth, RankedList, StoredRecordMeta } from '../engine/types.js'; /** * Journal corpus adapter + journal-specific ranked retrieval. * * This module is the SOLE owner of every journal concept: topic prefiltering, * cross-topic fallback, `superseded` visibility, typed-edge neighbour * expansion, tag overlap, and zero-padded node ids. The corpus-neutral engine * (`../engine/`) knows none of this — it only stores and lexically ranks * generic `records`. Two things live here: * * 1. `JournalCorpusAdapter` — implements the engine's `CorpusAdapter` * contract so `CorpusIndex` can enumerate/decode/index journal node * markdown as generic records (id/path/title/body) and lexically search * them (FTS5/BM25) without knowing what a "journal" is. `topic` and * `status` are supplied as the engine's real, INDEXED `topic`/`status` * columns (schema version 3) — so a query can prefilter/scope candidates * in SQL. Everything else journal-only (tags/links/description) rides * along as a JSON string in the record's opaque `adapterMeta` field — * persisted by the engine in the SAME row, SAME statement, and SAME * transaction as the record itself. There is no second table and no * second database connection: the engine never parses `adapterMeta`, it * only stores and returns it. * 2. `JournalIndexStore` — the public store journal callers have always * used. It wraps a `CorpusIndex` for content, lexical search, and * metadata, deserializing each record's `adapterMeta` back into journal * metadata on read. */ /** * The PUBLIC names for the derived index's filename and schema version. * * "Historical" until this audit, which reads as compatibility scaffolding safe to delete. It is * the opposite: `CORPUS_INDEX_DB_FILENAME` / `CORPUS_INDEX_SCHEMA_VERSION` are internal to * `../engine/index-store.ts` and referenced nowhere else, while these two are exported through * the core barrel, pinned by `core-barrel-surface.test.ts`, and used by `cli/journal-reindex.ts` * and the migration tests. Deleting them would be a breaking change to the published API in * service of removing an alias that is the only name anyone uses. */ export declare const JOURNAL_INDEX_DB_FILENAME = "index.db"; export declare const JOURNAL_INDEX_SCHEMA_VERSION = 3; export declare class JournalCorpusAdapter implements CorpusAdapter { readonly corpusId = "journal"; readonly root: string; constructor(opts: { journalRoot: string; }); private get nodesDir(); listFiles(): Promise; /** * Modification time of the `nodes/` directory itself — updates whenever a * node file is added or removed (standard filesystem behavior), so the * engine can use it as a cheap, exact "did anything change" pre-check * before paying for a full {@link listFiles} enumeration on every query * (see `CorpusIndex.ensureFreshRecords`). `null` if the directory is * unreadable (e.g. not yet created) — the engine falls back to its * unconditional `listFiles()` check in that case. */ rootMtimeMs(): Promise; decode(relPath: string, raw: string): Promise; /** * The engine owns only generic RRF mechanics. Journal-specific tag overlap * and typed-edge expansion are supplied here as ranked lists for it to fuse * with lexical order. Passing lexical order in lets the adapter preserve the * historical seed rule: lexical hits first, then tag hits, capped at ten. * `pool` records already carry their own `adapterMeta` — no separate lookup. * `pool` is metadata-only (no `body`): every signal here reads only `id` * and `adapterMeta`, never record text. */ signals(tokens: string[], pool: StoredRecordMeta[], lexicalOrder?: string[]): RankedList[]; } export interface IndexedLink { type: string; target: string; } /** One derived-index row, decoded back into structured form. */ export interface IndexedDocument { nodeId: string; nodePath: string; title: string; topic: string; status: string; tags: string[]; links: IndexedLink[]; /** One-line node summary from frontmatter `description`. */ description: string; /** * Concatenated `title\ndescription\ncontext\nconsequences`, used for FTS * indexing and for deriving a citation snippet (context/consequences excerpt). */ body: string; } /** * An {@link IndexedDocument} projection WITHOUT `body` — everything ranking * (topic prefilter, tag overlap, graph-neighbour expansion, RRF fusion) * needs, and nothing more. `body` is fetched separately, only for the * records that survive ranking (see {@link JournalIndexStore.documentsBodyByIds}). */ export type IndexedDocumentMeta = Omit; /** Result of a lexical FTS5 probe: node ids ordered best-first (lowest bm25). */ export type { IndexHealth }; export declare class JournalIndexStore { private readonly adapter; private readonly index; static open(opts: { journalRoot: string; }): Promise; private constructor(); ensureHealthy(): Promise; /** * Cheap per-query freshness gate — delegates to the engine's own count-based * check (schema invalid → rebuild; file count changed → incremental sync; * otherwise a no-op). */ ensureFresh(): Promise; /** Full rebuild: recreate the schema and re-derive every row from the node files. */ rebuildIndex(): Promise; syncIndexIncremental(): Promise; /** * Decode one engine record's metadata fields into an {@link * IndexedDocumentMeta}: `topic`/`status` come straight off the record's own * (real, indexed) columns; `tags`/`links`/`description` are deserialized * from its `adapterMeta` JSON — there is no separate metadata table or * connection to join against. Shared by every read path so this mapping is * written once. */ private toIndexedDocumentMeta; /** Every derived-index row, decoded back into structured form. */ allDocuments(): IndexedDocument[]; /** * Every derived-index row's metadata ONLY — same fields as * {@link allDocuments} minus `body`. This is the shape ranking (topic * prefilter, tag overlap, graph-neighbour expansion, RRF fusion) actually * needs; it never touches a record's body text. Use this on the query path * instead of {@link allDocuments}, whose `body` projection exists for * callers that need node text (e.g. {@link documentsBodyByIds}, or a * direct caller like the reindex CLI's node count). */ allDocumentsMeta(): IndexedDocumentMeta[]; /** * SQL-BOUNDED candidate metadata: lexical FTS5/BM25 match, scoped to an * optional exact `topic` and status visibility, capped at `opts.limit` rows * — the candidate-bounded counterpart to {@link allDocumentsMeta}. Returned * in bm25 (lexical relevance) order, best match first. This is the read a * per-query candidate search should use: unlike {@link allDocumentsMeta}, * its row count never grows with corpus or topic size. */ candidateDocumentsMeta(opts: { tokens: string[]; topic?: string; includeHistory: boolean; limit: number; }): IndexedDocumentMeta[]; /** * Metadata for exactly the given node ids — used for BOUNDED * graph-neighbour target lookups (a handful of specific link targets a * candidate pool's seed documents point to, capped by seeds x edges), never * for a whole-corpus read. Unmatched ids are silently omitted. */ documentsMetaByIds(nodeIds: string[]): IndexedDocumentMeta[]; /** * Body text for exactly the given node ids — the ONE targeted lookup a * query runs, once ranking has already narrowed the corpus down to the * ids it will actually return. Never call this with more ids than a query * is about to return; that would recreate the whole-corpus body read this * method exists to avoid. */ documentsBodyByIds(nodeIds: string[]): Map; /** * The same journal-owned ranked signals supplied to the engine's RRF, * built directly from the already-in-memory metadata `pool` — no index * read. `pool` already carries each document's decoded topic/status/tags/ * links, so those are simply re-serialized into the opaque `adapterMeta` * string shape {@link JournalCorpusAdapter.signals} expects, rather than * re-fetched from the database. */ rankedSignals(tokens: string[], pool: IndexedDocumentMeta[], lexicalOrder: string[]): RankedList[]; /** Reflected schema table list (for health/diagnostics tests). */ inspectSchema(): Promise<{ tables: string[]; schemaVersion: number; }>; close(): void; } /** * A retrieval candidate handed to the recall/record route for retrieve-then-judge. * `score` is the fused Reciprocal Rank Fusion score (higher is better). * `fallback` marks a cross-topic candidate surfaced only because the in-topic * pass produced fewer than {@link MIN_IN_TOPIC} candidates. */ export interface JournalCandidate { nodeId: string; nodePath: string; title: string; topic: string; status: string; tags: string[]; /** One-line node summary (frontmatter `description`). */ description: string; /** * Short excerpt of the node's context/consequences body so the LLM can cite * evidence without opening the node file. */ snippet: string; score: number; fallback: boolean; matchedVia: string[]; } /** Recall retrieval result: previews that fit the budget, plus what did not. */ export interface RecallCandidateSet { candidates: JournalCandidate[]; /** Ranked candidates dropped by {@link RECALL_PREVIEW_BUDGET_BYTES}. 0 when all fit. */ withheld: number; /** Total ranked candidates before the budget was applied. */ totalRanked: number; } /** * Recall retrieval. Returns previews bounded by * {@link RECALL_PREVIEW_BUDGET_BYTES} plus the count of ranked candidates the * budget dropped, so a caller can never mistake a trimmed set for a complete * one. Ranking is unchanged; only the assembled payload is bounded. */ export declare function searchCandidatesForRecall(store: JournalIndexStore, input: { prompt: string; topic?: string; includeHistory: boolean; }): Promise; /** * Single-record dedup retrieval. * * The `journal_record` route does NOT call this — it calls * {@link searchCandidatesForRecordBatch}, which runs health and freshness once for the whole * request. This remains the published single-record entry point and, more usefully, the ORACLE * the batch variant is checked against: `tests/journal/search.test.ts` asserts the batch returns * for record 1 exactly what this returns for record 1 alone. Keep them behaviourally identical. */ export declare function searchCandidatesForRecord(store: JournalIndexStore, input: { prompt: string; topic?: string; }): Promise; /** * Batch counterpart of {@link searchCandidatesForRecord}: health + freshness * run ONCE for the whole batch, not once per submitted record. A * `journal_record` request with N records previously paid health/freshness * work N times (once per {@link searchCandidatesForRecord} call) even though * nothing in the store changes between records within one request — searching * record 2 can never observe a staleness that record 1's check didn't already * resolve, so the repeat work was pure waste. Same per-search semantics as * {@link searchCandidatesForRecord} (record retrieval never surfaces * superseded history) for every record in the batch. */ export declare function searchCandidatesForRecordBatch(store: JournalIndexStore, records: Array<{ prompt: string; topic?: string; }>): Promise; //# sourceMappingURL=journal-adapter.d.ts.map