/** * The canonical memory-kind vocabulary — the kinds a bare `recall(query)` groups over. `code` * (ts/js/py) + `doc` (md) enter via `index()` (files, routed by `indexer.classify`); `fact` + * `episode` enter via `remember()` (written directly — slice 7, §3.2). `doc` is the one kind both * paths produce. A bare grouped `recall` returns a (possibly empty) list per kind. * @type {string[]} */ export const KINDS: string[]; /** * The kinds `remember()` accepts — the WRITE-side vocabulary ({@link KINDS} minus `code`, which * enters only via `index()` from files). `remember()` validates against this exact list, so it is the * single source of truth for "what can be written directly." Consumers that gate a config *before* * calling `remember` (e.g. a schema validator) should bind this instead of re-typing the subset, so * they can never drift if the write set widens or narrows. * @type {readonly string[]} */ export const WRITE_KINDS: readonly string[]; /** * The shared-tier scope sentinel (multis M3 fail-closed ask). Under `strictScope`, a missing scope * THROWS — so reading or writing the global knowledge base needs an explicit, unambiguous opt-in that * is never spelled the same as "I forgot." That is `GLOBAL`: pass it as a `scope` (to `recall`/`get`/ * `ingest`/`remember`) or bind a `ctx.scoped(GLOBAL)` view to act on the global tier deliberately. * * It is a **read/write sentinel, never a stored value**: on write it maps to "no `doc_scope` row" * (`scope IS NULL`, exactly today's global rows); on read it maps to "`ds.scope IS NULL` only." So it * needs no migration and leaves the `scope ∪ NULL` union untouched. A unique Symbol so it can never * collide with a tenant scope string. * @type {symbol} */ export const GLOBAL: symbol; /** * Thrown by `get(path, { startLine, endLine })` when the file changed on disk after it was indexed, so * the pointer's line range no longer describes the chunk it was issued for. * * This is deliberately loud rather than a `null`. A `null` reads as "not there"; the truth is "it IS * there, and I can no longer tell you *which* code it is" — and the tempting fallback (slice the lines * anyway) returns a *different symbol's body* with no error at all. That is the one outcome worse than * an exception, so we refuse. It is also fully recoverable: re-run `index()` (~6ms when nothing moved) * and `recall()` for a fresh pointer. */ export class StalePointerError extends Error { /** @param {string} path the repo-relative path whose content moved on */ constructor(path: string); /** @type {string} */ code: string; /** @type {string} */ path: string; } /** * @typedef {Object} LiteCtxConfig * @property {string} root repo root to index * @property {string[]} [include] file extensions to index (default: ts/js/py/md) * @property {string[]} [pathspecs] optional git pathspecs to scope the index (e.g. ["app/**\/*.js"]) * @property {string} [dbPath] SQLite file path (default: /.litectx/index.db) * @property {boolean} [embeddings] enable the opt-in semantic tier (default false). When on, * `index()` embeds each file and `recall()` fuses cosine into * the ranking. Requires the optional peer dep `@huggingface/transformers`. * @property {number} [embedWeight] semantic fusion weight (default 1.0); higher = more semantic * @property {string} [embedModel] transformers.js model id (default Xenova/all-MiniLM-L6-v2) * @property {{ embed(text: string): Promise }} [embedder] inject a custom/stub embedder * (advanced/testing); overrides the built-in model loading * @property {string} [owner] scope key (§4.4) — the actor that owns durable `fact`s (and * tags `episode`s). Default unset = global/unscoped: recall * sees & writes are owner-blind. A multi-tenant / shared-db host * sets it (e.g. git email or OS user, resolved host-side) so a * shared store isolates per actor; recall then returns own + * global (NULL-owner) memory only. * @property {boolean} [strictScope] fail-closed multi-tenant mode for the DOC axis (multis M3 ask). * Default false = today's behaviour (a missing/`null` doc `scope` * means "see everything" — correct single-tenant, a footgun on a * shared store). When true, a missing scope on `recall({kind:'doc'})`, * `get`, `ingest`, and `remember({kind:'doc'})` THROWS instead of * returning/writing every tenant's rows; the only ways to act are an * explicit tenant scope (`scope ∪ global`) or {@link GLOBAL} (the * shared tier). Governs the doc/blob axis ONLY — `fact`/`episode` * (the `owner`/`session` memory axis) and `code` are untouched. * @property {string} [session] scope key (§4.4) — the run that owns volatile `episode`s. * Default unset = durable/unscoped: recall sees all sessions' * episodes. A host running concurrent agents sets it so a run's * own episodes aren't buried by more-relevant other sessions * (gate #1, 2026-06-13). `fact`s ignore it (always cross-session). * @property {number} [episodeWindowDays] the rolling window (in days) an `episode` stays RETAINED and * promote-eligible (default {@link ACTIVE_EPISODE_DAYS} = 30). On * each episode write, episodes older than this self-prune; the same * window floors {@link LiteCtx#promotionCandidates}. One knob bounds * the agent scratchpad (no count cap). ⚠ This is NOT a free hygiene * dial — it is COUPLED to the promotion ladder: a window shorter than * an episode's promote-and-prove time can prune it (and drop it below * the promotion floor) BEFORE it reaches the threshold, so it never * promotes. Shorten it for data-minimization, lengthen it to retain * older episodes longer (the set grows); 30d is the safe default that * leaves room to promote-and-prove. Facts are durable and untouched. * @property {WriteGateLike} [writeGate] optional write-gate hook (CE-PRD §10.1) — when set, `remember()` * emits a `{type:"memory.write", …}` action and `await`s * `writeGate.check(action)` BEFORE persisting; a `deny` outcome * throws {@link WriteDeniedError} and the write does not commit * (`ask`/`allow` proceed). Duck-typed — bareguard's `Gate` when * embedded, any `.check`-shaped object standalone. litectx is not * coupled to a gate version. Default unset = no gate (byte-identical * to pre-hook writes). * @property {WriteAudit} [writeAudit] optional standalone audit sink (the paper-trail half §10.1) — * when set with `writeGate`, each write decision is recorded. A * host-supplied `redact` on it scrubs secrets (litectx ships none). * @property {boolean} [trace] when true, the instance is returned wrapped in `observe()` — every * CE verb call is recorded into `ctx.trace` (a `ContextGraph`). Off * by default = the bare instance, no proxy. See `src/contextgraph.js`. */ /** @typedef {import("./writegate.js").WriteGateLike} WriteGateLike */ /** * @typedef {Object} Item * @property {string} id the written-memory id, or the file's repo-relative path * @property {string} kind "code" | "doc" | "fact" | "episode" * @property {string} format "ts" | "js" | "py" | "md" | "text" | ... * @property {string} source "file" (indexed from disk) | "direct" (written via remember) * @property {string|null} provenance "human" | "agent" for written memory; null for indexed files * @property {number|null} occurredAt episode timestamp (epoch ms); null otherwise * @property {string|null} text the full body — written memory verbatim as remembered, files * read fresh from disk; null when the file is gone, or for a blob * (a byte-exact upload — its payload is in `bytes`, not text) * @property {Buffer|null} bytes a byte-exact upload's original bytes (R3); null for every other * kind and for file rows * @property {Record|null} meta opaque caller metadata (RT-3 #3) as supplied to * `remember`, returned verbatim; null for files and for memory * with none */ /** * One row of an {@link LiteCtx#enumerate} page — a pointer record (search-only fields like `score` * omitted; this is an unranked table read). Field name is `path` (the public id, decoded via `memId`), * uniform with `recall`/`recentMemory`/`get`. * @typedef {Object} EnumItem * @property {string} path the written-memory id (public, owner-prefix stripped) * @property {string} kind "fact" | "episode" * @property {string} format the stored format * @property {number|null} occurredAt episode timestamp (epoch ms); null for a fact * @property {string} [body] VERBATIM stored text — present only when `body:true` * @property {Record} [meta] opaque caller metadata, present only when the row carries it */ /** * @typedef {Object} IndexResult * @property {number} files total documents in the index after the pass * @property {number} added newly indexed files * @property {number} updated re-indexed files (content changed) * @property {number} removed files dropped (no longer present) * @property {number} unchanged files skipped (mtime or content unchanged) */ /** * The store. One instance = one SQLite-backed code+context graph; every verb * (recall/impact/get/remember/…) hangs off it. Local-first, single file. * @param {LiteCtxConfig} config requires `root`; `dbPath` defaults to `/.litectx/index.db`. * @category core * @when You need a litectx store — the entry point for every other primitive. * @fails Throws if `config.root` is missing. * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * await ctx.index() * const hits = await ctx.recall('rate limiter', { kind: 'code' }) */ export class LiteCtx { /** @param {LiteCtxConfig} config */ constructor(config: LiteCtxConfig); root: string; include: string[]; pathspecs: string[] | undefined; dbPath: string; owner: string | null; session: string | null; strictScope: boolean; episodeWindowDays: number; store: Store; embeddings: boolean; embedWeight: number; embedModel: string | undefined; /** @type {{ embed(text: string): Promise } | null} */ _embedder: { embed(text: string): Promise; } | null; /** @type {Map} LRU query-embedding cache */ _qcache: Map; /** @type {WriteGateLike | null} */ writeGate: WriteGateLike | null; /** @type {WriteAudit | null} */ writeAudit: WriteAudit | null; /** The embedder for this instance — injected, or lazily constructed when the tier is on. */ get embedder(): { embed(text: string): Promise; }; /** * Embed text, degrading gracefully when the optional model dependency (`@huggingface/transformers`) * can't load: disable the tier for this instance, warn once to stderr, and return `null` so * callers fall back to BM25. An injected embedder (tests) or a present dep never trips this. * @param {string} text @returns {Promise} */ _embedSafe(text: string): Promise; /** Embed a query with a small LRU cache (repeated queries skip the model). @param {string} q */ _embedQuery(q: string): Promise | null>; /** * Build or incrementally refresh the index over the configured root. * * By default only files whose content changed are re-read, and files that disappeared are * dropped. Pass `force` for a full rebuild, or `paths` (git pathspecs) to scope the pass — * a scoped pass never deletes files outside its scope. The two compose: `force` with `paths` * re-chunks the scoped files (ignoring their hashes) while leaving the rest of the index * untouched — it does NOT wipe the whole index. * * **Self-healing on upgrade.** An index also goes stale when *litectx itself* changes: the chunker * decides where a chunk starts, and mtime/size cannot see that a new chunker would have drawn the * boundary somewhere else. So an index is stamped with {@link indexStamp}, and a stamp mismatch * forces a full re-chunk exactly as `force` would. Without this an upgraded library keeps serving * boundaries its old self wrote — silently, forever. The rebuild clears file-sourced rows only; * written memory survives it (§3.2). * * **Cooperative yielding.** `index()` is `async`, but its work (tree-sitter chunking, SQLite * upserts) is synchronous CPU — so on a large or `force` pass it can hold the host event loop for * seconds, during which a co-hosted timer/socket cannot fire. Pass `yield: true` to release the * loop between per-file parses (via `setImmediate`), so the host breathes. It does NOT parallelise * or change the stored index — only *when* the CPU runs — so the result is byte-identical to the * default pass; a single file's parse and the one atomic `applyChanges` transaction remain * uninterrupted (their duration is the residual floor). Default `false` keeps today's behaviour. * A caller needing full isolation should run the instance in its own worker thread instead. * * @param {{ paths?: string[], force?: boolean, yield?: boolean }} [opts] * @returns {Promise} * @category index * @when Build or refresh the graph from source before recall/impact — call after files change. * @fails Throws `RipgrepMissingError` only via later `impact()`, not here; a bad `root` throws at construction. * @signature liteCtx.index(opts?: { paths?, force?, yield? }) => Promise * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const { added, updated, unchanged } = await ctx.index() * // incremental: only re-chunk two files * await ctx.index({ paths: ['src/a.js', 'src/b.js'] }) */ index(opts?: { paths?: string[]; force?: boolean; yield?: boolean; }): Promise; /** * Resolve a caller's `scope` arg for a READ (`recall`/`get`) into the store's tri-state filter, * applying the strictScope policy (multis M3 fail-closed ask). The whole point: a *missing* scope * and a *deliberate* all/global read must not share a spelling. * - {@link GLOBAL} → the shared tier only (`{ scope: null, seeAll: false, globalOnly: true }`). * - a tenant string → `scope ∪ global` (`{ scope, seeAll: false, globalOnly: false }`). * - omitted/`null` → under `strict`, THROW; otherwise the legacy see-all (`{ seeAll: true }`). * @param {string | symbol | null | undefined} scope * @param {boolean} strict enforce (throw on a missing scope) — the caller decides per-axis * @param {string} op label for the thrown error * @returns {{ scope: string|null, seeAll: boolean, globalOnly: boolean }} */ _resolveReadScope(scope: string | symbol | null | undefined, strict: boolean, op: string): { scope: string | null; seeAll: boolean; globalOnly: boolean; }; /** * Resolve a caller's `scope` arg for a WRITE (`ingest`/`remember` on the doc axis) into the stored * `doc_scope` value, applying strictScope. {@link GLOBAL} and omitted-when-not-strict both map to * `null` (the shared tier = no `doc_scope` row); a missing scope under `strict` THROWS — so an * accidental publish-to-everyone is impossible, the persistent-leak half of the ask. * @param {string | symbol | null | undefined} scope * @param {boolean} strict * @param {string} op * @returns {string | null} */ _resolveWriteScope(scope: string | symbol | null | undefined, strict: boolean, op: string): string | null; /** * Resolve a caller's `scope` arg for a memory-axis READ (`recall`/`reviewCandidates`/ * `promotionCandidates` over `fact`/`episode`) into the store's owner fence (multis M4). The memory * axis historically fenced ONLY by the instance `owner` set at construction; this lets one shared * instance fence per tenant by threading the scope through per call (via {@link scoped} or an explicit * `scope`). The mapping mirrors the doc-axis `_resolveReadScope` but targets `owner`: * - a tenant string → that owner ∪ global (`{ memOwner: scope, memSeeAll: false }`). * - {@link GLOBAL} → the shared tier only (`{ memOwner: null, memSeeAll: false }`). * - omitted/`null` → under `strictScope`, THROW (fail-closed, the M4 ask); otherwise fall back to the * INSTANCE owner — so a single-tenant instance (owner set at construction, strict off) is unchanged. * @param {string | symbol | null | undefined} scope * @param {string} op label for the thrown error * @returns {{ memOwner: string|null, memSeeAll: boolean }} */ _resolveMemReadScope(scope: string | symbol | null | undefined, op: string): { memOwner: string | null; memSeeAll: boolean; }; /** * Resolve a caller's `scope` arg for a memory-axis WRITE (`remember` over `fact`/`episode`) into the * stored `mem_scope.owner` (multis M4). {@link GLOBAL} → `null` (the shared tier); a tenant string → * that owner; omitted/`null` → under `strictScope` THROW (fail-closed), else the INSTANCE owner * (legacy single-tenant). So an accidental un-scoped tenant write is impossible under strict. * @param {string | symbol | null | undefined} scope * @param {string} op * @returns {string | null} */ _resolveMemWriteOwner(scope: string | symbol | null | undefined, op: string): string | null; /** * A scope-bound view (multis M3 fail-closed ask, layer c) — the doc-axis equivalent of binding * `owner`/`session` on the instance. `ctx.scoped('user:42')` returns a handle whose `recall`/`get`/ * `ingest`/`remember` carry that scope automatically, so "forgot to pass a scope" becomes a * non-existent code path (the per-call `scope` is gone, there is nothing to omit). This is the * blessed multi-tenant pattern: it works regardless of `strictScope`, but pairs with it (the flag * makes the BASE methods safe; the view makes the safe path the only path the caller touches). * * Pass {@link GLOBAL} for a shared-tier (KB) view. A bad bind (null/omitted/non-string-non-GLOBAL) * throws HERE, at creation — a scope-bound view with no scope is the very footgun this closes, so * it can never be constructed. The bound scope is fixed: the returned methods ignore any `scope` * passed in their opts. * @param {string | symbol} scope a tenant scope string, or {@link GLOBAL} for the shared tier * @returns {ScopedView} * @category core * @when Serve many tenants from one instance — bind a scope once and every verb on the view is fenced to it. * @fails Throws at creation on a bad bind (null / omitted / non-string non-GLOBAL) — a scope-less scoped view is impossible. * @signature liteCtx.scoped(scope: string | symbol) => ScopedView * @example * import { LiteCtx, GLOBAL } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const acme = ctx.scoped('tenant:acme') // every verb fenced to acme * await acme.remember('pref-1', 'prefers dark mode', { kind: 'fact' }) * const shared = ctx.scoped(GLOBAL) // shared knowledge-base tier */ scoped(scope: string | symbol): ScopedView; /** * Ranked recall over the index, scoped by memory `kind`. * * Kinds never share a ranking, so high-volume prose can't bury code (§5): each kind is * FTS-gated and BM25-ranked only against its own kind, in a separate query. Three modes: * - **single kind** (`kind: "code"`) → a flat ranked `Hit[]`, default depth `n = 10`. * - **multiple kinds** (`kind: ["code", "doc"]`) → results grouped per kind, default `n = 5` each. * - **omitted** (`recall(q)`) → grouped over all known {@link KINDS}, default `n = 5` each — * the safe default for a CLI or an agent that didn't state a kind (never a flattened ranking). * * `n` caps results **per kind**; raise it to dig deeper. There is no hard cap and no * pagination — a larger `n` is a larger context, which is the caller's budget to manage. * * **Async** since slice 6: the embeddings tier embeds the query at call time. With embeddings off * (the default) no model is touched — the work is synchronous, just wrapped in a resolved promise. * * Every hit carries a `chunk` pointer — the best-matching function/section inside the file * (chunk-granular recall; `null` for written memory, where the row is the unit). Ranking stays * file-level and is unchanged by this: the pointer localizes, it never reorders. * * `log: false` skips the recall audit log. The log is a **demand signal** — anything that isn't * real demand (dashboards, CI checks, batch tooling, read-only-db consumers) must not write to it. * * `body: true` inlines each hit's content as `hit.body` (off by default — recall returns pointers, * not payloads). litectx owns this because *where the body lives is kind-dependent*: written memory * comes back VERBATIM; a file hit returns its localized chunk's indexed text, or the whole file when * nothing localized. Opt in when mounting litectx as a memory store or feeding an assembler. (A blob * hit — a byte-exact upload, R3 — has no text body: `body` is null; fetch its bytes with {@link get}.) * * `scope` (multis M3 R2 / M4) fences BOTH per-upload axes: direct doc/blob rows to `scope ∪ null-global` * (a chat sees its uploads + the global KB, never another chat's) AND `fact`/`episode` rows to that * tenant's owner ∪ global (multis M4 — one shared instance fences memory per tenant; {@link GLOBAL} = * the shared tier only). Unset = unscoped = the instance owner's view (sees everything when the instance * is ownerless; under `strictScope`, a memory- or doc-touching recall with no scope THROWS — fail-closed). * Code/file rows are repo-global, unaffected. Expired rows (R5 `expiresAt`) are always excluded. * * @overload * @param {string} query * @param {{ kind: string, n?: number, log?: boolean, body?: boolean, scope?: string | symbol }} opts * @returns {Promise} */ recall(query: string, opts: { kind: string; n?: number; log?: boolean; body?: boolean; scope?: string | symbol; }): Promise; /** * @overload * @param {string} query * @param {{ kind?: string[], n?: number, log?: boolean, body?: boolean, scope?: string | symbol }} [opts] * @returns {Promise>} */ recall(query: string, opts?: { kind?: string[]; n?: number; log?: boolean; body?: boolean; scope?: string | symbol; } | undefined): Promise>; /** * The ONE place the index is checked before a stored chunk body is served — shared by * `get(path, {startLine, endLine})` and `recall({ body: true })`, so the drift guard cannot be true * of one and false of the other. * * A chunk's line range only means anything against the file the index actually saw. Two ways that * breaks, and they are **not** the same thing: * - **`drifted`** — the file still exists but its content changed, so the range now spans *different * code*. This is the dangerous one: slicing anyway returns another symbol's body, silently. The * stored body is also now a lie (it would hand an editing caller back its own pre-edit code). * - **`missing`** — the file is gone from disk entirely. Nothing can be misread as anything else, * and the stored body is the only record left. Callers legitimately differ on whether they want it, * so this reports the fact rather than deciding: `recall` serves it (its chunk bodies are * deliberately index-truth and survive a deletion), `get` returns `null` (it is disk-truth — a * whole-file `get` of a deleted path already yields `text: null`). * * @param {string} path repo-relative * @param {number} startLine 0-based, inclusive * @param {number} endLine 0-based, inclusive * @returns {{ ok: true, body: string | null } | { ok: false, reason: "missing" | "drifted" }} * `ok` with `body: null` means the file is current but no chunk sits at that range. */ _chunkState(path: string, startLine: number, endLine: number): { ok: true; body: string | null; } | { ok: false; reason: "missing" | "drifted"; }; /** * Fill each hit's `body` with its content (RT-3 inline-body, the opt-in for `recall({ body: true })`). * Kind-routed — the reason this is litectx's job, not an adapter's: written memory (`source:'direct'`) * returns its VERBATIM stored text (the FTS body is a processed search surface, never the deliverable); * an indexed file hit returns its localized chunk's indexed body via {@link LiteCtx#_chunkState}, which * verifies the file has not changed since it was indexed — a hit whose file drifted gets `body: null` * rather than the pre-edit text (serving that silently is exactly the bug the chunk fetch exists to * kill, and it must not survive on this path either). When nothing localized, the whole file is read * fresh from disk (matching {@link get}'s freshness). `null` when the file is gone, drifted, or the id * is unknown. Mutates in place; bounded disk reads (≤ hits, file-kind only). Note: does NOT log a * fetch — body-fill is part of recall, not a `get`, so it never pollutes the demand signal. * @param {Omit[]} hits any hit-like row (recall's `Hit`, or * `recentMemory`'s unranked scoreless row) — reads `path`/`chunk`, writes `body`; `score` unused * @returns {Omit[]} */ _attachBodies(hits: Omit[]): Omit[]; /** * Attach parsed opaque `meta` (RT-3 #3) to written-memory hits, in place — the read half of the * sealed passthrough. One batched lookup; a hit whose path carries no metadata (every file, and * memory written without meta) is left untouched, so this is a no-op on pure-code recall. Parsed * here because the facade owns the JSON boundary; the store only ever holds/returns the raw string. * @param {Omit[]} hits any hit-like row (recall's `Hit`, or * `recentMemory`'s unranked scoreless row) — reads `path`, writes `meta`; `score` unused * @returns {Omit[]} */ _attachMeta(hits: Omit[]): Omit[]; /** * Rank one kind. Dual path (BM25 + spreading) when `qvec` is null; tri-hybrid when it's the query * vector — a wider BM25-gated pool re-ranked by `norm(dual) + weight·norm(cosine)`, then sliced to * `n`. Cosine runs on the pool plus at most {@link KNN_K} nominees, so it stays O(pool), never * O(corpus) for files. * * **Written kinds (slice 11): cosine also NOMINATES, not just re-ranks.** For `fact`/`episode`, * up to {@link KNN_K} stored vectors nearest the query are unioned into the pool before fusion — * so a zero-shared-term paraphrase ("money back" → a refunds fact) is reachable at all. Nominees * enter at the pool's score floor and compete on semantics alone; lexical hits keep their head * start. `code`/`doc` stay strictly gate-then-rerank (their queries share identifiers with their * answers, and their corpora are where a full scan would cost). * @param {string|null} match FTS expression, or null when the query has no usable terms * @param {string} kind * @param {number} n * @param {Float32Array|null} qvec * @param {{ scope?: string|null, seeAll?: boolean, now?: number|null, memOwner?: string|null, memSeeAll?: boolean }} [filter] R2 scope + R5 expiry (docs/blobs) + per-call owner fence (fact/episode, multis M4) * @returns {import("./store.js").Hit[]} */ _rankKind(match: string | null, kind: string, n: number, qvec: Float32Array | null, filter?: { scope?: string | null; seeAll?: boolean; now?: number | null; memOwner?: string | null; memSeeAll?: boolean; }): import("./store.js").Hit[]; /** * The **impact** view (§7): for a symbol, its blast radius and change-risk bucket. Computed on * demand — callees via a tree-sitter walk of the symbol's body, callers via an `rg -w` sweep * confirmed with tree-sitter; no LSP. Built around the §7.2 asymmetry (over-count safe, * under-count dangerous): connectivity may be overstated, but "isolated / low-risk" only ever * ships **hedged**. Returns `null` when the symbol isn't defined in the index. * * Requires `rg` (ripgrep) on PATH: without it the caller sweep can't run and a silent 0-caller * result would be a §7.2 false isolation, so this **throws** {@link RipgrepMissingError} rather * than under-count silently. (`recall()`/`index()`/`get()` don't use `rg` and are unaffected.) * * @param {string} symbol the symbol name to assess * @returns {Promise} * @throws {RipgrepMissingError} when ripgrep (`rg`) is not on PATH * @category impact * @when Gauge the blast radius / change-risk of a symbol before editing it. The model calls this directly via MCP. * @fails Throws `RipgrepMissingError` when `rg` is not on PATH (rather than silently under-counting to a false "isolated"); returns `null` when the symbol isn't in the index. * @signature liteCtx.impact(symbol: string) => Promise * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const view = await ctx.impact('parseConfig') * if (view) console.log(view.risk, view.callers.length) // 'low' | 'med' | 'high' */ impact(symbol: string): Promise; /** * Describe one graph node — the substrate accessor (`getNode` returns STRUCTURE; `get` returns the * body). The graph is first-class public API: recall and impact are *views* over it, and so are the * future codegraph/contextgraph. Kind-agnostic — an indexed file's repo-relative path returns a * file node (its symbols as `chunks` + exact import-edge counts), a written-memory id returns a * zero-chunk, zero-edge node. Edge counts are the persisted `import` graph (exact); call * relationships are `impact()`'s on-demand job, never drawn as graph edges. Sync; `null` if unknown. * @param {string} id an indexed file's repo-relative path, or a written-memory id * @returns {import("./store.js").GraphNode | null} */ getNode(id: string): import("./store.js").GraphNode | null; /** * Walk the edge graph from `id` — the substrate navigator. BFS over persisted `import` edges (the * only persisted type; `call`/blast is `impact()`). `dir`: "out" = what `id` imports, "in" = what * imports it, "both" = the neighbourhood (default). `hops` is the depth (default 1, hard-capped at * 3 — navigation, not ranking; `truncated` flags the cap). Deduped, nearest-hop-wins, excludes the * seed. `edge` is generic so future non-code edges slot in unchanged. Sync. * @param {string} id * @param {{ edge?: string, dir?: "out"|"in"|"both", hops?: number }} [opts] * @returns {{ items: import("./store.js").RelatedNode[], truncated: boolean }} */ related(id: string, opts?: { edge?: string; dir?: "out" | "in" | "both"; hops?: number; }): { items: import("./store.js").RelatedNode[]; truncated: boolean; }; /** * Fetch one stored item's full record by id — the body-access counterpart to `recall` (slice 9). * Recall returns ranked pointers (paths/ids); `get` returns the thing itself. Any id works: * a written-memory id (`"fact:auth-uses-jwt"`) returns the text exactly as remembered, and an * indexed file's repo-relative path (`"src/auth.js"`) returns the file read fresh from disk — * the index stores the *searchable surface*, not a copy of your files. `text` is `null` only * when an indexed file has vanished from disk since the last `index()` pass. * * Each `get` appends an `action: 'fetch'` row to the audit log — a **tagged weak signal**, kept * apart from recall's demand signal: you fetch what recall just returned, so counting fetches as * demand would double-count every retrieval (the fetch-toll). Nothing reads the tag yet; it earns * weight (if any) at the action-signal bench. `log: false` opts out, same as `recall`. * * A **blob** (a byte-exact upload, R3) returns its original bytes in `bytes` with `text: null` — the * round-trip is byte-identical. An **expired** row (R5) returns `null`, exactly like recall hides it. * * `scope` (multis M3 R2) **fences the direct handle** like `recall({scope})` fences discovery: a `get` * for a doc/blob tagged with a *different* scope returns `null` (a global/null-scope row stays visible * to every scope; fact/episode/file rows are unaffected). This is the load-bearing half of "one customer * never sees another's" — recall alone fences search, but ids can be guessed, so a customer-reachable * fetch must pass the requesting scope. Pass {@link GLOBAL} to fetch only shared-tier rows. * * Under `strictScope` (multis M3 fail-closed ask), a **bare `get(id)` THROWS** — because `get` can't * know a guessable id's scope without fetching it, so a missing scope on a strict store is a leak, not * a convenience. Pass a tenant `scope` or `GLOBAL` to fetch. With strictScope off, `get(id)` is unfenced * by id (the legacy behaviour), exactly as before. * * **Fetching ONE chunk instead of the whole file.** Pass the `startLine`/`endLine` from a recall * hit's {@link ChunkRef} and `text` is that chunk's body alone — the answer a pointer promised, * without dragging its whole file through context. The lines are a *handle you hand back*, never a * range you compute: they mean nothing except against the file the index actually saw. * * Which is why this is gated on the file's content hash. If the file changed since it was indexed, * those line numbers now describe *different code* — slicing by them returns another symbol's body, * silently, with no error. So a drifted file throws {@link StalePointerError} rather than guess. When * the hash matches, the indexed chunk and the live file are identical by construction, so the stored * body is served verbatim (no re-parse). Symbol names deliberately play no part: they are duplicated * (`recall` exists on both `LiteCtx` and `ScopedView`), renamed, and absent on ~40% of chunks — the * hash is the only anchor that holds for every chunk in every language. * * Recover by re-running {@link index} (a no-op pass is ~6ms) and re-`recall`ing for a fresh pointer. * A line range that matches no chunk returns `null` — never a silent fallback to the whole file. * * Sync (no embedder involved). Returns `null` for an unknown id. * * @param {string} id a written-memory id, a stashed payload's id, or an indexed file's repo-relative path * @param {{ log?: boolean, scope?: string | symbol, startLine?: number, endLine?: number }} [opts] * `startLine`/`endLine` (0-based, inclusive) — a recall hit's `chunk` range; both or neither * @returns {Item | null} * @throws {StalePointerError} when a chunk is requested from a file that changed since indexing * @category recall * @when Fetch the full body behind a recall hit — a whole file/fact, or one chunk (code + its docstring) by line range. The model calls this directly via MCP. * @fails Throws `StalePointerError` when a chunk range is requested from a file that changed since indexing (refuses rather than return different code); returns `null` for an unknown id or a range matching no chunk. * @signature liteCtx.get(id: string, opts?: { startLine?, endLine?, scope? }) => Item | null * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const [hit] = await ctx.recall('backoff', { kind: 'code' }) * // echo the hit's chunk range back as the address — nothing widens it * const chunk = ctx.get(hit.path, hit.chunk) */ get(id: string, opts?: { log?: boolean; scope?: string | symbol; startLine?: number; endLine?: number; }): Item | null; /** * Write a directly-authored memory — a `fact`/`episode`/`doc` with no file behind it (§3.2). This * is the write counterpart to `index()`: knowledge that isn't a file enters here. Upsert by `id` * (also the update/forget handle — recommend namespacing it, e.g. `"fact:auth-uses-jwt"`). Stored * **whole** (never chunked). Written rows are `source='direct'`, so `index()` never reconciles them * away; recall finds them like any other kind. Embeds the text when the embeddings tier is on. * * **Upsert is tenant-fenced (multis M4 W4): the `(scope, id)` pair is the supersede key.** Re- * `remember`ing the same `id` under the same `scope` REPLACES the prior value in place (one row, the * latest — so a restated fact never piles up). The SAME `id` under a DIFFERENT `scope` is a SEPARATE * row: one tenant can never overwrite another's memory by id (the row's physical key folds in the * owner; the public `id` you get back is unchanged). Detecting "same subject, new value" is the * caller's job — litectx supplies the keyed, fenced upsert mechanism. * * @param {string} id caller key / identity (lands in the row's `path`) * @param {string} text the content * @param {{ kind?: string, format?: string, by?: string, occurredAt?: number, meta?: Record, injectionRisk?: "low"|"medium"|"high", scope?: string|symbol|null, expiresAt?: number|null }} [opts] * `kind` ∈ {fact, episode, doc} (default `fact`); `by` = provenance `"human"|"agent"` (default * `"agent"`); `injectionRisk` = OPTIONAL guardrails shape flag forwarded to a wired `writeGate` * (litectx core never computes it — a guardrails tier sets it; ignored when no `writeGate`); * `occurredAt` = episode timestamp (epoch ms, default now; ignored for non-episodes); * `format` defaults to `md` for docs, `text` otherwise. `meta` = an opaque caller dict (RT-3 #3) * stored verbatim and returned untouched by `get`/`recall` — small structured tags ({sessionId, * tag, …}), NEVER searched or ranked; park large payloads in `stash`, not here. Re-`remember`ing * without `meta` clears any prior meta. `scope` routes by kind: on a `doc` it tags the row's recall * scope (multis M3 R2; with `expiresAt`/R5 retention, both default null = global/forever); on a * `fact`/`episode` it sets the per-call `mem_scope.owner` (multis M4 — a tenant string fences that * memory to one tenant on a shared instance, {@link GLOBAL} writes the shared tier, omitted uses the * instance `owner`; under `strictScope`, omitted THROWS). Prefer a bound {@link scoped} view so the * scope can't be forgotten. Doc `scope` usually set via {@link ingest}, not here directly. * @returns {Promise} * @category memory * @when Persist a fact/episode/doc so it survives across sessions and is recallable by meaning. The model calls this directly via MCP. * @fails Throws when `kind` is not one of fact/episode/doc; under `strictScope`, throws when `scope` is omitted. Re-`remember`ing the same `(scope, id)` supersedes in place (no duplicate row). * @signature liteCtx.remember(id: string, text: string, opts?: { kind?, by?, occurredAt?, scope? }) => Promise * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * await ctx.remember('pref-theme', 'user prefers dark mode', { kind: 'fact', by: 'human' }) * await ctx.remember('ep-1', 'deploy failed on missing env var', { kind: 'episode' }) */ remember(id: string, text: string, opts?: { kind?: string; format?: string; by?: string; occurredAt?: number; meta?: Record; injectionRisk?: "low" | "medium" | "high"; scope?: string | symbol | null; expiresAt?: number | null; }): Promise; /** * Forget directly-written memory (§3.2). Pass an `id` to drop one item, or a query for bulk * invalidation. **Only ever removes `source='direct'` rows** — an indexed file is never touched. * **Memory-only:** a stash is not memory — clean parked payloads with {@link evict} (a `forget`-by-id * no longer reaches the stash table). Returns the count removed. * * Query shapes: * - `{ id }` / `{ idPrefix }` — **precise** owner-blind delete by caller key. `id` drops one row; * `idPrefix` drops a base id and all its `#` segments (the clean-re-ingest handle for a * multi-segment {@link ingest} doc). Explicit by-key targets, like the string `forget('id')` form — * **not** subject to the `strictScope` throw (that guards omission-based blind wipes, not by-id * deletes). Ids are guessable, so a shared-store host should namespace them. * - `{ kind?, by? }` — owner-BLIND **bulk** delete across every tenant (legacy; unchanged). Reaches all * `mem_scope.owner`s — only safe on a single-tenant instance. Under `strictScope` this THROWS. * - `{ scope, kind? }` — **tenant-fenced** (multis M4): deletes only that owner's `fact`+`episode` * rows, the delete-side mirror of the {@link recall} owner fence. A tenant string → `mem_scope.owner * = scope`; {@link GLOBAL} → the shared tier (`owner IS NULL`) ONLY — never a tenant's rows. Prefer * the bound {@link ScopedView#forget}. Mem-axis only: a tenant's `doc`/blob uploads (separate * `doc_scope` axis), other tenants' rows, and the stash are untouched. Under `strictScope` a * scope-less memory forget THROWS — a tenant-blind wipe is unexpressible by omission. * - `{ scope, id }` / `{ scope, idPrefix }` — **tenant-fenced delete-by-key** (Feature B, 0.27.0): * `{ id }`/`{ idPrefix }` COMBINE with the fence to drop one row / one id's segments for exactly that * owner — the delete-side mirror of the W4 `(scope, id)` upsert. The fence is STRUCTURAL (the * owner-qualified physical key): a foreign tenant's id matches nothing → **0**, the fence not * id-matching decides. `{ scope, by }` still THROWS (`by` is owner-blind provenance; combined with a * fence it is the omission-blind footgun — use base `{ by }` for an owner-blind provenance delete). * * @param {string | { kind?: string, by?: string, scope?: string | symbol, id?: string, idPrefix?: string }} sel * @returns {number} * @category memory * @when Delete written memory — by id, by kind, or tenant-fenced by scope (the correct compliance/erasure primitive: it deletes now). The model calls this directly via MCP. * @fails Under `strictScope`, a scope-less memory forget throws (a tenant-blind wipe is unexpressible by omission); combining `{ scope, by }` throws (owner-blind provenance + a fence is the omission footgun). Mem-axis only — never docs/blob/stash. * @signature liteCtx.forget(sel: string | { id?, kind?, by?, scope?, idPrefix? }) => number * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * ctx.forget('pref-theme') // one id * ctx.forget({ kind: 'episode' }) // all episodes * ctx.scoped('tenant:acme').forget({ scope: 'tenant:acme' }) // right-to-erasure for one tenant */ forget(sel: string | { kind?: string; by?: string; scope?: string | symbol; id?: string; idPrefix?: string; }): number; /** * Ingest an uploaded file (bytes + filename) — the third ingest path, distinct from {@link index} * (sweeps a disk root) and {@link remember} (stores text whole, unchunked). Built for transient chat * uploads (a buffer, not a repo file). Routed by extension (multis M3): * * - **md / pdf / docx** → converted to markdown, split into segments, each written as its own * `source='direct'` `doc` row — so it ranks alongside `md` docs in `recall(q,{kind:'doc'})`, * survives every `index()` pass, and carries its `format` ("md"|"pdf"|"docx") under `kind='doc'`. * - **txt / text / log / csv** → already plaintext (no parser, no peer dep): packed into * passage-sized segments (blank-line paragraphs, else lines), stored exactly like the above with * `format` ("txt"|"log"|"csv"; "text"→"txt"). CSV is chunked as raw text (no columnar parse). * - **everything else** (xlsx / xml / code / binary) → stored BYTE-EXACT as a blob; its * **filename** is indexed for recall but the body is never parsed/chunked, and {@link get} returns * the original bytes. Getting body-search for those types is the consumer's opt-in (send a chunkable type). * * Untrusted input is BOUNDED: oversized / over-page / slow / corrupt / encrypted / no-text inputs * throw a clear, specific error and write NOTHING (the index is left intact). The two parsers * (`pdfjs-dist`, `mammoth`) are optional peer deps, lazy-loaded on first chunkable ingest (a blob * needs neither). Every row may carry a `scope` (R2 — recall fences `scope ∪ null-global`) and an * `expiresAt` (R5 — excluded from recall/get once past, reclaimed by {@link purge}). * * **A wired `writeGate` screens the CHUNKABLE path only (per segment, via {@link remember}); a BLOB * write is NOT gated.** This is deliberate: the gate judges searchable *text* for injection-risk, and a * blob has none — its bytes are opaque and never reach an LLM (the only path that turns blob content * into context is converting + sending as md/pdf/docx, which IS the gated chunked route). Screen * uploads at the call site (size/type/AV) and treat retrieved bytes as untrusted on egress; don't rely * on `writeGate` for blobs. * * Re-ingesting the same `id` is an upsert: prior segments/blob of that document are dropped first, so * a shorter (or format-changed) re-ingest never leaves orphans. * * @param {Uint8Array} buffer the file bytes (e.g. a chat upload) * @param {{ filename?: string, format?: string, id?: string, scope?: string|symbol|null, expiresAt?: number|null, meta?: Record, maxSize?: number, maxPages?: number, parseTimeoutMs?: number }} [opts] * `filename` drives extension routing; `format` overrides it; `id` = stable base id (else derived * from the filename); `scope`/`expiresAt` = per-upload recall scope + retention (default null = * global/forever); `meta` = opaque passthrough; `maxSize`/`maxPages`/`parseTimeoutMs` = the * untrusted-input bounds (defaults 10 MB / 2000 / 30 s; `maxSize` also caps a blob). * @returns {Promise<{ id: string, kind: "doc", format: string, mode: "chunked" | "blob", chunks: number }>} * @category ingest * @when Store an uploaded document (pdf/docx/md/txt/csv → chunked + searchable; anything else → byte-exact blob) with an optional per-upload scope. * @fails Throws when a required optional peer dep is missing (pdf → `pdfjs-dist`, docx → `mammoth`) or input exceeds `maxSize`/`maxPages`; under `strictScope`, throws when `scope` is omitted. * @signature liteCtx.ingest(buffer: Uint8Array, opts?: { filename?, format?, id?, scope?, expiresAt? }) => Promise<{ id, kind, format, mode, chunks }> * @example * import { readFileSync } from 'node:fs' * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const res = await ctx.ingest(readFileSync('spec.pdf'), { filename: 'spec.pdf', scope: 'project:x' }) * // res.mode === 'chunked', res.chunks > 0 */ ingest(buffer: Uint8Array, opts?: { filename?: string; format?: string; id?: string; scope?: string | symbol | null; expiresAt?: number | null; meta?: Record; maxSize?: number; maxPages?: number; parseTimeoutMs?: number; }): Promise<{ id: string; kind: "doc"; format: string; mode: "chunked" | "blob"; chunks: number; }>; /** * Reclaim expired uploads (multis M3 R5) — the retention sweep's mechanism. Every direct doc/blob row * whose `expiresAt` has passed `now` (default `Date.now()`) is deleted and its storage (including the * byte-exact blob) reclaimed, leaving no orphans. The CONSUMER owns the schedule (when/how often); * litectx owns the delete. Note recall/get already EXCLUDE expired rows the instant they expire — so * this is a storage-reclamation pass, not a correctness gate. Returns the number of rows reclaimed. * @param {{ now?: number }} [opts] `now` = the cutoff (epoch ms); rows with `expiresAt <= now` go * @returns {number} * @category ingest * @when Reclaim storage from expired doc/blob uploads — a scheduled retention sweep (recall already excludes expired rows live). * @fails Does not throw; returns the count of rows reclaimed (0 when nothing has expired). * @signature liteCtx.purge(opts?: { now?: number }) => number * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const reclaimed = ctx.purge() // rows whose expiresAt has passed */ purge(opts?: { now?: number; }): number; /** * Park a payload in the keyed agent-context store and return its handle — the durable half of * **restorable compression** (R-C4). The caller drops a large payload (a tool result, a fetched * page, a file dump) from its context window, keeping only the cheap handle (`id`); {@link get} * rehydrates the full text on demand and {@link evict} drops it when truly done. A stash is **not * memory**: it is never indexed and never recalled (it lives in no FTS table, so recall can't * surface it on any kind) and never auto-pruned — it is addressable only by exact `id`. Upsert by * `id` (also the rehydrate/evict handle; namespace it, e.g. `"stash:toolresult-42"`). Sync — a * stash is never embedded (it isn't meaning-searchable), which is the whole point. * * @param {string} id caller-chosen handle / identity * @param {string} text the payload to park * @returns {void} * @category CE * @when Drop a large payload (tool result, page dump) from the context window, keeping only a cheap handle to rehydrate later. API-only — adopter code chooses this, never a model verb. * @fails Does not throw; upserts by `id` (a re-stash replaces). A stash is never indexed or recalled — reachable only by exact `id` via `get`/`peek`/`evict`. * @signature liteCtx.stash(id: string, text: string) => void * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * ctx.stash('stash:toolresult-42', hugeToolOutput) * // later: const full = ctx.get('stash:toolresult-42') */ stash(id: string, text: string): void; /** * Peek a stashed payload (R-I3 handle / lazy-load): a cheap **head+tail** preview of a parked blob * *without* rehydrating it — the read-half of {@link stash}. Where {@link get} pays the whole * payload's tokens back, `peek` returns only `{ id, bytes, head, tail, createdAt, truncated }`: a * fixed-length prefix *and suffix* (the conclusion — exit code, failing frame, closing structure — * lives at the end), the true byte size, the parked-at time, and whether a middle span is elided * (`tail` is empty when the head already holds the whole payload). The agent reasons over the handle * and calls {@link get} to load the full body *only if it decides it needs it*. The win is the * **bounded result** — only ~head+tail bytes return to the caller, never the whole blob, so the * payload stays out of its context/token budget. (Not a DB-time win: SQLite reads the column to slice * it, so peek's local compute scales with payload size — `get` it directly if you'll load it anyway.) * **Stash-only**: recall owns ranked retrieval over memory; a stash is a dumb keyed blob, so `peek` * carries no weights and no ranking. Null for an unknown id. * * @param {string} id a stashed payload's id (as passed to {@link stash}) * @returns {{ id: string, bytes: number, head: string, tail: string, createdAt: number, truncated: boolean } | null} * @category CE * @when Preview a stashed payload's head+tail without paying its full tokens — decide whether to rehydrate. API-only. * @fails Does not throw; returns `null` for an unknown id. * @signature liteCtx.peek(id: string) => { id, bytes, head, tail, createdAt, truncated } | null * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const p = ctx.peek('stash:toolresult-42') * if (p?.truncated) { const full = ctx.get('stash:toolresult-42') } */ peek(id: string): { id: string; bytes: number; head: string; tail: string; createdAt: number; truncated: boolean; } | null; /** * Evict parked stashes (R-C4 housekeeping) — the runtime's stash deleter, the cleanup half of * {@link stash}. **API-only** (§10.5: a stash is orchestration plumbing, never a model verb) and * **stash-only**: unlike {@link forget} (which invalidates durable memory), `evict` can never reach a * fact/episode — a bulk age/size sweep is safe by construction (only the `stash` table is touched). * Pass an `id` to drop one parked payload, or a policy: `{ olderThan }` (epoch-ms floor — evict anything * parked before it) and/or `{ maxCount }` (keep only the newest N, evict the rest). When both are given * they apply in turn (age first, then count). The runtime owns the *policy* (which/when); litectx owns * the *delete*. Returns the count removed. * * @param {string | { olderThan?: number, maxCount?: number }} sel * @returns {number} * @category CE * @when Drop parked stashes when done — one id, or a bulk age/size policy. API-only; stash-only (never reaches memory). * @fails Does not throw; returns the count removed. Cannot touch a fact/episode by construction — only the stash table. * @signature liteCtx.evict(sel: string | { olderThan?, maxCount? }) => number * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * ctx.evict('stash:toolresult-42') // one payload * ctx.evict({ maxCount: 100 }) // keep newest 100 */ evict(sel: string | { olderThan?: number; maxCount?: number; }): number; /** * Human-in-the-loop review candidates (§3.2): agent-asserted facts whose recall count has crossed * `threshold`. The intended loop is the **consumer's** — it shows each candidate to a human who * either validates it (re-`remember(id, text, { by: "human" })`, promoting it to durable/high-trust) * or invalidates it (`forget(id)`). litectx supplies only the candidate set + those two actions; * the threshold and the review flow are the consumer's. Review is earned by use, so a human never * sees every agent fact — only the ones that proved useful. * * `scope` (multis M4) fences the candidate set to one tenant on a shared instance, exactly like * `recall({ kind: 'fact', scope })`: a tenant string → that owner's facts only; {@link GLOBAL} → the * shared tier only; omitted → the instance owner (under `strictScope`, omitted THROWS — fail-closed). * Prefer the bound {@link ScopedView#reviewCandidates} so the scope can't be forgotten. * @param {number} [threshold=5] * @param {{ scope?: string | symbol }} [opts] * @returns {{ path: string, hits: number }[]} * @category memory * @when Surface agent-asserted facts that proved useful (recalled ≥ threshold) for a human to validate or discard. * @fails Under `strictScope`, throws when `scope` is omitted; otherwise returns `[]` when nothing crossed the threshold. * @signature liteCtx.reviewCandidates(threshold?: number, opts?: { scope? }) => { path, hits }[] * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * for (const c of ctx.reviewCandidates(5)) { * // show c.path to a human → validate (re-remember by:'human') or forget * } */ reviewCandidates(threshold?: number, opts?: { scope?: string | symbol; }): { path: string; hits: number; }[]; /** * Episode promotion candidates (§14 #4 view #4, slice 5b) — the agent-side first rung of the * promotion ladder. Returns agent-written `episode`s recalled at least `threshold` times within the * rolling active window (`episodeWindowDays`, default 30 — older episodes have decayed out and * self-prune on the next episode write). The window floor here is the SAME knob the prune uses, so a * candidate is never surfaced after it would have been pruned. The intended loop is the **consumer's agent**: read each * candidate (`get(id)`), distil a durable `fact` via `remember(id, text, { kind: "fact", by: * "agent" })` — which then rides the existing `reviewCandidates(5)` → human-validate path. litectx * **flags, never summarizes** (no extraction LLM): it supplies the trigger; the agent writes the * fact. The count gates **distillation, never ranking** — a frequently-recalled episode does not * rank higher (that would be the feedback loop §4 forbids). Threshold defaults higher than facts' * review (10 vs 5): episodes are noisier and more numerous. * * Unlike `reviewCandidates` (where a human re-`remember` flips provenance and drops the row), * distilling does not remove an episode — it stays a candidate until it ages out of the window (or * the consumer `forget`s it post-distillation). Re-distilling is harmless: the agent's fact id is a * stable handle, so a second pass upserts the same fact rather than duplicating it. * * `scope` (multis M4) fences the candidate set to one tenant on a shared instance, exactly like * {@link reviewCandidates} (tenant string → that owner; {@link GLOBAL} → shared tier; omitted → the * instance owner, or THROW under `strictScope`). Prefer the bound {@link ScopedView#promotionCandidates}. * @param {number} [threshold=10] * @param {{ scope?: string | symbol }} [opts] * @returns {{ path: string, hits: number }[]} * @category memory * @when Find episodes recalled often enough to distil into durable facts — the agent-side rung of the promotion ladder. Exposed to the model via MCP. * @fails Under `strictScope`, throws when `scope` is omitted; otherwise returns `[]`. litectx flags candidates, never summarizes them (no extraction LLM). * @signature liteCtx.promotionCandidates(threshold?: number, opts?: { scope? }) => { path, hits }[] * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * for (const c of ctx.promotionCandidates(10)) { * const ep = ctx.get(c.path) // read it, distil, then: * // await ctx.remember(factId, distilled, { kind: 'fact', by: 'agent' }) * } */ promotionCandidates(threshold?: number, opts?: { scope?: string | symbol; }): { path: string; hits: number; }[]; /** * "What was I working on" (§14 #4 view #3, slice 5a): the code/doc chunks litectx witnessed edited * most recently — newest first — inside a recency window. Each `index()` pass that sees a chunk's * body change (added or modified vs the stored node) logs an edit; a cold first/`force` build logs * nothing (loading isn't editing), so this stays empty until real edits are observed. * * An **isolated** read by design: it reads the witnessed edit log and never the ranking path, so it * cannot regress recall — the edit→recall re-rank ships at zero (falsified repo-dependent, §14 #4). * The edit signal's home is here (next-use / "where was I"), not in search scores. * * @param {{ days?: number, since?: number, limit?: number }} [opts] * `since` (epoch ms) sets the window floor explicitly; otherwise `days` back from now (default 7). * `limit` caps rows (default 20). * @returns {{ id: string, symbol: string|null, kind: string, lastEditedAt: number, edits: number }[]} * `id` is the chunk's file path (feed it to `get`); `symbol` localizes within the file (null for a * file's anonymous chunks, collapsed to one row); `edits` is how many index passes (sessions) * changed it in the window; sorted by `lastEditedAt` desc. * @category memory * @when Answer "what was I working on" — the code/doc chunks litectx witnessed edited most recently. * @fails Does not throw; empty until real edits are observed (a cold/`force` first build logs none — loading isn't editing). * @signature liteCtx.recentActivity(opts?: { days?, since?, limit? }) => { id, symbol, kind, lastEditedAt, edits }[] * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const recent = ctx.recentActivity({ days: 3 }) // newest edits first */ recentActivity(opts?: { days?: number; since?: number; limit?: number; }): { id: string; symbol: string | null; kind: string; lastEditedAt: number; edits: number; }[]; /** * Recent written memory, newest first — the recency sibling of `recall` for the empty-FTS-match * fallback. When a query carries no usable term (an all-stopword "what did I say"), `recall` returns * `[]` (no relevance to rank on); call `recentMemory` to ground the agent on its latest memory for the * scope instead. **The consumer owns the policy** (when to fall back); litectx owns the mechanism — so * it is a separate verb, not a `recall` flag (which would mix recency into a relevance ranking and let * it pollute the demand signal). * * **Two axes, picked by `kind` (multis M3 doc → M4 R3 memory):** * - `kind` omitted or `'doc'` → the DOC axis (default; byte-identical to the original verb). Direct * `doc` rows (written via {@link ingest}/{@link remember}; blobs by filename), ordered by write time, * fenced to `scope ∪ null-global` on `doc_scope.scope`, **expiry-aware** (R5 — expired excluded). Under * `strictScope` a missing `scope` THROWS (same as `recall({kind:'doc'})`/`get`/`ingest`). * - `kind: 'fact' | 'episode'` (or an array of them) → the MEMORY axis (R3). `fact`/`episode` rows for * the tenant, ordered by **`occurred_at` (episodes) / `created_at` (facts)** newest-first, fenced on * `mem_scope.owner` (`tenant ∪ shared`, session-aware) — the SAME fence as `recall`/`get` on memory, so * it can never surface another tenant's rows. Under `strictScope` a missing `scope` THROWS. No per-row * expiry on this axis (episode staleness is the rolling-window prune — `episodeWindowDays`, default * 30; pruned rows are already gone). * * The two axes resolve scope differently (`doc_scope` vs `mem_scope.owner`), so a single call mixes ONE * axis only: `doc` with `fact`/`episode` together THROWS — call once per axis (they are distinct stores). * Pass {@link GLOBAL} for the shared tier on either. * * Each row is a `recall`-shaped hit (`{ path, kind, format }`) plus `createdAt` (epoch ms; `null` if * written before the column shipped — sorted last), `occurredAt` on memory rows (epoch ms; `null` for a * fact), the opaque `meta` when present (where the caller parks `role`/turn markers for faithful history * reconstruction), and `body` when `body:true` (VERBATIM stored text; `null` for a blob — fetch its bytes * with {@link get}). It does NOT log a recall: recency is not query-demand, so counting it would inflate * `use` for whatever is newest. * * @param {{ scope?: string | symbol, kind?: string | string[], n?: number, body?: boolean }} [opts] * @returns {(Omit & { createdAt: number|null, occurredAt?: number|null })[]} * @category memory * @when Ground on the latest written memory when a query has no rankable term (all-stopword "what did I say") and `recall` returns `[]`. Exposed to the model via MCP. * @fails Under `strictScope`, throws when `scope` is omitted; throws if one call mixes the doc axis with fact/episode (distinct scope stores). Logs no recall (recency is not demand). * @signature liteCtx.recentMemory(opts?: { kind?, scope?, n?, body? }) => (Hit & { createdAt, occurredAt? })[] * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const latest = ctx.recentMemory({ kind: 'episode', n: 5, body: true }) */ recentMemory(opts?: { scope?: string | symbol; kind?: string | string[]; n?: number; body?: boolean; }): (Omit & { createdAt: number | null; occurredAt?: number | null; })[]; /** * Exhaustive, scope-aware, deterministic, paginated read of ONE memory kind — the structural opposite * of {@link recall} (FTS-gated + ranked + capped, so it MISSES the tail by design; measured 0.05–0.24 * recall on a "how many" ask). For batch "count / all of them" operations (bareagent RLM `scan` over * accrued memory): no query, no ranking, no embedder — an ordered `rowid` table read that runs * identically with the embeddings tier on or off. **API-only, never model-callable** (like `stash`): * the deterministic scan orchestrator is the only caller — an LLM must never say "dump everything". * * Unioning every page (`offset` 0 until `nextOffset === null`) yields EXACTLY the rows of `kind` * visible to this scope — gapless, no dupes (the load-bearing property). Scope-fenced on * `mem_scope.owner` via the SAME resolver {@link recall}/{@link count} use, so a scoped instance sees * its own ∪ shared only (pass {@link GLOBAL} for the shared tier; under `strictScope` a missing scope * THROWS). `total` is the scoped count of the kind; `nextOffset = offset + items.length` while rows * remain, else `null`. `body:true` inlines the VERBATIM stored text (same path as `recall({body:true})`). * Writes NO recall audit log — a full scan is batch tooling, not user demand. * * v1 is the **memory axis only** (`fact`/`episode` — the scan targets). `code`/`doc` enumeration (a * second expiry-aware `doc_scope` path, file-granular) lands when a codebase-scan consumer exists. * * Async to match the `recall` family's signature (no embedder is actually touched). * * @param {{ kind: 'fact'|'episode', scope?: string | symbol, offset?: number, limit?: number, body?: boolean }} opts * @returns {Promise<{ items: EnumItem[], total: number, offset: number, nextOffset: number | null }>} * @category memory * @when Read ALL memory of one kind, gapless + paginated, for "count / all of them" questions recall can't answer (it's ranked + capped). API-only. * @fails Throws when `kind` isn't fact/episode, or `offset`/`limit` are not valid non-negative/positive integers; under `strictScope`, throws when `scope` is omitted. * @signature liteCtx.enumerate(opts: { kind: 'fact'|'episode', scope?, offset?, limit?, body? }) => Promise<{ items, total, offset, nextOffset }> * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * let offset = 0, all = [] * do { const p = await ctx.enumerate({ kind: 'fact', offset }); all.push(...p.items); offset = p.nextOffset } while (offset !== null) */ enumerate(opts: { kind: "fact" | "episode"; scope?: string | symbol; offset?: number; limit?: number; body?: boolean; }): Promise<{ items: EnumItem[]; total: number; offset: number; nextOffset: number | null; }>; /** @returns {number} total stored items — indexed documents + written memory */ size(): number; /** * Count a tenant's written memory by kind (multis M4 O1) — for "you have N facts / M episodes / K docs * here" surfaces (`/memory`, `/docs`) without pulling rows. Tenant-fenced + expiry-aware on the SAME * predicates as {@link recall}/{@link recentMemory}: `fact`/`episode` fence on `mem_scope.owner`, * direct `doc` on `doc_scope.scope` (expired excluded). Unlike `recentMemory`, a count is purely * additive, so `kind` MAY span both axes in one call — each kind is counted under its own fence and * the totals summed (a tenant's facts ∪ shared + its live docs ∪ shared). `scope` resolves once per * axis touched (so under `strictScope` a missing scope THROWS exactly as the read verbs do); pass * {@link GLOBAL} for the shared tier. `code`/file rows are repo-global — out of scope here (use * {@link size} for the grand total). * @param {{ scope?: string | symbol, kind?: string | string[] }} [opts] `kind` ⊆ {fact, episode, doc}; * omitted → all three (the tenant's whole writable memory). Single kind or array. * @returns {number} * @category memory * @when Report how much memory a tenant holds ("N facts / M episodes / K docs") without pulling rows. * @fails Throws when `kind` isn't a subset of fact/episode/doc; under `strictScope`, throws when `scope` is omitted. * @signature liteCtx.count(opts?: { scope?, kind? }) => number * @example * import { LiteCtx } from 'litectx' * const ctx = new LiteCtx({ root: process.cwd() }) * const facts = ctx.count({ kind: 'fact' }) * const all = ctx.count() // fact + episode + doc for this tenant */ count(opts?: { scope?: string | symbol; kind?: string | string[]; }): number; close(): void; } /** * A scope-bound facade over a {@link LiteCtx} (multis M3 fail-closed ask). Created by {@link LiteCtx#scoped}, * never directly. Every doc-axis verb carries the view's bound scope automatically; there is no per-call * `scope` to pass, so it cannot be forgotten — the structural fix that mirrors how the memory axis binds * `owner`/`session` once on the instance. The bound scope is final: any `scope` in a call's opts is ignored. */ export class ScopedView { /** @param {LiteCtx} ctx the underlying instance @param {string | symbol} scope the bound scope (string | GLOBAL) */ constructor(ctx: LiteCtx, scope: string | symbol); /** @type {LiteCtx} */ _ctx: LiteCtx; /** @type {string | symbol} */ _scope: string | symbol; /** Scope-bound {@link LiteCtx#recall}. @param {string} query @param {{ kind?: string | string[], n?: number, log?: boolean, body?: boolean }} [opts] */ recall(query: string, opts?: { kind?: string | string[]; n?: number; log?: boolean; body?: boolean; }): Promise | Promise>; /** * Scope-bound {@link LiteCtx#get}. Carries the chunk fetch (`startLine`/`endLine`) through unchanged. * @param {string} id * @param {{ log?: boolean, startLine?: number, endLine?: number }} [opts] */ get(id: string, opts?: { log?: boolean; startLine?: number; endLine?: number; }): Item | null; /** Scope-bound {@link LiteCtx#recentMemory}. @param {{ kind?: string | string[], n?: number, body?: boolean }} [opts] */ recentMemory(opts?: { kind?: string | string[]; n?: number; body?: boolean; }): (Omit & { createdAt: number | null; occurredAt?: number | null; })[]; /** Scope-bound {@link LiteCtx#count} (multis M4 O1). @param {{ kind?: string | string[] }} [opts] */ count(opts?: { kind?: string | string[]; }): number; /** Scope-bound {@link LiteCtx#enumerate} (bareagent RLM scan). @param {{ kind: 'fact'|'episode', offset?: number, limit?: number, body?: boolean }} opts */ enumerate(opts: { kind: "fact" | "episode"; offset?: number; limit?: number; body?: boolean; }): Promise<{ items: EnumItem[]; total: number; offset: number; nextOffset: number | null; }>; /** Scope-bound {@link LiteCtx#reviewCandidates} (multis M4). @param {number} [threshold=5] */ reviewCandidates(threshold?: number): { path: string; hits: number; }[]; /** Scope-bound {@link LiteCtx#promotionCandidates} (multis M4). @param {number} [threshold=10] */ promotionCandidates(threshold?: number): { path: string; hits: number; }[]; /** Scope-bound {@link LiteCtx#ingest}. @param {Uint8Array} buffer @param {{ filename?: string, format?: string, id?: string, expiresAt?: number|null, meta?: Record, maxSize?: number, maxPages?: number, parseTimeoutMs?: number }} [opts] */ ingest(buffer: Uint8Array, opts?: { filename?: string; format?: string; id?: string; expiresAt?: number | null; meta?: Record; maxSize?: number; maxPages?: number; parseTimeoutMs?: number; }): Promise<{ id: string; kind: "doc"; format: string; mode: "chunked" | "blob"; chunks: number; }>; /** Scope-bound {@link LiteCtx#remember}. @param {string} id @param {string} text @param {{ kind?: string, format?: string, by?: string, occurredAt?: number, meta?: Record, injectionRisk?: "low"|"medium"|"high", expiresAt?: number|null }} [opts] */ remember(id: string, text: string, opts?: { kind?: string; format?: string; by?: string; occurredAt?: number; meta?: Record; injectionRisk?: "low" | "medium" | "high"; expiresAt?: number | null; }): Promise; /** Scope-bound {@link LiteCtx#forget} (multis M4) — deletes ONLY the bound tenant's `fact`+`episode` memory. `{ id }` / `{ idPrefix }` (Feature B, 0.27.0) delete one row / one id's segments BY KEY, tenant-fenced (a foreign tenant's id → 0, the fence not id-matching decides — the delete-side mirror of the W4 `(scope, id)` upsert); a bare call (or `{ kind }`) wipes the whole bound tenant. `{ by }` is still rejected (owner-blind provenance — use base `ctx.forget({ by })`). The bound scope can't be omitted (fail-closed under strictScope). @param {{ kind?: string, id?: string, idPrefix?: string }} [sel] @returns {number} */ forget(sel?: { kind?: string; id?: string; idPrefix?: string; }): number; } export { RipgrepMissingError } from "./impact.js"; export { Store } from "./store.js"; export { liteCtxAsStore } from "./memory-store.js"; export type LiteCtxConfig = { /** * repo root to index */ root: string; /** * file extensions to index (default: ts/js/py/md) */ include?: string[] | undefined; /** * optional git pathspecs to scope the index (e.g. ["app/**\/*.js"]) */ pathspecs?: string[] | undefined; /** * SQLite file path (default: /.litectx/index.db) */ dbPath?: string | undefined; /** * enable the opt-in semantic tier (default false). When on, * `index()` embeds each file and `recall()` fuses cosine into * the ranking. Requires the optional peer dep `@huggingface/transformers`. */ embeddings?: boolean | undefined; /** * semantic fusion weight (default 1.0); higher = more semantic */ embedWeight?: number | undefined; /** * transformers.js model id (default Xenova/all-MiniLM-L6-v2) */ embedModel?: string | undefined; /** * inject a custom/stub embedder * (advanced/testing); overrides the built-in model loading */ embedder?: { embed(text: string): Promise; } | undefined; /** * scope key (§4.4) — the actor that owns durable `fact`s (and * tags `episode`s). Default unset = global/unscoped: recall * sees & writes are owner-blind. A multi-tenant / shared-db host * sets it (e.g. git email or OS user, resolved host-side) so a * shared store isolates per actor; recall then returns own + * global (NULL-owner) memory only. */ owner?: string | undefined; /** * fail-closed multi-tenant mode for the DOC axis (multis M3 ask). * Default false = today's behaviour (a missing/`null` doc `scope` * means "see everything" — correct single-tenant, a footgun on a * shared store). When true, a missing scope on `recall({kind:'doc'})`, * `get`, `ingest`, and `remember({kind:'doc'})` THROWS instead of * returning/writing every tenant's rows; the only ways to act are an * explicit tenant scope (`scope ∪ global`) or {@link GLOBAL} (the * shared tier). Governs the doc/blob axis ONLY — `fact`/`episode` * (the `owner`/`session` memory axis) and `code` are untouched. */ strictScope?: boolean | undefined; /** * scope key (§4.4) — the run that owns volatile `episode`s. * Default unset = durable/unscoped: recall sees all sessions' * episodes. A host running concurrent agents sets it so a run's * own episodes aren't buried by more-relevant other sessions * (gate #1, 2026-06-13). `fact`s ignore it (always cross-session). */ session?: string | undefined; /** * the rolling window (in days) an `episode` stays RETAINED and * promote-eligible (default {@link ACTIVE_EPISODE_DAYS} = 30). On * each episode write, episodes older than this self-prune; the same * window floors {@link LiteCtx#promotionCandidates}. One knob bounds * the agent scratchpad (no count cap). ⚠ This is NOT a free hygiene * dial — it is COUPLED to the promotion ladder: a window shorter than * an episode's promote-and-prove time can prune it (and drop it below * the promotion floor) BEFORE it reaches the threshold, so it never * promotes. Shorten it for data-minimization, lengthen it to retain * older episodes longer (the set grows); 30d is the safe default that * leaves room to promote-and-prove. Facts are durable and untouched. */ episodeWindowDays?: number | undefined; /** * optional write-gate hook (CE-PRD §10.1) — when set, `remember()` * emits a `{type:"memory.write", …}` action and `await`s * `writeGate.check(action)` BEFORE persisting; a `deny` outcome * throws {@link WriteDeniedError} and the write does not commit * (`ask`/`allow` proceed). Duck-typed — bareguard's `Gate` when * embedded, any `.check`-shaped object standalone. litectx is not * coupled to a gate version. Default unset = no gate (byte-identical * to pre-hook writes). */ writeGate?: import("./writegate.js").WriteGateLike | undefined; /** * optional standalone audit sink (the paper-trail half §10.1) — * when set with `writeGate`, each write decision is recorded. A * host-supplied `redact` on it scrubs secrets (litectx ships none). */ writeAudit?: WriteAudit | undefined; /** * when true, the instance is returned wrapped in `observe()` — every * CE verb call is recorded into `ctx.trace` (a `ContextGraph`). Off * by default = the bare instance, no proxy. See `src/contextgraph.js`. */ trace?: boolean | undefined; }; export type WriteGateLike = import("./writegate.js").WriteGateLike; export type Item = { /** * the written-memory id, or the file's repo-relative path */ id: string; /** * "code" | "doc" | "fact" | "episode" */ kind: string; /** * "ts" | "js" | "py" | "md" | "text" | ... */ format: string; /** * "file" (indexed from disk) | "direct" (written via remember) */ source: string; /** * "human" | "agent" for written memory; null for indexed files */ provenance: string | null; /** * episode timestamp (epoch ms); null otherwise */ occurredAt: number | null; /** * the full body — written memory verbatim as remembered, files * read fresh from disk; null when the file is gone, or for a blob * (a byte-exact upload — its payload is in `bytes`, not text) */ text: string | null; /** * a byte-exact upload's original bytes (R3); null for every other * kind and for file rows */ bytes: Buffer | null; /** * opaque caller metadata (RT-3 #3) as supplied to * `remember`, returned verbatim; null for files and for memory * with none */ meta: Record | null; }; /** * One row of an {@link LiteCtx#enumerate} page — a pointer record (search-only fields like `score` * omitted; this is an unranked table read). Field name is `path` (the public id, decoded via `memId`), * uniform with `recall`/`recentMemory`/`get`. */ export type EnumItem = { /** * the written-memory id (public, owner-prefix stripped) */ path: string; /** * "fact" | "episode" */ kind: string; /** * the stored format */ format: string; /** * episode timestamp (epoch ms); null for a fact */ occurredAt: number | null; /** * VERBATIM stored text — present only when `body:true` */ body?: string | undefined; /** * opaque caller metadata, present only when the row carries it */ meta?: Record | undefined; }; export type IndexResult = { /** * total documents in the index after the pass */ files: number; /** * newly indexed files */ added: number; /** * re-indexed files (content changed) */ updated: number; /** * files dropped (no longer present) */ removed: number; /** * files skipped (mtime or content unchanged) */ unchanged: number; }; import { Store } from "./store.js"; import { WriteAudit } from "./writegate.js"; export { splitIdent, keywords, ftsMatch } from "./tokenize.js"; export { Embedder, cosine } from "./embedder.js"; export { compress, COMPRESS_LEVELS } from "./compress.js"; export { assemble, summaryWindow, trim } from "./assemble.js"; export { toWriteAction, WriteAudit, WriteDeniedError } from "./writegate.js"; export { observe, ContextGraph, PRIMITIVES, VERBS_BY_PRIMITIVE, PRIMITIVE } from "./contextgraph.js";