/** * postgresIndex.ts * * Supabase Postgres-backed search and listing index for wiki pages. * * Mirrors the SqliteSearchIndex public surface so callers can swap between * backends without changing their code. Uses the Supabase PostgREST REST API * (no extra SDK dependency — same raw-fetch pattern as SupabaseStorageClient). * * Auto-selected when STORAGE_BACKEND=supabase. Override with SEARCH_BACKEND=sqlite|scan. * * Table schema: scripts/supabase-pages-index.sql */ import type { WikiFrontmatter } from "../wiki/frontmatter.js"; import type { IndexedSearchResult, IndexedSearchResponse } from "./sqliteIndex.js"; export interface PageListEntry { name: string; title: string; summary: string; updated: string; status: string; tags: string[]; created: string; modified: string; revision?: string; } /** A cross-account project-share grant (BRA-368): read as `owner`'s own pages under `folder`. */ export interface GrantScope { userId: string; folder: string; } export interface ListPagesOptions { tag?: string; status?: string; userId?: string; /** When set, only pages with access_level ≤ this value are returned. */ accessLevel?: number; /** ISO datetime — only pages with modified > changedAfter are returned. */ changedAfter?: string; /** When true, returns only name + modified (skips heavy fields). */ compact?: boolean; /** * Active project-share grants (BRA-368) for the requesting user. When set, * results are merged with each grant owner's pages under their granted * folder prefix — otherwise a grantee's shared content is invisible to * listing (only reachable by exact path). */ grantScopes?: GrantScope[]; /** Internal: restricts a single-scope query to pages under this folder prefix. Set by listPages() for each grant scope — not for direct callers. */ pathPrefix?: string; /** Restrict results to pages under this folder prefix (without trailing slash), e.g. "projects/brainsapp". */ folder?: string; /** Max rows to return. Unset means unbounded (existing behavior) — callers should pass one. */ limit?: number; /** Rows to skip, for pagination. Only exact for the no-grants case; grant-scoped merges are best-effort. */ offset?: number; } export interface SearchOptions { query: string; tag?: string; maxResults: number; userId?: string; /** When set, only pages with access_level ≤ this value are returned. */ accessLevel?: number; /** * Restrict candidates to a single Lobe (folder prefix). Pass the folder path * without a trailing slash, e.g. `"projects/brainsapp"`. Only pages whose * name starts with `folder/` are considered, eliminating cross-Lobe noise at * scale. Omit to search the full corpus. */ folder?: string; /** * Active project-share grants (BRA-368) for the requesting user. When set, * each grant owner's pages under their granted folder are also searched * and merged into the results — otherwise a grantee's shared content is * invisible to search (only reachable by exact path). */ grantScopes?: GrantScope[]; } export interface HybridSearchResult extends IndexedSearchResult { hybridScore: number; ftsScore: number; vecScore: number; } export interface HybridSearchResponse { results: HybridSearchResult[]; totalMatches: number; queryMs: number; backend: "postgres-hybrid"; } export interface ArchivedPageRow { id: string; originalPath: string; originalFolder: string; title: string; tags: string[]; content: string; revision: string; sizeBytes: number; reason: string; deletedAt: string; deletedBy: string; restoredAt: string; restoredBy: string; } export interface ArchivePageInput { userId: string; originalPath: string; originalFolder: string; title?: string; tags?: string[]; content: string; revision?: string; sizeBytes?: number; reason?: string; deletedBy: string; } export declare function getPostgresSearchIndex(): PostgresSearchIndex; export declare class PostgresSearchIndex { private readonly supabaseUrl; private readonly serviceRoleKey; private readonly baseUrl; private readonly linksBaseUrl; private readonly archivedPagesBaseUrl; private readonly eventsBaseUrl; private readonly relevanceSignalsBaseUrl; private readonly headers; constructor(); isConfigured(): boolean; /** * Lightweight connectivity check. Uses a HEAD + count=exact to verify the * Supabase REST endpoint is reachable without fetching any row data. */ ping(userId?: string): Promise<{ ok: boolean; rowCount?: number; error?: string; }>; /** * Count active (non-archived) pages for a user. Returns 0 when not configured * or when the request fails. Uses HEAD + count=exact to avoid fetching rows. */ countActivePages(userId: string): Promise; /** * Append one activity event. * * Replaces the read-modify-write of log.md that every page write used to * perform: Storage has no append primitive, so appending to a 460 KB file * meant moving ~920 KB and took 2.6-3.6s measured against production. An * indexed insert is O(1) (GH #417). * * Throws on failure so callers can decide — the write paths log a warning * rather than failing the page write, but a silent no-op is what let the * old log.md problem go unnoticed, so the error is not swallowed here. */ appendEvent(input: { userId: string; type: string; title: string; detail?: string; /** Actor that performed the write (GH #498) — falls back to userId when omitted. */ authorSubject?: string; }): Promise; /** * Read activity events for one user since a cutoff date (inclusive, * `YYYY-MM-DD`). Newest first. Returns null when Postgres is unavailable so * the caller can fall back to parsing log.md rather than silently reporting * no activity. */ listEvents(userId: string, sinceDate: string, limit?: number): Promise | null>; /** * Fetch every row matching `params`, following limit/offset until exhausted. * * PostgREST clamps a result set to its configured max-rows (1000 here) and * returns 200 OK. A truncated response is byte-identical to a complete one — * no error, no flag — so any "fetch everything then aggregate in JS" call site * silently computes on partial input and returns a confident wrong answer. * That has now happened four times: the bundle folder filter undercounted 164 * against 213 (#432), listProjectInventory lost a whole project whose pages * all fell in the truncated tail (#445), and audit_index reported ~174 * fabricated unindexed pages by diffing a capped snapshot against full * storage (#449). * * Callers that need a whole table must route through here rather than issuing * an unbounded request. Returns null on transport/HTTP failure, which callers * must treat as "unknown" — never as "empty", since that is exactly how a * partial read becomes a wrong answer. * * Pages by keyset (`name > last seen`), not by offset, and always sorts by * `name.asc`. Two reasons, both of which offset paging gets wrong: * * - **Never infer the end from a short page.** `rows.length < pageSize` is * only "the end" if pageSize happens to be ≤ the server's max-rows. That * is a Supabase dashboard setting on a different system, and staging and * prod are configured independently. If max-rows is ever lowered below * pageSize, the first page comes back short and an offset walk stops * there — silently returning a truncated list from the very helper written * to prevent that, with no round-number tell to notice it by. Keyset walks * until a page comes back empty, so it is correct whatever the cap is. * - **Offset is not stable against concurrent writes.** A page created with * an earlier-sorting name mid-walk shifts every later row down by one and * drops exactly one real page from the snapshot. During an audit that page * then looks unindexed, and cleanup would "repair" it by overwriting its * DB row from Storage. Keyset resumes from a value, not a position, so an * insert behind the cursor cannot displace anything. * * Requires `name` in the select — it is the cursor. Returns null if it is * missing rather than looping. */ private fetchAllRows; /** * Every indexed page name for a user, complete — not capped at max-rows. * * Exists because audit_index compares the index against storage, and a * truncated index snapshot makes every missing row look like an unindexed * page. Applies the same exclusions as listPages so the two agree on what * counts as a page. * * Returns null if any page of the walk fails. Callers must not fall back to a * partial list — an incomplete audit is worse than no audit, because it * invites a "repair" of pages that were never broken. */ listAllPageNames(userId?: string): Promise; /** * One row per project: slug, page count, and the root readme's title/aliases. * * Two requests rather than one per project. Aliases come from the readme's * frontmatter and are how a nickname reaches the right project — the * cressida/nessie case, where the slug never appears in what the user typed. * * Anchored to `projects//readme.md` exactly. A looser match picks up * nested readmes — cetera-pds alone has nine — and reports one project many * times (GH #430). */ listProjectInventory(userId: string): Promise | null>; /** * Full-text hits per project, for when the slug does not appear in what the * user typed. Returns a slug -> hit-count map. */ countProjectContentHits(context: string, userId: string): Promise>; countArchivedPages(userId: string): Promise; archivePage(input: ArchivePageInput): Promise<{ id: string; }>; listArchivedPages(options: { userId: string; limit?: number; includeRestored?: boolean; originalPath?: string; }): Promise; getArchivedPageForRestore(input: { userId: string; archiveId?: string; originalPath?: string; deletedAt?: string; }): Promise; markArchivedPageRestored(archiveId: string, userId: string, restoredBy: string): Promise; /** * Insert or update a page in the index. * Non-blocking: callers should fire-and-forget with `.catch(() => {})`. */ upsertPage(name: string, frontmatter: WikiFrontmatter, content: string, revision: string, modifiedTime: string, userId?: string, contentModifiedTime?: string, /** Actor that performed the write (GH #498) — falls back to userId (storage owner) when omitted. */ authorSubject?: string): Promise; /** * Conditionally update a page row only if it still has `expectedRevision`. * Returns false (no throw) when the row's revision has moved or the row is * gone — that's a concurrent-write signal, not an error. Used by * patch_page's CAS retry loop (GH #500): unlike upsertPage's unconditional * merge-duplicates upsert, this scopes the write with a `revision=eq.` filter * so a concurrent writer between read and write can't be silently clobbered. */ casUpdatePage(name: string, frontmatter: WikiFrontmatter, content: string, expectedRevision: string, newRevision: string, modifiedTime: string, userId?: string, contentModifiedTime?: string, /** Actor that performed the write (GH #498) — falls back to userId (storage owner) when omitted. */ authorSubject?: string): Promise; /** * Insert a brand-new page row only if one doesn't already exist. * Returns false (no throw) when a concurrent writer already created the * row — that's the insert-side counterpart of casUpdatePage (GH #500): * patch_page's "first write, no row to CAS against" branch previously fell * back to upsertPage's unconditional merge-duplicates upsert, so two * concurrent first writes on the same new page could silently clobber one * another with no conflict ever reported. `resolution=ignore-duplicates` * makes the INSERT a no-op on a (user_id, name) collision instead of * merging, and `return=representation` lets us tell a real insert apart * from a no-op by whether a row came back. */ casInsertPage(name: string, frontmatter: WikiFrontmatter, content: string, newRevision: string, modifiedTime: string, userId?: string, contentModifiedTime?: string, /** Actor that performed the write (GH #498) — falls back to userId (storage owner) when omitted. */ authorSubject?: string): Promise; /** * Generate an embedding for a page and store it in the embedding column. * Called fire-and-forget from upsertPage. */ generateAndStoreEmbedding(name: string, title: string, summary: string, content: string, userId?: string): Promise; /** * Write a pre-computed embedding vector to the pages table. * Used by generateAndStoreEmbedding and the backfill script. */ storeEmbedding(name: string, userId: string, embedding: number[]): Promise; /** * Remove a page from the index by name. * Non-blocking: callers should fire-and-forget with `.catch(() => {})`. */ deletePage(name: string, userId?: string): Promise; /** * Remove multiple pages and their wiki_links in two batch HTTP requests per chunk. * Avoids N×2 serial round-trips when purging or reconciling many rows at once. */ deletePages(names: string[], userId?: string): Promise; syncPageLinks(name: string, content: string, userId?: string): Promise<{ linkCount: number; }>; /** * Resolve bare-slug [[wiki-link]] targets (e.g. "foo.md" with no folder) to * their real page name by suffix-matching against known pages. Only slugs * with exactly one match are resolved; ambiguous (multiple matches) or * unknown slugs are left out of the returned map, and the caller keeps its * best-effort root-level guess as the indexed target. */ private resolveBareSlugTargets; getOutboundLinkCount(name: string, userId?: string): Promise; getBacklinks(name: string, userId?: string, accessLevel?: number): Promise; getRelatedPages(name: string, userId?: string, accessLevel?: number): Promise; /** * Filter a list of page names to only those where `access_level <= accessLevel` * for the given user. Used to enforce Hive mode governance in link-graph operations. * Queries the `wiki_pages` table (columns: `name`, `user_id`, `access_level`). */ private filterPagesByAccessLevel; extractLinksForStalePages(filter: { tags?: string[]; folder?: string; userId?: string; }): Promise<{ pagesConsidered: number; pagesProcessed: number; linksExtracted: number; }>; private deleteLinksForSource; private stampLinksExtractedAt; private fetchPageContent; /** * Read a single page row from the DB for the hot path (read_page, patch_page). * Returns null when not configured, row not found, or on any error so the * caller can fall back to Supabase Storage transparently. */ fetchPageRow(name: string, userId?: string): Promise<{ fileId: string; name: string; content: string; revision: string; modifiedTime: string; authorSubject?: string; } | null>; /** * Batch content read for export (GH #726 / BRA-4220): `buildExportZip` * previously called {@link fetchPageRow} once per page — 732 pages meant * 732 serial-ish PostgREST round trips (bounded only by READ_CONCURRENCY), * measured at ~27-30s end to end. Chunks via PostgREST `name=in.(...)`, * same pattern as {@link deletePages}/`checkTaxonomy`, so a large export is * a handful of round trips instead of one per page. * * Returns whatever it found — a name missing from the result Map (a bad * chunk, a row that vanished between listing and reading) is the caller's * signal to fall back to fetchPageRow for that one page, not an error here. */ fetchPageRows(names: string[], userId?: string): Promise>; /** * List all pages, with optional tag and status filters. * Returns lightweight entries — no content. */ listPages(options?: ListPagesOptions): Promise; private listPagesSingleScope; /** * Same filters as {@link listPages}, but walks the full result set via * keyset pagination instead of a single capped request. Ignores `limit`/ * `offset` — a caller asking for "every page matching this filter" cannot * be honestly served a capped page, so those options are not applied here. * * Returns null if any page of the walk (own scope or any grant scope) * fails. Callers must not fall back to the partial result — a truncated * bundle/export reported as complete is the exact failure mode this * exists to prevent (GH #452). */ listAllPages(options?: ListPagesOptions): Promise; private listAllPagesSingleScope; /** * Targeted existence check for health_check's taxonomy verification — point * lookups instead of diffing against a bulk listPages() fetch. A bulk fetch * ordered by modified.desc is bounded by PostgREST's default row cap, so on * wikis past ~1000 pages, old/rarely-touched seed pages silently age out of * the window and false-flag as "missing" (see GH #358). Point lookups by * exact name / folder prefix are correct regardless of wiki size. * * Checks both the clean name and the legacy `users//`-prefixed * variant some accounts have from an older populate script, matching the * normalization the bulk-fetch approach used to do. */ checkTaxonomy(userId: string, requiredSeedPages: string[], requiredFolders: string[]): Promise<{ missingSeedPages: string[]; missingFolders: string[]; }>; /** * Full-text search using Postgres tsvector / tsquery. * Returns ranked results compatible with IndexedSearchResponse. */ search(options: SearchOptions): Promise; private searchSingleScope; /** * Re-rank FTS results using search_relevance_signals: a page this same user * actually opened after this exact search before is promoted ahead of * ts_rank/recency order. First real consumer of the table (GH #501 / * BRA-4571) — it was written on every read-after-search and never read back. * * Scoped to (auth_subject, query_text) rather than page_name alone: a page * that is generically popular for other queries is not evidence it answers * *this* query, and cross-user boosting would leak one user's click * behavior into another user's ranking. query_text is matched case- * insensitively via `ilike` (no wildcards) but not whitespace-normalized — * a near-identical re-phrasing of the query simply won't match, which * degrades to no boost rather than a wrong one. * * Fails open: any error, timeout, or non-2xx response from the signals * table leaves ts_rank order untouched. This is a ranking hint, not a * correctness dependency — search must not break because this table is * unreachable or not yet migrated on an older deployment. */ private applyRelevanceBoost; /** * Count search_relevance_signals rows per page, for one user, since a * cutoff timestamp. First consumer outside ranking (GH #497 milestone 1, * `stale_pages`) — identifies pages with zero recent engagement. * * Scoped by auth_subject: the signals table has no wiki/project column (see * scripts/search-relevance-signals.sql), so subject-scoping here is what * keeps a cold-page report from ever correlating one caller's engagement * against another caller's page name. Callers must additionally restrict * `pageNames` to pages already known to belong to this same caller — this * method does not verify page ownership itself. * * Throws on a non-2xx response or network error, matching the other * `search_relevance_signals` accessor (applyRelevanceBoost fails open at * the *call site*, not inside the accessor) — callers that need a * degrade-not-throw contract must catch this themselves. */ countRelevanceSignalsSince(options: { userId: string; pageNames: string[]; sinceIso: string; }): Promise>; private runFtsQuery; /** * Relevance-ordered FTS via the search_pages_ranked RPC (orders by ts_rank in * Postgres). Returns null when the function isn't deployed (PGRST202 / 404) so * the caller can fall back to the REST filter. */ private runRankedQuery; /** * Re-trigger the FTS update trigger for all rows where fts_vector is NULL. * Needed when rows were inserted before the trigger was applied (e.g. after a * Supabase project migration). Safe to call multiple times — a second call is * a no-op because no null rows remain. */ reindexFts(userId?: string): Promise<{ updated: number; }>; /** * Score-based hybrid retrieval: FTS (weight 0.3) + vector cosine similarity (weight 0.7). * Calls the search_pages_hybrid stored function deployed by embedding-migration.sql. * Returns null when the function isn't deployed or embedding is unavailable. */ searchHybrid(options: SearchOptions): Promise; private searchHybridSingleScope; /** * Checks whether a column actually exists on a live table (GH #801). * * PostgREST doesn't expose `information_schema` over the REST API, so this * probes the table directly with a zero-row select and reads PostgREST's * `42703` (undefined_column) error code rather than querying the catalog. * Returns `true`/`false` when the check is conclusive; throws when the * probe itself fails for an unrelated reason (network error, auth failure, * unknown table) since that's not something a caller can safely treat as * "column exists". */ hasColumn(table: string, column: string): Promise; /** No-op: connection is stateless (each call is an independent HTTP request). */ close(): void; } interface FtsRow { name: string; title: string; summary: string; updated: string | null; modified: string | null; tags: string[]; content?: string; } /** Lowercase, strip punctuation, split into tokens. */ export declare function tokenizeQuery(query: string): string[]; /** Drop stopwords — but never return empty (keep originals if all were stopwords). */ export declare function stripStopwords(tokens: string[]): string[]; /** * Tokens usable for a *matched-on* relevance signal: stopwords and sub-3-char * tokens excluded, since both produce false-positive substring hits (e.g. the * stopword "is" or "or" appearing inside "History"/"Editor"). */ export declare function matchableTokens(query: string): string[]; /** * Prefix match on a word boundary (mirrors the `token:*` prefix semantics used * in the tsquery builder below) — not a bare substring test. Prevents a token * like "type" from being credited for matching inside "stereotype". */ export declare function tokenMatchesText(text: string, token: string): boolean; /** Stable sort by descending term-overlap; ties keep the incoming (recency) order. */ export declare function rankByOverlap(rows: FtsRow[], tokens: string[]): FtsRow[]; export {}; //# sourceMappingURL=postgresIndex.d.ts.map