/** * @fileoverview Local mirror of the Project Gutenberg catalog — an embedded SQLite + * FTS5 index of every book record, harvested from the bulk per-book RDF archive. * * Gutendex is a single live upstream on the catalog read path, and it has been down * for extended stretches. This service owns the mirror instance, the RDF ingester, and * the hydrated read helpers the tool layer serves from; the live Gutendex client stays * in place as the fallback for a cold mirror or an ID the mirror does not hold. * * A full harvest costs a ~121 MB download and ~79,000 XML parses, so it runs * out-of-band via the `mirror:init` / `mirror:refresh` scripts rather than at startup. * @module services/catalog-mirror/catalog-mirror-service */ import { type Mirror, type MirrorRunOptions, type MirrorStatus, type QueryOptions } from '@cyanheads/mcp-ts-core/mirror'; import type { ServerConfig } from '../../config/server-config.js'; import type { Book } from '../gutendex/types.js'; /** Result of a hydrated mirror query. */ export interface CatalogQueryResult { books: Book[]; /** Total matches before `limit` / `offset`. */ total: number; } /** Health report for the mirror database, as `mirror:verify` prints it. */ export interface CatalogMirrorReport { /** * Resume position of an interrupted run. Read from the persisted sync state rather * than {@link MirrorStatus}, which does not carry the volatile cursor. */ cursor: string | undefined; /** `PRAGMA integrity_check` + `quick_check` outcome. */ integrity: { ok: boolean; results: string[]; }; /** Rows currently stored. */ rows: number; status: MirrorStatus; } export declare class CatalogMirrorService { readonly mirror: Mirror; private readonly archiveUrl; private readonly pageSize; constructor(serverConfig: ServerConfig, options?: { pageSize?: number; }); /** * Whether the mirror has ever completed a full sync. Keys off the durable completion * marker, not live status, so a mirror mid-refresh — or one whose last refresh failed * — keeps serving the dataset it already has. */ ready(): Promise; /** Sync state for `mirror:verify` and operational reporting. */ status(): Promise; /** Run a full (`init`) or incremental (`refresh`) harvest. */ runSync(options: MirrorRunOptions): ReturnType; /** Release the SQLite handle. */ close(): Promise; /** Row count, sync state, and SQLite integrity — the mirror's operational health. */ verify(): Promise; /** Resume position persisted by an interrupted run, if any. */ cursor(): Promise; /** * Fetch books by Gutenberg ID, preserving the requested order and skipping IDs the * mirror does not hold — the caller decides whether a miss falls through to the live * catalog. * * Uses a filtered query rather than the store's `getByIds`, which takes string keys * and re-keys its results by the raw column value: against this table's `INTEGER` * primary key the SQL matches but the string-to-number lookup does not, so every row * is dropped. A numeric `in` filter has no such mismatch. */ getBooks(ids: number[]): Promise; /** * Run a mirror query and hydrate the rows into normalized books. The caller owns the * FTS expression and filters; this only crosses the row/record boundary. */ queryBooks(options: QueryOptions): Promise; /** * The ingester: stream the bulk RDF archive, parse each per-book document, and yield * pages of rows. * * `cursor` is `|`. Archive order is stable for a * given rebuild but is not sorted, so position is the only usable resume key — and it * is only valid against the rebuild it was recorded from, which is why the timestamp * travels with it. Resuming against a newer archive discards the position and re-reads * from the top rather than skipping entries whose contents have shifted. The download * always restarts either way: a bzip2 stream has no seekable index. * * `checkpoint` is the archive's own `Last-Modified`. Project Gutenberg rebuilds the * archive daily; a refresh against an unchanged archive has nothing to harvest and * costs one HEAD request. It advances only on the final page, so an interrupted run * never leaves the mirror claiming to hold an archive it did not finish reading. */ private harvest; /** IDs the mirror still holds that the archive no longer carries — books withdrawn upstream. */ private staleIds; } export declare function initCatalogMirrorService(serverConfig: ServerConfig, options?: { pageSize?: number; }): CatalogMirrorService; export declare function getCatalogMirrorService(): CatalogMirrorService; /** * Register and start the in-process refresh cron, when one is configured. * * Off unless `GUTENBERG_MIRROR_REFRESH_CRON` is set, because a refresh is a full * re-harvest — the archive publishes no incremental delta — and that is a poor * neighbour for a process that is also serving requests. Deployments with a host * scheduler should run `mirror:refresh` out-of-band instead. */ export declare function startCatalogMirrorRefresh(serverConfig: ServerConfig): Promise; /** * Stop and unregister the refresh cron, when one was registered. * * The framework clears every scheduled job during shutdown, but only after the * `teardown` hook has run — and teardown is where the mirror's SQLite handle is * released. Stopping the job first is what keeps a refresh tick from re-opening * the database on a process that is on its way out. */ export declare function stopCatalogMirrorRefresh(): void; //# sourceMappingURL=catalog-mirror-service.d.ts.map