/** * Client-side sync manifest generator (sync-reconciliation-audit US-002). * * WHAT THIS IS FOR * ---------------- * The reconciliation audit compares three pictures of a scope: what the vault * holds, what this machine's journal (the "ledger") believes it synced, and * what is actually ON DISK. Only the client can produce the last two, and only * together do they distinguish the failure modes that matter: * * - a file on disk with no ledger row → the daemon stopped tracking it * (wedged watcher, crashed push) — invisible to any server-side check; * - a ledger row with no file on disk → the local copy vanished; * - both present and in agreement → healthy. * * That is exactly the `disk` / `ledger` / `both` split the wire contract * encodes, and it is why this builder unions the two sources instead of * walking the disk alone. * * WHAT THIS DELIBERATELY DOES NOT DO * ---------------------------------- * - It never reads `sync-journal.*.json` by path. The v3 journal's real * state lives in a digest-named store; the legacy filename is only a * locator. `listJournals()` is the ONLY supported way to enumerate shards. * - It never re-implements ignore rules. `createIgnoreFilter()` owns the * layered DEFAULT_IGNORES / .hqignore / .hqinclude semantics; a second * implementation here would drift and manufacture false findings for every * file the two disagree about. * - It never reads file CONTENT unless it has to. The walk is stat-only; a * hash is computed only when the ledger has no usable one (see * {@link canReuseLedgerHash}). On a healthy machine that is ~zero files. * * COST DISCIPLINE * --------------- * This runs on the user's laptop, in the background, beside their editor. The * walk yields to the event loop every `yieldEvery` files, honours an * `AbortSignal`, and hard-stops at `maxWallMs` — marking the manifest * `truncated` rather than pretending a partial picture is complete. A * truncated manifest never emits `removedPaths` and never overwrites the * snapshot, because "I stopped early" and "these files are gone" are the same * observation from the walker's point of view and only one of them is true. */ import { type JournalSummary } from "../journal.js"; import { type SyncManifestEntry, type SyncManifestMode, type SyncManifestScope, type SyncManifestSource, type SyncManifestUpload, type SyncManifestWalkStats } from "./contract.js"; import { type ManifestSnapshot } from "./snapshot-store.js"; /** * The contract's {@link SyncManifestWalkStats} has NO `truncated` field, and * the contract file is byte-mirrored with hq-pro — adding one here would break * a test in BOTH repos. * * The client still needs to say "this manifest is partial", so the flag rides * as an EXTRA field on the walk stats. The contract validator is explicitly * tolerant of unknown extra fields (a newer client must not fail an older * server), so a chunk carrying it still validates today and the field becomes * meaningful to the server the moment the contract adopts it. */ export interface BuildWalkStats extends SyncManifestWalkStats { truncated: boolean; /** * Directories the walk could not `readdir` (EACCES, a raced unmount, a * transient FUSE error). Everything beneath them is INVISIBLE this cycle — * which is not the same as absent — so the builder suppresses ledger-only * verdicts and removals under those prefixes (see `lostVisibilityPrefixes`). * Reported so an operator can tell "clean scope" from "clean scope, plus a * subtree I was never allowed to look at". */ unreadableDirs: number; } /** * Scope plus the local knowledge the wire contract deliberately omits. The * contract carries `companyUid` (the server's identifier); the local walk * needs the SLUG, because that is what names both the on-disk directory * (`companies/{slug}`) and the journal shard. */ export interface BuildManifestScope extends SyncManifestScope { /** Company directory/journal slug. Required for `kind: "company"`. */ slug?: string; } /** * How aggressively the builder may open files. * * - `never` (the DEFAULT, and what the sync-runner tail step uses) — never * open a file. A hash is reused from the journal when its (size, mtime) * pair still matches; otherwise the entry ships WITHOUT one, meaning * "present, hash unknown". This is a pure stat walk: bounded, cheap, and * safe to run after every sync cycle. * - `auto` — like `never`, but may hash up to `maxHashedFiles` files (and * for up to `maxHashMs`) per pass; everything past the cap ships unhashed. * Successive passes therefore fill the picture in without any single pass * costing minutes. * - `full` — hash every file the journal cannot cover, unbounded. Only ever * from an explicit human invocation (`hq sync manifest --full-hash`), * never from the daemon. This was the previous `auto`, and on a 690k-file * vault with a cold journal it measured ~9m40s / 2.9 GB RSS. * - `ledger-only` — never open a file AND DROP entries with no reusable * hash. Retained for the stat-walk benchmark, which wants to time the walk * rather than sha256 throughput. Not for production use: dropping an entry * reports a file the client holds as one it does not. */ export type ManifestHashPolicy = "auto" | "ledger-only" | "full" | "never"; /** * Default per-pass ceiling on files the `auto` policy may READ. * * Sized so the hash phase stays comparable to the stat walk it follows rather * than dwarfing it: a few thousand small files is sub-second, while the * unbounded version of this loop measured in minutes on a real vault. */ export declare const DEFAULT_MANIFEST_MAX_HASHED_FILES = 2000; /** Default per-pass wall-clock ceiling on the `auto` policy's hash phase. */ export declare const DEFAULT_MANIFEST_MAX_HASH_MS = 5000; export interface BuildSyncManifestOptions { scope: BuildManifestScope; hqRoot: string; stateDir: string; installationId: string; machineId: string; source: SyncManifestSource; sequence: number; /** Injected clock (epoch ms) — tests pin it; production omits it. */ now?: number; abortSignal?: AbortSignal; /** Server said `resend_full`, or the caller wants a fresh baseline. */ forceFull?: boolean; maxWallMs?: number; yieldEvery?: number; hashPolicy?: ManifestHashPolicy; /** `auto` only: files this pass may read. Defaults to {@link DEFAULT_MANIFEST_MAX_HASHED_FILES}. */ maxHashedFiles?: number; /** `auto` only: wall clock this pass may spend hashing. Defaults to {@link DEFAULT_MANIFEST_MAX_HASH_MS}. */ maxHashMs?: number; /** * Test seam: stands in for the real per-slug journal read. * * Shaped as the whole-machine list for historical reasons — the builder now * reads only the ONE shard it needs (see `readLedger`), and when this seam is * supplied it simply picks that slug out of the fabricated list. */ listJournalsImpl?: () => JournalSummary[]; /** Tuning seam for the ignore filter's path memo (see MANIFEST_IGNORE_MEMO_LIMIT). */ ignoreMemoLimit?: number; /** * Cloud-backed slugs, forwarded to `computePersonalVaultPaths()` for the * PERSONAL scope only. Same set the runner derives from live membership. * Optional: `companies/manifest.yaml`'s `cloud_uid` markers already exclude * designated companies without it, so an omitted set is conservative rather * than wrong. */ teamSyncedSlugs?: ReadonlySet; /** Skip persisting the new snapshot (used by read-only callers and benches). */ persistSnapshot?: boolean; /** * Serialised-byte budget for one chunk. Forwarded to the planner; see * {@link ChunkInput.chunkByteBudget}. */ chunkByteBudget?: number; /** * Test seam: entries per chunk, clamped to the contract ceiling. * * Without it the ONLY way to produce a multi-chunk pass is a 50 000-file * fixture, so every multi-chunk rule (ascending chunkIndex, a byte-identical * header on every chunk, a mid-pass abort) is untestable — and an untestable * rule is one a refactor can delete without a red test. */ maxEntriesPerChunk?: number; } /** * Test-only references sampled at the resident boundary immediately before a * manifest result is returned. The production builder never retains this * object; the hook is unset outside the isolated memory benchmark. */ export interface BuildManifestMemoryReferencesForTest { ledgerEntries: Map; walkRecords: Map; entries: SyncManifestEntry[]; snapshotEntries: ManifestSnapshot["entries"]; changed?: SyncManifestEntry[]; present?: Set; removed?: string[]; } export declare function setBuildManifestMemoryReferencesHookForTest(hook: ((references: BuildManifestMemoryReferencesForTest) => void) | undefined): void; /** * A chunk set that has been PLANNED but not materialised. * * WHY THIS IS NOT JUST `SyncManifestUpload[]` * ------------------------------------------ * The chunk set is pure derived data: every chunk is a window onto `entries` * plus a header that is identical across the pass. Holding all of it for the * duration of the upload buys nothing — the pass sends one chunk at a time — * and it grows with the scope, so it is a copy that gets more expensive * exactly on the machines that can least afford it. * * So the plan reports the two things a caller needs UP FRONT — how many chunks * there are, and what mode they are in — and materialises a chunk only when * one is asked for. The upload loop asks for chunk `i`, serialises it, POSTs * it, and drops it before asking for `i + 1`, so the transient cost is one * chunk plus its JSON rather than the whole set plus one JSON. * * BE HONEST ABOUT THE SIZE OF THIS WIN. A chunk's `entries` is a `slice`, so * the set that used to be held was 200,000 POINTERS (~1.6 MB), not a second * copy of the entries. It is a real bound and it is the RIGHT shape, but it is * not what made the operator's 199k scope peak at 642 MB — measured, this * changes peak RSS by nothing on a 200k-file fixture. The dominant terms are * the walk records, the entry objects and the snapshot map, all O(entries) and * all necessarily live at once because the delta decision needs the complete * entry set before the first chunk can be sized. Cutting those needs a * two-pass build; this keeps the upload half from adding to them. * * `chunkCount` MUST be known before the first POST and MUST NOT change during * it: the server hashes it (with `generatedAt`, `sequence`, `machineId`, * `mode` and `scope`) into the snapshotId, so a pass whose chunk count moved * mid-flight would derive two different snapshots for one upload. It is fixed * here, after the walk and the delta decision, and every chunk this plan mints * carries the same frozen header. */ export interface ManifestChunkPlan { /** Frozen for the whole pass; safe to send in every chunk header. */ chunkCount: number; /** Total entries across every chunk. */ entryCount: number; /** * The serialised-byte budget this plan was cut against, AFTER clamping. * * Reported rather than assumed so the upload pass can log the number that * actually shaped the chunks (which is not necessarily the number it asked * for — see the clamp in {@link planManifestChunks}) and so a 413 can be * diagnosed against a real value instead of a default. */ chunkByteBudget: number; /** * Largest chunk this plan will mint, in serialised UTF-8 bytes, as measured * by the sizing pass. * * This is the plan's own answer to "will the server take this?", available * WITHOUT materialising a single chunk. It can legitimately exceed * {@link ManifestChunkPlan.chunkByteBudget} in exactly one case: a single * entry larger than the whole budget, which cannot be split further and is * sent alone rather than dropped. */ maxChunkBytes: number; /** How many of those entries carry `source: "disk"` (see the upload guard). */ diskEntryCount: number; mode: SyncManifestMode; /** Materialise ONE chunk. The caller is expected to drop it before the next. */ chunk(chunkIndex: number): SyncManifestUpload; } export interface BuildSyncManifestResult { /** * Every chunk, materialised on access. * * A getter, not a field: `dryRun`/print and the tests legitimately want the * whole set, and the production upload path legitimately must never build * it. Reading this property on a 199k-entry scope allocates the full chunk * set — use {@link chunkPlan} on any path that runs on a user's machine. */ readonly chunks: SyncManifestUpload[]; /** The streaming surface the upload pass uses. */ chunkPlan: ManifestChunkPlan; mode: SyncManifestMode; walkStats: BuildWalkStats; /** The snapshot describing this manifest — persisted unless truncated. */ snapshot: ManifestSnapshot; /** Paths that could not be represented on the wire (see `toManifestPath`). */ unrepresentablePaths: number; /** Entries dropped because a `ledger-only` policy left them without a hash. */ skippedUnhashed: number; /** Entries emitted WITHOUT a hash (stat-only, or past this pass's hash cap). */ unhashedFiles: number; /** Entries dropped because the file changed while it was being hashed. */ droppedStatRaces: number; } /** Default wall-clock cap for one scope's walk. */ export declare const DEFAULT_MANIFEST_MAX_WALL_MS = 30000; /** Default number of files walked between event-loop yields. */ export declare const DEFAULT_MANIFEST_YIELD_EVERY = 500; /** * Ignore-filter memo size for a manifest walk — deliberately far below the * filter's own 50k default. * * That default is tuned for the sync engine, which probes the SAME paths on * every cycle and lives off the cache hits. A manifest walk visits each path * exactly once, so the memo can never hit; all it does is retain a key string * per file (and keep the `ignore` library's internal per-path cache alive * beside it) for the whole walk — tens of megabytes of resident memory on a * 70k-file root, bought for zero hits. A small window keeps both layers * bounded at the cost of a periodic matcher rebuild. */ export declare const MANIFEST_IGNORE_MEMO_LIMIT = 2000; export type ManifestUploadOutcome = "ok" | "soft_skip" | "resend_full" | "error"; export interface ManifestUploadClassification { outcome: ManifestUploadOutcome; } /** * Classify the manifest upload route's response. * * The one non-obvious rule: **404 is a SOFT SKIP, not an error.** hq-cloud * ships to end-user machines and updates independently of the server, so a * client that knows about the manifest route will routinely meet a deployment * that does not host it yet. Treating that as an error would light up every * such user's logs (and their support tickets) with a failure that describes * a perfectly healthy machine. It is logged at info and the cycle moves on. * * A 409 — or any 2xx/4xx body asking for `resend_full` — means the server lost * its delta base, so the next manifest must be FULL. */ export declare function classifyManifestUploadResponse(status: number, body?: unknown): ManifestUploadClassification; /** * Build the manifest chunks for one scope. * * Never throws for environmental reasons (unreadable journal, unreadable * directory, undecodable filename, snapshot write failure) — those degrade to * a narrower but honest manifest. It DOES throw for programming errors: * a company scope without a uid or slug, or a personal scope carrying a * company uid, are caller bugs that must fail loudly rather than upload a * mislabelled tenant's file list. */ export declare function buildSyncManifest(options: BuildSyncManifestOptions): Promise; export interface ChunkInput { entries: SyncManifestEntry[]; removedPaths: string[]; walkStats: BuildWalkStats; ignoreRules: string[]; mode: SyncManifestMode; baseSnapshotId?: string; generatedAt: string; installationId: string; machineId: string; scope: SyncManifestScope; source: SyncManifestSource; sequence: number; ledgerUnavailable?: { errorClass: string; }; /** Test seam; clamped to `SYNC_MANIFEST_MAX_ENTRIES_PER_CHUNK`. */ maxEntriesPerChunk?: number; /** * Serialised-byte budget for ONE chunk, clamped to * `SYNC_MANIFEST_MAX_CHUNK_BYTES`. Omitted means * `SYNC_MANIFEST_DEFAULT_CHUNK_BYTE_BUDGET`. * * Production DOES set this — unlike `maxEntriesPerChunk`, it is not a test * seam. The upload pass halves it after a 413 and persists the reduced * value, so a scope whose entries are unusually fat self-heals instead of * failing the same way every pass. */ chunkByteBudget?: number; } /** * Split entries into contract-sized chunks. ALWAYS emits at least one chunk: * an empty scope is a real, meaningful observation ("this machine has nothing * here"), and swallowing it would look identical to "the client never ran". * * `removedPaths` ride the FIRST chunk only, so the server never has to * de-duplicate a removal repeated across chunks. */ export declare function chunkManifest(input: ChunkInput): SyncManifestUpload[]; /** * Plan the chunk set without building it. * * Everything a chunk header carries is decided HERE, once — `chunkCount` * included — so every chunk the plan later mints is byte-identical in the * fields the server hashes into the snapshotId. Only the `entries` window and * `chunkIndex` differ between chunks, which is the whole reason a chunk can be * built on demand and thrown away instead of being held for the pass. * * ALWAYS plans at least one chunk: an empty scope is a real, meaningful * observation ("this machine has nothing here"), and swallowing it would look * identical to "the client never ran". */ export declare function planManifestChunks(input: ChunkInput): ManifestChunkPlan; //# sourceMappingURL=build-manifest.d.ts.map