/** * Client-side manifest UPLOAD pass (sync-reconciliation-audit US-004). * * WHY THIS SEAM LIVES IN hq-cloud * ------------------------------- * Two clients upload manifests — the desktop sync daemon (hq-sync) after a * sync cycle, and `hq sync manifest` in the CLI — and the PRD's decision * record settles it as "both, via shared hq-cloud code". The rules that make * an upload SAFE are not obvious and must not be re-derived twice: * * - a snapshot may become the local delta BASE only after the server says * `materialised: true` (a `completed` fold that was refused is not a base); * - the delta base is the snapshotId the SERVER returns, never the local * `snap-*` handle the builder minted (the server derives its own id); * - `sequence` must strictly increase per (installation, scope), including * across passes that failed before they uploaded anything; * - every chunk of one pass must be byte-identical in the fields the server * hashes into the snapshotId (`generatedAt`, `sequence`, `chunkCount`, * `machineId`, `mode`, `scope`); * - a `resend_full` at ANY chunk ends the pass and invalidates the base. * * Two implementations of those rules would drift, and the failure mode of the * drift is silent server-side corruption of a scope's materialised view — the * exact thing this audit exists to detect. * * WHAT THIS DELIBERATELY DOES NOT OWN * ----------------------------------- * - **Auth and the base URL.** The transport is INJECTED. hq-cloud has no * business deciding which API host a daemon talks to or how a JWT is * minted; the clients already own both, and owning them here would force * this module to import the vault client (and with it the AWS SDK) into * what is otherwise a lean, hot-path module graph. * - **Scheduling.** The caller decides when a pass runs; this function only * refuses to run one too often (see the 24h throttle). * - **Wiring.** The hq-cli `sync manifest` command, its `hq doctor` check, * and the hq-sync daemon hook land SEPARATELY, in those repos, on top of * {@link runManifestUploadPass} and {@link readManifestUploadStatus}. * * FAILURE POSTURE * --------------- * This never throws at the caller for anything that happens at runtime. The * daemon calls it after a sync pass and must continue regardless: an audit * that can break sync is strictly worse than no audit. Build errors, transport * rejections, aborts and unexpected statuses all come back as * `{ status: "failed" }`. Only invalid OPTIONS — a programmer error, not a * runtime condition — throw. */ import { type BuildManifestScope, type ManifestHashPolicy } from "./build-manifest.js"; import type { JournalSummary } from "../journal.js"; import { type SyncManifestMode, type SyncManifestSource, type SyncManifestUpload } from "./contract.js"; import { MANIFEST_SNAPSHOT_MAX_AGE_MS } from "./snapshot-store.js"; /** * Env var that disables manifest uploads entirely. A user (or a support * engineer) must be able to switch this feature off on a machine WITHOUT * downgrading hq-cloud, and the switch must short-circuit before the walk — * the walk, not the POST, is the expensive half. */ export declare const MANIFEST_UPLOAD_DISABLED_ENV = "HQ_SYNC_MANIFEST_DISABLED"; /** * Default minimum spacing between upload passes for one scope. * * The audit is a slow-moving safety net, not telemetry: a daily picture is * enough to catch a wedged watcher, and a 70k-file walk per sync cycle is not * something a laptop should pay for. */ export declare const DEFAULT_MANIFEST_UPLOAD_MIN_INTERVAL_MS: number; /** * Wall-clock budget for the ENTIRE upload phase (all chunks), after which the * pass stops sending and reports a non-throwing failure. * * Why the whole phase and not per chunk: the caller's actual requirement is * "the daemon's `await` returns" — the desktop daemon runs this after every * sync pass, and a transport that never settles would otherwise wedge the pass * (and the sync cycle behind it) forever. A per-chunk budget would not bound * that: a 30-chunk manifest would still admit 30 x budget of hanging. One * budget for the phase bounds the pass by construction, whatever the chunk * count. * * 60s is generous next to a single small JSON POST and mean next to a sync * cycle, so it only ever fires on a genuinely wedged transport — never on a * merely slow link, where the pass would be retried next cycle anyway. * * Note this bounds only the UPLOAD; `maxWallMs` bounds the filesystem walk. * They are separate budgets because they fail for unrelated reasons (a slow * disk vs. a black-holed socket) and a caller may reasonably tune one alone. */ export declare const DEFAULT_MANIFEST_UPLOAD_WALL_MS = 60000; /** * First retry floor after a pass FAILED without reaching the server. * * The 24h throttle is armed by `lastUploadAt`, which only a pass that actually * reached the server writes. That is correct — a network blip must not * suppress the audit for a day — but taken alone it means a permanently broken * endpoint (a wrong base URL, an unenrolled installation answering 403, a 5xx * outage) buys a FULL tree walk, and under `hashPolicy: "auto"` a full content * hash, on EVERY sync cycle. On a large vault that is minutes of IO and * gigabytes of RSS per cycle, forever, for an audit that cannot succeed. * * So a failed pass arms its own, much shorter floor, doubling per consecutive * failure up to {@link MANIFEST_FAILURE_BACKOFF_MAX_MS}. One hour keeps a * transient outage cheap to recover from (the next hourly cycle retries) while * capping a hard failure's cost at ~1 walk/hour instead of 1 walk/cycle. */ export declare const MANIFEST_FAILURE_BACKOFF_BASE_MS: number; /** Ceiling for the failure backoff — the same daily spacing a healthy scope gets. */ export declare const MANIFEST_FAILURE_BACKOFF_MAX_MS: number; /** * Share of `source: "disk"` entries above which a FULL manifest is treated as * an implausible picture of a scope that was previously tracked. * * Set hard against 1.0 rather than at a "suspicious" level like 0.9 on * purpose. A legitimate manifest CAN be very disk-heavy — a user who just * restored a backup, or a first pass after a journal repair — and refusing * those would suppress the audit exactly when it is most useful. The failure * this guards is not "mostly disk": it is "the ledger vanished", which is * total. 0.995 leaves room for the handful of `ledger`/`both` entries a * partially-readable ledger still contributes without admitting a merely * lopsided scope. */ export declare const MANIFEST_IMPLAUSIBLE_DISK_RATIO = 0.995; /** * Floor the 413 self-heal will not halve below. * * The halving exists to rescue a scope whose entries are fatter than the * default budget assumed; it is not a search for the smallest body the server * will take. Below ~256 KiB the chunk count for a large scope climbs past * `SYNC_MANIFEST_MAX_CHUNK_COUNT` and the pass would start failing for a * second, unrelated reason — and a 413 that survives a budget this small is * not a sizing problem at all (a proxy with its own tiny limit, a gateway * refusing the route), so halving further would just burn passes hiding it. */ export declare const MANIFEST_CHUNK_BYTE_BUDGET_FLOOR: number; /** * The budget to plan the NEXT pass against after this one was refused 413. * * Halving, not a fixed step, because the client has no information about how * far over the line it was: API Gateway's 413 is opaque, and even the * server's own rejection reports its ceiling rather than the body's size. A * geometric retreat reaches any workable budget in a handful of passes from * any starting point, and it converges — a linear step sized for the common * case is either too slow for the bad case or overshoots the common one. */ export declare function nextManifestChunkByteBudget(current: number): number; /** * Retry floor for a scope that has failed `consecutiveFailures` times in a row. * * `1 → 1h, 2 → 2h, 3 → 4h … capped at 24h`. Zero (or anything nonsensical) * means "no floor": a scope with no failure history is governed by the normal * `lastUploadAt` throttle alone. */ export declare function manifestFailureBackoffMs(consecutiveFailures: number): number; /** * Only an explicit, unambiguous value disables. Empty string, `0`, `false` and * anything else mean "not set": env vars are routinely exported empty by * shell wrappers, and treating that as "off" would silently disable the * feature on machines nobody meant to opt out. */ /** * The kill switch, readable by a CALLER. * * `runManifestUploadPass` enforces it itself, but a caller that assembles the * pass (resolving identity, minting a transport, choosing scopes) needs to * check it BEFORE doing any of that, so an opted-out machine pays literally * nothing. Exported rather than duplicated so the two can never disagree about * what counts as "disabled". */ export declare function isManifestUploadDisabled(env: NodeJS.ProcessEnv): boolean; /** Raw HTTP response, reduced to the two things the pass reasons about. */ export interface ManifestUploadResponse { status: number; body?: unknown; } /** * Injected by hq-sync / hq-cli: POSTs one chunk to * `POST /v1/sync-manifest/upload` with the caller's own JWT and base URL. * * It must RESOLVE with the status for any HTTP answer (including 4xx/5xx) and * may reject only for transport-level failures; a rejection is caught and * reported as a failed pass either way. */ export type ManifestUploadTransport = (chunk: SyncManifestUpload, ctx: { attempt: number; }) => Promise; export interface ManifestUploadLogger { info?: (message: string, meta?: Record) => void; warn?: (message: string, meta?: Record) => void; error?: (message: string, meta?: Record) => void; } export interface RunManifestUploadPassOptions { scope: BuildManifestScope; hqRoot: string; stateDir: string; /** The SAME installation id the client-health heartbeat reports. */ installationId: string; machineId: string; source: SyncManifestSource; transport: ManifestUploadTransport; /** Injected clock (epoch ms) — tests pin it; production omits it. */ now?: number; /** Injected env — tests pin it; production reads `process.env`. */ env?: NodeJS.ProcessEnv; /** `--full`: skip the delta base and send a fresh baseline. */ forceFull?: boolean; /** `--print`: build only. Never uploads, never persists a snapshot. */ dryRun?: boolean; /** * Explicit, human-initiated invocations (`hq sync manifest`) may bypass the * throttle. The DAEMON path must not — that is the whole point of it. */ ignoreThrottle?: boolean; minIntervalMs?: number; abortSignal?: AbortSignal; /** Wall-clock bound on the tree WALK (forwarded to the builder). */ maxWallMs?: number; /** * Forwarded to the builder. Omitted means the builder's stat-only default, * which is what the daemon must use. `full` is reserved for an explicit * human invocation (`hq sync manifest --full-hash`). */ hashPolicy?: ManifestHashPolicy; /** `auto` only, forwarded to the builder: files this pass may read. */ maxHashedFiles?: number; /** `auto` only, forwarded to the builder: wall clock spent hashing. */ maxHashMs?: number; /** * Wall-clock bound on the whole UPLOAD phase. See * {@link DEFAULT_MANIFEST_UPLOAD_WALL_MS}. Measured against the real clock, * not `now` — `now` pins the manifest's CONTENT (so a pass is reproducible), * while this is about how long a caller is prepared to be blocked. */ uploadWallMs?: number; logger?: ManifestUploadLogger; /** Test seam, forwarded to the builder: stands in for the real `listJournals()`. */ listJournalsImpl?: () => JournalSummary[]; /** * Test seam, forwarded to the builder: entries per chunk. Production leaves * it unset and gets the contract ceiling; tests shrink it so the multi-chunk * paths (ordering, shared header, mid-pass abort) are actually reachable. */ maxEntriesPerChunk?: number; /** * Serialised-byte budget for one chunk. * * Unlike `maxEntriesPerChunk` this is NOT purely a test seam — an explicit * value here overrides the self-healed budget stored on the snapshot, which * is how an operator un-sticks a scope that halved its way down to the floor * against a transient gateway fault. Omitted (the normal case) means "the * stored budget, or the contract default". */ chunkByteBudget?: number; } export type ManifestUploadPassStatus = "disabled" | "throttled" | "locked" | "uploaded" | "printed" | "soft_skipped" | "resend_full_scheduled" /** * The tree walk did not finish (wall clock or abort), so the manifest * describes only the part of the scope the walker reached. NOTHING is * uploaded on this path: a truncated walk of a large vault can legitimately * yield ZERO entries, and a full-mode manifest with zero entries tells the * server this client holds nothing at all — which it would then act on. The * pass burns its sequence, arms the failure backoff, and retries later. */ | "walk_truncated" /** * The built manifest is a FULL statement that this machine holds nothing it * has ever tracked — every entry `source: "disk"` — while a prior local * snapshot for the same scope says otherwise. NOTHING is uploaded. * * Found on a real machine: every generation of the scope's v3 ledger store * (`snapshot-*.bin` and `wal-*.bin`) was truncated and no legacy shard * existed, so the ledger read came back empty, all 188,997 files walked read * as on-disk-never-tracked, and the pass uploaded that as the truth. The * ledger read itself is fixed (an unreadable ledger is no longer an empty * one), but this is the backstop for the class: whatever makes the ledger * silently disappear next time, a scope does not go from "fully tracked" to * "nothing tracked" in one cycle. */ | "ledger_implausible" | "failed"; export type ManifestUploadFailureKind = "build_error" | "transport_error" | "aborted" | "upload_timeout" | "chunk_too_large" | "manifest_too_large" /** Every chunk was accepted but the server refused (or never finished) the fold. */ | "not_materialised" | "rejected"; export interface ManifestUploadFailure { kind: ManifestUploadFailureKind; /** HTTP status, when the failure came from a response rather than a throw. */ status?: number; /** Error CLASS name or server reason code — never a message with paths in it. */ detail?: string; chunkIndex?: number; } export interface RunManifestUploadPassResult { status: ManifestUploadPassStatus; scopeKey: string; snapshotId?: string; mode?: SyncManifestMode; chunkCount?: number; uploadedChunks?: number; /** The byte budget this pass planned against; useful in a 413 post-mortem. */ chunkByteBudget?: number; /** True when the NEXT pass must send a full manifest. */ nextModeFull?: boolean; /** Populated only for `dryRun` — the built chunks, unsent. */ chunks?: SyncManifestUpload[]; error?: ManifestUploadFailure; } /** Per-scope upload state, for `hq doctor`'s `manifest` check. */ export interface ManifestUploadStatus { scopeKey: string; lastUploadAt: string | null; snapshotId: string | null; sequence: number; /** False when the stored record is bookkeeping-only (next pass is full). */ baseUsable: boolean; } /** * The `command` recorded in the lock file. It shows up verbatim in * `OperationLockedError` messages and in `hq doctor`-style lock dumps, so it * names the operation a human would recognise. */ export declare const MANIFEST_UPLOAD_LOCK_COMMAND = "sync-manifest"; /** * Build and upload one scope's manifest. * * Chunks go up SEQUENTIALLY in ascending `chunkIndex`: the server folds them * into one snapshot keyed by a hash of the shared header fields, and a * concurrent send would make a partial-failure state (some chunks in, some * not) impossible to reason about from the client side. */ export declare function runManifestUploadPass(options: RunManifestUploadPassOptions): Promise; /** * Read per-scope upload state without touching the network or the disk tree. * * This is the seam `hq doctor`'s `manifest` check consumes: it answers "when * did this machine last tell the server what it holds?" for each scope, which * is the one question that distinguishes "no findings" from "no data". * Missing/older snapshots report `lastUploadAt: null` — never uploaded. */ export declare function readManifestUploadStatus(stateDir: string, scopes: readonly BuildManifestScope[]): ManifestUploadStatus[]; /** Re-exported so callers do not need a second import for the staleness rule. */ export { MANIFEST_SNAPSHOT_MAX_AGE_MS }; //# sourceMappingURL=upload-manifest.d.ts.map