/** * The v2 telemetry transport: an append-only claim protocol * (docs/CROSS_PROCESS_FLUSH_DESIGN.md §3.2, stage 2). * * Three independent actors share this file format — this module (W1, the CLI), * packages/orchestrator/src/state-machine.ts appendTelemetryQueueEntry (W2, * append-only, no network, must not depend on this package), and * templates/cursor-hooks/session-start.mjs (W3, dependency-free mirror of the * flusher below, per the license-gate spec-plus-mirror discipline). Change the * protocol here only with matching changes there. * * Why not the v1 mutable array: every v1 operation was read-modify-write, so a * concurrent append could be clobbered by a flusher's save (lost append), two * flushers could POST the same snapshot (double-send), and a reader could parse * a half-written array, hit the corrupt-file catch, and "start fresh" — losing * the whole queue (torn read). The journal closes all three structurally: * appends are single O_APPEND writes, flushers get exclusivity by atomic * rename, and a torn LINE is skippable without losing the rest of the file. * * MUTUAL EXCLUSION (ES-P0-JOURNAL-WINDOWS-EXACTLY-ONCE): flushers serialize on * an exclusive-create lock file, NOT on rename. Concurrent renames of one * source on Windows can both report success while only one destination * materialises (measured, s72), so rename cannot be the fencing primitive. * fs.openSync(path, "wx") maps to CREATE_NEW, which IS kernel-atomic on both * Windows and POSIX: exactly one creator wins. The rename inside claimJournal * is demoted to a data move that only ever happens while holding the lock, so * the anomalous concurrent-rename case cannot arise between current-version * flushers. Defense in depth, outermost first: the lock (primary fence), a * read-back of the lock token after create (closes the stale-takeover * create/unlink race), a token re-verify immediately before the rename, and * the post-rename statSync (backstop for version skew with 4.30.x flushers * that do not take the lock). * * Residual windows, accepted and bounded — every one degrades to DUPLICATION, * never loss, and the stage-1 server dedup (event_id idempotency, 4.29.5) * absorbs duplicates: (1) an HTTP POST whose client-side timeout fires after * the server already processed the batch is an irreducible at-least-once * delivery; the batch re-appends with its ORIGINAL event_ids and the ingest * keeps one row. (2) A claim file the OS refuses to unlink (Windows EPERM) is * left for the orphan sweep, which re-appends its content. (3) During version * skew, an old flusher that renames without the lock re-opens the s72 rename * anomaly against a new flusher; the statSync backstop turns that into a * skipped cycle exactly as 4.30.1 did. (4) A stale-lock takeover racing a * sub-millisecond fresh acquisition can, in the worst interleaving, orphan the * fresh owner's claim; the sweep recovers it by append. Duplication is always * preferred over loss (D-3). */ export declare const JOURNAL_BASENAME = ".telemetry-journal.jsonl"; export declare const V1_QUEUE_BASENAME = ".telemetry-queue.json"; /** Claim files: `.telemetry-journal.-.flushing`. */ export declare const CLAIM_PATTERN: RegExp; /** A claim this old belongs to a crashed flusher — its events go back to the live journal. */ export declare const ORPHAN_AGE_MS: number; /** * Past this age a claim is recovered even if a process with the owner's pid * still runs: pids recycle, and an eternal liveness veto would strand the * segment forever. No flush cycle approaches an hour, so a same-pid process * this much later is somebody else. */ export declare const HARD_ORPHAN_AGE_MS: number; /** Enforced at CLAIM time, not append time — appenders stay O(1) and dependency-free (§3.3). */ export declare const QUEUE_CAP = 200; /** The flush lock: exclusive-create arbitration among flushers. Writers never touch it. */ export declare const FLUSH_LOCK_BASENAME = ".telemetry-journal.flush-lock"; /** * A lock this old belongs to a crashed flusher and may be taken over. A flush * cycle is bounded by batches-of-50 x 5s timeouts over a 200-cap claim, well * under this; a live slow flusher merely costs its rivals skipped cycles (C-2). */ export declare const LOCK_STALE_MS: number; export interface JournalEvent { event_type: string; skill_id: string; event_id?: string; session_id?: string; duration_ms?: number; success?: boolean; metadata?: Record; } export declare function journalFile(): string; export declare function v1QueueFile(): string; export declare function flushLockFile(): string; /** One JSON object per line, trailing newline — the whole batch in ONE write. */ export declare function toJournalLines(events: object[]): string; /** A torn line is skippable damage (§3.2) — parse what parses, drop the rest. */ export declare function parseJournalLines(text: string): JournalEvent[]; export declare function claimBasename(pid: number, ts: number): string; /** Owner token stored in the lock file: `--`. */ export declare function lockToken(pid: number, ts: number, nonce: string): string; export declare function parseLockTimestamp(content: string): number; /** * Staleness from the timestamp embedded in the token (a crashed flusher can't * refresh anything); mtime is the fallback for content that doesn't parse. */ export declare function isStaleLock(content: string, mtimeMs: number, now: number): boolean; /** Owner pid from a claim basename, NaN when the name doesn't parse. */ export declare function claimOwnerPid(basename: string): number; /** * Liveness check for recovery: a claim whose owner still runs is NOT orphaned * regardless of age (a debugger-paused flusher must not have its segment * recovered out from under it). EPERM means "alive but not ours to signal". */ export declare function isPidAlive(pid: number): boolean; /** * Age from the timestamp embedded in the claim name — a crashed flusher can't * update an mtime, and copying tools can. mtime is the fallback for a name * whose timestamp doesn't parse. */ export declare function isOrphanedClaim(basename: string, mtimeMs: number, now: number): boolean; /** Cap at claim time: oldest dropped, count reported to the caller (§3.3). */ export declare function capClaimedEvents(events: T[], cap?: number): { kept: T[]; dropped: number; }; /** * Acquire the flush lock; null means another flusher holds it (skip the * cycle, C-2). Exclusive-create ("wx" → CREATE_NEW) is the arbiter — the one * primitive that IS kernel-atomic across processes on Windows, unlike * concurrent rename. After winning the create, the token is READ BACK: a * rival taking over a stale lock unlinks by PATH, so in the worst * interleaving our fresh lock file can be replaced between our write and our * first use — whoever's token is in the file owns the lock, everyone else * backs off. Takeover of a stale lock re-reads immediately before the unlink * and only then loops back to the create, which re-arbitrates. */ export declare function acquireFlushLock(now?: number): string | null; /** True while the lock file still carries OUR token. */ export declare function verifyFlushLock(token: string): boolean; /** Release only what is still ours — never unlink a rival's takeover. */ export declare function releaseFlushLock(token: string): void; /** Lock-free append — the only write path for W1/W2. Never reads. */ export declare function appendJournalEvents(events: object[]): void; /** * D-4: drain the v1 array file — read once, append each event to the journal * (fresh event_id where absent, so the stage-1 dedup covers a concurrent * double-drain), then delete. The delete is the last v1 mutation anywhere in * the codebase. A v1 file that does not PARSE is left alone: deleting on a * torn read is exactly the amplification §1.1 describes, and a live v1 writer * will overwrite it with a valid array on its next append. */ export declare function drainV1Queue(): void; /** * §3.2 recovery: claims older than ORPHAN_AGE_MS whose owner process is GONE * are re-appended to the live journal (append, never overwrite — D-3), then * removed. Two gates protect a segment owned by another process: the age gate * (unchanged), and a liveness check on the pid embedded in the claim name — a * paused or debugged flusher older than the age gate is still not recovered * out from under a running owner. Recovery MUST run under the flush lock * (callers pass their held token), so it can never race a live flusher's * claim/release either. Append lands BEFORE the remove, so a crash between * the two duplicates rather than loses. Bounded (one directory listing) and * loud (onRecover) — never deletes unsent. */ export declare function sweepOrphanedClaims(heldToken: string, onRecover?: (basename: string, eventCount: number) => void, now?: number): void; /** * The claim: move the journal aside for exclusive draining. REQUIRES the * flush lock — the token is re-verified immediately before the rename, so the * rename only ever runs under mutual exclusion and the Windows concurrent- * rename anomaly (both renames of one source report success, one destination * materialises — measured, s72) cannot arise between current-version * flushers. Null means no journal, no lock, or Windows refusing the rename * while an appender briefly holds the file — all "do not flush this cycle", * none an error (C-2). */ export declare function claimJournal(heldToken: string, now?: number): string | null; /** * Release a claim after flushing: re-read for surplus lines a racing POSIX * appender landed after the claimer's read (fd opened pre-rename), append * them back to the live journal, THEN unlink. The unlink happens ONLY when * the read and the restore both succeeded — a claim we could not fully read, * or whose surplus we could not re-append, is LEFT for the orphan sweep * (its already-sent portion resends and the stage-1 dedup absorbs it; * unlinking it would be the loss the design exists to kill). A refused * unlink likewise leaves the file for the sweep. */ export declare function releaseClaim(claimedPath: string, consumedCount: number): void; //# sourceMappingURL=flush-journal.d.ts.map