// Typed records become obligatory on agent→agent traffic (Phase 5.1 Task 12). // // THE MECHANISM ALREADY SHIPPED AND ADOPTION WAS THE WHOLE GAP. // `hooks/tier.mjs:injectLine` slims a multi-line message to *first line + // `[+N lines · record: · retrieve_message id=]`* — but ONLY when // `record.type` is a string. Measured on the groundwork room 2026-08-30: 206 // messages, 175 multi-line, 33 typed, 33 slimmed; the coordinator had sent 101 // and typed 0. The renderer was never the problem. // // SO THE RULE IS ENFORCED AT THE SEND, NOT THE RENDER. An untyped agent→agent // message must not be able to EXIST, or the next reader re-derives the type // from prose and we are back to parsing English. Prose is a second source of // truth about what a message meant. // // Three things this module is careful about, each one a stated acceptance // property rather than an implementation taste: // // 1. STAGED, WITH A NAMED DATE. Five fleets share this bus and 7 of 13 server // processes predate 0.26.7. A refusal shipped before a verified restart // rejects messages from agents that CANNOT comply — the sender gets an // error it has no code path to satisfy. So: WARN until the cutover, REFUSE // after it, and the warning names the date in every single message so the // deadline arrives having been announced ~every send rather than once. // // 2. THE REFUSAL NAMES A TYPE THE SENDER MAY ACTUALLY USE. `go` and `scope` // are coordinator-restricted and `verdict` is gate-runner-restricted // (RECORD_AUTHORITY): the aide hit exactly this, being told to use `scope` // by one check and refused it by another. Naming a forbidden type is // unusable guidance, so every suggestion is filtered through what this // sender's role may emit before it is offered. // // 3. `fyi` IS AN HONEST CATCH-ALL. If the nine types do not cover a legitimate // send, agents mistype into whatever passes and the type becomes noise — // which costs more than the untyped message did. Anything unrecognised // suggests `fyi` and says so plainly, never a forced `decision`/`risk`. // The named cutover. Until this instant an untyped agent→agent send WARNS and // is stored; from it, the same send is refused. // // A DATE, NOT A VERSION, because the constraint being waited on is a fleet-wide // RESTART, and no version string on the wire can answer "has every server // restarted" — 0.26.7 is published while servers predating it still run. Time // is the only clock every one of those processes shares. export const TYPED_RECORD_CUTOVER_ISO = "2026-09-15T00:00:00.000Z"; export const TYPED_RECORD_CUTOVER_MS = Date.parse(TYPED_RECORD_CUTOVER_ISO); export type TypedRecordMode = "warn" | "refuse"; // `AGENT_COORD_TYPED_RECORDS=warn|refuse` overrides the date. It exists for two // callers and no others: tests (which must exercise both sides of a date that // has not arrived) and an operator who needs to pull the refusal back for a // fleet mid-incident. It is NOT the exemption — an exemption is per-agent and // visible (see `proseOnly`); this switch is fleet-wide and invisible, which is // exactly why it is not how a less capable model gets on the bus. export function typedRecordMode(now: number = Date.now()): TypedRecordMode { const env = process.env.AGENT_COORD_TYPED_RECORDS; if (env === "warn" || env === "refuse") return env; return now >= TYPED_RECORD_CUTOVER_MS ? "refuse" : "warn"; } // ---------- suggesting the type the sender should have used ---------- // A rejection that says only "record required" costs a round trip; one that // says "looks like a `done` — it cites a PR" costs none. A FAILURE PATH MUST BE // MORE INFORMATIVE THAN THE SUCCESS PATH HERE, because it fires during a // migration, at agents that are mid-slice and did nothing wrong yesterday. // // Read from the text the fleet already writes: the six canonical prefixes are // in every card and on ~every message, so the type is usually already declared // in English one token in. const PREFIX_TYPES: Array<[RegExp, string, string]> = [ [/^\s*DONE\b/i, "done", "the message opens with the DONE: prefix"], [/^\s*BLOCKER\b/i, "blocker", "the message opens with the BLOCKER: prefix"], [/^\s*RISK\b/i, "risk", "the message opens with the RISK: prefix"], [/^\s*FYI\b/i, "fyi", "the message opens with the FYI: prefix"], [/^\s*AGENT_ACTION\b/i, "action", "the message opens with the AGENT_ACTION: prefix"], [/^\s*DAVID_DECISION\b/i, "decision", "the message opens with the DAVID_DECISION: prefix"], [/^\s*GO\b/i, "go", "the message opens with the GO: prefix"], [/^\s*SCOPE(?:\s+CHANGE)?\b/i, "scope", "the message opens with the SCOPE: prefix"], ]; // Weaker than a prefix and only consulted when there is no prefix at all. const SHAPE_TYPES: Array<[RegExp, string, string]> = [ [/\b(PASS|FAIL)\b.*\b[0-9a-f]{7,40}\b/, "verdict", "it reads as a gate verdict over a commit sha"], [/^\s*(PASS|FAIL)\b/, "verdict", "it opens with a PASS/FAIL gate result"], [/(^|\s)(#\d+|[\w.-]+\/[\w.-]+#\d+|https:\/\/github\.com\/\S+\/pull\/\d+)/, "done", "it cites a PR"], ]; export type RecordSuggestion = { type: string; /** Why this type — quoted back to the sender so the guess is auditable. */ why: string; /** Set when the best-fitting type is one this sender's role may not emit. */ downgradedFrom?: string; }; /** * The record type this message most likely should have carried, CONSTRAINED to * what this sender is allowed to emit. * * `mayNotEmit` is the restricted set this role is refused (recordAuthorityFor). * A suggestion landing in it is downgraded to `fyi` and the downgrade is * reported rather than hidden — being quietly steered off `go` reads as the * heuristic being bad, when in fact the role simply does not own that type. */ export function suggestRecordType(text: string, mayNotEmit: readonly string[] = []): RecordSuggestion { const forbidden = new Set(mayNotEmit); const firstLine = (text ?? "").split("\n", 1)[0] ?? ""; const hit = PREFIX_TYPES.find(([re]) => re.test(firstLine)) ?? SHAPE_TYPES.find(([re]) => re.test(firstLine) || re.test(text ?? "")); if (!hit) { return { type: "fyi", why: "nothing in the text names a kind — 'fyi' is the honest catch-all and is never wrong on purpose", }; } const [, type, why] = hit; if (forbidden.has(type)) { return { type: "fyi", downgradedFrom: type, why: `${why}, but '${type}' is restricted to roles this sender does not hold — 'fyi' carries the same slimming`, }; } return { type, why }; } /** * The one line of guidance appended to both the warning and the refusal. Same * words either side of the cutover ON PURPOSE: an agent that reads it during * the warn window has already been told the exact call that will keep working. */ export function typedRecordGuidance(s: RecordSuggestion): string { const cite = s.type === "done" ? `, cites: [{kind:'pr', ref:''}]` // a `done` is refused without one anyway : ""; return ( `looks like a '${s.type}' — ${s.why}. ` + `Add record: {type:'${s.type}', payload:{summary:''}${cite}} to this same call; ` + `'text' is untouched by the record and your wording always wins.` ); } // ---------- measuring adoption, and the inverse failure ---------- // SUCCESS IS THE TRAFFIC TABLE RE-MEASURED, NOT THIS TASK MERGED. The baseline // captured the day the rule was made: 212 room messages, 37 typed (17%), // 425,903 bytes sitting below line 1 — bytes every reader pays for and no // reader asked for. // // AND THE INVERSE FAILURE LOOKS EXACTLY LIKE SUCCESS ON A COVERAGE NUMBER. If // `fyi` becomes nearly everything, coverage reads ~100% while agents are // mistyping to get past the gate and the type has stopped carrying // information — that is the catch-all rule (12.5) failing, not holding. So the // distribution is reported alongside the percentage and is what the verdict // keys on. A coverage-only check would certify the failure it exists to catch. // // The threshold is deliberately generous: `fyi` IS the honest answer for a // large share of real traffic, so this fires on domination, not on presence. export const FYI_DOMINANCE_THRESHOLD = 0.75; export type TypedRecordStats = { total: number; typed: number; coverage: number; /** Bytes below the first line of UNTYPED messages — the cost still being paid. */ unslimmedBytes: number; /** count by record.type, plus `untyped`. */ distribution: Record; fyiShareOfTyped: number; verdict: "healthy" | "adopting" | "degenerate"; note: string; }; /** `messages` is any array of stored messages ({text, record?}). */ export function typedRecordStats(messages: ReadonlyArray<{ text?: string; record?: { type?: string } }>): TypedRecordStats { const distribution: Record = {}; let typed = 0; let unslimmedBytes = 0; for (const m of messages) { const type = typeof m.record?.type === "string" ? m.record.type : undefined; distribution[type ?? "untyped"] = (distribution[type ?? "untyped"] ?? 0) + 1; if (type) { typed++; continue; } const text = m.text ?? ""; const nl = text.indexOf("\n"); if (nl !== -1) unslimmedBytes += Buffer.byteLength(text.slice(nl + 1), "utf8"); } const total = messages.length; const coverage = total ? typed / total : 0; const fyiShareOfTyped = typed ? (distribution.fyi ?? 0) / typed : 0; // Order matters: degeneracy is checked BEFORE coverage, or a room that is // 100% `fyi` reports "healthy" — the exact misreading this guards. const degenerate = typed >= 10 && fyiShareOfTyped > FYI_DOMINANCE_THRESHOLD; const verdict = degenerate ? "degenerate" : coverage >= 0.95 ? "healthy" : "adopting"; const note = degenerate ? `'fyi' is ${(fyiShareOfTyped * 100).toFixed(0)}% of typed messages — coverage looks solved while the type has ` + `stopped carrying information. Agents are typing to pass the gate, which is 12.5 FAILING, not holding.` : verdict === "healthy" ? `${(coverage * 100).toFixed(0)}% typed with a spread distribution — the reader-side cost is paid down and the type still discriminates.` : `${(coverage * 100).toFixed(0)}% typed; ${unslimmedBytes.toLocaleString("en-US")} bytes still sit below line 1 in untyped messages, ` + `carried by every reader.`; return { total, typed, coverage: Number(coverage.toFixed(3)), unslimmedBytes, distribution, fyiShareOfTyped: Number(fyiShareOfTyped.toFixed(3)), verdict, note }; }