/** * Shared plumbing for the Claude-backed role implementations: the * construction context they all take, and the few helpers that keep every * role a thin composition over the leaf executor (claude-step.ts) — * phrase the task, run the session, parse the final text. */ import { type StepExecutor } from "./claude-step.js"; import type { CommandCatalog, CommandResult, CommandRunner } from "./commands.js"; import type { WorkKind } from "./model-roster.js"; import type { WorkOrder } from "./prompts.js"; import type { RepoLayout } from "./repo-layout.js"; import type { ChecklistItem, ClientPlatform, TestRun } from "./types.js"; /** * A single-test command whose suite discovered NO test matching the item * is a STRUCTURAL failure: the checklist item is misplaced — its id * names no spec this suite executes — and no amount of implementation * fixing can ever turn it green. Failing fast with the real cause beats * burning the repair budget (live-run lesson: a fake-vs-real parity item * landed in an e2e checklist and ate three repair sessions on * Playwright's "No tests found"). */ /** * Map a single-test run to a TestRun, GUARDING the phantom-pass hole: a * title-filtered run that matched no test is neither a legitimate red nor a * pass — the written test does not actually run. This returns a red TestRun * flagged `noTestsFound` (with a diagnostic the loop surfaces) rather than * THROWING: a phantom test is a writer mistake the loop can repair by * rewriting within budget — it must not crash the whole run (it did: a live * run ended when a writer's reported title didn't match the spec). */ export declare function toTestRunGuardingNoTests(result: CommandResult, item: ChecklistItem, suite: string, testTitle?: string): TestRun; export interface RoleContext { /** Absolute repository root — every session's working directory. */ repoRoot: string; layout: RepoLayout; commands: CommandCatalog; runner: CommandRunner; /** * Resolves a work order for a role's INTERNAL model needs — the hybrid * executors whose interface method carries no order (e.g. a local * acceptance run that must judge its own failure). Wired by * buildLiveDeps as workOrderFor over the run's deps. */ orderFor: (kind: WorkKind) => WorkOrder; step?: StepExecutor; } export declare function stepOf(context: RoleContext): StepExecutor; /** * Runs are UNSUPERVISED: no human is available mid-run, so a session that * hits a judgment call decides it and reports the decision — it never * stops to ask. Appended to EVERY session's task (stepOf), harvested and * stripped from every report (surfaceDilemmas), so parsers downstream see * clean output while the console carries the decision unmissably. */ export declare const DILEMMA_PROTOCOL: string; /** * Print a decided dilemma louder than anything else in the run log — the * unsupervised user's contract is "decide, but make sure I cannot miss * that you did". */ export declare function reportDilemma(context: string, dilemma: string): void; /** Harvest DILEMMA lines from a session report: log each, return the rest. */ export declare function surfaceDilemmas(context: string, report: string): string; /** * Orders for client-bound kinds arrive with the all-clients scope; a role * that knows its platform narrows to its own subtree before launching. */ export declare function narrowedToPlatform(order: WorkOrder, layout: RepoLayout, platform: ClientPlatform): WorkOrder; /** * Every session does EXACTLY its step's job: the algorithm runs the * neighboring steps as their own sessions, and a step that "helpfully" * does their work corrupts the phase (a spec written during DSL-adding * made a later step believe a whole slice was already done). */ export declare const STEP_DISCIPLINE: string; /** * The run instructions travel to every session (algorithm.md: * "Instructions travel"). Empty instructions add nothing. */ export declare function withInstructions(task: string, instructions: readonly string[]): string; /** Failure outputs travel into repair sessions, capped like fixDanglingReference. */ export declare function tailOf(text: string, max?: number): string; /** * A test-first loop's premature pass has TWO possible causes, and only a * judgment can tell them apart: the item's behavior may ALREADY be * implemented (a stale review finding, a resumed run, work an earlier * item pulled in) — then the fresh test is a legitimate regression test * and the item is covered — or the test is VACUOUS and must be * rewritten. Shared by the client and backend agents; the caller * resolves the order (failure-judgment: cheap tier, judgment work). */ export declare function judgePrematurePass(step: StepExecutor, order: WorkOrder, item: ChecklistItem, repoRoot: string): Promise; /** Map a catalog command's result to the algorithm's TestRun shape. */ export declare function toTestRun(result: CommandResult): TestRun; /** * Fill a catalog command's single (whatever its name — * , , ) with a value, shell-quoted. * Throws when the command has none — a catalog/checklist mismatch must be * loud, not a command that silently runs everything. * * Checklist convention this backs: ChecklistItem.id is the REPO-RELATIVE * SPEC FILE covering the item — exactly what the catalog's single-test * forms take. Several items may share a file; when the item also carries * a testTitle the titled command form narrows to the one test * (fillTitledCommand), otherwise the file-level run is the verdict's * granularity. */ export declare function fillCommand(command: string, value: string): string; /** * Fill a titled single-test command's AND * placeholders (by name — order in the catalog string does not matter). * The file narrows discovery, the title narrows execution to the ONE * test the write-test step reported — so a test that was never written * surfaces as "no tests found" instead of hiding behind the file's * other, green tests. */ export declare function fillTitledCommand(command: string, specFile: string, testTitle: string): string; /** * Every runner's title filter (jest -t, vitest -t, playwright -g) is a * REGEX matched against the test's full name. Titles are prose — and the * multi-line Given/When/Then convention makes their whitespace SHAPE * (leading newline, per-line indent) an implementation detail the * write-test step cannot be trusted to reproduce byte-for-byte. So the * filter escapes every regex metacharacter and matches any whitespace * run against any other: a single-line report finds the multi-line test * (live-run lesson: a green, existing test read as "never written" * because the reported title's whitespace differed, and the item burned * its whole budget on the mismatch). */ export declare function titleToNamePattern(title: string): string; /** * The task-text block demanding a write-test session report the exact * title of the test it added — the loop records it on the item, and the * single-test executors filter by it (fillTitledCommand). The plain-words * constraint exists because runners treat the filter as a REGEX * (Playwright -g, jest -t): metacharacters in a title would silently * match nothing and read as a missing test. */ export declare const TEST_TITLE_REPORT: string; /** Parse a write-test session's {"testTitle"} report; defects are loud. */ export declare function parseTestTitle(report: string, what: string): string;