/** * S1 value-verdict harness — the RESUMABLE FIRM-RUN DRIVER. SPEC-S1-value-harness.md §11 ("the human's post-step * that wires this scaffold into a runnable batch CLI"). Enumerates the firm cells {C1,C2,C4} × the arms each trap * declares × N seeds, runs each MISSING cell via `runArm` against a REAL `ProfileDeps` (live-deps.ts), records an * append-only JSONL LEDGER keyed by (taskId, arm, seed), and at the end builds the `ValueJudgmentReport` via * `buildS1Report`. The report IS the output — this driver never asserts which arm wins. * * 🔴 RESUMABLE (clay's hard requirement): on start the driver READS the ledger, builds the done-set, and runs ONLY * missing cells. A completed SCORED cell is NEVER re-run. A cell that throws / times out is recorded * `runStatus:"infra-failed"` (EXCLUDED from scoring, NOT a fake loss) AND left re-runnable on the next pass — so * a hang/crash mid-run resumes cleanly. Per-cell wall-clock timeout → kill the cell, mark infra-failed, move on. * * 🔴 --dry-run: a MOCK ProfileDeps (no E2B/DeepSeek) proves the driver loop + ledger + resume + buildReport work * DETERMINISTICALLY (this is what the test + CI exercise; the REAL run is the owner's gated step). * * 🔴 ENV GATING: the live path requires E2B_API_KEY + DEEPSEEK_API_KEY; absent → the driver refuses the live path * with a clear message (never a silent fake run). The deterministic --dry-run never needs keys. */ import { Runner } from "@sema-agent/core"; import { type Arm } from "./arms.js"; import { type TrapSpec } from "./tasks.js"; import { type BenchBudget } from "./runner-ctx.js"; import { type RawRow, type ValueJudgmentReport } from "./row.js"; import { type LiveRuntimeConfig, type LiveCell } from "./live-deps.js"; /** A ledger key uniquely identifies one cell (the resume primary key). */ export interface CellKey { taskId: string; arm: Arm; seed: number | string; } /** One ledger record on disk: the emitted RawRow (a completed cell) — re-aggregated on the final pass. */ export type LedgerRecord = RawRow; /** The driver options. */ export interface RunFirmOptions { /** Seeds per (trap, arm). Default 5 (clay: comprehensive → CLI `--seeds N`). */ seeds?: number; /** Optional cell filters for a controlled live run (single-cell smoke / re-run one trap-arm). `only` = a trapId * (e.g. "C1"), `arm` = a lowercase arm tag ("solo"|"sup"|"team"). The ledger keys are unchanged → a filtered run * still resumes/merges into the same run. */ only?: string; arm?: string; /** Per-cell wall-clock timeout ms. A cell that exceeds it is killed → infra-failed → re-runnable. Default 15min. */ cellTimeoutMs?: number; /** The append-only JSONL ledger path. */ ledgerPath: string; /** The output report JSON path. */ reportPath: string; /** The shared budget every arm runs on (the §3.2 fairness root). */ budget: BenchBudget; /** Provenance. */ runId: string; gitSha: string; /** DRY-RUN: inject a MOCK deps factory (no E2B/DeepSeek). The live path uses buildLiveDeps when this is absent. */ mockDepsFactory?: (trap: TrapSpec, seed: number | string, cellId: string) => LiveCell; /** The live runtime config (required for the live path; ignored in dry-run). */ liveRuntime?: LiveRuntimeConfig; /** Observability sink. */ log?: (msg: string, meta?: Record) => void; } /** The cell + the seam through which it runs (the SUP arm needs a per-cell CheckpointStore). */ interface PlannedCell extends CellKey { trap: TrapSpec; } /** Enumerate the firm cells = FIRM_TRAPS × trap.arms × seeds. Optional `only` (trapId) / `arm` filters slice the * set for a controlled live run — a single-cell smoke (`--only C1 --arm solo --seeds 1`) before the full sweep, or * re-running one trap/arm. The ledger keys are unchanged, so a filtered run still resumes/merges into the same run. */ export declare function enumerateCells(seeds: number, filter?: { only?: string; arm?: string; }): PlannedCell[]; /** A stable string key for a cell (the ledger dedupe key). */ export declare function cellKeyStr(k: CellKey): string; /** Read the ledger, returning ALL records + the done-set of SCORED-or-otherwise-completed cell keys. A record with * `runStatus:"infra-failed"` is NOT in the done-set (it is re-runnable — the resume contract). */ export declare function readLedger(ledgerPath: string): Promise<{ records: LedgerRecord[]; done: Set; }>; /** * Run the firm batch. Returns the final `ValueJudgmentReport` (also written to `reportPath`). Resumable: re-running * continues where it stopped, never re-running a completed scored cell. */ export declare function runFirm(opts: RunFirmOptions): Promise<{ report: ValueJudgmentReport; scored: number; excluded: number; ranThisPass: number; }>; /** Parse a CLI flag as a positive integer; default when the flag is absent; exit(2) on a present-but-invalid value * (0, negative, NaN, fractional) rather than silently coercing it. */ export declare function positiveIntArg(raw: string | undefined, def: number, flag: string): number; /** CLI entry. */ export declare function main(argv?: string[]): Promise; export { buildLiveDeps, liveRuntimeConfigFromEnv } from "./live-deps.js"; export { Runner }; //# sourceMappingURL=run-firm.d.ts.map