{"version":3,"file":"eval-harness.d.ts","sourceRoot":"","sources":["../../../src/core/search/eval-harness.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AASH,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,mCAAmC,CAAC;AAE1E,OAAO,EAAE,KAAK,UAAU,EAAE,KAAK,SAAS,EAAE,KAAK,eAAe,EAAiB,MAAM,WAAW,CAAC;AAEjG,+DAA+D;AAC/D,MAAM,WAAW,aAAa;IAC7B,KAAK,EAAE,MAAM,CAAC;IACd,SAAS,EAAE,MAAM,CAAC;IAClB,SAAS,EAAE,MAAM,CAAC;IAClB,UAAU,EAAE,MAAM,CAAC;IACnB,UAAU,EAAE,MAAM,CAAC;IACnB,GAAG,EAAE,MAAM,CAAC;IACZ,wCAAwC;IACxC,CAAC,EAAE,MAAM,CAAC;IACV,uEAAuE;IACvE,QAAQ,EAAE,MAAM,CAAC;CACjB;AAED,2EAA2E;AAC3E,MAAM,WAAW,cAAc;IAC9B,WAAW,EAAE,MAAM,CAAC;IACpB,uDAAuD;IACvD,SAAS,EAAE,MAAM,CAAC;IAClB,iEAAiE;IACjE,SAAS,EAAE,MAAM,CAAC;IAClB;iEAC2D;IAC3D,qBAAqB,EAAE,OAAO,CAAC;IAC/B;0DACsD;IACtD,WAAW,EAAE,OAAO,CAAC;IACrB;;8EAE0E;IAC1E,cAAc,EAAE,MAAM,EAAE,CAAC;IACzB;;;;;;;OAOG;IACH,eAAe,CAAC,EAAE,mBAAmB,CAAC;IACtC;oEACgE;IAChE,UAAU,EAAE,MAAM,CAAC;IACnB;;;;;;OAMG;IACH,mBAAmB,EAAE,MAAM,CAAC;IAC5B;yDACqD;IACrD,QAAQ,EAAE;QACT,SAAS,EAAE,OAAO,CAAC;QACnB,MAAM,CAAC,EAAE,MAAM,CAAC;QAChB,0DAA0D;QAC1D,UAAU,CAAC,EAAE,MAAM,CAAC;QACpB,KAAK,EAAE,MAAM,CAAC;QACd,wEAAwE;QACxE,UAAU,CAAC,EAAE,MAAM,CAAC;QACpB,aAAa,CAAC,EAAE,MAAM,CAAC;QACvB;;;;;;;;;;;WAWG;QACH,OAAO,CAAC,EAAE,MAAM,CAAC;QACjB;2EACmE;QACnE,QAAQ,CAAC,EAAE,MAAM,CAAC;KAClB,CAAC;IACF;kDAC8C;IAC9C,YAAY,CAAC,EAAE;QAAE,SAAS,EAAE,OAAO,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,CAAC;IACrD;;;;;;;;;;;OAWG;IACH;;;;OAIG;IACH,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,MAAM,CAAC,EAAE;QACR,4EAA4E;QAC5E,YAAY,EAAE,MAAM,CAAC;QACrB,gEAAgE;QAChE,YAAY,EAAE,MAAM,CAAC;QACrB;;qDAE6C;QAC7C,WAAW,EAAE,OAAO,CAAC;KACrB,CAAC;IACF,OAAO,EAAE;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,MAAM,CAAA;KAAE,CAAC;CAC1D;AAED,MAAM,WAAW,aAAa;IAC7B,UAAU,EAAE,cAAc,CAAC;IAC3B,OAAO,EAAE;QAAE,UAAU,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;QAAC,aAAa,EAAE,MAAM,CAAA;KAAE,CAAC;IACxF,OAAO,EAAE,SAAS,UAAU,EAAE,CAAC;IAC/B,UAAU,EAAE,aAAa,EAAE,CAAC;IAC5B,QAAQ,EAAE,KAAK,CAAC;QAAE,EAAE,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,eAAe,EAAE,CAAA;KAAE,CAAC,CAAC;CAC3E;AAMD;0EAC0E;AAC1E,wBAAgB,mBAAmB,CAAC,QAAQ,EAAE,MAAM,GAAG,MAAM,CA+B5D;AAED,MAAM,WAAW,YAAY;IAC5B,qCAAqC;IACrC,GAAG,EAAE,MAAM,CAAC;IACZ,GAAG,EAAE,MAAM,CAAC;IACZ,eAAe,EAAE,OAAO,CAAC;IACzB,KAAK,EAAE,OAAO,CAAC;IACf;yDACqD;IACrD,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB;6CACyC;IACzC,SAAS,CAAC,EAAE,mBAAmB,CAAC;IAChC,gDAAgD;IAChD,OAAO,EAAE,MAAM,IAAI,CAAC;CACpB;AAED,6EAA6E;AAC7E,MAAM,WAAW,sBAAsB;IACtC,qEAAqE;IACrE,YAAY,EAAE,MAAM,CAAC;IACrB,8EAA4E;IAC5E,YAAY,EAAE,SAAS,MAAM,EAAE,CAAC;IAChC,oEAAoE;IACpE,IAAI,EAAE,MAAM,CAAC;CACb;AAED,iFAAiF;AACjF,MAAM,WAAW,mBAAmB;IACnC,YAAY,EAAE,MAAM,CAAC;IACrB,yEAAyE;IACzE,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;IAClB,YAAY,EAAE,MAAM,CAAC;IACrB,iEAAiE;IACjE,aAAa,EAAE,MAAM,CAAC;IACtB,IAAI,EAAE,MAAM,CAAC;CACb;AAgHD;;;;;;;;;;;;;;;;;;GAkBG;AACH,eAAO,MAAM,iBAAiB,EAAE,SAAS,MAAM,EAK9C,CAAC;AAEF;;;;;;;;;;;;;GAaG;AACH,wBAAgB,SAAS,CAAC,QAAQ,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,SAAS,EAAE,SAAS,CAAC,EAAE,sBAAsB,GAAG,YAAY,CA2ErH;AAYD,wBAAgB,iBAAiB,CAChC,QAAQ,EAAE,MAAM,EAChB,MAAM,EAAE,YAAY,EACpB,SAAS,EAAE,MAAM,EACjB,OAAO,EAAE,gBAAgB,GAAG,SAAS,EACrC,eAAe,CAAC,EAAE,MAAM,EACxB,aAAa,CAAC,EAAE,gBAAgB,EAChC,QAAQ,CAAC,EAAE,MAAM,EACjB,MAAM,CAAC,EAAE,cAAc,CAAC,QAAQ,CAAC,EACjC,aAAa,CAAC,EAAE,MAAM,GACpB,cAAc,CAiChB;AAED,wBAAgB,gBAAgB,CAAC,OAAO,EAAE,SAAS,SAAS,EAAE,GAAG,aAAa,CAAC,SAAS,CAAC,CAQxF;AAED,MAAM,WAAW,mBAAmB;IACnC,GAAG,EAAE,MAAM,CAAC;IACZ,OAAO,EAAE,SAAS,SAAS,EAAE,CAAC;IAC9B,OAAO,EAAE,SAAS,UAAU,EAAE,CAAC;IAC/B,OAAO,CAAC,EAAE,gBAAgB,CAAC;IAC3B;wEACoE;IACpE,aAAa,CAAC,EAAE,gBAAgB,CAAC;IACjC,OAAO,CAAC,EAAE,CAAC,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,SAAS,KAAK,IAAI,CAAC;CACpD;AAED,wBAAsB,YAAY,CAAC,OAAO,EAAE,mBAAmB,GAAG,OAAO,CAAC;IACzE,UAAU,EAAE,aAAa,EAAE,CAAC;IAC5B,QAAQ,EAAE,aAAa,CAAC,UAAU,CAAC,CAAC;CACpC,CAAC,CA4CD;AAED,wBAAgB,oBAAoB,CAAC,UAAU,EAAE,SAAS,aAAa,EAAE,GAAG,MAAM,CAcjF","sourcesContent":["/**\n * Eval harness: corpus pinning, provenance capture, and run records.\n *\n * The scoring math lives in `eval.ts`; this module is everything around it\n * that makes a number *comparable to a later number*. Three problems it\n * exists to solve, all of which bit the first eval round\n * (docs/hybrid-retrieval-design.md, \"Eval results\"):\n *\n *  1. **The corpus is the repo.** Retrieval is measured over hoocode itself,\n *     so every commit moves the thing being measured. A baseline taken today\n *     and a rerun taken after a retrieval change differ by both the change\n *     and the intervening commits, and nothing in the output says so. Fix:\n *     run against a detached git worktree pinned to an explicit SHA, and put\n *     that SHA in the record.\n *  2. **Nothing was recorded.** Results were printed to a terminal and\n *     hand-copied into a markdown table with no repo SHA, no embedder\n *     identity, and no index state. Fix: emit a machine-readable run record.\n *  3. **A degraded run looks like a real one.** With no embsearch binary the\n *     semantic and hybrid rows silently degrade to lexical, producing a table\n *     that is all-lexical but reads like a full sweep. Fix: `embedder` in the\n *     record, plus a per-row degraded count that the writer refuses to hide.\n */\n\nimport { createHash } from \"node:crypto\";\nimport { existsSync, readdirSync, readFileSync, statSync } from \"node:fs\";\nimport { execFileSync } from \"child_process\";\nimport { rmSync } from \"fs\";\nimport { tmpdir } from \"os\";\nimport path from \"path\";\nimport { chunkFile } from \"../embsearch/chunker.js\";\nimport type { EmbsearchService } from \"../embsearch/embsearch-service.js\";\nimport { scanRepo } from \"../embsearch/repo-scan.js\";\nimport { type EvalConfig, type EvalQuery, type EvalQueryResult, evaluateQuery } from \"./eval.js\";\n\n/** Metrics aggregated per config across the whole gold set. */\nexport interface EvalAggregate {\n\tlabel: string;\n\trecallAt1: number;\n\trecallAt5: number;\n\trecallAt10: number;\n\trecallAt50: number;\n\tmrr: number;\n\t/** Queries scored under this config. */\n\tn: number;\n\t/** How many of them ran degraded (requested retriever unavailable). */\n\tdegraded: number;\n}\n\n/** Everything needed to decide whether two run records may be compared. */\nexport interface EvalProvenance {\n\ttimestampMs: number;\n\t/** SHA of the corpus actually indexed and searched. */\n\tcorpusSha: string;\n\t/** Ref the caller asked for, before resolution (e.g. \"HEAD\"). */\n\tcorpusRef: string;\n\t/** True when the corpus came from the live working tree rather than a\n\t *  pinned worktree — results are then not reproducible. */\n\tcorpusFromWorkingTree: boolean;\n\t/** Uncommitted changes present at run time. Only meaningful (and only\n\t *  possible) when `corpusFromWorkingTree` is true. */\n\tcorpusDirty: boolean;\n\t/** Files removed from the corpus before indexing — see\n\t *  {@link CORPUS_EXCLUSIONS}. A score against a different exclusion list is\n\t *  a score against a different corpus, so it is recorded, not assumed. */\n\tcorpusExcluded: string[];\n\t/**\n\t * Set only when the run scored a deliberately shrunk corpus.\n\t *\n\t * A smaller distractor pool makes every query easier, so these metrics are\n\t * higher than a full-corpus run's and are **not** comparable to one. They\n\t * are comparable to another subsampled run with the same target and seed,\n\t * which is what makes this useful for screening model arms.\n\t */\n\tcorpusSubsample?: CorpusSubsampleInfo;\n\t/** SHA of the tree whose retrieval code ran. Usually equals `corpusSha`,\n\t *  but differs when pinning an old corpus with today's code. */\n\tharnessSha: string;\n\t/**\n\t * Content hash of `src/core/search` + the chunker. Every tuning constant\n\t * that shapes a result — the fusion cap, top-k depths, rerank weights,\n\t * chunk sizing — lives in those files, so a changed hash means the\n\t * numbers are not comparable, without this module having to maintain a\n\t * hand-copied (and inevitably stale) list of constants.\n\t */\n\tretrievalSourceHash: string;\n\t/** Embedding backend state. `available: false` means every semantic and\n\t *  hybrid row in this record degraded to lexical. */\n\tembedder: {\n\t\tavailable: boolean;\n\t\treason?: string;\n\t\t/** Indexed chunk count when the index reached `ready`. */\n\t\tchunkCount?: number;\n\t\tphase: string;\n\t\t/** Binary that served the embeddings, and its self-reported version. */\n\t\tbinaryPath?: string;\n\t\tbinaryVersion?: string;\n\t\t/**\n\t\t * Model id the daemon reported — the thing that actually identifies\n\t\t * which model produced these scores.\n\t\t *\n\t\t * This used to be inferred from `binaryVersion`, on the reasoning that\n\t\t * the model was baked into the binary at build time. `--model <dir>`\n\t\t * ends that: one binary now serves any number of models, so two arms of\n\t\t * a model comparison would have carried identical provenance and been\n\t\t * indistinguishable in the record. The id is a hash over the model's\n\t\t * whole spec (pooling, token limit, prefixes), so a change to any of\n\t\t * them shows up here.\n\t\t */\n\t\tmodelId?: string;\n\t\t/** Model directory passed as `--model`, when the run overrode the\n\t\t *  bundled model. Absent means the binary's own model was used. */\n\t\tmodelDir?: string;\n\t};\n\t/** Daemon-side BM25 hybrid store, when the run included one. Absent means\n\t *  the record has no `daemon-hybrid` rows. */\n\tdaemonHybrid?: { available: boolean; phase: string };\n\t/**\n\t * Wall time, split at the seam between building the index and scoring the\n\t * gold set.\n\t *\n\t * Recorded because the cost side of a model comparison is almost entirely\n\t * indexing, and a single total cannot show it: the first such comparison\n\t * could only report \"17 -> 60 min\" for whole runs and had to note that the\n\t * figure was \"not isolated from query work\", which left the headline cost\n\t * of the change unmeasured. These two numbers are machine- and\n\t * load-dependent and say nothing about retrieval quality; they are a budget,\n\t * not a metric.\n\t */\n\t/**\n\t * Chunker character cap, when an arm overrode it. Absent means the shipped\n\t * `CHUNK_MAX_CHARS`. Records differing here are not comparable: the chunks\n\t * are different text, so every id, span and vector differs.\n\t */\n\tchunkMaxChars?: number;\n\ttiming?: {\n\t\t/** Seconds spent bringing the index(es) to `ready`, model load included. */\n\t\tindexSeconds: number;\n\t\t/** Seconds spent running every config over every gold query. */\n\t\tquerySeconds: number;\n\t\t/** True when one hybrid store served both the dense and BM25 roles\n\t\t *  rather than the corpus being embedded twice. Runs with this false\n\t\t *  paid roughly double the indexing time. */\n\t\tsharedStore: boolean;\n\t};\n\truntime: { node: string; platform: string; arch: string };\n}\n\nexport interface EvalRunRecord {\n\tprovenance: EvalProvenance;\n\tgoldSet: { queryCount: number; byClass: Record<string, number>; goldSpanCount: number };\n\tconfigs: readonly EvalConfig[];\n\taggregates: EvalAggregate[];\n\tperQuery: Array<{ id: string; class: string; results: EvalQueryResult[] }>;\n}\n\nfunction git(repoRoot: string, args: string[]): string {\n\treturn execFileSync(\"git\", [\"-C\", repoRoot, ...args], { encoding: \"utf-8\" }).trim();\n}\n\n/** Hash every retrieval-shaping source file, so a tuning change is visible as\n *  a changed provenance field rather than an unexplained metric shift. */\nexport function hashRetrievalSource(repoRoot: string): string {\n\tconst roots = [\n\t\tpath.join(repoRoot, \"packages/coding-agent/src/core/search\"),\n\t\tpath.join(repoRoot, \"packages/coding-agent/src/core/embsearch/chunker.ts\"),\n\t];\n\tconst files: string[] = [];\n\tconst walk = (target: string): void => {\n\t\tlet stat: ReturnType<typeof statSync>;\n\t\ttry {\n\t\t\tstat = statSync(target);\n\t\t} catch {\n\t\t\treturn;\n\t\t}\n\t\tif (stat.isDirectory()) {\n\t\t\tfor (const entry of readdirSync(target).sort()) walk(path.join(target, entry));\n\t\t} else if (target.endsWith(\".ts\")) {\n\t\t\tfiles.push(target);\n\t\t}\n\t};\n\tfor (const root of roots) walk(root);\n\n\tconst hash = createHash(\"sha256\");\n\tfor (const file of files) {\n\t\t// Eval-only modules are excluded: changing how we measure must not look\n\t\t// like changing what we measure.\n\t\tconst base = path.basename(file);\n\t\tif (base.startsWith(\"eval\")) continue;\n\t\thash.update(path.relative(repoRoot, file).replace(/\\\\/g, \"/\"));\n\t\thash.update(readFileSync(file));\n\t}\n\treturn hash.digest(\"hex\").slice(0, 16);\n}\n\nexport interface PinnedCorpus {\n\t/** Directory to index and search. */\n\tcwd: string;\n\tsha: string;\n\tfromWorkingTree: boolean;\n\tdirty: boolean;\n\t/** Files removed from the corpus before indexing. Empty when the corpus is\n\t *  the live working tree, which is never mutated. */\n\texcluded: string[];\n\t/** Present only on a subsampled run. Its presence is what marks a record as\n\t *  incomparable to a full-corpus one. */\n\tsubsample?: CorpusSubsampleInfo;\n\t/** Removes the worktree, if one was created. */\n\tdispose: () => void;\n}\n\n/** Request to shrink the corpus to a chunk budget. See {@link pinCorpus}. */\nexport interface CorpusSubsampleRequest {\n\t/** Approximate chunk budget. Gold-bearing files are kept past it. */\n\ttargetChunks: number;\n\t/** Files that must survive regardless of budget — the gold-bearing ones. */\n\tkeepRelPaths: readonly string[];\n\t/** Seed for the distractor draw, so a budget reproduces exactly. */\n\tseed: number;\n}\n\n/** What a subsampled run did, recorded so it can never be read as a full one. */\nexport interface CorpusSubsampleInfo {\n\ttargetChunks: number;\n\t/** Chunks actually kept. Exceeds the target when gold files alone do. */\n\tchunkCount: number;\n\tfilesKept: number;\n\tfilesDropped: number;\n\t/** Gold-bearing files, all of which are kept unconditionally. */\n\tgoldFilesKept: number;\n\tseed: number;\n}\n\n/**\n * Cut `dir` down to a chunk budget, in place.\n *\n * Counts chunks with the indexer's own chunker rather than estimating from\n * file size, because the budget is meant to predict indexing time and\n * indexing time is per chunk. Gold-bearing files are never candidates for\n * removal: dropping one would make its queries unanswerable and score the\n * arm on a corpus that cannot contain the answer.\n *\n * Deletion is what makes this apply to every leg at once. Filtering the\n * indexer's file list instead would shrink the dense and BM25 legs while grep\n * still walked the full tree, and the legs would then be answering about\n * different corpora.\n */\nfunction applySubsample(dir: string, request: CorpusSubsampleRequest): CorpusSubsampleInfo {\n\tconst gold = new Set(request.keepRelPaths);\n\tconst counts = new Map<string, number>();\n\tfor (const file of scanRepo(dir).files) {\n\t\t// The scanner skips `.git` as a *directory*, but a linked worktree's\n\t\t// `.git` is a file holding the path to the real gitdir — so it comes back\n\t\t// as an ordinary indexable file. Deleting it detaches the worktree from\n\t\t// the repo, and `worktree remove` then fails on a tree git can no longer\n\t\t// validate. Never a candidate.\n\t\tif (file.rel === \".git\" || file.rel.startsWith(`.git${path.sep}`) || file.rel.startsWith(\".git/\")) {\n\t\t\tcontinue;\n\t\t}\n\t\tlet content: string;\n\t\ttry {\n\t\t\tcontent = readFileSync(path.join(dir, file.rel), \"utf8\");\n\t\t} catch {\n\t\t\tcontinue;\n\t\t}\n\t\t// Deliberately the *default* cap, never the arm's.\n\t\t//\n\t\t// The budget picks which files survive, and a chunk-cap sweep must\n\t\t// compare caps over identical source text. Counting at the arm's cap\n\t\t// would let a cap-2000 arm — whose chunks are bigger, so fewer fit the\n\t\t// budget — keep far more of the repo than a cap-1000 arm, and the two\n\t\t// would then differ by corpus as well as by cap. Sizing is a secondary\n\t\t// concern to that: bigger caps produce fewer chunks and index faster\n\t\t// anyway, so the budget only ever overestimates their cost.\n\t\tconst n = chunkFile(file.rel, content).length;\n\t\tif (n > 0) counts.set(file.rel, n);\n\t}\n\n\tconst keep = new Set<string>();\n\tlet chunks = 0;\n\tlet goldFilesKept = 0;\n\tfor (const rel of counts.keys()) {\n\t\tif (!gold.has(rel)) continue;\n\t\tkeep.add(rel);\n\t\tchunks += counts.get(rel) ?? 0;\n\t\tgoldFilesKept++;\n\t}\n\n\t// Shuffle the distractors rather than taking the scan's order, which is\n\t// directory order — that would keep a few whole subtrees and drop the rest,\n\t// making the sample a slice of the repo instead of a sample of it.\n\tconst rng = seededRandom(request.seed);\n\tconst others = [...counts.keys()].filter((rel) => !gold.has(rel));\n\tfor (let i = others.length - 1; i > 0; i--) {\n\t\tconst j = Math.floor(rng() * (i + 1));\n\t\t[others[i], others[j]] = [others[j], others[i]];\n\t}\n\tfor (const rel of others) {\n\t\tif (chunks >= request.targetChunks) break;\n\t\tkeep.add(rel);\n\t\tchunks += counts.get(rel) ?? 0;\n\t}\n\n\tlet filesDropped = 0;\n\tfor (const rel of counts.keys()) {\n\t\tif (keep.has(rel)) continue;\n\t\ttry {\n\t\t\trmSync(path.join(dir, rel), { force: true });\n\t\t\tfilesDropped++;\n\t\t} catch {\n\t\t\t// A file the scanner listed but cannot be removed stays in the\n\t\t\t// corpus; it inflates the sample slightly and is not worth failing\n\t\t\t// the run over.\n\t\t}\n\t}\n\n\treturn {\n\t\ttargetChunks: request.targetChunks,\n\t\tchunkCount: chunks,\n\t\tfilesKept: keep.size,\n\t\tfilesDropped,\n\t\tgoldFilesKept,\n\t\tseed: request.seed,\n\t};\n}\n\n/**\n * Deterministic PRNG (mulberry32).\n *\n * `Math.random()` would make a \"reproducible\" subsample a different corpus on\n * every run, which is the one property this must not have.\n */\nfunction seededRandom(seed: number): () => number {\n\tlet a = seed >>> 0;\n\treturn () => {\n\t\ta = (a + 0x6d2b79f5) >>> 0;\n\t\tlet t = a;\n\t\tt = Math.imul(t ^ (t >>> 15), t | 1);\n\t\tt ^= t + Math.imul(t ^ (t >>> 7), t | 61);\n\t\treturn ((t ^ (t >>> 14)) >>> 0) / 4294967296;\n\t};\n}\n\n/**\n * Files that describe this eval rather than being searched by it.\n *\n * The fixtures hold all 62 query strings verbatim, so every query is a perfect\n * lexical match against its own entry, and the design note quotes the same\n * queries while discussing the classes they belong to. Measured before this\n * exclusion existed: **56 of 62 queries had one of these files in the top 10,\n * 27 of 62 had one as the #1 result, and they consumed 133 of the 620\n * top-10 slots** — a fifth of the window, spent on the eval reading itself.\n *\n * That is not a ranking artifact a reranker can fix: it displaces real answers\n * out of the window entirely, which is why two boundary-class queries were\n * absent from the top *50* rather than merely buried. Retrieving your own\n * question is not retrieval, so the corpus is scored without them.\n *\n * Removed from the pinned worktree before indexing, never from the repo — and\n * recorded in the run's provenance so a score is never silently taken against\n * a different corpus than it claims.\n */\nexport const CORPUS_EXCLUSIONS: readonly string[] = [\n\t\"packages/coding-agent/test/fixtures/search-eval.json\",\n\t\"packages/coding-agent/test/fixtures/search-eval-live.json\",\n\t\"packages/coding-agent/test/fixtures/search-eval-baseline.json\",\n\t\"docs/hybrid-retrieval-design.md\",\n];\n\n/**\n * Materialize the corpus to evaluate.\n *\n * With a `ref`, checks out a detached worktree at that commit so the corpus is\n * byte-identical on every rerun. Without one, falls back to the live working\n * tree and reports `dirty` so the record shows the run was not reproducible.\n *\n * `subsample` shrinks the corpus to a chunk budget, for screening runs where a\n * full arm costs too much to iterate on. It keeps every gold-bearing file and\n * draws distractors deterministically. This is a real change to what is being\n * measured — a smaller distractor pool makes retrieval easier and inflates\n * every metric — so it is recorded in the record and folded into the worktree\n * path, and it needs a `ref`.\n */\nexport function pinCorpus(repoRoot: string, ref: string | undefined, subsample?: CorpusSubsampleRequest): PinnedCorpus {\n\tconst dirty = git(repoRoot, [\"status\", \"--porcelain\"]).length > 0;\n\tif (!ref) {\n\t\tif (subsample) {\n\t\t\t// Subsampling deletes files. Against the live checkout that is the\n\t\t\t// user's source tree, so this refuses rather than asks.\n\t\t\tthrow new Error(\n\t\t\t\t\"subsampling requires --corpus-ref: it deletes files, and the working tree is not ours to cut\",\n\t\t\t);\n\t\t}\n\t\t// The live working tree is the user's checkout; deleting files from it to\n\t\t// tidy a measurement would be an unforgivable trade. Working-tree runs\n\t\t// are already stamped non-reproducible, so they carry the contamination.\n\t\treturn {\n\t\t\tcwd: repoRoot,\n\t\t\tsha: git(repoRoot, [\"rev-parse\", \"HEAD\"]),\n\t\t\tfromWorkingTree: true,\n\t\t\tdirty,\n\t\t\texcluded: [],\n\t\t\tdispose: () => {},\n\t\t};\n\t}\n\n\tconst sha = git(repoRoot, [\"rev-parse\", ref]);\n\t// Deterministic path, not mkdtemp: the embedding store is keyed by a hash of\n\t// the corpus directory, so a fresh temp path every run would re-embed all\n\t// ~17k chunks (minutes) instead of reusing the store built for this exact\n\t// SHA. The worktree is still removed afterwards; only the store persists.\n\t//\n\t// The subsample is part of the key. Without it a screening run and a full\n\t// run at the same SHA would share this path *and* the store derived from it,\n\t// so the second would silently score the first's index — a wrong number that\n\t// looks entirely normal.\n\t// No chunk cap in the key: the file set is cap-independent by construction\n\t// (see `applySubsample`), so cap arms share one worktree. The *store* key\n\t// does carry the cap, because the vectors differ.\n\tconst subsampleKey = subsample ? `-fast${subsample.targetChunks}s${subsample.seed}` : \"\";\n\tconst dir = path.join(tmpdir(), `hoocode-search-eval-${sha.slice(0, 12)}${subsampleKey}`);\n\tif (existsSync(dir)) {\n\t\t// Left behind by an interrupted run — drop it so `worktree add` succeeds.\n\t\ttry {\n\t\t\tgit(repoRoot, [\"worktree\", \"remove\", \"--force\", dir]);\n\t\t} catch {\n\t\t\trmSync(dir, { recursive: true, force: true });\n\t\t\tgit(repoRoot, [\"worktree\", \"prune\"]);\n\t\t}\n\t}\n\tgit(repoRoot, [\"worktree\", \"add\", \"--detach\", dir, sha]);\n\n\tconst excluded: string[] = [];\n\tfor (const rel of CORPUS_EXCLUSIONS) {\n\t\tconst target = path.join(dir, rel);\n\t\tif (existsSync(target)) {\n\t\t\trmSync(target, { force: true });\n\t\t\texcluded.push(rel);\n\t\t}\n\t}\n\n\tconst subsampleInfo = subsample ? applySubsample(dir, subsample) : undefined;\n\n\treturn {\n\t\tcwd: dir,\n\t\tsha,\n\t\tfromWorkingTree: false,\n\t\tdirty: false,\n\t\texcluded,\n\t\tsubsample: subsampleInfo,\n\t\tdispose: () => {\n\t\t\ttry {\n\t\t\t\tgit(repoRoot, [\"worktree\", \"remove\", \"--force\", dir]);\n\t\t\t} catch {\n\t\t\t\trmSync(dir, { recursive: true, force: true });\n\t\t\t}\n\t\t},\n\t};\n}\n\n/** `<binary> --version`, or undefined when it cannot be run. */\nfunction probeBinaryVersion(binaryPath: string | undefined): string | undefined {\n\tif (!binaryPath) return undefined;\n\ttry {\n\t\treturn execFileSync(binaryPath, [\"--version\"], { encoding: \"utf-8\" }).trim();\n\t} catch {\n\t\treturn undefined;\n\t}\n}\n\nexport function collectProvenance(\n\trepoRoot: string,\n\tcorpus: PinnedCorpus,\n\tcorpusRef: string,\n\tservice: EmbsearchService | undefined,\n\tembsearchBinary?: string,\n\thybridService?: EmbsearchService,\n\tmodelDir?: string,\n\ttiming?: EvalProvenance[\"timing\"],\n\tchunkMaxChars?: number,\n): EvalProvenance {\n\tconst state = service?.getState();\n\tconst phase = state?.phase ?? \"absent\";\n\treturn {\n\t\ttimestampMs: Date.now(),\n\t\tcorpusSha: corpus.sha,\n\t\tcorpusRef,\n\t\tcorpusFromWorkingTree: corpus.fromWorkingTree,\n\t\tcorpusDirty: corpus.dirty,\n\t\tcorpusExcluded: corpus.excluded,\n\t\tcorpusSubsample: corpus.subsample,\n\t\tharnessSha: git(repoRoot, [\"rev-parse\", \"HEAD\"]),\n\t\tretrievalSourceHash: hashRetrievalSource(repoRoot),\n\t\tembedder: {\n\t\t\t// `ready` is the only phase the service reaches with a real embedder:\n\t\t\t// it rejects the mock backend at startup, so availability here also\n\t\t\t// certifies the numbers came from a genuine ONNX build.\n\t\t\tavailable: service?.isAvailable() ?? false,\n\t\t\treason: state && \"reason\" in state ? state.reason : undefined,\n\t\t\tchunkCount: state?.phase === \"ready\" ? state.chunkCount : undefined,\n\t\t\tphase,\n\t\t\tbinaryPath: embsearchBinary,\n\t\t\tbinaryVersion: probeBinaryVersion(embsearchBinary),\n\t\t\tmodelId: service?.modelId(),\n\t\t\tmodelDir,\n\t\t},\n\t\tdaemonHybrid: hybridService\n\t\t\t? { available: hybridService.isAvailable(), phase: hybridService.getState().phase }\n\t\t\t: undefined,\n\t\ttiming,\n\t\tchunkMaxChars,\n\t\truntime: { node: process.version, platform: process.platform, arch: process.arch },\n\t};\n}\n\nexport function summarizeGoldSet(dataset: readonly EvalQuery[]): EvalRunRecord[\"goldSet\"] {\n\tconst byClass: Record<string, number> = {};\n\tlet goldSpanCount = 0;\n\tfor (const query of dataset) {\n\t\tbyClass[query.class] = (byClass[query.class] ?? 0) + 1;\n\t\tgoldSpanCount += query.gold.length;\n\t}\n\treturn { queryCount: dataset.length, byClass, goldSpanCount };\n}\n\nexport interface RunEvalSuiteOptions {\n\tcwd: string;\n\tdataset: readonly EvalQuery[];\n\tconfigs: readonly EvalConfig[];\n\tservice?: EmbsearchService;\n\t/** Second service backed by a daemon-side BM25 hybrid store, for the\n\t *  `daemon-hybrid` configs. Absent means those rows are omitted. */\n\thybridService?: EmbsearchService;\n\tonQuery?: (index: number, query: EvalQuery) => void;\n}\n\nexport async function runEvalSuite(options: RunEvalSuiteOptions): Promise<{\n\taggregates: EvalAggregate[];\n\tperQuery: EvalRunRecord[\"perQuery\"];\n}> {\n\tconst { cwd, dataset, configs, service } = options;\n\tconst totals = new Map<string, EvalAggregate>();\n\tconst perQuery: EvalRunRecord[\"perQuery\"] = [];\n\n\tfor (const [index, evalQuery] of dataset.entries()) {\n\t\toptions.onQuery?.(index, evalQuery);\n\t\tconst results = await evaluateQuery(cwd, evalQuery, configs, service, options.hybridService);\n\t\tperQuery.push({ id: evalQuery.id, class: evalQuery.class, results });\n\t\tfor (const result of results) {\n\t\t\tconst total = totals.get(result.label) ?? {\n\t\t\t\tlabel: result.label,\n\t\t\t\trecallAt1: 0,\n\t\t\t\trecallAt5: 0,\n\t\t\t\trecallAt10: 0,\n\t\t\t\trecallAt50: 0,\n\t\t\t\tmrr: 0,\n\t\t\t\tn: 0,\n\t\t\t\tdegraded: 0,\n\t\t\t};\n\t\t\ttotal.recallAt1 += result.recallAt1;\n\t\t\ttotal.recallAt5 += result.recallAt5;\n\t\t\ttotal.recallAt10 += result.recallAt10;\n\t\t\ttotal.recallAt50 += result.recallAt50;\n\t\t\ttotal.mrr += result.mrr;\n\t\t\ttotal.n++;\n\t\t\tif (result.degraded) total.degraded++;\n\t\t\ttotals.set(result.label, total);\n\t\t}\n\t}\n\n\tconst aggregates = configs\n\t\t.map((config) => totals.get(config.label))\n\t\t.filter((total): total is EvalAggregate => total !== undefined)\n\t\t.map((total) => ({\n\t\t\t...total,\n\t\t\trecallAt1: total.recallAt1 / total.n,\n\t\t\trecallAt5: total.recallAt5 / total.n,\n\t\t\trecallAt10: total.recallAt10 / total.n,\n\t\t\trecallAt50: total.recallAt50 / total.n,\n\t\t\tmrr: total.mrr / total.n,\n\t\t}));\n\n\treturn { aggregates, perQuery };\n}\n\nexport function formatAggregateTable(aggregates: readonly EvalAggregate[]): string {\n\tconst pct = (x: number) => `${Math.round(x * 100)}%`.padStart(5);\n\tconst lines = [\n\t\t\"config           |  R@1  |  R@5  | R@10  | R@50  |  MRR  | notes\",\n\t\t\"-----------------|-------|-------|-------|-------|-------|------\",\n\t];\n\tfor (const a of aggregates) {\n\t\tconst notes = a.degraded === a.n ? \"degraded to lexical\" : a.degraded > 0 ? `${a.degraded}/${a.n} degraded` : \"\";\n\t\tlines.push(\n\t\t\t`${a.label.padEnd(16)} | ${pct(a.recallAt1)} | ${pct(a.recallAt5)} | ${pct(a.recallAt10)} | ` +\n\t\t\t\t`${pct(a.recallAt50)} | ${a.mrr.toFixed(3)} | ${notes}`,\n\t\t);\n\t}\n\treturn lines.join(\"\\n\");\n}\n"]}