import { z } from 'zod'; import { type MinedProvenanceRecord } from '../compiler-schema.js'; import type { ApiUsageLedger, DraftSourceKind, DropLedger } from './ledgers.js'; import type { SplitArtifact } from './split.js'; /** A single parsed review-thread comment (provider-neutral; mirrors the CLI `groupIntoThreads` shape). */ export interface ReviewThreadComment { author: string; /** The RAW comment body, kept verbatim for audit (slice β: `normalizedBody` is the de-chromed twin). */ body: string; /** * Author classification (slice β, strategy#709). `'bot'` iff the author is a * recognized review-FINDING bot (`reviewBotIdentity` — gemini-code-assist / * coderabbitai); `'human'` otherwise (a human author OR an unrecognized * automation account — the latter is excluded from the substantive count by the * separate `isBotIdentity` denylist check, and is rare on inline review threads). * Set at the CLI mapping boundary via core's `classifyAuthorKind`; the count + * source-tag READ it (its single classification home). */ authorKind: AuthorKind; /** * The de-chromed body the extractor actually consumes (slice β). For a `'bot'` * comment this is `normalizeReviewChrome(body)` (severity badges / `
` * collapsibles / footer chrome stripped); for a `'human'` comment it equals the * raw `body` (CRLF→LF + trim only — human prose carries no review-bot chrome). * The extractor prompt renders THIS, and `extractorInputKey` digests THIS (not * the raw body), so the key reflects exactly what the LLM saw (panel OQ-β3). */ normalizedBody: string; } /** Recognized-review-bot (`'bot'`) vs human/unrecognized (`'human'`) — see `ReviewThreadComment.authorKind`. */ export type AuthorKind = 'bot' | 'human'; /** * A single review thread on a file path. * * `isResolved` / `isOutdated` (slice 5a, mmnto-ai/totem#2201) are the per-thread * resolution signal the live `ReviewThreadSource` adapter SURFACES from the * GitHub `reviewThreads` payload — it does NOT filter on them. CRITICAL contract * (the contract-owner ruling): the adapter fetches resolved/outdated threads WITH * their flags and hands them to core; CORE decides eligibility + drop-ledgers (so * every resolution rejection is auditable, §8 "every rejection ledgered"). A * server-side / client-side `isResolved:false` pre-filter is FORBIDDEN — it would * make the rejection unledgered (a silenced §6/FM violation). */ export interface ReviewThread { path: string; comments: ReviewThreadComment[]; /** * GitHub `reviewThreads.isResolved` — the author marked this thread resolved. * Slice γ (strategy#709): RESOLVED no longer excludes a thread — a resolved * thread is the highest-signal LEGITIMACY marker (a defect the reviewer raised * AND the author confirmed by fixing). Surfaced for audit/identity; `isOutdated` * is now the SOLE eligibility filter. See `eligibleThreads`. */ isResolved: boolean; /** GitHub `reviewThreads.isOutdated` — the thread's diff hunk no longer matches HEAD. */ isOutdated: boolean; } /** * The CONTENT side of a train PR, returned by the injected `ReviewThreadSource`. * Content-only (ADR-111 §6): it never influences corpus membership / the split / * control selection — the offline `selectionRule` is the sole membership oracle. */ export interface ReviewThreadContent { pr: number; /** Lowercase 40-hex merge-commit SHA (lc is squash-merge) — becomes the candidate's `provenance.commitSha`. */ mergeCommitSha: string; threads: ReviewThread[]; } /** * The fetch outcome. §6 BINDING: distinguish "never fetched" (`unreachable`) * from "fetched but unusable" (`unparseable`) — they route to different drop * reason codes, so the §8 done-criterion can tell a broken fetch from thin * content. A discriminated result keeps that distinction at the source layer * rather than collapsing both into a `null`. */ export type FetchResult = { kind: 'ok'; content: ReviewThreadContent; } | { kind: 'unreachable'; detail?: string; } | { kind: 'unparseable'; detail?: string; }; /** * A transient, stage-internal Extract output — NOT the §3 `CandidateRuleRecord`, * NOT persisted, NOT a ledger row. Slice-3's classifier maps `DraftCandidate → * CandidateRuleRecord` by adding the structural/behavioral disposition + its * classifier-ledger reference. The miner's SOLE OUTPUT envelope remains the * `CandidateRuleRecord`, minted in slice 3 — this is just the funnel value that * flows Extract → Classify. */ export interface DraftCandidate { provenance: MinedProvenanceRecord; /** * The LLM-drafted lesson-markdown body (ADR-103 compiler input). `unverified` * and Stage-4-gated downstream — lesson-markdown is the syntax, not a * Pipeline-1 trust class. Guaranteed non-empty and carrying a usable * `**Pattern:**` / yaml rule by the syntactic preflight. */ dslSource: string; /** * The SUBSTRATE provenance (slice β, strategy#709): whether the eligible threads * this PR drafted from carried `human`, `bot` (recognized review-bot), or `mixed` * comments. A TRANSIENT diagnostic (Tenet-19, not an FM falsifier) carried to * Classify, which serializes it onto the §8 emission ledger (panel OQ-β4 — NOT * the reused `ProvenanceRecord`/legitimacy stamp). PR-level coarse-grained (a * single draft can derive from a mix), so a per-candidate tag mirrors its PR. */ sourceKind: DraftSourceKind; } /** * Injected review-thread fetch port (ADR-111 §6 content-only). Core-defined, * CLI-implemented — keeps core network-free. ASYNC: the CLI impl wraps the * GitHub API (network IO). MUST be called for train PRs only; the orchestrator * guarantees that by iterating the train slice. */ export interface ReviewThreadSource { fetch(pr: number): Promise; } /** * The `DraftExtractor` port's return: the zero-or-more draft bodies PLUS, when the * list is empty, WHY (`noDraftCause`). The cause is the extract-stage twin of the * classifier's `dispositionSource` (a non-FM Tenet-19 diagnostic) — a bare `[]` * conflated ≥6 causes (model declined / parser rejected a valid draft / transient * invoke failure) the funnel could not tell apart. INVARIANT (refined): a cause is * present IFF `drafts` is empty — a non-empty result carries drafts and no cause; * an empty result MUST name its cause. Parsed at the core boundary so a * contract-violating port (cause-without-empty, or empty-without-cause) fails loud * before the drop ledger, exactly as `ClassifierResultSchema.parse` guards classify. */ export declare const DraftResultSchema: z.ZodEffects; noDraftCause: z.ZodOptional>; }, "strip", z.ZodTypeAny, { drafts: string[]; noDraftCause?: "legacy-unknown" | "invoke-error" | "empty-output" | "none-sentinel" | "unparseable-shape" | "non-array" | "all-filtered" | undefined; }, { drafts: string[]; noDraftCause?: "legacy-unknown" | "invoke-error" | "empty-output" | "none-sentinel" | "unparseable-shape" | "non-array" | "all-filtered" | undefined; }>, { drafts: string[]; noDraftCause?: "legacy-unknown" | "invoke-error" | "empty-output" | "none-sentinel" | "unparseable-shape" | "non-array" | "all-filtered" | undefined; }, { drafts: string[]; noDraftCause?: "legacy-unknown" | "invoke-error" | "empty-output" | "none-sentinel" | "unparseable-shape" | "non-array" | "all-filtered" | undefined; }>; export type DraftResult = z.infer; /** * Injected draft-DSL extractor port. ASYNC: the CLI impl wraps the LLM call * (network IO). List-shaped (fold 1): one thread can carry multiple structural * invariants, so it returns ZERO-or-more draft bodies in `DraftResult.drafts`. The * LLM lives behind this at the CLI layer (draft-only, Tenet-15); a deterministic * fixture impl drives tests. The miner is BLIND to seed classes (§7 / FM f): the * port is never handed one. * * Error contract: returns `{ drafts: [], noDraftCause }` when it cannot draft — * INCLUDING on its own internal/transient failure (the CLI adapter catches its * LLM/network errors and surfaces `{ drafts: [], noDraftCause: 'invoke-error' }`). * It MUST NOT throw for a per-PR content failure: an empty list is a loud, * cause-tagged drop below (FM-i-creditable), whereas a throw would abort the whole * mining run. Keeping per-PR error handling in the adapter keeps the core * orchestrator Tenet-4-clean (no swallowing catch); a contract-violating throw * therefore propagates loudly rather than being silently absorbed. */ export interface DraftExtractor { draft(content: ReviewThreadContent): Promise; } /** Dependencies for a single Extract-stage run. */ export interface ExtractStageDeps { source: ReviewThreadSource; extractor: DraftExtractor; /** * §7 seed-blindness fact, established in-run: `true` iff a seed class WAS * supplied to the extractor (which would falsify FM f). Carried here; slice 3 * SERIALIZES it into the §8 emission ledger's `extractionInputsAttestation` * (single persisted home, Tenet 20). Slice 2 establishes the fact; it does not * grow a second store for it. */ seedClassesProvided: boolean; } /** The Extract stage's output: transient drafts + the two ledgers Extract owns. */ export interface ExtractStageResult { /** Transient draft candidates carried forward to slice-3 Classify. */ drafts: DraftCandidate[]; /** Drop ledger — the sole disposition for any content/provenance/draft failure (§6). */ dropLedger: DropLedger; /** API-usage ledger — every train-slice fetch; `heldOutFetchCount` MUST be 0 (FM h). */ apiUsageLedger: ApiUsageLedger; /** In-run seed-blindness fact; slice 3 persists it into the emission ledger. */ seedBlindness: { seedClassesProvided: boolean; }; } /** * Classify a comment author (slice β, strategy#709). `'bot'` iff it is a recognized * review-FINDING bot (`reviewBotIdentity` allowlist — gemini/CR); `'human'` * otherwise. The SINGLE classification home: the CLI mapping boundary stamps each * comment's `authorKind` via this, and the count + source-tag read that field. */ export declare function classifyAuthorKind(author: string): AuthorKind; /** * Run the deterministic Stage-1 Extract over a frozen split. Deterministic given * its deps: identical `split` + deps → identical drafts, drops, and ledgers (the * train slice is awaited sequentially, so ordering is stable). The * live LLM and GitHub IO are injected ports, so this orchestration is fully * CI-locked with a fixture extractor + a strict-spy fetch source. * * Per train PR (and ONLY train PRs): log the fetch → fetch → on unreachable / * unparseable-at-source, loud-drop → eligibility gate (slice γ: drop * `outdated-rejected` when the outdated filter empties an otherwise-substantive * thread, else `truncated` when thin to begin with) → completeness-check (≥1 * substantive comment on the survivors) → build provenance → draft zero-or-more * bodies from the SURVIVING threads only → preflight each → carry a * `DraftCandidate` or loud-drop. Every train PR ends with at least one draft or * one drop (FM i, slice-2 half). */ export declare function runExtractStage(split: SplitArtifact, deps: ExtractStageDeps): Promise; //# sourceMappingURL=extract.d.ts.map