import { R as ResolvedScoreDefinition, S as ScoreResult, e as ScorerRole, a as ScoreDefinition } from '../decision-summary-CksWu77D.js'; export { f as EVAL_RUN_DECISION_SUMMARY_SCHEMA_ID, g as EVAL_RUN_DECISION_SUMMARY_SCHEMA_VERSION, h as EVAL_RUN_DECISION_UNDECIDED_REASONS, i as EVAL_RUN_DECISION_UNDECIDED_REASON_LABELS, j as EVAL_RUN_DECISION_VERDICTS, k as EVAL_RUN_DECISION_VERDICT_LABELS, l as EVAL_RUN_DECISION_VERDICT_SOURCES, m as EVAL_RUN_DECISION_VERDICT_SOURCE_LABELS, n as EVAL_RUN_MEASUREMENT_UNITS, o as EVAL_RUN_MEASUREMENT_UNIT_LABELS, p as EvalRunDecisionAssemblyInput, q as EvalRunDecisionChain, r as EvalRunDecisionCounts, s as EvalRunDecisionDiagnostic, t as EvalRunDecisionDiagnostics, u as EvalRunDecisionEvidence, v as EvalRunDecisionIterationInput, w as EvalRunDecisionRunInput, d as EvalRunDecisionSummary, x as EvalRunDecisionSummarySchemaVersion, y as EvalRunDecisionUndecided, z as EvalRunDecisionUndecidedReason, A as EvalRunDecisionVerdict, B as EvalRunDecisionVerdictSource, C as EvalRunMeasurementUnit, E as EvaluationConfigSnapshot, M as MAX_ERROR_LENGTH, D as MAX_EVIDENCE_ENTRIES, F as MAX_EVIDENCE_ENTRY_LENGTH, G as MAX_EVIDENCE_REASONS, H as MAX_EVIDENCE_REASON_CHARS, I as MAX_RATIONALE_LENGTH, J as MAX_SCORER_ID_LENGTH, P as PREDICATES_VERSION, K as STAGE_ANALYZER_VERSION, L as STAGE_METADATA_KEYS, N as STAGE_REASONS, b as ScoreRawOutcome, O as ScoreStatus, Q as ScorerContextV1, T as ScorerErrorPolicy, U as ScorerIdSource, V as StageAuthoredCase, W as StageDerivation, X as StageDerivationInput, Y as StageEvidence, Z as StageEvidenceRefs, _ as StagePredicateResultLike, $ as StagePromptSummaryLike, a0 as StageReason, a1 as StageRenderObservationLike, c as StageResultRow, a2 as StageSetupPhaseSignal, a3 as StageSetupSignals, a4 as StageSpanLike, a5 as StageToolErrorLike, a6 as assembleEvalRunDecisionSummary, a7 as decisionDiagnosticFailureCategory, a8 as decisionDiagnosticFirstFailedStage, a9 as deriveStageResults, aa as evalIterationTracePath, ab as evalRunDecisionChainSchema, ac as evalRunDecisionCountsSchema, ad as evalRunDecisionDiagnosticSchema, ae as evalRunDecisionDiagnosticsSchema, af as evalRunDecisionEvidenceSchema, ag as evalRunDecisionSummarySchema, ah as evalRunDecisionSummaryStructuralSchema, ai as evalRunDecisionUndecidedReasonSchema, aj as evalRunDecisionUndecidedSchema, ak as evalRunDecisionVerdictSchema, al as evalRunDecisionVerdictSourceSchema, am as evalRunMeasurementUnitSchema, an as measurementUnitLabel, ao as stageDerivationSchema, ap as stageDerivationToMetadata, aq as stageReasonSchema, ar as stageResultRowSchema } from '../decision-summary-CksWu77D.js'; export { C as CanonicalJsonError, D as DECISION_LABEL_VOCABULARIES, a as DECISION_SUMMARY_FALLBACK_NEXT_ACTION, E as EVAL_VERDICT_DECISION_REASON_LABELS, F as FAILURE_CATEGORY_LABELS, N as NEXT_ACTION_BY_FAILURE_CATEGORY, S as STAGE_REASON_LABELS, b as STAGE_STATE_LABELS, U as USER_VALUE_STAGE_LABELS, c as aggregateEvaluationConfigHash, d as allGatingScorersPassed, e as buildEvaluationConfigSnapshot, f as canonicalDigest, g as canonicalJson, h as definitionHash, i as errorScoreResult, j as evaluationConfigHash, k as evaluationConfigSnapshotSchema, l as finalizeScoreResult, n as notApplicableScoreResult, r as resolveScoreDefinition, m as resolvedScoreDefinitionSchema, s as scoreDefinitionSchema, o as scorePassed, p as scoreResultArraySchema, q as scoreResultSchema, t as scoreStatusSchema, u as scorerErrorPolicySchema, v as scorerIdSourceSchema, w as scorerRoleSchema, x as sha256Hex, y as skippedScoreResult } from '../decision-labels-KRlswm7m.js'; import { f as EvalToolCallMatchResult, c as EvalExpectedToolCall, a as EvalMatchOptions, g as EvalSuiteFileToolPolicy } from '../index-fZyfLCHE.js'; export { h as EVAL_RATE_MEASUREMENT_STATES, i as EVAL_RUN_VERDICTS, j as EVAL_SUITE_SCHEMA_ID, k as EVAL_SUITE_SCHEMA_VERSION, l as EVAL_TASK_DECISION_REASONS, m as EVAL_TRIAL_EXCLUSION_REASONS, n as EVAL_VALIDITY_DECISION_REASONS, o as EVAL_VERDICT_DECISION_REASONS, p as EVAL_VERDICT_POLICY_SCHEMA_ID, q as EVAL_VERDICT_POLICY_VERSION, r as EvalCaseVerdictAggregation, s as EvalRateMeasurement, t as EvalRateMeasurementState, u as EvalRunVerdict, v as EvalSuiteFile, w as EvalSuiteFileCase, x as EvalSuiteFileCaseImport, y as EvalSuiteFileDefaults, z as EvalSuiteFileHost, A as EvalSuiteFileProvenance, B as EvalSuiteFileServer, D as EvalSuiteFileTarget, G as EvalSuiteFileValidity, H as EvalTaskDecisionReason, J as EvalTrialExclusionReason, K as EvalTrialExclusions, N as EvalValidityCoverage, O as EvalValidityDecisionReason, e as EvalVerdictDecision, P as EvalVerdictDecisionReason, Q as EvalVerdictPolicyVersion, R as EvalVerdictValidity, T as FAILURE_CATEGORIES, F as FailureCategory, V as IMPORT_MAPPING_STATUSES, W as ITERATION_STATUSES, X as ImportMappingStatus, I as IterationStatus, Y as MAX_BATCH_CREATE_CASES, Z as MAX_CASE_ASSERTIONS, _ as MAX_REPETITIONS, $ as MAX_SUITE_FILE_CASES, a0 as MAX_SUITE_FILE_TITLE_CHARS, a1 as RESERVED_CAPTURE_LEVELS, a2 as RESERVED_MODES, a3 as RESERVED_REPORTING_MODES, a4 as ResolvedEvalValidityPolicy, a5 as STAGE_STATES, d as StageState, a6 as USER_VALUE_STAGES, U as UserValueStage, a7 as evalCaseVerdictAggregationSchema, a8 as evalCaseVerdictAggregationStructuralSchema, a9 as evalFractionSchema, aa as evalRateMeasurementSchema, ab as evalRateMeasurementStateSchema, ac as evalRateMeasurementStructuralSchema, ad as evalRunVerdictSchema, ae as evalSuiteFileCaseImportSchema, af as evalSuiteFileCaseSchema, ag as evalSuiteFileDefaultsSchema, ah as evalSuiteFileHostSchema, ai as evalSuiteFileProvenanceSchema, aj as evalSuiteFileSchema, ak as evalSuiteFileServerSchema, al as evalSuiteFileStructuralSchema, am as evalSuiteFileTargetSchema, an as evalSuiteFileToolPolicySchema, ao as evalSuiteFileValiditySchema, ap as evalTrialExclusionReasonSchema, aq as evalTrialExclusionsSchema, ar as evalValidityCoverageSchema, as as evalVerdictDecisionReasonSchema, at as evalVerdictDecisionSchema, au as evalVerdictDecisionStructuralSchema, av as evalVerdictPolicyVersionSchema, aw as failureCategorySchema, ax as importMappingStatusSchema, ay as isEvalRunVerdict, az as isEvalTrialExclusionReason, aA as isEvalValidityDecisionReason, aB as isEvalVerdictDecisionReason, aC as isEvalVerdictPolicyV2, aD as iterationStatusSchema, aE as resolvedEvalValidityPolicySchema, aF as stageStateSchema, aG as userValueStageSchema } from '../index-fZyfLCHE.js'; import { c as PredicateResult, b as Predicate } from '../types-BBk0lTBo.js'; import { z } from 'zod'; import 'ai'; import '../public-types-CX8stXC3.js'; import '../types-HXAijHji.js'; import '../types-CI0Xyszt.js'; import '@modelcontextprotocol/client'; import '@modelcontextprotocol/client/stdio'; /** * Pure mappers from every legacy verdict source into the one contract shape. * * This module is browser-safe and intentionally has no node-only deps. * * Scoring is not a fifth verdict system beside `test()`, the tool-call matcher, * predicates and judges — it is the shape those four are projected into. These * adapters are that projection, and keeping them pure (no evaluation, no I/O) * is what lets the runner evaluate ONCE and emit two views: the legacy compat * field and the score row, which therefore cannot disagree. * * Each source contributes both halves: a `*ScoreDefinition` builder (what the * scorer is, including the `implementationHash` derived from its real config) * and a `from*` result mapper. */ /** Stable id of the scorer that projects `config.expectedToolCalls`. */ declare const TOOL_MATCH_SCORER_ID = "tool-match"; /** Stable id of the scorer that projects the legacy `test()` boolean. */ declare const LEGACY_TEST_SCORER_ID = "legacy:test"; /** Version of the legacy-boolean projection itself. */ declare const LEGACY_TEST_VERSION = "1"; /** Version of the tool-match projection; tracks the matcher's semantics. */ declare const TOOL_MATCH_VERSION = "1"; /** * Positional id minted for a predicate the author did not name. * * UNSTABLE by construction — inserting a predicate above index 2 renumbers it — * which is why the definition records `idSource: "generated"` and why gates * refuse to select one. Anything gated in CI needs an explicit id. */ declare function generatedPredicateScorerId(predicate: Predicate, ordinal: number): string; /** * The definition for one authored predicate. * * `implementationHash` is the canonicalized predicate itself: editing * `responseContains "refund issued"` to `"refund processed"` changes what the * scorer does, so it must change the evaluation config hash even though the * scorer id, version and threshold are untouched. */ declare function predicateScoreDefinition(predicate: Predicate, options: { id?: string; ordinal: number; role?: ScorerRole; }): ScoreDefinition; /** * Project a predicate verdict. Boolean in, `0|1` out against a threshold of * `1`, so a predicate reads on the dashboard exactly like every other scorer. * The evaluator's `reason` is the load-bearing diagnostic and survives as the * rationale. */ declare function scoreResultFromPredicateResult(definition: ResolvedScoreDefinition, result: PredicateResult): ScoreResult; /** * The definition for the `expectedToolCalls` matcher. * * `implementationHash` covers BOTH the expectations and the resolved matcher * policy: flipping `toolCallOrder` from `"ignore"` to `"strict"` changes the * verdict on an unchanged transcript, so it is an evaluation-config change. */ declare function toolMatchScoreDefinition(options: { expectedToolCalls: EvalExpectedToolCall[]; matchOptions: Required>; role?: ScorerRole; /** * Negative-case polarity. Part of the HASH, not just the behaviour: a * negative tool-match ("pass iff nothing was called") and a positive one * are different scorers, and digesting them identically would let a case * flip polarity while its `definitionHash` claimed nothing changed — so a * run comparison would read the flip as a regression rather than as a * changed definition. * * Omitted from the payload when false/absent so every EXISTING definition * hashes exactly as before; only a negative case gets a new digest. */ isNegativeTest?: boolean; }): ScoreDefinition; /** * Project the tool-call matcher's verdict. The `extra[]` list is reported in * the rationale but does not by itself decide the outcome — that is the * matcher's `passed`, which already applies the `maxExtraToolCalls` policy. */ declare function fromToolMatchResult(definition: ResolvedScoreDefinition, match: EvalToolCallMatchResult): ScoreResult; /** * The definition for the legacy `test()` boolean. * * `implementationHash` is a CONSTANT, and deliberately so: the test body is an * opaque closure with no serializable configuration. Hashing `String(fn)` was * considered and rejected — it varies with transpilation and minification, so * the same test would digest differently from a `tsx` run and a bundled run, * and every case would read as `configChanged` when nothing changed. The * consequence is stated rather than hidden: two different `test()` bodies * produce the same `implementationHash`, and changes to a test body do not * reach the evaluation config hash. */ declare function legacyTestScoreDefinition(options?: { role?: ScorerRole; }): ScoreDefinition; /** Project the legacy `test()` boolean into a score. */ declare function fromLegacyTestOutcome(definition: ResolvedScoreDefinition, passed: boolean): ScoreResult; /** * Project one hosted goalCompletion case row. * * The row's own `passed` is READ AND DISCARDED. The threshold on the definition * is authoritative (the hosted generator already works this way), so a judge * that reports `{score: 0.2, passed: true}` cannot smuggle a pass through — the * derived value is what lands. */ declare function fromGoalCompletionCase(definition: ResolvedScoreDefinition, row: { caseKey?: string; score: number; passed?: boolean; reason?: string; rubricHits?: string[]; }): ScoreResult; /** * Project one graded rubric criterion. A criterion is boolean, so it maps the * same way a predicate does: `0|1` against a threshold of `1`, carrying the * evaluator's sentence and any per-turn scope. */ declare function fromCriterionResult(definition: ResolvedScoreDefinition, row: { criterionId: string; passed: boolean; reason?: string; scope?: PredicateResult["scope"]; }): ScoreResult; /** * Opaque identity for contract objects — cases and suites. * * This module is browser-safe and intentionally has no node-only deps. * * ── Why ids are opaque ──────────────────────────────────────────────────────── * * Four id regimes already exist and all of them must remain valid: Convex * document ids on hosted cases, `ui_` on cases authored in the * inspector, ids minted here, and whatever a user hand-writes in a suite file. * So the validator constrains the CHARACTER SET and the LENGTH, and nothing * else — it never requires one of our prefixes. The prefixes {@link CASE_ID_PREFIX} * / {@link SUITE_ID_PREFIX} exist for humans grepping logs, not for machines * dispatching on them; a validator that demanded them would reject every id the * platform already issued. * * The charset is deliberately the URL/filename-safe one: ids end up in file * paths, YAML keys, CLI arguments and dashboard URLs, and a whitespace or * control character in any of those is a bug that surfaces far from where it * was authored. * * ── Why identity is DECLARED, never derived ────────────────────────────────── * * Nothing here derives an id from a title, a name, or a content hash. That is * the bug being retired: hosted history is joined on case identity, so an id * derived from display text forks a case's whole history the moment somebody * fixes a typo in its name. Minting is a convenience for authoring; once minted * the id is committed alongside the case and survives every rename. */ /** Max length of an opaque id. Comfortably above a Convex id or a `ui_`. */ declare const MAX_OPAQUE_ID_LENGTH = 128; /** * The one identity rule: URL-safe characters, 1..128 of them. * * Accepts every id regime in play (Convex ids, `ui_`, minted ids, * hand-authored ids) and rejects whitespace, control characters, path * separators and quoting metacharacters. */ declare const opaqueIdSchema: z.ZodString; type OpaqueId = z.infer; /** Prefix stamped on minted CASE ids. Human-facing only — never dispatched on. */ declare const CASE_ID_PREFIX = "c_"; /** Prefix stamped on minted SUITE ids. Human-facing only — never dispatched on. */ declare const SUITE_ID_PREFIX = "s_"; /** Random characters in a minted id (nanoid's default entropy). */ declare const MINTED_ID_ENTROPY_CHARS = 21; /** * A fresh case id: `c_` + 21 URL-safe characters of CSPRNG entropy. * * Mint ONCE and commit the result — an id regenerated on every run is not an * identity, and would fork history exactly the way a name-derived id does. */ declare function mintCaseId(): string; /** A fresh suite id: `s_` + 21 URL-safe characters of CSPRNG entropy. */ declare function mintSuiteId(): string; /** True when `value` satisfies {@link opaqueIdSchema}. */ declare function isOpaqueId(value: unknown): value is string; /** * Unified test-step model — the authored unit of an MCP-app synthetic test. * * This module is browser-safe and intentionally has no node-only deps. * * A test case is an ORDERED `TestStep[]` (Datadog-Synthetics-style): you record * a scenario by interacting with the live app, and assertions are first-class * steps interleaved inline. * * The four authored kinds: * - `prompt` — a user message; the model decides which tools to call. * - `toolCall` — a deterministic, model-free tool call. * - `interact` — ONE pure widget action (click/type/key/scroll/wait). No assertions. * - `assert` — the ONE place assertions live: a model-level `Predicate` * (toolCalledWith / widgetRendered / responseContains / …) OR a * DOM-level `WidgetAssertion` (textVisible / elementVisible / …). * * NAMING: this union is `TestStep`, NOT `Step`. The AI SDK already owns "step" * = one LLM round-trip (onStepFinish / stepNumber). Our authoring unit is a * different level — a single `prompt` TestStep may expand into several AI SDK * steps at runtime. The persisted/UI field stays `steps`. * * ── Why this lives in the SDK contract, not in the inspector app ───────────── * * The step union is part of the evaluation CONTRACT: it is what a suite file * carries, what the hosted API accepts, and what a code-first author writes. * It used to live in `mcpjam-inspector/shared/steps.ts`, which the SDK cannot * import (the dependency direction is shared → sdk, never the reverse), so * putting a suite-file schema in the SDK would have required a hand-mirrored * SECOND copy inside the SDK — a third sibling beside `shared/` and Convex. * The definition therefore moved HERE and `shared/steps.ts` re-exports it, so * there is exactly one definition and the inspector's 50-odd consumers are * unchanged. * * Mirrored by the Convex validator in mcpjam-backend `convex/lib/steps.ts` * (same hand-mirroring arrangement as `scriptedSteps` / `probeConfig` / the * predicate validators) — edit both in the same PR. * * ── Every object DECLARED here is `.strict()` ──────────────────────────────── * * A field this union does not declare is an ERROR, never a silently dropped * key. Two reasons, and the second is the one that forced the change: * * 1. The Convex mirror is built from `v.object`, which rejects unknown fields. * A permissive schema here accepted step payloads the backend refuses, so * the two validators disagreed about which files are valid — and the * disagreement was invisible until ingest. * 2. Steps are the surface an IMPORTER writes. A converter that mis-maps a * source field into a step (`text` where the contract says `prompt`) must * fail at the line that is wrong, not produce a step that runs and asserts * nothing. Silently discarding half of what was read is the exact failure * the closed-schema rule exists to prevent, and step level is where a * mis-mapped field actually lands. * * Two things are deliberately NOT closed: * * - `toolCallStep.arguments` — the tool's OWN argument object. Its keys come * from the server's input schema, not from this contract; closing it would * mean this file had to know every tool's arguments. * - The reused `predicateSchema` inside an `assert` step. That union is a * separate contract module (`../predicates/types.ts`) with its own Convex * mirror and its own parity fixtures, and it is authored from many more * surfaces than steps (swarm rubrics, the Checks panel, suite defaults). * Closing it is a change to THAT contract, made there with its own consumer * audit — not a side effect of closing this one. */ /** Max chars for a step's free text (`type` text, assertion text/value). */ declare const MAX_SCRIPTED_STEP_TEXT_CHARS = 5000; /** Max explicit `wait` duration (ms). */ declare const MAX_SCRIPTED_WAIT_MS = 30000; /** Tool-call render budget ceiling — matches the backend validator's cap. */ declare const MAX_PROBE_RENDER_TIMEOUT_MS = 120000; /** * Max serialized size (chars) of a tool call's pinned arguments — matches * `MAX_PROBE_ARGS_CHARS` in the mcpjam-backend validator * (`convex/lib/probeConfig.ts`). Arguments are stored verbatim and snapshotted * into every iteration, so an unbounded blob would bloat rows. */ declare const MAX_PROBE_ARGS_CHARS = 100000; /** * A bundle of semantic locators for one target element. At least one of * role/text/css/testId must be present; they are resolved in priority order * (testId → role → text → css) by the harness. `nth` disambiguates when a * locator matches multiple elements. * * Locators are intentionally a BUNDLE of semantic reference points rather than * coordinates: the widget authored against (client preview render) and the * widget executed against (headless harness render) are different render * instances, so only semantic locators transfer. */ declare const elementLocatorSchema: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; type ElementLocator = z.infer; declare const TEST_STEP_KINDS: readonly ["prompt", "toolCall", "interact", "assert"]; type TestStepKind = (typeof TEST_STEP_KINDS)[number]; declare const promptStepSchema: z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"prompt">; prompt: z.ZodString; }, z.core.$strict>; type PromptStep = z.infer; declare const toolCallStepSchema: z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"toolCall">; serverId: z.ZodOptional; serverName: z.ZodString; toolName: z.ZodString; arguments: z.ZodRecord; renderTimeoutMs: z.ZodOptional; }, z.core.$strict>; type ToolCallStep = z.infer; declare const interactActionSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{ kind: z.ZodLiteral<"click">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; clickType: z.ZodOptional>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"type">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"key">; key: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"scroll">; direction: z.ZodEnum<{ up: "up"; down: "down"; }>; amount: z.ZodOptional; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"wait">; ms: z.ZodNumber; }, z.core.$strict>], "kind">; type InteractAction = z.infer; declare const interactStepSchema: z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"interact">; toolName: z.ZodString; action: z.ZodDiscriminatedUnion<[z.ZodObject<{ kind: z.ZodLiteral<"click">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; clickType: z.ZodOptional>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"type">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"key">; key: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"scroll">; direction: z.ZodEnum<{ up: "up"; down: "down"; }>; amount: z.ZodOptional; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"wait">; ms: z.ZodNumber; }, z.core.$strict>], "kind">; }, z.core.$strict>; type InteractStep = z.infer; /** * DOM/widget-level assertions evaluated against the live widget by the headless * harness (NOT the transcript predicate engine). `toolName` is always the WIDGET * being asserted against. `widgetToolCalled.calledToolName` is the tool the * widget invoked (distinct from the widget's own tool). * * ── Why this is a SEPARATE union from `Predicate`, and stays one ───────────── * * The dividing line is what the assertion is evaluated AGAINST, and therefore * what can re-run it later: * * - A `Predicate` is evaluated against a persisted TRANSCRIPT. Anything * holding one — the eval runner, the swarm checks runner, an on-demand * re-grade months later — can re-derive the same verdict from stored data * by calling the SDK's pure evaluator. That is why swarm rubrics, whole-run * checks and the per-session Checks panel are all predicate-shaped. * - A `WidgetAssertion` is evaluated against a LIVE DOM inside the browser * harness, at the moment the step runs. Nothing is persisted that could * reproduce it: the transcript records that a widget rendered, not what was * on screen inside it. Re-running the assertion means re-running the * session. * * That is the whole reason `widgetRendered` is a `Predicate` while * `textVisible` is not, even though both sound like claims about a view. * "Did the host mount this widget?" is answered by a render observation the * transcript already carries; "is the word 'Refunded' visible in it?" is * answered only by looking at the live document. * * The consequence is not cosmetic: swarms have no `TestStep[]` and no browser * harness, so they cannot author or evaluate widget assertions at all. Merging * the two unions would put kinds in the swarm rubric menu that can never * produce a verdict there. * * Re-evaluate this split if a persisted-DOM-snapshot capability lands (a * serialized document per render observation, not just a screenshot). At that * point widget assertions become transcript-replayable and the argument for * two vocabularies goes away. */ declare const widgetAssertionSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{ kind: z.ZodLiteral<"textVisible">; toolName: z.ZodString; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementVisible">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementHidden">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"inputValue">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; equals: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"widgetToolCalled">; toolName: z.ZodString; calledToolName: z.ZodString; }, z.core.$strict>], "kind">; type WidgetAssertion = z.infer; declare const stepAssertionPayloadSchema: z.ZodUnion; toolName: z.ZodString; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementVisible">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementHidden">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"inputValue">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; equals: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"widgetToolCalled">; toolName: z.ZodString; calledToolName: z.ZodString; }, z.core.$strict>], "kind">, z.ZodIntersection; toolName: z.ZodString; args: z.ZodObject<{ args: z.ZodRecord; argumentMatching: z.ZodOptional>; }, z.core.$strip>; minCount: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolCalledAtLeastOnce">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolNeverCalled">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"firstToolWas">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseContains">; needle: z.ZodString; caseSensitive: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseMatches">; pattern: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"noToolErrors">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"finalAssistantMessageNonEmpty">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"tokenBudgetUnder">; tokens: z.ZodNumber; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRendered">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRenderLatencyUnder">; ms: z.ZodNumber; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetNoConsoleErrors">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"turnCountUnder">; turns: z.ZodNumber; }, z.core.$strip>], "type">, z.ZodObject<{ kind: z.ZodOptional; }, z.core.$strip>>]>; type StepAssertionPayload = WidgetAssertion | Predicate; declare const assertStepSchema: z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"assert">; assertion: z.ZodUnion; toolName: z.ZodString; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementVisible">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementHidden">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"inputValue">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; equals: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"widgetToolCalled">; toolName: z.ZodString; calledToolName: z.ZodString; }, z.core.$strict>], "kind">, z.ZodIntersection; toolName: z.ZodString; args: z.ZodObject<{ args: z.ZodRecord; argumentMatching: z.ZodOptional>; }, z.core.$strip>; minCount: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolCalledAtLeastOnce">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolNeverCalled">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"firstToolWas">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseContains">; needle: z.ZodString; caseSensitive: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseMatches">; pattern: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"noToolErrors">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"finalAssistantMessageNonEmpty">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"tokenBudgetUnder">; tokens: z.ZodNumber; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRendered">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRenderLatencyUnder">; ms: z.ZodNumber; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetNoConsoleErrors">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"turnCountUnder">; turns: z.ZodNumber; }, z.core.$strip>], "type">, z.ZodObject<{ kind: z.ZodOptional; }, z.core.$strip>>]>; }, z.core.$strict>; type AssertStep = z.infer; declare const testStepSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"prompt">; prompt: z.ZodString; }, z.core.$strict>, z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"toolCall">; serverId: z.ZodOptional; serverName: z.ZodString; toolName: z.ZodString; arguments: z.ZodRecord; renderTimeoutMs: z.ZodOptional; }, z.core.$strict>, z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"interact">; toolName: z.ZodString; action: z.ZodDiscriminatedUnion<[z.ZodObject<{ kind: z.ZodLiteral<"click">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; clickType: z.ZodOptional>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"type">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"key">; key: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"scroll">; direction: z.ZodEnum<{ up: "up"; down: "down"; }>; amount: z.ZodOptional; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"wait">; ms: z.ZodNumber; }, z.core.$strict>], "kind">; }, z.core.$strict>, z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"assert">; assertion: z.ZodUnion; toolName: z.ZodString; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementVisible">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementHidden">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"inputValue">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; equals: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"widgetToolCalled">; toolName: z.ZodString; calledToolName: z.ZodString; }, z.core.$strict>], "kind">, z.ZodIntersection; toolName: z.ZodString; args: z.ZodObject<{ args: z.ZodRecord; argumentMatching: z.ZodOptional>; }, z.core.$strip>; minCount: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolCalledAtLeastOnce">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolNeverCalled">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"firstToolWas">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseContains">; needle: z.ZodString; caseSensitive: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseMatches">; pattern: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"noToolErrors">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"finalAssistantMessageNonEmpty">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"tokenBudgetUnder">; tokens: z.ZodNumber; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRendered">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRenderLatencyUnder">; ms: z.ZodNumber; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetNoConsoleErrors">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"turnCountUnder">; turns: z.ZodNumber; }, z.core.$strip>], "type">, z.ZodObject<{ kind: z.ZodOptional; }, z.core.$strip>>]>; }, z.core.$strict>], "kind">; type TestStep = z.infer; /** Max steps per case — keeps snapshotted rows bounded. */ declare const MAX_TEST_STEPS = 200; declare const stepsSchema: z.ZodArray; prompt: z.ZodString; }, z.core.$strict>, z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"toolCall">; serverId: z.ZodOptional; serverName: z.ZodString; toolName: z.ZodString; arguments: z.ZodRecord; renderTimeoutMs: z.ZodOptional; }, z.core.$strict>, z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"interact">; toolName: z.ZodString; action: z.ZodDiscriminatedUnion<[z.ZodObject<{ kind: z.ZodLiteral<"click">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; clickType: z.ZodOptional>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"type">; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"key">; key: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"scroll">; direction: z.ZodEnum<{ up: "up"; down: "down"; }>; amount: z.ZodOptional; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"wait">; ms: z.ZodNumber; }, z.core.$strict>], "kind">; }, z.core.$strict>, z.ZodObject<{ id: z.ZodString; kind: z.ZodLiteral<"assert">; assertion: z.ZodUnion; toolName: z.ZodString; text: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementVisible">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"elementHidden">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"inputValue">; toolName: z.ZodString; target: z.ZodObject<{ role: z.ZodOptional; exact: z.ZodOptional; }, z.core.$strict>>; text: z.ZodOptional; css: z.ZodOptional; testId: z.ZodOptional; nth: z.ZodOptional; }, z.core.$strict>; equals: z.ZodString; }, z.core.$strict>, z.ZodObject<{ kind: z.ZodLiteral<"widgetToolCalled">; toolName: z.ZodString; calledToolName: z.ZodString; }, z.core.$strict>], "kind">, z.ZodIntersection; toolName: z.ZodString; args: z.ZodObject<{ args: z.ZodRecord; argumentMatching: z.ZodOptional>; }, z.core.$strip>; minCount: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolCalledAtLeastOnce">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"toolNeverCalled">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"firstToolWas">; toolName: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseContains">; needle: z.ZodString; caseSensitive: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"responseMatches">; pattern: z.ZodString; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"noToolErrors">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"finalAssistantMessageNonEmpty">; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"tokenBudgetUnder">; tokens: z.ZodNumber; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRendered">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetRenderLatencyUnder">; ms: z.ZodNumber; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"widgetNoConsoleErrors">; toolName: z.ZodOptional; }, z.core.$strip>, z.ZodObject<{ type: z.ZodLiteral<"turnCountUnder">; turns: z.ZodNumber; }, z.core.$strip>], "type">, z.ZodObject<{ kind: z.ZodOptional; }, z.core.$strip>>]>; }, z.core.$strict>], "kind">>; declare const isPromptStep: (s: TestStep) => s is PromptStep; declare const isToolCallStep: (s: TestStep) => s is ToolCallStep; declare const isInteractStep: (s: TestStep) => s is InteractStep; declare const isAssertStep: (s: TestStep) => s is AssertStep; /** True when `assertion` is a DOM-level WidgetAssertion (vs a transcript Predicate). */ declare function isWidgetAssertion(a: StepAssertionPayload): a is WidgetAssertion; type ToolSafetyClassification = "readOnly" | "destructive" | "unknown"; /** * The reason vocabulary as VALUES, so an out-of-process consumer can validate a * reason it read back off the wire instead of trusting the string. */ declare const TOOL_POLICY_DECISION_REASONS: readonly ["denyList", "allowList", "destructiveDefaultDeny", "readOnlyModeClassified", "readOnlyModeUnclassified", "modeDefault", "unknownAtLaunch"]; type ToolPolicyDecisionReason = (typeof TOOL_POLICY_DECISION_REASONS)[number]; declare function isToolPolicyDecisionReason(value: unknown): value is ToolPolicyDecisionReason; type ToolPolicyDecision = { allowed: boolean; reason: ToolPolicyDecisionReason; classification: ToolSafetyClassification; }; /** * Server-provided annotations are advisory and UNTRUSTED. Missing, malformed, * or contradictory values must never be treated as evidence that a tool is * safe to run. */ declare function classifyToolSafety(annotations: Record | undefined): ToolSafetyClassification; /** * A fully-resolved policy decision for every tool known at launch, so an * out-of-process enforcement point (the harness MCP proxy) performs a MAP * LOOKUP and never a classification. Annotations never travel with it. */ type ToolPolicySnapshot = { mode: "default" | "readOnly"; /** * Tool name → the denying decision. Allowed tools are absent. * * The classification travels with the reason because a block has to be * reported as the same `PolicyBlockRecord` the in-process gate produces, and * that record carries the classification; re-deriving it from the reason * would fabricate it (`denyList` says nothing about the annotations). */ denied: Record; /** * Every tool name the snapshot decided, denied or not. * * Required to enforce `unknownTool`: with `denied` alone, a tool that * appeared AFTER launch is indistinguishable from one that was decided and * allowed, so the enforcement point could not tell the two apart and * `unknownTool` would be unenforceable. */ known: string[]; /** What to do with a tool that was not known at launch. */ unknownTool: "deny" | "allow"; }; /** * Resolve `policy` against every tool known at launch, using `decideToolPolicy` * itself — one pure function, no forked precedence at the enforcement point. * * `unknownTool` is `deny` whenever a policy is present: a tool that appears * after launch has no decision, and permitting it would let a `tools/list` * change defeat the policy. */ declare function buildToolPolicySnapshot(args: { policy: EvalSuiteFileToolPolicy; tools: ReadonlyArray<{ name: string; annotations?: Record | undefined; }>; }): ToolPolicySnapshot; /** * Decide one call against an already-resolved snapshot — a MAP LOOKUP, so an * out-of-process enforcement point never classifies anything. * * A tool absent from `snapshot.known` did not exist at launch, so no decision * was ever made for it: denied with `unknownAtLaunch` unless the snapshot says * unknown tools are allowed. */ declare function decideToolPolicyFromSnapshot(args: { snapshot: ToolPolicySnapshot; toolName: string; }): { allowed: true; } | { allowed: false; reason: ToolPolicyDecisionReason; classification: ToolSafetyClassification; }; declare function decideToolPolicy(args: { toolName: string; annotations?: Record; policy: EvalSuiteFileToolPolicy; }): ToolPolicyDecision; /** * The eval suite file's JSON Schema (draft 2020-12). * * STRUCTURAL contract only. Cross-field rules the zod validator enforces — * unique case ids, unique step ids within a case, and a per-case `import` * block requiring top-level `provenance` — do not project into JSON Schema. * Validate with `evalSuiteFileSchema` when you have the SDK; use this when you * only have a JSON Schema validator. */ declare const evalSuiteFileJsonSchema: Record; /** * The eval run verdict decision's JSON Schema (draft 2020-12). * * STRUCTURAL contract only. Every arithmetic and phase-ordering rule the zod * validator enforces — a rate equalling its own quotient, a verdict following * from the trial counts, validity being decided before the task verdict — does * not project into JSON Schema. Validate with `evalVerdictDecisionSchema` when * you have the SDK; use this when you only have a JSON Schema validator. */ declare const evalVerdictPolicyJsonSchema: Record; export { type AssertStep, CASE_ID_PREFIX, type ElementLocator, EvalSuiteFileToolPolicy, type InteractAction, type InteractStep, LEGACY_TEST_SCORER_ID, LEGACY_TEST_VERSION, MAX_OPAQUE_ID_LENGTH, MAX_PROBE_ARGS_CHARS, MAX_PROBE_RENDER_TIMEOUT_MS, MAX_SCRIPTED_STEP_TEXT_CHARS, MAX_SCRIPTED_WAIT_MS, MAX_TEST_STEPS, MINTED_ID_ENTROPY_CHARS, type OpaqueId, type PromptStep, ResolvedScoreDefinition, SUITE_ID_PREFIX, ScoreDefinition, ScoreResult, ScorerRole, type StepAssertionPayload, TEST_STEP_KINDS, TOOL_MATCH_SCORER_ID, TOOL_MATCH_VERSION, TOOL_POLICY_DECISION_REASONS, type TestStep, type TestStepKind, type ToolCallStep, type ToolPolicyDecision, type ToolPolicyDecisionReason, type ToolPolicySnapshot, type ToolSafetyClassification, type WidgetAssertion, assertStepSchema, buildToolPolicySnapshot, classifyToolSafety, decideToolPolicy, decideToolPolicyFromSnapshot, elementLocatorSchema, evalSuiteFileJsonSchema, evalVerdictPolicyJsonSchema, fromCriterionResult, fromGoalCompletionCase, fromLegacyTestOutcome, fromToolMatchResult, generatedPredicateScorerId, interactActionSchema, interactStepSchema, isAssertStep, isInteractStep, isOpaqueId, isPromptStep, isToolCallStep, isToolPolicyDecisionReason, isWidgetAssertion, legacyTestScoreDefinition, mintCaseId, mintSuiteId, opaqueIdSchema, predicateScoreDefinition, promptStepSchema, scoreResultFromPredicateResult, stepAssertionPayloadSchema, stepsSchema, testStepSchema, toolCallStepSchema, toolMatchScoreDefinition, widgetAssertionSchema };