{
  "$schema": "https://anthropic.com/schemas/agent-routing-evals.v1.json",
  "version": "0.2.0",
  "description": "Routing-accuracy corpus for the adia-ui-factory consumer agent roster (4 cards materialized: app-planning-agent + screen-composition-agent + surface-qa-agent (2026-07-16) + ui-architect (2026-08-13, gh#1183)). For each user phrase, declares the expected agent that should activate (or null for phrases that intentionally route nowhere). Scored by `scripts/skills/run-agent-evals.mjs --agents-dir packages/plugins/adia-ui-factory/agents`. Boris falsification test T1 (agent routing). First eval pass 2026-06-08.",
  "scope": "Materialized factory agents: app-planning-agent, screen-composition-agent, surface-qa-agent (the reviewer seat — was a 'mode' of screen-composition-agent until 2026-07-16, when the factory-audit's generator≠critic finding materialized it as its own read-only card, gh#259 Wave 3), ui-architect (the whole-deliverable coordinator seat, gh#1183 — dispatches the other three, never duplicates their charters); app-migrator is deferred. Trimmed to the T1 keystone evidence (2026-06-08).",
  "scoring_notes": "The scorer uses TF-IDF-style token overlap between the phrase and each agent card's description (+triggers if present). This is a HEURISTIC eval — real orchestrator routing may differ. Treat misroutes as a signal to tighten the card's description field, not as ground truth. Scores ≥ MIN_SCORE_THRESHOLD (2.0) + ≥ MIN_MATCH_COUNT (2) required to activate.",
  "phrases": [
    {
      "id": "architect-01",
      "phrase": "I have a brief for a patient-facing app, where do we start?",
      "expected": "app-planning-agent",
      "rationale": "Classic brief-intake — architect's primary trigger."
    },
    {
      "id": "architect-02",
      "phrase": "turn this PRD into a screen plan",
      "expected": "app-planning-agent",
      "rationale": "Plan-from-brief — architect owns Intent+Plan slice."
    },
    {
      "id": "architect-03",
      "phrase": "what shell should I use for this admin dashboard?",
      "expected": "app-planning-agent",
      "rationale": "Shell classification is an architect gate."
    },
    {
      "id": "architect-04",
      "phrase": "design this auth flow before we build it",
      "expected": "app-planning-agent",
      "rationale": "Flow design precedes composition — architect's §SpecToUi."
    },
    {
      "id": "architect-05",
      "phrase": "classify the rendering mode for this Next.js app",
      "expected": "app-planning-agent",
      "rationale": "Rendering-mode classification is architect's four-classifier step."
    },
    {
      "id": "architect-06",
      "phrase": "I need a plan for 4 screens before we start composing",
      "expected": "app-planning-agent",
      "rationale": "Multi-screen decomposition — architect, not composer."
    },
    {
      "id": "architect-07",
      "phrase": "produce an orientation record for this greenfield project",
      "expected": "app-planning-agent",
      "rationale": "Orientation Record is architect's literal output."
    },
    {
      "id": "architect-08",
      "phrase": "clarify this vague intent before screen-composition-agent starts",
      "expected": "app-planning-agent",
      "rationale": "Intent disambiguation = architect's stop-and-ask gate."
    },
    {
      "id": "architect-09",
      "phrase": "walk me through §SpecToUi before we emit any components",
      "expected": "app-planning-agent",
      "rationale": "§SpecToUi now lives in app-planning-agent's own preloaded domain-planning skill (gh#1207), not a bare methodology rung — architect is the pre-rename seat name."
    },
    {
      "id": "architect-10",
      "phrase": "what's the project shape — single surface or shared foundation?",
      "expected": "app-planning-agent",
      "rationale": "Project-shape classification — architect's second classifier."
    },
    {
      "id": "composer-01",
      "phrase": "compose the patient-labs Live tab screen",
      "expected": "screen-composition-agent",
      "rationale": "Compose a screen from primitives — composer's primary task."
    },
    {
      "id": "composer-02",
      "phrase": "scaffold a new SPA app for this brief",
      "expected": "screen-composition-agent",
      "rationale": "Scaffold host (SPA/SSR) is composer's first EXECUTE step."
    },
    {
      "id": "composer-03",
      "phrase": "wire the data stream on the recommendations card",
      "expected": "screen-composition-agent",
      "rationale": "Data/LLM wiring — composer's EXECUTE slice."
    },
    {
      "id": "composer-04",
      "phrase": "build the composed screen from the architect's plan",
      "expected": "screen-composition-agent",
      "rationale": "Execute from plan — composer consumes Orientation Record."
    },
    {
      "id": "composer-05",
      "phrase": "implement the admin shell with nav and main content area",
      "expected": "screen-composition-agent",
      "rationale": "Shell composition is composer's scope (shells skill)."
    },
    {
      "id": "composer-06",
      "phrase": "validate the LLM-generated A2UI before writing it to the file",
      "expected": "screen-composition-agent",
      "rationale": "Trust gate is explicitly composer's (validate-before-write)."
    },
    {
      "id": "composer-07",
      "phrase": "the screen is scaffolded, now compose the components",
      "expected": "screen-composition-agent",
      "rationale": "Post-scaffold compose step — composer."
    },
    {
      "id": "composer-08",
      "phrase": "set up the SPA routing and content-less router-ui",
      "expected": "screen-composition-agent",
      "rationale": "SPA host wiring — spa skill, composer's EXECUTE."
    },
    {
      "id": "composer-09",
      "phrase": "write the patient-labs-embed component using adia-ui primitives",
      "expected": "screen-composition-agent",
      "rationale": "Primitive-based component authoring — composer's core."
    },
    {
      "id": "composer-10",
      "phrase": "the trust gate failed — hold the LLM output for human review",
      "expected": "screen-composition-agent",
      "rationale": "Fail-closed trust gate action — composer's §Trust rule."
    },
    {
      "id": "null-01",
      "phrase": "add a new primitive to packages/web-components/",
      "expected": null,
      "note": "In-tree authoring — forge/author, not a factory agent."
    },
    {
      "id": "null-02",
      "phrase": "cut a v0.8.0 release of the framework",
      "expected": null,
      "note": "Framework release — forge/package-release-agent, not a factory agent."
    },
    {
      "id": "null-03",
      "phrase": "why did the zettel eval coverage drop?",
      "expected": null,
      "note": "A2UI pipeline internals — forge/a2ui-maintenance-agent."
    },
    {
      "id": "null-04",
      "phrase": "rotate the SSH keys on the exe.dev VM",
      "expected": null,
      "note": "Infra ops — forge/repo-steward, not factory."
    },
    {
      "id": "null-05",
      "phrase": "what's the weather in Helsinki?",
      "expected": null,
      "note": "Off-topic."
    },
    {
      "id": "null-06",
      "phrase": "fix a bug in the segmented-ui JavaScript",
      "expected": null,
      "note": "Framework-internal bug fix — forge/author."
    },
    {
      "id": "null-07",
      "phrase": "harvest a new A2UI training chunk",
      "expected": null,
      "note": "Corpus authoring — forge/a2ui-maintenance-agent."
    },
    {
      "id": "ambig-02",
      "phrase": "compose this",
      "expected": null,
      "ambiguous_between": [
        "screen-composition-agent",
        "app-planning-agent"
      ],
      "note": "Bare verb, no context. Should not activate either card — MIN_SCORE_THRESHOLD / MIN_MATCH_COUNT guard."
    },
    {
      "id": "verifier-01",
      "phrase": "screen-composition-agent says the claims screen is done — QA it",
      "expected": "surface-qa-agent",
      "rationale": "Definition-of-done dispatch — the reviewer grades work the builder produced."
    },
    {
      "id": "verifier-02",
      "phrase": "run the browser gate on the new settings surface and give me the VerifyProof",
      "expected": "surface-qa-agent",
      "rationale": "Names the seat's deliverable directly."
    },
    {
      "id": "verifier-03",
      "phrase": "is this surface ready to ship? check a11y and the screenshot",
      "expected": "surface-qa-agent",
      "rationale": "Ship-readiness ask — probe + judgment half, no fixing."
    },
    {
      "id": "verifier-04",
      "phrase": "the QA found a 0x0 host — fix the missing import",
      "expected": "screen-composition-agent",
      "rationale": "Adversarial: the FIX routes to the builder; the verifier holds no Write tool."
    },
    {
      "id": "verifier-05",
      "phrase": "review this PRD before we plan the screens",
      "expected": "app-planning-agent",
      "rationale": "Adversarial: document intake is the architect's slice, not surface QA."
    },
    {
      "id": "coordinator-01",
      "phrase": "build me an internal claims-review tool, nothing fancy",
      "expected": "ui-architect",
      "rationale": "Whole-deliverable ask, no plan yet — GEAR 1, the coordinator's entry point."
    },
    {
      "id": "coordinator-02",
      "phrase": "here's the PRD, figure out what screens we need and build it",
      "expected": "ui-architect",
      "rationale": "PRD handed over expecting the agent to 'figure it out' — the coordinator's literal GEAR 2 charter."
    },
    {
      "id": "coordinator-03",
      "phrase": "make an app that lets patients review their labs and message their care team",
      "expected": "ui-architect",
      "rationale": "'Make an app that...' — whole-deliverable novice ask, not a single-screen composition."
    },
    {
      "id": "coordinator-04",
      "phrase": "take this spec doc and build the whole thing end to end, screens and all",
      "expected": "ui-architect",
      "rationale": "Explicit end-to-end ask across multiple surfaces from one document — GEAR 2."
    },
    {
      "id": "coordinator-05",
      "phrase": "coordinate the build across all the surfaces in this PRD and open the PRs",
      "expected": "ui-architect",
      "rationale": "Names the coordinator's own deliverable — multi-surface dispatch plus tracked, mergeable PRs."
    },
    {
      "id": "coordinator-06",
      "phrase": "some kind of dashboard thing for tracking our support tickets, not sure exactly what yet",
      "expected": "ui-architect",
      "rationale": "Classic novice/vague brief needing the clarify-then-build GEAR 1 loop, not a bare classify."
    },
    {
      "id": "coordinator-null-01",
      "phrase": "what shell should I use for this admin dashboard?",
      "expected": "app-planning-agent",
      "note": "A lone classify/plan ask with no build attached stays app-planning-agent's territory — the coordinator is never triggered by orientation alone."
    },
    {
      "id": "coordinator-null-02",
      "phrase": "compose the patient-labs Live tab screen",
      "expected": "screen-composition-agent",
      "note": "One already-planned screen — composition, not whole-deliverable coordination."
    },
    {
      "id": "coordinator-null-03",
      "phrase": "is this surface ready to ship? check a11y and the screenshot",
      "expected": "surface-qa-agent",
      "note": "Single-surface QA ask — no coordination across multiple surfaces or seats implied."
    }
  ],
  "expected_baseline": {
    "measured": "2026-07-16",
    "overall_accuracy": 0.606,
    "per_agent_f1": {
      "app-planning-agent": 0.78,
      "screen-composition-agent": 0.27,
      "surface-qa-agent": 0.86,
      "no-agent": 0.62,
      "ui-architect": 0.73
    },
    "floor": "ADVISORY pending the W1 trigger-vocabulary decision (gh#268)",
    "attribution": "The retired 0.952 baseline was measured against v1 cards carrying `triggers:` fields; the v2 cards removed them and screen-composition-agent's recall collapsed to 18% under the token-overlap heuristic — a scorer artifact, not a live routing regression (the real harness routes on full descriptions). Restore trigger vocabulary or ratify a new floor in W1; do not tune card prose to game the heuristic meanwhile.",
    "known_scorer_limits": [
      "verifier-04: 'fix the missing import' — the card's own 'never fixes' supplies the fix token; expected stays screen-composition-agent",
      "verifier-05: PRD-intake phrase — architect's triggers-era vocabulary gone; expected stays app-planning-agent",
      "coordinator-01..06/null-01..03: added 2026-08-13 (gh#1183) alongside ui-architect itself — measured separately (eval:agent-routing:factory, 2026-08-13, overall 57.1%/42 phrases) after tightening ui-architect's description to clear app-planning-agent's fp=0 (the ADR/#80 requirement not to steal app-planning-agent's territory); app-planning-agent/surface-qa-agent F1 unchanged from the 2026-07-16 run, screen-composition-agent's pre-existing scorer collapse (gh#268) is unaffected by this addition"
    ]
  },
  "notes": "Carried to adia-factory (v2 estate) 2026-07-11; agent names unchanged."
}
