{
  "$schema": "https://json-schema.org/draft/2020-12/schema",
  "name": "demo-audit routing accuracy corpus",
  "version": "2.1.0",
  "purpose": "Routing-eval corpus for demo-audit. Each phrase declares the skill (expected), a forbidden skill (expected_not, for phrases the source data only ever asserted as \"not this skill\"), or neither. Scored by scripts/skills/run-skill-evals.mjs (TF-IDF token overlap over per-skill description+triggers).",
  "scoring_notes": "Heuristic signal, not ground truth. Treat misroutes as a prompt to tighten the skill description, never as a reason to keyword-stuff it. Real harness routing is LLM-driven.",
  "scope": "demo-audit routing, does this phrase activate demo-audit?",
  "phrases": [
    {
      "id": "demo-audit-pos-01",
      "phrase": "run a dogfood sweep before we cut the release",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-02",
      "phrase": "find broken demos across the component site",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-03",
      "phrase": "audit native primitive leaks in the admin app",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-04",
      "phrase": "check for attr-quote typos that broke the rendered HTML",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-05",
      "phrase": "run the app-shell QA sweep after the apps/ structural refactor",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-06",
      "phrase": "audit admin-shell composition for missing canonical parts",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-07",
      "phrase": "sweep card anatomy for a collapsed header wrapper",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-08",
      "phrase": "run npm run dogfood:status and post the findings in the PR description",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-09",
      "phrase": "is this surface clean before we merge",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-10",
      "phrase": "find token drift and contrast collapse across the components",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-11",
      "phrase": "audit the demo pages for broken components after the token refactor",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-pos-12",
      "phrase": "check for unregistered custom elements on the component pages",
      "expected": "demo-audit"
    },
    {
      "id": "demo-audit-neg-01",
      "phrase": "score the gen-ui gallery output quality against the rubric",
      "expected_not": "demo-audit"
    },
    {
      "id": "demo-audit-neg-02",
      "phrase": "review the rubric score for this generated screen and root-cause the gap",
      "expected_not": "demo-audit"
    },
    {
      "id": "demo-audit-neg-03",
      "phrase": "add a new primitive component to packages/web-components",
      "expected_not": "demo-audit"
    },
    {
      "id": "demo-audit-neg-04",
      "phrase": "cut the next release and publish the lockstep packages",
      "expected_not": "demo-audit"
    },
    {
      "id": "demo-audit-neg-05",
      "phrase": "deploy the latest build to ui-kit.exe.xyz",
      "expected": "site-deployment"
    },
    {
      "id": "demo-audit-neg-06",
      "phrase": "zettel coverage dropped on the nightly eval, find out why",
      "expected_not": "demo-audit"
    },
    {
      "id": "demo-audit-neg-07",
      "phrase": "fix the openai stopReason mapping in the llm client",
      "expected_not": "demo-audit"
    },
    {
      "id": "demo-audit-neg-08",
      "phrase": "compose the billing settings screen from the catalog",
      "expected_not": "demo-audit"
    }
  ],
  "_measured_historical": {
    "as_of": "2026-07-18",
    "scorer": "routing_eval.py (nonoun-plugins/forge)",
    "f1": 0.857,
    "precision": 1.0,
    "recall": 0.75,
    "exit_code": 0,
    "note": "measured by the external routing_eval.py (nonoun-plugins/forge) against the pre-conversion positives/negatives shape; historical record only, not regenerated by run-skill-evals.mjs."
  }
}
