{
  "$schema": "https://json-schema.org/draft/2020-12/schema",
  "name": "shell-selection routing accuracy corpus",
  "version": "2.1.0",
  "purpose": "Routing-eval corpus for shell-selection (adia-ui-factory plugin). Each phrase declares whether shell-selection SHOULD be the routing target. Scored by the estate's TF-IDF routing-eval runner against the skill's description (heuristic token overlap).",
  "scoring_notes": "Heuristic signal, not ground truth. Treat misroutes as a prompt to tighten the skill description, never as a reason to keyword-stuff it. Real harness routing is LLM-driven.",
  "license": "internal",
  "scope": "shell-selection routing \u2014 does this phrase activate shell-selection?",
  "phrases": [
    {
      "id": "shells-admin-01",
      "phrase": "I need a sidebar + topbar layout with a command palette for this app",
      "expected": "shell-selection",
      "expected_shape": "pick-admin",
      "rationale": "The exact 'sidebar + topbar layout' trigger phrase named in the description \u2014 this is the fence-test the audit flagged: it must land on shell-selection (picking page-chrome), not screen-composition (which builds screen content, not the frame around it)."
    },
    {
      "id": "shells-admin-02",
      "phrase": "set up the full app frame \u2014 sidebar(s) plus topbar plus command palette \u2014 for the dashboard",
      "expected": "shell-selection",
      "expected_shape": "pick-admin",
      "rationale": "Restates the admin-shell decision-table signal ('full app frame \u2014 sidebar(s) + topbar + command palette + pages') in the requester's own words."
    },
    {
      "id": "shells-chat-01",
      "phrase": "build an LLM conversation surface with a thread and a composer",
      "expected": "shell-selection",
      "expected_shape": "pick-chat",
      "rationale": "Matches chat-shell's decision-table signal: an LLM conversation surface (thread + composer)."
    },
    {
      "id": "shells-chat-02",
      "phrase": "wire up the chat-shell's thread and composer chrome",
      "expected": "shell-selection",
      "expected_shape": "pick-chat",
      "rationale": "Names chat-shell directly and asks for its chrome, not the conversation content inside it."
    },
    {
      "id": "shells-editor-01",
      "phrase": "I need a design tool with a center canvas and resizable side panes",
      "expected": "shell-selection",
      "expected_shape": "pick-editor",
      "rationale": "Matches editor-shell's decision-table signal: a design tool \u2014 center canvas + resizable side panes."
    },
    {
      "id": "shells-editor-02",
      "phrase": "add focus mode to the editor-shell's canvas + panes layout",
      "expected": "shell-selection",
      "expected_shape": "pick-editor",
      "rationale": "Focus mode is an editor-shell chrome feature named in the skill body, not screen content."
    },
    {
      "id": "shells-simple-01",
      "phrase": "scaffold a minimal centered chrome for the marketing landing page",
      "expected": "shell-selection",
      "expected_shape": "pick-simple",
      "rationale": "Matches simple-shell's decision-table signal: marketing/error/landing/auth \u2014 minimal centered chrome."
    },
    {
      "id": "shells-simple-02",
      "phrase": "pick a shell for the auth error page \u2014 it just needs minimal centered chrome",
      "expected": "shell-selection",
      "expected_shape": "pick-simple",
      "rationale": "Explicit shell-selection ask against the simple-shell signal."
    },
    {
      "id": "shells-embed-01",
      "phrase": "embed this surface into a host page \u2014 size and center the light-DOM element",
      "expected": "shell-selection",
      "expected_shape": "pick-embed",
      "rationale": "The exact 'embed this surface' trigger phrase named in the description, matched to embed-shell's decision-table row."
    },
    {
      "id": "shells-embed-02",
      "phrase": "wire the embed-shell cluster for the widget we're dropping into a third-party site",
      "expected": "shell-selection",
      "expected_shape": "pick-embed",
      "rationale": "Names embed-shell directly; cluster registration is shell-selection/composition territory."
    },
    {
      "id": "shells-debug-01",
      "phrase": "the sidebar's [collapsed] attribute doesn't collapse the admin-shell layout \u2014 help me debug the shell markup",
      "expected": "shell-selection",
      "expected_shape": "shell-debugging",
      "rationale": "The exact 'shell markup debugging' trigger phrase named in the description, tied to the shared state-is-an-attribute convention."
    },
    {
      "id": "shells-debug-02",
      "phrase": "adia-lint flagged a legacy <aside data-sidebar> shape in our shell \u2014 fix the markup",
      "expected": "shell-selection",
      "expected_shape": "shell-debugging",
      "rationale": "Legacy data-attribute shapes are called out explicitly in the skill body as retired shell-tier vocabulary; fixing them is shell markup debugging."
    },
    {
      "id": "shells-generic-01",
      "phrase": "use a shell for this new app frame",
      "expected": "shell-selection",
      "expected_shape": "shell-selection-generic",
      "rationale": "The literal 'use a shell' trigger phrase from the description with no shell yet named \u2014 canonical cold-start-within-shells ask."
    },
    {
      "id": "shells-generic-02",
      "phrase": "which shell fits an app that needs both a big canvas and side panes",
      "expected": "shell-selection",
      "expected_shape": "shell-selection-generic",
      "rationale": "A decision-cue question aimed at the pick-the-shell table without naming a shell \u2014 still shell-selection' job, not a specific-shell shape."
    },
    {
      "id": "shells-adv-01",
      "phrase": "compose the screens that go inside the admin-shell's page area",
      "expected": "screen-composition",
      "rationale": "Adversarial \u2014 the audit's core fence question, resolved the other direction: screens INSIDE an already-chosen shell are screen-composition's job per the description's own NOT-fence."
    },
    {
      "id": "shells-adv-02",
      "phrase": "wire the SSR host so the framework's route outlet renders inside the shell",
      "expected": "host-wiring",
      "rationale": "Adversarial \u2014 host/SSR route-outlet wiring is explicitly carved out to host-wiring in both the description and the shared-conventions section (NEVER mount <router-ui> under SSR \u2014 the framework outlet owns the route)."
    },
    {
      "id": "shells-adv-03",
      "phrase": "the admin-shell's sidebar-collapse behavior regressed \u2014 fix the bug in packages/web-modules source",
      "expected": "primitive-authoring",
      "rationale": "Adversarial \u2014 a shell BUG in the framework monorepo itself (editing packages/web-modules source) is forge-side authoring, not factory-side shell selection/composition.",
      "expected_alternative_raw": "primitive-authoring (adia-ui-forge plugin)"
    },
    {
      "id": "shells-adv-04",
      "phrase": "wire the state/data hydration for the shell's sidebar list",
      "expected": "data-wiring",
      "rationale": "Adversarial \u2014 data/state pattern choice is data-wiring's territory; shell-selection only reflects state as an attribute the consumer already owns."
    },
    {
      "id": "shells-adv-05",
      "phrase": "add the LLM streaming client to power the chat-shell's composer",
      "expected": "llm-wiring",
      "rationale": "Adversarial \u2014 the chat LLM client/proxy contract is explicitly deferred to llm-wiring in the shell's own references section, even though it's chat-shell-adjacent."
    },
    {
      "id": "shells-adv-06",
      "phrase": "scaffold a brand new app's on-disk structure before we even pick a shell",
      "expected": "project-scaffolding",
      "rationale": "Adversarial \u2014 app structure/scaffolding precedes shell selection and belongs to project-scaffolding; no shell has been named or implied."
    }
  ],
  "minimums_per_spec": {
    "trigger_phrases": 14,
    "adversarial_phrases": 6,
    "task_shapes_covered": "pick-admin, pick-chat, pick-editor, pick-simple, pick-embed, shell-debugging, shell-selection-generic",
    "adversarial_fraction": "30%"
  },
  "evaluator_notes": {
    "current_state": "20 cases (14 trigger + 6 adversarial). Each of the 7 task shapes (5 named shells + shell-debugging + shell-selection-generic) has 2 trigger cases.",
    "promotion_criteria": "Promote to a CI gate (warn \u2192 hard-fail) when F1 >= 0.85 on this corpus across 3+ consecutive runs without description edits.",
    "review_cadence": "Re-run on every shell-selection description edit. Add 1-2 new cases per checkpoint reflecting newly-observed routing failure modes.",
    "known_limitations": "The audit's core open question \u2014 whether 'sidebar + topbar layout' reaches shell-selection or gets pulled toward screen-composition \u2014 is encoded as shells-admin-01/02 (positive) and shells-adv-01 (the compose-side adversarial); the heuristic TF-IDF scorer may not separate 'layout' (shell chrome) from 'screen'/'form'/'dashboard' (compose content) as cleanly as an LLM router would, since both descriptions share UI-layout vocabulary."
  }
}
