{
  "skill_name": "prompt-generator",
  "target_output_spec": "TARGET_OUTPUT.md",
  "source": "https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices#evaluation-and-iteration",
  "evals": [
    {
      "id": 1,
      "name": "fresh_chat_brief_goal",
      "scenario": "Scenario 1",
      "prompt": "/prompt-generator Write a system prompt for a Python linting agent that auto-fixes code style issues in this repo",
      "files": [],
      "expected_behavior": [
        "EnterPlanMode is invoked before any AskUserQuestion",
        "All questions delivered via AskUserQuestion — zero questions in direct chat text",
        "AskUserQuestion contains 2-4 questions, each with 2-4 options, recommended option first",
        "After internal drafting finishes, one assistant turn shows ### Outcome preview with bullets only: what executing the generated prompt will produce, primary inputs or tools, done-when, short sample (TARGET_OUTPUT.md)",
        "That same turn uses AskUserQuestion (2-4 options): recommended option first = user confirms the preview matches their intent and proceeds to the final handoff; two options offer scope or emphasis shifts grounded in discovery; one option collects free-text refinements and triggers another drafting pass (at most three such preview rounds unless the user raises the cap in chat)",
        "Final response: Audit line, one Markdown code fence tagged xml with the full prompt, then ## Outcome digest after the closing fence",
        "No second outer ```xml fence in the digest (samples use four spaces or tilde fences only)",
        "Fenced block contains XML sections derived from the approved plan's heading structure",
        "Prompt generation delegated to a subagent (Agent tool call visible in the flow)"
      ]
    },
    {
      "id": 2,
      "name": "session_handoff",
      "scenario": "Scenario 2",
      "prompt": "[Preceded by 20+ turns debugging a theme export race condition, modifying download_manager.py and orchestrator.py, deciding on retry logic] /prompt-generator Generate a handoff prompt so a new session can continue this work",
      "files": [
        "packages/samsung-automation/download_manager.py",
        "packages/samsung-automation/orchestrator.py"
      ],
      "expected_behavior": [
        "AskUserQuestion has 1-2 questions — lighter than Scenario 1",
        "Generated prompt <background> includes: session state, decisions, files modified, next steps",
        "No redundant plan-mode exploration for information already in conversation",
        "Handoff prompt is self-contained — a new session can resume without prior context",
        "Prior decisions preserved in the handoff, not lost or paraphrased away",
        "After internal drafting finishes, ### Outcome preview bullets plus AskUserQuestion as in eval id 1 (confirm match as recommended first option; two contextual alternates; free-text refine option; preview loop cap)",
        "Final output: 1-liner audit + fenced XML prompt + ## Outcome digest after the fence"
      ]
    },
    {
      "id": 3,
      "name": "long_unstructured_input",
      "scenario": "Scenario 3",
      "prompt": "/prompt-generator i need a prompt for an agent that goes through our samsung seller portal automation scripts and finds all the places where we hardcoded timeouts or selectors and then extracts them into config files, the scripts are in packages/samsung-automation and they use playwright and theres shared_utils that already has some config patterns i think, also make sure it doesnt break existing tests and follows our TDD approach and code rules",
      "files": [],
      "expected_behavior": [
        "First AskUserQuestion question confirms extracted intent — not generic",
        "Ambiguities surfaced as specific options, not open-ended questions",
        "Plan mode exploration verifies references from input (shared_utils, config patterns)",
        "ALL requirements from unstructured input captured (timeouts, selectors, config extraction, TDD, code rules, test safety) — none dropped",
        "After internal drafting finishes, ### Outcome preview bullets plus AskUserQuestion as in eval id 1 (confirm match as recommended first option; two contextual alternates; free-text refine option; preview loop cap)",
        "Final output: 1-liner audit + fenced XML prompt + ## Outcome digest after the fence"
      ]
    },
    {
      "id": 4,
      "name": "noisy_context_no_degradation",
      "scenario": "Scenario 4",
      "prompt": "[Preceded by 80+ turns: failed git push, hook debugging, unrelated Samsung portal discussion, Python tracebacks, Midjourney tangent, 15+ empty Grep results] /prompt-generator Write a system prompt for a code review agent that checks for security vulnerabilities",
      "files": [],
      "expected_behavior": [
        "Output format matches Scenario 1: 1-liner audit + fenced XML prompt + ## Outcome digest after the fence",
        "After internal drafting finishes, ### Outcome preview bullets plus AskUserQuestion as in eval id 1 (confirm match as recommended first option; two contextual alternates; free-text refine option; preview loop cap)",
        "Prompt content about code review and security — zero contamination from prior noise",
        "No references to prior errors, tangents, or unrelated tool calls in the prompt",
        "XML structure complete and well-formed — no truncation from context pressure",
        "Subagent delegation visible (Agent tool call with curated context, not raw conversation)",
        "Plan mode entry scoped to the current task — prior noisy context excluded from plan"
      ]
    },
    {
      "id": 5,
      "name": "no_tool_calls_after_fence",
      "scenario": "Structural invariant A (Issue #41 Eval A)",
      "prompt": "/prompt-generator Create a prompt for an agent that traces a routing bug across shared_utils/export_handler.py, orchestrator.py, and download_manager.py — find where extract_apk is called and whether it handles APK signature check failures",
      "files": ["packages/samsung-automation/shared_utils/export_handler.py"],
      "expected_behavior": [
        "No tool_use blocks appear after the first fence marker of the canonical prompt artifact",
        "All plan-mode exploration and AskUserQuestion interactions precede the subagent invocation",
        "AskUserQuestion interactions precede the subagent; Outcome preview AskUserQuestion precedes the final Audit line and xml fence",
        "Review the last successful Audit + fenced xml pair; blocked retry attempts preserved by exported conversation logs do not count as additional delivered artifacts"
      ]
    },
    {
      "id": 6,
      "name": "fenced_block_closes_cleanly",
      "scenario": "Structural invariant B (Issue #41 Eval B)",
      "prompt": "/prompt-generator Write a detailed agent-harness prompt for a TDD bug-fix workflow that traces a routing error across 5+ files, with state management for multi-window execution and structured test tracking",
      "files": [],
      "expected_behavior": [
        "The canonical prompt artifact has one opening xml fence and one matching closing fence; exported conversation logs are normalized to that same boundary before review",
        "Every XML tag properly opened and closed",
        "No truncation at numbered-list bullets (the Issue #41 failure mode)",
        "No mid-sentence cuts or incomplete sections",
        "Artifact is copy-pasteable as-is without manual repair"
      ]
    },
    {
      "id": 7,
      "name": "discovery_complete_gate",
      "scenario": "Structural invariant C (Issue #41 Eval C)",
      "prompt": "/prompt-generator Create a prompt for an agent that refactors the Samsung theme scoring pipeline — but I'm not sure if the scoring logic is in theme_scorer.py or distributed across multiple files",
      "files": [],
      "expected_behavior": [
        "Plan mode exploration attempts to locate scoring logic before prompt generation",
        "If resolved: prompt references concrete file paths from plan-mode exploration",
        "If unresolved: prompt contains <open_question> in the relevant plan-derived section for downstream agent",
        "No re-entry to plan mode after the canonical artifact fence starts",
        "AskUserQuestion may surface the uncertainty if plan-mode exploration was inconclusive; when exploration resolves concrete paths before the artifact, absence of <open_question> is expected"
      ]
    },
    {
      "id": 8,
      "name": "no_mid_artifact_hedging",
      "scenario": "Structural invariant D (Issue #41 Eval D)",
      "prompt": "/prompt-generator Write a comprehensive agent prompt for migrating all 12 Samsung portal automation scripts from hardcoded selectors to centralized config, covering full test suite update",
      "files": [],
      "expected_behavior": [
        "Zero instances of 'let me also check', 'actually', 'one more consideration' inside fenced block",
        "No tentative language ('might be', 'possibly', 'I think') in instructions or constraints",
        "All uncertainty expressed as <open_question> tags, not inline hedges",
        "Prompt reads as confident complete instructions, not a draft-in-progress"
      ]
    },
    {
      "id": 9,
      "name": "zero_negative_phrasing_in_output",
      "scenario": "Content quality gate A (anti-pattern elimination)",
      "prompt": "/prompt-generator Write a system prompt for an agent that reviews TypeScript code for type safety, enforces strict null checks, and ensures all function signatures have complete type annotations",
      "files": [],
      "expected_behavior": [
        "Fenced prompt artifact contains zero hard anti-pattern keywords: 'no', 'not', 'don't', 'do not', 'never', 'avoid', 'without', 'refrain', 'stop', 'prevent', 'exclude', 'prohibit', 'forbid', 'reject'",
        "Zero indirect anti-patterns: 'instead of X' (implies X is bad), 'rather than X', 'as opposed to'",
        "Every instruction phrased as a positive directive: what TO do, what TO produce, what TO enforce",
        "Constraints section uses affirmative boundaries: 'only X', 'always X', 'ensure X', 'require X' — positive framing throughout",
        "Example: 'Ensure all functions have explicit return types' passes; 'Do not leave return types implicit' fails; 'Avoid missing return types' fails",
        "Applies to all plan-derived sections inside the fenced block"
      ]
    },
    {
      "id": 10,
      "name": "required_sections_present_in_artifact",
      "scenario": "Section completeness gate (render-survival)",
      "prompt": "/prompt-generator Write a system prompt for a Python linting agent that auto-fixes code style issues in this repo",
      "files": [],
      "expected_behavior": [
        "The approved plan produces at least one heading that maps to each major concern of the task (role/identity, background/context, instructions/steps, constraints/boundaries, output format)",
        "The fenced XML block contains opening and closing tags for every section derived from the approved plan's heading structure",
        "Each plan-derived section contains substantive content (minimum one sentence each)",
        "The Stop hook section-presence check passes for this output (no missing section tags relative to the approved plan)",
        "Sections derived from the plan appear in a logical order matching the plan's heading sequence"
      ]
    },
    {
      "id": 11,
      "name": "section_missing_triggers_hook_block",
      "scenario": "Section completeness gate — failure path",
      "prompt": "Synthetic eval: assistant final message is prompt-workflow shaped (overall_status, checklist, scope anchors, runtime signals) with a fenced Markdown XML block whose body omits a section that was present as a heading in the approved plan; observer asserts Stop hook behavior and successful retry.",
      "files": [],
      "expected_behavior": [
        "The Stop hook runs _check_required_xml_sections and returns a block decision naming the plan-derived section that is missing from the artifact",
        "The model retry includes all plan-derived sections with both opening and closing tags",
        "The retry output passes the section-presence gate (empty missing list from missing_required_xml_sections)"
      ]
    },
    {
      "id": 12,
      "name": "render_survival_file_fallback",
      "scenario": "Render-layer mitigation",
      "prompt": "/prompt-generator Write a comprehensive agent prompt for migrating a large Prisma schema and all related API routes, with step-by-step rollout, rollback, and verification — artifact sized like the migration prompt that triggered chat render stripping.",
      "files": [],
      "expected_behavior": [
        "When the artifact exceeds a size threshold or contains XML section tag names that collide with HTML5 elements (section, summary, details, header, footer, main, aside, article, nav, figure), the orchestrator writes the full artifact to a file under data/prompts/ or a user-specified path",
        "The file contains the complete XML with all tags preserved as literal text",
        "The user-facing message states the file path and briefly inventories which required sections the artifact contains"
      ]
    },
    {
      "id": 13,
      "name": "nested_inner_fence_does_not_truncate_xml_for_hooks",
      "scenario": "Structural invariant E — nested Markdown fences inside ```xml",
      "prompt": "/prompt-generator Include <illustrations> with a bash snippet using triple-backtick fences inside the XML, mirroring real prompts that previously hid </illustrations> in chat and broke hook extraction.",
      "files": [],
      "expected_behavior": [
        "prompt_workflow_gate_core.extract_fenced_xml_content includes text after inner ```bash ... ``` lines up to the final closing ``` of the xml fence",
        "missing_required_xml_sections sees closing tags for all plan-derived sections when those appear after nested fences",
        "SKILL.md §7 states ordered authoring steps for <illustrations>: four-space-indented sample lines, then tilde fences, then a complete triple-backtick pair when required"
      ]
    },
    {
      "id": 14,
      "name": "outcome_preview_gate_and_digest_placement",
      "scenario": "Outcome preview + post-fence digest (refinement contract)",
      "prompt": "/prompt-generator Write a short user-task prompt for triaging GitHub issues by label in this repo",
      "files": [],
      "expected_behavior": [
        "Subagent returns final XML plus preview summary fields for orchestrator use",
        "### Outcome preview markdown block precedes AskUserQuestion; bullets cover executor output, inputs or tools, done when, sample excerpt (~20 lines max)",
        "AskUserQuestion: recommended first option labels accepting the described outcome and proceeding (SKILL.md may phrase this as 'Ship this outcome profile' or equivalent); plus two contextual alternates and a free-text refinement path; at most three preview rounds unless user extends cap in chat",
        "At most three preview refinement loops unless user raises cap in chat",
        "Final handoff order: Audit line, single ```xml fence, ## Outcome digest, then optional hook validation block (defined in SKILL.md Terminology) after digest",
        "extract_fenced_xml_content returns only the XML body (digest uses no second ```xml fence)"
      ]
    },
    {
      "id": 15,
      "name": "outcome_digest_required_bullets_present",
      "scenario": "Outcome digest structural completeness",
      "prompt": "/prompt-generator Write a system prompt for an agent that audits a repository's system prompt against its rules/ files and recommends merging redundant content",
      "files": [],
      "expected_behavior": [
        "## Outcome digest section present immediately after the closing xml fence with zero intervening prose",
        "Four required bold headers present in order: **What it does**, **Key inputs**, **Done when**, **Quick sample**",
        "Each header is followed by at least one sentence or bullet of substantive content",
        "No second ```xml fence anywhere inside the ## Outcome digest section (four-space indent, tilde fence, or complete triple-backtick pair when a code sample is needed)"
      ]
    },
    {
      "id": 16,
      "name": "outcome_digest_quick_sample_is_executor_output",
      "scenario": "Quick sample content — executor output, not prompt revision commentary",
      "prompt": "/prompt-generator Refine my code review prompt to incorporate the missing CODE_RULES.md sections into <code_quality> and demote the rules file to an abbreviated pointer",
      "files": [],
      "expected_behavior": [
        "**Quick sample** content represents approximately 20 lines of what the downstream executor would produce (structured findings, categorized recommendations, code diffs, or checklists) — not a description of edits made to the prompt XML",
        "Quick sample contains zero instances of phrases that describe prompt authoring: 'moved into', 'extracted as', 'replaced the softer', 'N changes from the original', 'updated to match', 'now names'",
        "Quick sample format is consistent with the output shape described in **What it does** — if the prompt produces categorized recommendations, the sample shows categorized recommendations",
        "Digest describes what executing the prompt produces, not what prompt-generator did to the XML"
      ]
    },
    {
      "id": 17,
      "name": "outcome_digest_no_change_log_framing",
      "scenario": "Outcome digest framing — executor perspective, not author perspective",
      "prompt": "/prompt-generator Update my Python code reviewer prompt: add a security-vulnerability section and add a done-when criterion that the agent reports zero critical findings",
      "files": [],
      "expected_behavior": [
        "**What it does** describes what the executor running the prompt will do — not what prompt-generator changed in the XML",
        "**Key inputs** names files, tools, or context the executor needs — not a list of sections that were added or modified",
        "**Done when** states a checkable executor success condition — not 'prompt validated at exit 0' or 'all sections present'",
        "No bullet in any digest section uses past-tense authoring framing: 'was added', 'has been updated', 'now includes', 'changed from'",
        "Digest reads from the perspective of someone deciding whether to run the prompt, not someone reviewing what prompt-generator produced"
      ]
    }
  ]
}
