{
  "skill_name": "pmin",
  "target_output_spec": "skills/prompt-generator/TARGET_OUTPUT.md",
  "source": "https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices#evaluation-and-iteration",
  "evals": [
    {
      "id": 1,
      "name": "pmin_formats_input_does_not_execute",
      "scenario": "Scope boundary — format only, no task execution",
      "prompt": "/pmin i want to do a final review of our software-engineer file. do a complete audit, thorough. make sure you read every single character and line. target: only include things that apply broadly. Keep it concise. For each line, ask: \"Would removing this cause Claude to not follow my required specifications?\" If not, cut it.",
      "files": [],
      "expected_behavior": [
        "Output is exactly one xml fence followed immediately by ## Outcome digest — zero prose before the fence, zero prose after the digest",
        "The xml fence contains a structured XML prompt wrapping the user's audit request — not the result of performing the audit",
        "Zero tool calls made: no file reads, no Glob, no Zoekt searches, no subagent spawns",
        "The assistant produces no audit output (tables, findings, executive summaries, action lists) before, inside, or after the xml fence",
        "## Outcome digest contains all four required bullets: **What it does**, **Key inputs**, **Done when**, **Quick sample**"
      ]
    },
    {
      "id": 2,
      "name": "pmin_single_pass_zero_tool_calls",
      "scenario": "Zero-tool single pass — no AskUserQuestion, no plan mode, no preview gate",
      "prompt": "/pmin You are a Python test writer. Given a function signature and its docstring, produce a pytest test file with happy-path and one edge-case test. Use descriptive test names.",
      "files": [],
      "expected_behavior": [
        "No tool calls of any kind: no Read, Glob, Grep, Zoekt, AskUserQuestion, EnterPlanMode, ExitPlanMode, Agent, or Task",
        "No Outcome preview gate turn before the final handoff",
        "Output is one xml fence followed immediately by ## Outcome digest — complete in a single turn"
      ]
    },
    {
      "id": 3,
      "name": "pmin_outcome_digest_four_bullets",
      "scenario": "Outcome digest structural completeness",
      "prompt": "/pmin Review all Python files in packages/ for type hint completeness. For each file, list functions missing return types or parameter types. Output a markdown table: File | Function | Missing.",
      "files": [],
      "expected_behavior": [
        "## Outcome digest present immediately after the closing xml fence with zero intervening prose",
        "Four required bold headers present in order: **What it does**, **Key inputs**, **Done when**, **Quick sample**",
        "Each header is followed by at least one sentence or bullet of substantive content",
        "No second ```xml fence anywhere inside the ## Outcome digest section"
      ]
    },
    {
      "id": 4,
      "name": "pmin_outcome_digest_not_a_table",
      "scenario": "Outcome digest format — bullets, not markdown table",
      "prompt": "/pmin Audit .claude/system-prompts/software-engineer.xml line by line. For each section, decide: KEEP, CUT, or MOVE. Output a table: Section | Decision | Rationale.",
      "files": [],
      "expected_behavior": [
        "## Outcome digest uses four bullet sections (**What it does**, **Key inputs**, **Done when**, **Quick sample**) — not a markdown table",
        "The digest format matches the Outcome digest section in TARGET_OUTPUT.md: bullets under each of the four required headers",
        "No part of the digest previews or replicates the output table the executor prompt would produce"
      ]
    },
    {
      "id": 5,
      "name": "pmin_quick_sample_is_executor_output",
      "scenario": "Quick sample content — executor output, not prompt commentary",
      "prompt": "/pmin Refine this system prompt to incorporate the missing CODE_RULES.md sections into <code_quality> and demote the rules file to an abbreviated pointer",
      "files": [],
      "expected_behavior": [
        "**Quick sample** content represents approximately 20 lines of what the downstream executor would produce — not a description of edits made to the prompt XML",
        "Quick sample contains zero prompt-authoring phrases: 'moved into', 'extracted as', 'replaced the softer', 'N changes from the original', 'updated to match', 'now names'",
        "Digest reads from the perspective of someone deciding whether to run the prompt"
      ]
    },
    {
      "id": 6,
      "name": "pmin_no_prose_before_or_after",
      "scenario": "Clean output boundaries",
      "prompt": "/pmin You are a git commit message writer. Given a diff, produce a conventional commit message with subject line and body.",
      "files": [],
      "expected_behavior": [
        "Zero prose before the opening xml fence",
        "Zero prose between the closing xml fence and ## Outcome digest",
        "Zero prose after the final bullet of ## Outcome digest",
        "No 'Here is the formatted prompt:' or similar framing sentence before the fence"
      ]
    },
    {
      "id": 7,
      "name": "pmin_no_nested_backtick_fences",
      "scenario": "Fence safety — input with shell code blocks must not produce nested backtick fences inside the xml fence",
      "prompt": "/pmin You are a focused implementation agent working in the backfill-scripts repository. Execute each task below in sequence without asking clarifying questions unless a command would be destructive and irreversible.\n\n## Task 1 — Create a fresh branch off origin/main\n\nRun:\n```\ngit fetch origin\ngit checkout -b fix/backfill-status-enum origin/main\n```\n\nConfirm the branch is checked out and clean before proceeding.\n\n## Task 2 — Fix the backfill script\n\nEdit `reports/backfill_release_versions_from_exports.py`. Locate the UPDATE statement that sets `status = 'active'` and change it to `status = 'Released'`. All themes that have a matched export file should be marked Released.\n\n## Task 3 — Commit and push\n\n```\ngit add reports/backfill_release_versions_from_exports.py\ngit commit -m \"fix: use Released status in backfill UPDATE statement\"\ngit push -u origin fix/backfill-status-enum\n```\n\nOutput results in order: branch confirmation, the edited UPDATE line, push confirmation URL.",
      "files": [],
      "expected_behavior": [
        "No triple-backtick code fences inside the xml fence",
        "Shell commands and code samples use 4-space indented text",
        "Output is exactly one xml fence followed immediately by ## Outcome digest — zero prose before the fence, zero prose after the digest"
      ]
    },
    {
      "id": 8,
      "name": "pmin_compound_task_formats_not_executes",
      "scenario": "Scope boundary — compound investigate+implement task formatted as prompt artifact, not executed",
      "prompt": "/pmin investigate the auth flow in the codebase and implement rate-limit checks on the login endpoint",
      "files": [],
      "expected_behavior": [
        "Output is exactly one xml fence followed immediately by ## Outcome digest — zero prose before the fence, zero prose after the digest",
        "Zero tool calls of any kind: no Read, Glob, Grep, Zoekt, AskUserQuestion, EnterPlanMode, ExitPlanMode, Agent, or Task",
        "The xml fence contains a structured XML prompt that an executor would run to investigate and implement — not investigation output, not file contents, not a findings report, not a diff, and not implemented code",
        "Outside of **Quick sample**, no investigation output, findings, analysis, or implementation work appears anywhere in the response",
        "Action verbs in the input ('investigate', 'find', 'read', 'add', 'implement', 'fix') do not trigger tool calls or any form of inline task execution",
        "## Outcome digest contains all four required bullets: **What it does**, **Key inputs**, **Done when**, **Quick sample**"
      ]
    }
  ]
}
