/** * Phase 4 (caveman): Output format validation for live-session workers. * * Validates that worker output conforms to the structured output contract * for the given role. If validation fails, returns structured error info * that can be used for retry or fallback. * * Inspired by caveman's validate.py — check structural preservation * (headings, code blocks, URLs) after compression. */ /** * Why relax: real worker LLMs emit markdown handoffs (`## Handoff`, * `### Summary`, `## Follow-ups`, `- bullet`, `**bold**`) instead of the * caveman formats (`file:line — text`, `PASS:`/`FAIL:`, emoji findings) * these patterns originally required. Each role pattern below ORs its * strict contract with a markdown-structured alternation so structured * handoffs validate while empty/garbage output still fails. * * What changed: added the MARKDOWN_STRUCTURED alternation to every role. * What is preserved: the strict patterns still match verbatim, and the * structural-preservation checks (code blocks, URLs, headings) in * validateWorkerOutput are UNCHANGED. */ /** Accepts atx headings (`## X`), bold (`**x**`), bullets (`- x` / `* x`), and numbered lists (`1. x`) */ const MARKDOWN_STRUCTURED = /^(?:#{1,6}\s|\*\*|[-*]\s|\d+\.\s)/m; /** Strict per-role contract patterns (kept verbatim; `.source` is embedded in the alternations below) */ const STRICT_ROLE_PATTERNS: Record = { explorer: /^(\S+:\d+|Defs:|Refs:|Callers:|Tests:|Sites:|No match\.|totals:)/m, executor: /^(\S+:\d+(-\d+)? — .{1,80}\.|verified:|too-big\.|needs-confirm\.|ambiguous\.|regressed\.)/m, reviewer: /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu, "security-reviewer": /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu, verifier: /^(PASS:|FAIL:)/m, }; /** Role-specific output format patterns — constructed fresh per call to avoid /g lastIndex leak. * Each factory: strict contract OR markdown-structured alternation. */ const ROLE_PATTERN_DEFS: Record RegExp> = { explorer: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.explorer.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"), executor: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.executor.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"), reviewer: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.reviewer.source})|(?:${MARKDOWN_STRUCTURED.source})`, "mu"), "security-reviewer": () => new RegExp(`(?:${STRICT_ROLE_PATTERNS["security-reviewer"].source})|(?:${MARKDOWN_STRUCTURED.source})`, "mu"), verifier: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.verifier.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"), }; /** Fresh RegExp factories for structural preservation checks (avoids /g lastIndex leak) */ const makeUrlRe = () => /\bhttps?:\/\/[^\s<>)\]"',;]+/gi; const makeFencedCodeRe = () => /```[\s\S]*?```/g; const makeInlineCodeRe = () => /`[^`\n]+`/g; const makeHeadingRe = () => /^#{1,6}\s+.+/gm; export interface OutputValidationResult { /** Whether the output passes validation */ valid: boolean; /** Whether the output follows the role's contract format */ formatMatch: boolean; /** Whether structural elements (code, URLs, headings) are preserved */ structurePreserved: boolean; /** Specific issues found */ issues: string[]; } /** * Validate worker output against role-specific contract + structural preservation. */ export function validateWorkerOutput(role: string, output: string): OutputValidationResult { const issues: string[] = []; // Empty output always fails if (!output?.trim()) { return { valid: false, formatMatch: false, structurePreserved: false, issues: ["Empty output"], }; } // Check role-specific format const patternFactory = ROLE_PATTERN_DEFS[role]; const pattern = patternFactory ? patternFactory() : undefined; const formatMatch = !pattern || pattern.test(output); if (!formatMatch) { issues.push(`Output does not match expected ${role} contract format`); } // Check structural preservation (code blocks, URLs, headings) let structurePreserved = true; const trimmedOutput = output.trim(); // Detect if output was truncated mid-code-block const opens = (trimmedOutput.match(/```/g) ?? []).length; if (opens % 2 !== 0) { structurePreserved = false; issues.push("Unclosed code block — output may be truncated"); } // Check for malformed URLs const urls = trimmedOutput.match(makeUrlRe()) ?? []; for (const url of urls) { if (url.endsWith(".") || url.endsWith(",")) { structurePreserved = false; issues.push(`URL with trailing punctuation: ${url.slice(-20)}`); } } return { valid: formatMatch && structurePreserved, formatMatch, structurePreserved, issues, }; } /** * bug-026 sub-issue A: strict log-noise line shapes observed in the corrupted * `02_explore-core.txt` result artifact (run team_20260815144514) — extension/ * MCP session-log stderr that the child-executor result fallback chain leaked * into the result artifact when the worker payload was corrupted/empty. * * A trimmed non-empty line counts as log noise ONLY if it matches one of: * 1. a bracket-tag extension log line: `[oc-go] ...`, `[pi-qwen-mm] ...`, * `[pi-qwen-mm] [core] [stderr] ...`, and — Finding #2 (2026-09-29 * battery) — colon-bearing extension tags like * `[pi-crew:crash-recovery.reconcileStaleRuns] ...` or * `[pi-crew:crew-vibes.publish-quota-status] ...`. The leading tag * starts with a lowercase char (deliberately excludes capitalized * bracketed prose like `[Note] ...` / `[Note: ...]`); the rest of the tag * may carry `:`, `-`, `_`, `.`, digits, and inner capitals (extension * subsystem names are camelCase, e.g. `reconcileStaleRuns`), optionally * followed by known sub-tags (core/mcp/stderr/stdout/warn/info/error/debug); * 2. a Python `warnings.warn(` line (deprecation-warning continuation); * 3. a timestamped logging line: `2026-08-15 21:49:02,986 WARNING ...`. * * Anything else — prose, markdown, file:line results, `OK done.` — is NOT * noise, so a single such line marks the artifact as usable content. * * NOTE: a `true` return is NOT by itself failure evidence; the caller * (post-execution.ts) applies the two-gate rule (authoritative output * sources empty AND artifact log-noise-only) before failing a task. */ const BRACKET_TAG_LOG_LINE = /^\[[a-z0-9][a-zA-Z0-9_.:\-]*\](?:\s*\[(?:core|mcp|stderr|stdout|warn|info|error|debug)\])*(?:\s.*)?$/; const PYTHON_WARNING_LINE = /(?:^|\s)warnings\.warn\(/; const TIMESTAMPED_LOG_LINE = /^\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}(?:[.,]\d{1,6})?(?:Z|[+-]\d{2}:?\d{2})?\s+[A-Za-z]+\b/; /** * Detect a "stderr-only" result artifact: every non-empty trimmed line is * strict log noise (see the pattern docs above). Empty/whitespace-only input * returns false — emptiness is the caller's separate, explicit check. * Conservative by design: any prose/markdown line makes it return false. */ export function isStderrOnlyResult(text: string): boolean { if (!text?.trim()) return false; let noiseLines = 0; for (const line of text.split("\n")) { const trimmed = line.trim(); if (!trimmed) continue; const isNoise = BRACKET_TAG_LOG_LINE.test(trimmed) || PYTHON_WARNING_LINE.test(trimmed) || TIMESTAMPED_LOG_LINE.test(trimmed); if (!isNoise) return false; noiseLines++; } return noiseLines > 0; } /** * Extract structured findings from reviewer output. * Returns array of { file, line, severity, message } objects. */ export function parseReviewerFindings(output: string): Array<{ file: string; line: number; severity: string; message: string }> { const findings: Array<{ file: string; line: number; severity: string; message: string; }> = []; const lines = output.split("\n"); const SEVERITY_MAP: Record = { "🔴": "bug", "🟡": "risk", "🔵": "nit", "❓": "question", }; for (const line of lines) { // Match: path/to/file.ts:42: 🔴 bug: problem. fix. const match = line.match(/^([^:\s]+):(\d+):\s+(\p{Emoji_Presentation}) (\w+):\s+(.+)/u); if (match) { findings.push({ file: match[1], line: Number(match[2]), severity: SEVERITY_MAP[match[3]] ?? match[3], message: match[5].trim(), }); } } return findings; } /** * Extract explorer results from structured output. * Returns array of { file, line, symbol, note } objects. */ export function parseExplorerResults(output: string): Array<{ file: string; line: number; symbol: string; note: string }> { const results: Array<{ file: string; line: number; symbol: string; note: string; }> = []; const lines = output.split("\n"); for (const line of lines) { // Match: path/to/file.ts:42 — `symbol` — note const match = line.match(/^[- ]*(\S+):(\d+)\s*[—–-]\s*`([^`]+)`\s*[—–-]\s*(.+)/); if (match) { results.push({ file: match[1], line: Number(match[2]), symbol: match[3], note: match[4].trim(), }); } } return results; } /** * Validate that compressed prose preserves structural elements from original. * Returns list of specific issues (empty = valid). */ export function validateCompressionPreservation(original: string, compressed: string): string[] { const issues: string[] = []; // Check code blocks preserved const origBlocks = original.match(makeFencedCodeRe()) ?? []; const compBlocks = compressed.match(makeFencedCodeRe()) ?? []; if (origBlocks.length !== compBlocks.length) { issues.push(`Code block count: ${origBlocks.length} → ${compBlocks.length}`); } for (let i = 0; i < Math.min(origBlocks.length, compBlocks.length); i++) { if (origBlocks[i] !== compBlocks[i]) { issues.push(`Code block ${i + 1} content changed`); } } // Check URLs preserved const origUrls = new Set(original.match(makeUrlRe()) ?? []); const compUrls = new Set(compressed.match(makeUrlRe()) ?? []); for (const url of origUrls) { if (!compUrls.has(url)) { issues.push(`URL lost: ${url.slice(0, 60)}...`); } } // Check inline code preserved const origInline = original.match(makeInlineCodeRe()) ?? []; const compInline = compressed.match(makeInlineCodeRe()) ?? []; const origInlineSet = new Set(origInline); const compInlineSet = new Set(compInline); for (const code of origInlineSet) { if (!compInlineSet.has(code)) { issues.push(`Inline code lost: ${code}`); } } // Check headings preserved const origHeadings = original.match(makeHeadingRe()) ?? []; const compHeadings = compressed.match(makeHeadingRe()) ?? []; if (origHeadings.length !== compHeadings.length) { issues.push(`Heading count: ${origHeadings.length} → ${compHeadings.length}`); } return issues; }