import * as fs from "node:fs"; import * as path from "node:path"; import type { ExtensionAPI } from "@mariozechner/pi-coding-agent"; import { detectProject as detectProjectFromModule } from "../src/project-detector/detector"; import { loadMergedPolicy as loadMergedPolicyFromModule, type LayeredPolicy } from "../src/policy/policy-loader"; import { classifyCommand } from "../src/command-risk/classifier"; import { suggestInstalledResources } from "../src/extension-advisor/advisor"; import { validateArtifact } from "../src/artifacts/validator"; import { detectCorrectionCandidate, promoteRuleToProjectPolicy, type CandidateRule } from "../src/memory-rules/promotion"; import { createTaskState, transitionTaskState, approveWorkstream, detectDrift, type TaskStateSnapshot, type WorkstreamCategory } from "../src/task-state/state-machine"; // ============================================================================ // INLINED PERSISTENCE LAYER — no external imports (bundler compatibility) // ============================================================================ const CURRENT_SCHEMA = 1 as const; const STATE_FILE = "minimax-persist.json"; interface ValidationRecord { cmd: string; lastRun: string; lastResult: "pass" | "fail"; task?: string; } interface GrinderRecord { path: string; lastGrindSent: string; grindCount: number; lastGrindIteration: number; } interface PersistentAutomationState { schemaVersion: 1; validations: ValidationRecord[]; grinds: GrinderRecord[]; sessionChangedPaths: string[]; sessionGrindIteration: number; task?: TaskStateSnapshot; } function isRecord(v: unknown): v is Record { return typeof v === "object" && v !== null && !Array.isArray(v); } function freshState(): PersistentAutomationState { return { schemaVersion: CURRENT_SCHEMA, validations: [], grinds: [], sessionChangedPaths: [], sessionGrindIteration: 0, task: createTaskState() }; } function statePath(cwd: string): string { return path.join(cwd, ".pi", "agent", "minimax-state", STATE_FILE); } function stateRoot(cwd: string): string { return path.join(cwd, ".pi", "agent", "minimax-state"); } async function loadState(cwd: string): Promise { const p = statePath(cwd); let raw: string; try { raw = fs.readFileSync(p, "utf-8"); } catch (err: any) { if (err?.code === "ENOENT") return freshState(); throw err; } let parsed: unknown; try { parsed = JSON.parse(raw); } catch { return freshState(); } if (!isRecord(parsed) || parsed["schemaVersion"] !== CURRENT_SCHEMA || !isRecord(parsed["state"])) return freshState(); return parsed["state"] as PersistentAutomationState; } async function saveState(cwd: string, state: PersistentAutomationState): Promise { const root = stateRoot(cwd); const p = statePath(cwd); const snap = JSON.stringify({ schemaVersion: CURRENT_SCHEMA, state }, null, 2) + "\n"; const tmp = path.join(root, `.${STATE_FILE}.${process.pid}.${Date.now()}.tmp`); fs.mkdirSync(root, { recursive: true }); fs.writeFileSync(tmp, snap, "utf-8"); fs.renameSync(tmp, p); } function getLastValidation(state: PersistentAutomationState, cmd: string): ValidationRecord | undefined { return state.validations.find((v) => v.cmd === cmd); } function upsertValidation(state: PersistentAutomationState, record: ValidationRecord): void { const idx = state.validations.findIndex((v) => v.cmd === record.cmd); if (idx >= 0) state.validations[idx] = record; else state.validations.push(record); } function getGrindRecord(state: PersistentAutomationState, filePath: string): GrinderRecord | undefined { return state.grinds.find((g) => g.path === filePath); } function upsertGrindRecord(state: PersistentAutomationState, rec: GrinderRecord): void { const idx = state.grinds.findIndex((g) => g.path === rec.path); if (idx >= 0) state.grinds[idx] = rec; else state.grinds.push(rec); } function isGrindDebounced(state: PersistentAutomationState, filePath: string, debounceWindowMs = 5 * 60 * 1000): boolean { const rec = getGrindRecord(state, filePath); if (!rec) return false; return Date.now() - new Date(rec.lastGrindSent).getTime() < debounceWindowMs; } function trackChangedPath(state: PersistentAutomationState, p: string): void { if (!state.sessionChangedPaths.includes(p)) state.sessionChangedPaths.push(p); } function nextGrindIteration(state: PersistentAutomationState): number { return ++state.sessionGrindIteration; } function grindIterationExceeded(state: PersistentAutomationState, max: number): boolean { return state.sessionGrindIteration >= max; } function resetSessionState(state: PersistentAutomationState): void { state.sessionChangedPaths = []; state.sessionGrindIteration = 0; } // ============================================================================ // INLINED STATE CACHE — singleton per cwd (no external imports) // ============================================================================ let _cachedState: PersistentAutomationState | null = null; let _cachedCwd = ""; async function getState(cwd: string): Promise { if (cwd !== _cachedCwd || !_cachedState) { _cachedCwd = cwd; _cachedState = await loadState(cwd); } return _cachedState; } async function persistState(cwd: string, state: PersistentAutomationState): Promise { await saveState(cwd, state); } // ============================================================================ // PATH HELPERS // ============================================================================ function getPackageRoot(): string { try { const { fileURLToPath } = require("node:url"); const filePath = fileURLToPath(import.meta.url); return path.resolve(path.dirname(filePath), ".."); } catch { // Fallback: manual parse for environments where fileURLToPath fails const u = new URL(import.meta.url); const pathname = decodeURIComponent(u.pathname); const filePath = pathname.replace(/^\/([A-Za-z]:)/, "$1"); return path.resolve(path.dirname(filePath), ".."); } } // Embedded fallback contract. Kept in sync with docs/AGENTS.md so the extension // always has a working contract even when the package layout makes filesystem // resolution unreliable (e.g. bundled installs, alternate pi install paths). const EMBEDDED_CONTRACT = `# Global Agent Contract ## Default Posture - Act before explaining when tools can ground the answer. - Read before editing and verify after meaningful changes. - Match effort to task complexity and risk. - Prefer the smallest safe change that solves the real problem. - Reuse existing patterns before inventing new abstractions. - Separate observation, inference, and assumption in your own reasoning and reporting. ## Solver Loop For non-trivial work: 1. Define the outcome in operational terms. 2. Inspect the repo and current environment before choosing an approach. 3. Find the spine: entry points, data flow, state boundaries, persistence, and user-visible behavior. 4. Build the smallest vertical slice that proves the solution works. 5. Verify at the surface where the user experiences the change. 6. Expand scope only after the core slice is working. ## Scope Control - Do exactly the slice the user asked for. - Do not turn planning into implementation or explanation into edits. - Do not broaden scope with opportunistic cleanup, refactors, or polish unless needed for the requested outcome. - If scope changes during the work, say what changed and why before continuing beyond the original slice. - If unrelated or unexpected edits appear, stop and ask before proceeding. ## Stuck Loop And Retry Policy - After two failed verification attempts on the same hypothesis, stop repeating the same fix. - Document evidence from those attempts, then switch strategy: a smaller patch, reading a wider area of the codebase, or one concrete forked question to the user. - Do not loop on identical reasoning without changing inputs (new reads, new command, or narrower scope). ## Mid Task Checkpointing - On long or multi-step work, checkpoint before expanding scope: restate the goal, list files touched, checks already run, and what remains. - Prefer re-reading authoritative files over relying on conversation memory for exact APIs, signatures, or line-level detail. ## Tool And Scaffold Discipline - Do not invent tool names, wrappers, or APIs that are not present in the current environment. - Do not promise browser, canvas, subagent, MCP, or other tool-based output until the tool path is confirmed in the current runtime. - Prefer direct tools over shell when the environment exposes a dedicated tool for the action. - Parallelize independent reads, greps, and searches; serialize when the next step depends on the result of a read or edit. - Verify new packages, frameworks, and toolchains against current sources before recommending them. - Use official CLI or \`create\` or \`init\` scaffolding paths when they exist. - Do not hand-write manifests, boilerplate, or generated project structure when an official scaffold exists. - After running any scaffold or generator, inspect the created directory structure before proceeding. ## Security And Destructive Preflight - Before destructive or high-impact actions (\`rm -rf\`, dropping databases, production deploys, irreversible data migration, or changing secrets and credentials): obtain explicit user confirmation when the environment allows; do not proceed on assumption. - Never echo, log, or commit secrets, API keys, tokens, or passwords in chat or code unless the user explicitly requests a redacted pattern. ## Freshness And Honesty - When facts may be stale or fast-moving, check current docs or web sources before speaking with confidence. - If you did not verify a claim, say that directly instead of implying certainty. - Do not use fake \`\` blocks, inflated self-descriptions, or confident filler in place of grounded evidence. - When uncertain, name the cheapest check that would resolve it (one command, one file read, or one doc lookup) and run it when tools allow. ## Status And Verification Contract (ENFORCED) ### Allowed status labels Only use these labels in updates and closeouts: - \`changed\` - \`verified\` - \`unverified\` - \`blocked\` - \`assumption\` ### Hard rules - **Never** say \`done\`, \`fixed\`, \`working\`, or \`resolved\` unless you immediately list the proof (tool evidence). - If you made any claim that depends on the filesystem, network, tests, build, or runtime behavior, you **must** ground it with a tool result. - For any substantive task, you **must** end your response with a **Status Report** section using the labels above. - When verification is missing, you must say **\`unverified\`** (do not imply confidence). ### Minimum proof by change type - Localized edit: re-read or one targeted static check - Backend/logic/API change: targeted test/command/script/runtime request - UI/interaction change: user-surface verification + static checks - Integration-sensitive change: build/typecheck + one focused behavior check - New app/scaffold: install succeeds + start/health succeeds + production build succeeds + one happy-path works ### Closeout template (use for substantive work) - **Summary** - **Files touched** - **Verification evidence** (commands, manual checks) - **Risks and unverified items** ## Communication - Lead with actions, findings, and results. - Keep progress updates short and high signal. - Prefer milestone updates over step-by-step narration. - Report new information, blockers, scope changes, and verification results. - When blocked, state the blocker, evidence, and smallest next step; if two attempts on the same hypothesis failed, switch strategy per the stuck-loop policy instead of retrying blindly. ## Durable Design Preferences - Avoid generic "AI slop" UI patterns; commit to a clear aesthetic direction before building. - Keep UI constraints framework-agnostic and responsive across desktop and mobile. - Use real SVG icons such as Lucide, Heroicons, or Phosphor instead of emoji. - Use real imagery, product screenshots, or purposeful decorative graphics instead of blank placeholders. - Keep section containers and horizontal padding aligned consistently across a page. - Center hero sections optically and structurally; do not bias them with asymmetric padding. - Do not default to overused fonts such as \`Inter\`, \`Roboto\`, \`Arial\`, or \`Space Grotesk\` unless explicitly requested. - Treat motion as a real design tool: purposeful entrances, scroll reveals, and hover feedback when appropriate.`; interface ContractLoadResult { text: string; source: "file" | "embedded"; warning?: string; } function readContractText(): ContractLoadResult { let attemptedPath = ""; try { const root = getPackageRoot(); attemptedPath = path.join(root, "docs", "AGENTS.md"); const fileText = fs.readFileSync(attemptedPath, "utf-8").trim(); if (fileText.length > 0) { return { text: fileText, source: "file" }; } return { text: EMBEDDED_CONTRACT, source: "embedded", warning: `AGENTS.md at ${attemptedPath} is empty; using embedded contract.`, }; } catch (err: any) { return { text: EMBEDDED_CONTRACT, source: "embedded", warning: attemptedPath ? `Could not read ${attemptedPath} (${err?.code || err?.message || "error"}); using embedded contract.` : `Could not resolve package root for AGENTS.md (${err?.message || "error"}); using embedded contract.`, }; } } type ProjectType = | "generic" | "node" | "frontend" | "go" | "wails" | "rust" | "tauri" | "python" | "php" | "laravel" | "postgres" | "docker"; interface DetectedProject { root: string; types: ProjectType[]; markers: string[]; confidence: number; } type PolicyDecision = "allow" | "warn" | "preflight" | "block" | "ask" | "require_approval"; type PolicyScope = "global" | "workspace" | "project" | "task"; interface PolicyRule { id: string; description?: string; enabled?: boolean; when?: { projectType?: string; tool?: string; commandRegex?: string; }; action: PolicyDecision; message?: string; } interface LayeredPolicy { version: number; name: string; scope: PolicyScope; project?: { name?: string; root?: string; types?: string[]; }; workflow?: Record; commandPolicy?: Record; rules: PolicyRule[]; artifactPolicy?: Record; todoPolicy?: Record; memoryPolicy?: Record; extensionPolicy?: Record; } const DEFAULT_POLICY: LayeredPolicy = { version: 1, name: "default policy", scope: "task", project: { name: "auto", root: "auto", types: ["auto"] }, workflow: { requireDiagnosisBeforeEdit: false, requirePlanBeforeRiskyCommand: false, oneNextStepAfterFailure: true, stopOnTaskDrift: false, stopOnPopupError: false, stopOnEmptyOutput: false, stopOnArtifactMismatch: false, }, commandPolicy: { safeRead: "allow", build: "preflight", install: "require_reason", destructive: "require_approval", deploy: "require_explicit_user_request", database: "require_backup_or_dry_run", network: "require_reason", unknown: "ask", }, rules: [], artifactPolicy: { verifyOutputExists: false, verifyOutputTimestamp: false, verifyHashWhenCopyingArtifacts: false, verifyNonEmptyOutputDir: false, }, todoPolicy: { maxActiveTodos: 3, blockUnrelatedTodos: false, requireUserApprovalForNewWorkstream: false, }, memoryPolicy: { detectCorrections: false, proposeRulePromotion: false, autoPromote: false, storeCandidateRules: false, }, extensionPolicy: { preferInstalledExtensions: true, requireAuditBeforeNewExtension: false, }, }; function detectProject(cwd: string): DetectedProject { const detected = detectProjectFromModule(cwd); return { root: detected.root, types: detected.types as ProjectType[], markers: detected.evidenceFiles, confidence: detected.confidence, }; } function loadMergedPolicy(cwd: string, project: DetectedProject, taskPolicy?: Partial): LayeredPolicy { return loadMergedPolicyFromModule(cwd, { root: project.root, types: project.types as any, confidence: project.confidence, evidenceFiles: project.markers, recommendedProfiles: project.types.filter((t) => t !== "generic") as string[], }, taskPolicy || null).merged; } // ============================================================================ // TOOL OP TRACKING (in-memory per-request) // ============================================================================ type ToolOp = { toolName: string; path?: string; editsPaths?: string[]; index: number; command?: string; isError?: boolean; evidence?: string; hint?: string }; type TurnToolState = { ops: ToolOp[] }; function asPath(v: any): string | undefined { return typeof v === "string" ? v : undefined; } function unique(items: T[]): T[] { return [...new Set(items)]; } function isValidationLikeCommand(cmd: string): boolean { return /\b(tsc|eslint|vitest|jest|pytest|cargo test|go test|npm test|npm run (test|build|lint|typecheck)|pnpm (test|build|lint)|yarn (test|build|lint)|bun test)\b/i.test(cmd); } function extractToolResultText(result: any): string { if (!result) return ""; if (typeof result === "string") return result; const parts: string[] = []; if (typeof result.stdout === "string") parts.push(result.stdout); if (typeof result.stderr === "string") parts.push(result.stderr); if (typeof result.output === "string") parts.push(result.output); if (Array.isArray(result.content)) { for (const item of result.content) if (item && typeof item.text === "string") parts.push(item.text); } return parts.join("\n").trim(); } function firstFailureLine(text: string): string { const lines = text.split(/\r?\n/).map((l) => l.trim()).filter(Boolean); return lines.find((l) => /\berror\b|failed|exception|traceback|cannot find|not assignable|does not exist/i.test(l)) || lines[0] || "verification command failed"; } function hintForFailure(evidence: string): string { if (/TS2322|not assignable to type/i.test(evidence)) return "Type mismatch: align the declared type with the assigned value, or convert the value before assignment."; if (/TS2304|Cannot find name/i.test(evidence)) return "Missing symbol: define it, import it, or correct the identifier spelling/scope."; if (/TS2339|Property .* does not exist/i.test(evidence)) return "Invalid property access: update the object type or use an existing property."; if (/Cannot find module|module not found/i.test(evidence)) return "Module resolution issue: check install status, import path, and tsconfig/module settings."; if (/command not found|not recognized as an internal or external command/i.test(evidence)) return "Missing command: verify the tool is installed and available on PATH, or use the nearest documented repo command."; if (/ENOENT|no such file or directory/i.test(evidence)) return "Missing path: verify the file or directory exists before rerunning the command."; if (/EACCES|permission denied|access is denied/i.test(evidence)) return "Permission issue: check file ownership, locks, and whether the command needs a different writable location."; if (/EADDRINUSE|address already in use/i.test(evidence)) return "Port conflict: stop the existing process or choose a free port before rerunning."; if (/AssertionError|expected .* to/i.test(evidence)) return "Test assertion failed: inspect the expected versus actual values, then fix the smallest behavior mismatch."; if (/detached HEAD|working tree|uncommitted changes/i.test(evidence)) return "Git state issue: inspect status, preserve local changes, then retry the git operation from a clean state."; if (/syntax|unexpected token/i.test(evidence)) return "Syntax issue: inspect the reported line/column and fix the malformed expression or delimiter."; return "Read the first failure line, make one minimal fix, then rerun the same verification command."; } function appendTextToMessage(message: any, text: string): any { if (typeof message.content === "string") return { ...message, content: message.content + text }; if (Array.isArray(message.content)) return { ...message, content: [...message.content, { type: "text", text }] }; return { ...message, content: String(message.content || "") + text }; } // ============================================================================ // COMMAND POLICY DECISION // ============================================================================ type CommandCategoryName = "SAFE_READ" | "BUILD" | "TEST" | "LINT" | "INSTALL" | "DESTRUCTIVE" | "DEPLOY" | "DATABASE" | "NETWORK" | "UNKNOWN"; function policyKeyForCategory(category: CommandCategoryName): string { switch (category) { case "SAFE_READ": return "safeRead"; case "BUILD": case "TEST": case "LINT": return "build"; case "INSTALL": return "install"; case "DESTRUCTIVE": return "destructive"; case "DEPLOY": return "deploy"; case "DATABASE": return "database"; case "NETWORK": return "network"; case "UNKNOWN": return "unknown"; default: return "unknown"; } } function defaultDecisionForCategory(category: CommandCategoryName): string { switch (category) { case "SAFE_READ": return "allow"; case "BUILD": case "TEST": case "LINT": return "preflight"; case "INSTALL": return "require_reason"; case "DESTRUCTIVE": return "require_approval"; case "DEPLOY": return "require_explicit_user_request"; case "DATABASE": return "require_backup_or_dry_run"; case "NETWORK": return "require_reason"; case "UNKNOWN": return "ask"; default: return "ask"; } } interface CommandPolicyOutcome { block: true; reason: string; } function applyCommandPolicyDecision( decision: string, category: CommandCategoryName, classificationReason: string, ctx: any, ): CommandPolicyOutcome | undefined { const lower = category.toLowerCase().replace("_", " "); switch (decision) { case "allow": return undefined; case "warn": ctx.ui?.notify?.(`Warning (${lower}): ${classificationReason}`, "warning"); return undefined; case "preflight": ctx.ui?.notify?.(`Preflight note (${lower}): ${classificationReason}`, "info"); return undefined; case "require_backup_or_dry_run": ctx.ui?.notify?.(`${category} note: confirm backup or dry-run plan before executing migration/mutation commands.`, "warning"); return undefined; case "ask": return { block: true, reason: `Blocked: ${classificationReason} Ask user or explain intent before rerunning.` }; case "block": return { block: true, reason: `Blocked: ${classificationReason}` }; case "require_reason": return { block: true, reason: `Blocked: ${lower} command requires reason. State why this command is needed, then rerun.` }; case "require_approval": return { block: true, reason: `Blocked: ${lower} command requires explicit user approval.` }; case "require_explicit_user_request": return { block: true, reason: `Blocked: ${lower} command requires explicit user request.` }; default: return { block: true, reason: `Blocked: unrecognized policy decision '${decision}' for ${lower}. Treat as ask.` }; } } // ============================================================================ // WORKSTREAM CLASSIFICATION (task-state integration) // ============================================================================ function classifyWorkstream(promptText: string): WorkstreamCategory { const p = (promptText || "").toLowerCase(); if (/\b(deploy|publish|release|ship|push to (prod|production))\b/.test(p)) return "deploy"; if (/\b(install|setup|scaffold|bootstrap|configure)\b/.test(p)) return "installer"; if (/\b(audit|review|inspect|analyze|check policy)\b/.test(p)) return "audit"; if (/\b(ci|cicd|pipeline|workflow|github actions|gitlab ci)\b/.test(p)) return "ci"; if (/\b(bug|error|fix|debug|crash|fail|broken|tidak jalan)\b/.test(p)) return "debug-build"; if (/\b(local fix|patch|hotfix)\b/.test(p)) return "local-fix"; if (/\b(refactor|implement|build|develop|create|add feature|tambah)\b/.test(p)) return "coding"; return "unknown"; } // ============================================================================ // ADVISOR INTEGRATION // ============================================================================ function buildAdvisorDirective(promptText: string, policy: LayeredPolicy): string { if (!promptText) return ""; const prefer = (policy.extensionPolicy as any)?.preferInstalledExtensions; if (prefer === false) return ""; const advice = suggestInstalledResources(promptText); if (!advice.suggestions.length) return ""; return `\n\n## Installed Resource Hint (always-on)\n\n${advice.reminder}\n${advice.suggestions.map((s) => "- " + s).join("\n")}\n`; } // ============================================================================ // ARTIFACT VALIDATION INTEGRATION // ============================================================================ const COMMON_OUTPUT_DIRS = ["dist", "build", "out", ".next", "target/release", "target/debug", "public/build"]; function findExistingOutputDirs(cwd: string): string[] { const existing: string[] = []; for (const rel of COMMON_OUTPUT_DIRS) { const full = path.join(cwd, rel); try { if (fs.existsSync(full) && fs.statSync(full).isDirectory()) existing.push(full); } catch {} } return existing; } function runArtifactValidation(cwd: string, policy: LayeredPolicy, commandStartedAt: number): { ok: boolean; reasons: string[]; validated: string[] } { const ap = (policy.artifactPolicy || {}) as Record; const verifyExists = ap.verifyOutputExists === true; const verifyNonEmpty = ap.verifyNonEmptyOutputDir === true; const verifyTimestamp = ap.verifyOutputTimestamp === true; if (!verifyExists && !verifyNonEmpty && !verifyTimestamp) { return { ok: true, reasons: [], validated: [] }; } const dirs = findExistingOutputDirs(cwd); if (dirs.length === 0) { return verifyExists ? { ok: false, reasons: ["No standard output directory (dist/build/out) found after build."], validated: [] } : { ok: true, reasons: [], validated: [] }; } const reasons: string[] = []; for (const dir of dirs) { const result = validateArtifact({ outputDir: verifyNonEmpty ? dir : undefined, commandStartedAt: verifyTimestamp ? commandStartedAt : undefined, }); if (!result.ok) reasons.push(`${dir}: ${result.reasons.join("; ")}`); } return { ok: reasons.length === 0, reasons, validated: dirs }; } // ============================================================================ // VALIDATION COMMAND DETECTION // ============================================================================ function detectValidationCommands(cwd: string, changedFiles: string[] = []): string[] { try { const pkgPath = path.join(cwd, "package.json"); if (fs.existsSync(pkgPath)) { const pkg = JSON.parse(fs.readFileSync(pkgPath, "utf-8")); const scripts = (pkg && pkg.scripts) || {}; const cmds: string[] = []; if (scripts.test) cmds.push("npm test"); if (scripts.lint) cmds.push("npm run lint"); if (scripts.build) cmds.push("npm run build"); return cmds; } } catch {} try { const pyproject = path.join(cwd, "pyproject.toml"); if (fs.existsSync(pyproject)) return ["python -m pytest -q"]; } catch {} try { const makefile = path.join(cwd, "Makefile"); if (fs.existsSync(makefile)) return ["make test"]; } catch {} try { const cargoToml = path.join(cwd, "Cargo.toml"); if (fs.existsSync(cargoToml)) return ["cargo test"]; } catch {} try { const goMod = path.join(cwd, "go.mod"); if (fs.existsSync(goMod)) return ["go test ./..."]; } catch {} return []; } // ============================================================================ // STATUS REPORT BUILDER // ============================================================================ function buildAutoStatusReport(state: TurnToolState): string | null { const ops = state.ops; if (ops.length === 0) return null; const writes = ops.filter((o) => o.toolName === "write" && o.path).map((o) => ({ path: o.path!, index: o.index })); const reads = ops.filter((o) => o.toolName === "read" && o.path).map((o) => ({ path: o.path!, index: o.index })); const edits = ops.filter((o) => o.toolName === "edit" && (o.path || o.editsPaths?.length)).flatMap((o) => { const paths = o.path ? [o.path] : (o.editsPaths || []); return paths.map((p) => ({ path: p, index: o.index })); }); const changedPaths = unique([...writes.map((w) => w.path), ...edits.map((e) => e.path)]); const lines: string[] = ["\n\nStatus Report (auto)"]; for (const p of changedPaths) lines.push(`- changed: ${p}`); for (const w of writes) { const hasReadBack = reads.some((r) => r.path === w.path && r.index > w.index); lines.push(hasReadBack ? `- verified: wrote then read-back ${w.path}` : `- unverified: wrote ${w.path} but no read-back proof`); } for (const e of edits) { const hasReadBefore = reads.some((r) => r.path === e.path && r.index < e.index); const hasReadAfter = reads.some((r) => r.path === e.path && r.index > e.index); lines.push( hasReadBefore && hasReadAfter ? `- verified: edited with read-before & read-after ${e.path}` : `- unverified: edited ${e.path} without full read-before/read-after proof`, ); } for (const fail of ops.filter((o) => o.toolName === "bash" && o.isError && o.command && o.evidence)) { lines.push(`- blocked: verification failed: ${fail.command}`); lines.push(`- evidence: ${fail.evidence}`); lines.push(`- hint: ${fail.hint || "Read the failure output, apply one minimal fix, then rerun the command."}`); } return lines.join("\n"); } // ============================================================================ // EXTENSION ENTRY POINT // ============================================================================ export default function globalContract(pi: ExtensionAPI) { pi.registerFlag("minimax-grind", { description: "Enable/disable MiniMax auto-grind (auto run validations after changes)", type: "boolean", default: true, }); pi.registerFlag("minimax-grind-max", { description: "Max auto-grind follow-up iterations per user request", type: "string", default: "2", }); const { text: contract, warning: contractWarning, source: contractSource } = readContractText(); let notified = false; let grindFiredThisSession = false; let correctionLoopCount = 0; const MAX_CORRECTION_LOOPS = 3; // Per-request (in-memory) tool tracking let opIndex = 0; let turnState: TurnToolState = { ops: [] }; const lastReadByPath = new Map(); const pendingReadBacks: string[] = []; // files written this turn that need read-back // Persistent state reference let persistStateRef: PersistentAutomationState | null = null; let persistCwd = ""; async function ensureState(cwd: string): Promise { if (cwd !== persistCwd || !persistStateRef) { persistCwd = cwd; persistStateRef = await getState(cwd); } return persistStateRef!; } async function flushState(): Promise { if (persistStateRef) await persistState(persistCwd, persistStateRef); } // === input — reset per-user-request state === pi.on("input", async (event, ctx) => { // Skip reset for messages that this extension itself injected (grind, force-verify). // Pi marks extension-originated messages with event.source === "extension". if (event && (event as any).source === "extension") { return; } grindFiredThisSession = false; correctionLoopCount = 0; opIndex = 0; turnState = { ops: [] }; lastReadByPath.clear(); pendingReadBacks.length = 0; const state = await ensureState(ctx.cwd); resetSessionState(state); // Memory-rules + task-state integration. Both gated by policy. const promptText = String((event as any)?.text || ""); if (promptText) { const project = detectProject(ctx.cwd); const policy = loadMergedPolicy(ctx.cwd, project); // Workstream tracking (task-state) const workstream = classifyWorkstream(promptText); if (!state.task) state.task = createTaskState(workstream); if (workstream !== "unknown") { const driftCheck = detectDrift(state.task, workstream); if (driftCheck.drift && (policy.workflow as any)?.stopOnTaskDrift && ctx.hasUI) { const ok = await ctx.ui.confirm("Task drift detected", `${driftCheck.reason} Approve switch to '${workstream}'?`); if (ok) { state.task = approveWorkstream(state.task, workstream); state.task = transitionTaskState({ ...state.task, currentWorkstream: workstream }, "DIAGNOSE", `Approved drift to ${workstream}`); } } else if (!driftCheck.drift) { state.task.currentWorkstream = workstream; } } // Correction detection (memory-rules) if ((policy.memoryPolicy as any)?.detectCorrections) { const candidate = detectCorrectionCandidate(promptText); if (candidate) { if ((policy.memoryPolicy as any)?.autoPromote) { try { const written = promoteRuleToProjectPolicy(ctx.cwd, candidate); ctx.ui?.notify?.(`Auto-promoted rule '${candidate.id}' to ${written}`, "info"); } catch (err: any) { ctx.ui?.notify?.(`Rule promotion failed: ${err?.message || err}`, "warning"); } } else if ((policy.memoryPolicy as any)?.proposeRulePromotion && ctx.hasUI) { const ok = await ctx.ui.confirm("Promote correction to policy rule?", candidate.summary); if (ok) { try { const written = promoteRuleToProjectPolicy(ctx.cwd, candidate); ctx.ui?.notify?.(`Rule '${candidate.id}' added to ${written}`, "info"); } catch (err: any) { ctx.ui?.notify?.(`Rule promotion failed: ${err?.message || err}`, "warning"); } } } } } } await flushState(); }); // === tool_execution_start — cache args for this call id === const toolArgsById = new Map(); pi.on("tool_execution_start", async (event) => { toolArgsById.set(event.toolCallId, { args: event.args || {}, startedAt: Date.now() }); }); // === tool_execution_end — track ops + update persistent state === pi.on("tool_execution_end", async (event, ctx) => { const toolName = event.toolName; const cached = toolArgsById.get(event.toolCallId) || { args: {}, startedAt: Date.now() }; const args = cached.args; const startedAt = cached.startedAt; toolArgsById.delete(event.toolCallId); // Read if (toolName === "read") { const p = asPath(args.path); turnState.ops.push({ toolName, path: p, index: opIndex++ }); if (p) lastReadByPath.set(p, opIndex); return; } // Write if (toolName === "write") { const p = asPath(args.path); turnState.ops.push({ toolName, path: p, index: opIndex++ }); if (p) { if (!pendingReadBacks.includes(p)) pendingReadBacks.push(p); const state = await ensureState(ctx.cwd); trackChangedPath(state, p); const existing = getGrindRecord(state, p); if (existing) { existing.grindCount = 0; existing.lastGrindIteration = 0; upsertGrindRecord(state, existing); } await flushState(); } return; } // Edit if (toolName === "edit") { const p = asPath(args.path); turnState.ops.push({ toolName, path: p, index: opIndex++ }); if (p) { const state = await ensureState(ctx.cwd); trackChangedPath(state, p); const existing = getGrindRecord(state, p); if (existing) { existing.grindCount = 0; existing.lastGrindIteration = 0; upsertGrindRecord(state, existing); } await flushState(); } return; } // Bash: detect validation command results if (toolName === "bash") { const cmd = String(args.command || "").trim(); if (!cmd) return; if (event.isError && isValidationLikeCommand(cmd)) { const output = extractToolResultText(event.result); const evidence = firstFailureLine(output); turnState.ops.push({ toolName, command: cmd, isError: true, evidence, hint: hintForFailure(evidence), index: opIndex++ }); } // Artifact validation: after a successful BUILD command, optionally check output dirs if (!event.isError) { const cls = classifyCommand(cmd); if (cls.category === "BUILD") { const project = detectProject(ctx.cwd); const policy = loadMergedPolicy(ctx.cwd, project); const artifact = runArtifactValidation(ctx.cwd, policy, startedAt); if (!artifact.ok) { const evidence = artifact.reasons.join("; "); turnState.ops.push({ toolName: "bash", command: cmd, isError: true, evidence, hint: "Build reported success but output artifacts failed validation. Inspect the output directory and policy expectations.", index: opIndex++, }); } } } const state = await ensureState(ctx.cwd); const matchedCmd = state.validations.find((v) => v.cmd === cmd); if (matchedCmd) { matchedCmd.lastRun = new Date().toISOString(); matchedCmd.lastResult = event.isError ? "fail" : "pass"; await flushState(); } const validationCmds = detectValidationCommands(ctx.cwd, state.sessionChangedPaths); if (!matchedCmd && validationCmds.some((v) => cmd === v || cmd.startsWith(v + " ")) && cmd !== "read" && cmd !== "edit" && cmd !== "write") { const rec: ValidationRecord = { cmd, lastRun: new Date().toISOString(), lastResult: event.isError ? "fail" : "pass" }; upsertValidation(state, rec); await flushState(); } } }); // === tool_call — hard safety gates + policy gate === pi.on("tool_call", async (event, ctx) => { if (event.toolName === "bash") { const cmd = String((event as any).input?.command || "").trim(); if (/(^|\s)(>|>>|2>|&>)(\s|$)/.test(cmd)) { return { block: true, reason: "Blocked: use write/edit tool for file modifications (no shell redirection)." }; } const project = detectProject(ctx.cwd); const policy = loadMergedPolicy(ctx.cwd, project); const classification = classifyCommand(cmd); if (project.types.includes("wails") && /^go\s+build\b/i.test(cmd)) { return { block: true, reason: "Blocked: Wails project detected. Use 'wails build -clean' instead of plain 'go build'." }; } const commandPolicy = (policy.commandPolicy || {}) as Record; const policyKey = policyKeyForCategory(classification.category as CommandCategoryName); const decision = commandPolicy[policyKey] || defaultDecisionForCategory(classification.category as CommandCategoryName); const outcome = applyCommandPolicyDecision(decision, classification.category as CommandCategoryName, classification.reason, ctx); if (outcome) return outcome; } if (event.toolName === "edit") { const p = String((event as any).input?.path || ""); if (!p) return; if (!lastReadByPath.has(p)) { return { block: true, reason: `Blocked: must read ${p} before editing.` }; } } if (event.toolName === "write") { const p = String((event as any).input?.path || ""); if (!p) return; const abs = path.isAbsolute(p) ? p : path.join(ctx.cwd, p); if (fs.existsSync(abs) && !lastReadByPath.has(p)) { return { block: true, reason: `Blocked: must read ${p} before overwriting an existing file.` }; } } }); // === context — prune old tool results to reduce noise for weak models === pi.on("context", async (event, _ctx) => { const messages = event.messages; if (!messages || messages.length < 8) return; // nothing to prune for short conversations // Keep last 6 messages full. For older messages, collapse tool results to summaries. const keepFullCount = 6; const cutoff = messages.length - keepFullCount; const pruned = messages.map((msg: any, i: number) => { if (i >= cutoff) return msg; // keep recent messages intact if (msg.role !== "toolResult") return msg; // Collapse tool result content to a 1-line summary const toolName = msg.toolName || "tool"; const isError = msg.isError || false; const contentText = Array.isArray(msg.content) ? msg.content.map((c: any) => c.text || "").join(" ").slice(0, 120) : String(msg.content || "").slice(0, 120); const summary = `[${toolName}${isError ? " ERROR" : ""}: ${contentText}${contentText.length >= 120 ? "..." : ""}]`; return { ...msg, content: [{ type: "text", text: summary }], }; }); return { messages: pruned }; }); // === turn_end — forced read-back after writes + auto-correction on failure === pi.on("turn_end", async (event, ctx) => { // 1. Forced read-back: if files were written this turn but not read back, steer agent to read them const unreadWrites = pendingReadBacks.filter((p) => { const readIdx = lastReadByPath.get(p); const writeOp = turnState.ops.find((o) => o.toolName === "write" && o.path === p); return !readIdx || !writeOp || readIdx < writeOp.index; }); if (unreadWrites.length > 0) { const readBackMsg = [ { type: "text", text: `Read back these files to confirm writes are correct:\n${unreadWrites.map((p) => "- read " + p).join("\n")}` }, ]; try { await pi.sendUserMessage(readBackMsg, { deliverAs: "steer" }); } catch {} pendingReadBacks.length = 0; return; } pendingReadBacks.length = 0; // 2. Auto-correction loop: if this turn had a failed validation, send evidence+hint back const failedOps = turnState.ops.filter((o) => o.isError && o.evidence && o.hint); if (failedOps.length > 0 && correctionLoopCount < MAX_CORRECTION_LOOPS) { correctionLoopCount++; const lastFail = failedOps[failedOps.length - 1]; const correctionMsg = [ { type: "text", text: `Verification failed (attempt ${correctionLoopCount}/${MAX_CORRECTION_LOOPS}).` }, { type: "text", text: `Command: ${lastFail.command}` }, { type: "text", text: `Evidence: ${lastFail.evidence}` }, { type: "text", text: `Hint: ${lastFail.hint}` }, { type: "text", text: "Apply ONE minimal fix based on the evidence above, then rerun the same command." }, ]; try { await pi.sendUserMessage(correctionMsg, { deliverAs: "steer" }); } catch {} } }); // === session_start — notify user once === pi.on("session_start", async (_event, ctx) => { if (!ctx.hasUI || notified) return; notified = true; const label = contractSource === "embedded" ? "Global contract pack loaded (embedded)" : "Global contract pack loaded"; ctx.ui.notify(label, "info"); if (contractWarning) { ctx.ui.notify(contractWarning, "warning"); } }); // === before_agent_start — inject contract + auto skill routing === pi.on("before_agent_start", async (event, ctx) => { if (!contract) return; const state = await ensureState(ctx.cwd); const skills = event.systemPromptOptions.skills || []; const findSkill = (name: string) => skills.find((s: any) => s.name === name); const promptText = event.prompt || ""; const wantsMedia = /\b(image|gambar|video|audio|tts|voice|musik|music|transcribe|transkripsi)\b/i.test(promptText); const wantsIncident = /\b(error|exception|traceback|stack trace|failed|gagal|crash|timeout|500|bug)\b/i.test(promptText); const wantsIterate = /\b(refactor|optimi[sz]e|improve|iterate|polish|cleanup|perf|performance)\b/i.test(promptText); const wantsDesign = /\b(ui|ux|design|landing page|hero section|layout|typography|warna|color palette)\b/i.test(promptText); const wantsResearch = /\b(research|riset|investigate|deep dive|compare|survey|analisis|analysis|benchmark)\b/i.test(promptText); const wantsWeb = /\b(web|website|url|link|latest|current|news|docs?|documentation|search web|web search|cari info|cari di web|cari di x|search x|x\.com|twitter|x\/twitter)\b/i.test(promptText); const wantsBrowser = /\b(browser|chrome|web page|lihat browser|klik|click|screenshot|inspect page|debug frontend|isi form|fill form)\b/i.test(promptText); const requestedSkills: Array<{ name: string; path: string }> = []; if (wantsIncident) { const s = findSkill("minimax-incident-triage-harness"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } if (wantsMedia) { const s = findSkill("minimax-multimodal-toolkit"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } if (wantsWeb) { const s = findSkill("minimax-web-ops"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } if (wantsResearch) { const s = findSkill("minimax-deep-research"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } if (wantsIterate) { const s = findSkill("minimax-m2-self-evolution"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } if (wantsDesign) { const s = findSkill("minimax-anti-slop-design"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } if (wantsBrowser) { const s = findSkill("minimax-browser-ops"); if (s) requestedSkills.push({ name: s.name, path: s.filePath }); } const skillDirective = requestedSkills.length > 0 ? `\n\n## Auto Skill Routing (always-on)\n\nYou MUST start by reading and following these skills (in order):\n${requestedSkills.map((s) => `- ${s.name}: ${s.path}`).join("\n")}\n` : ""; const validationCmds = detectValidationCommands(ctx.cwd, state.sessionChangedPaths); const validationDirective = validationCmds.length > 0 ? `\n\n## Auto Verification Loop (always-on)\n\nAfter meaningful code changes, run (in order) until the smallest relevant proof passes (or report blocked):\n${validationCmds.map((c) => `- ${c}`).join("\n")}\n` : ""; // Installed-resource advisor — gated by policy.extensionPolicy.preferInstalledExtensions const advisorPolicy = loadMergedPolicy(ctx.cwd, detectProject(ctx.cwd)); const advisorDirective = buildAdvisorDirective(promptText, advisorPolicy); const ENFORCEMENT = ` ## Output Enforcement (always-on) - If you used any tool OR made any claim that depends on tool evidence, end with a **Status Report** using only: changed / verified / unverified / blocked / assumption. - **verified** is only allowed when the proof exists (e.g., read-back after write; read-before & read-after around edit; test command output for test claims). - If verification was not performed, explicitly label it **unverified**. `; const AUTO_WORKFLOW = ` ## Automatic Workflow Directives (always-on) These directives are active every turn. You do not need a slash command to invoke them. Apply the relevant ones based on what the user asks; do not announce that you are running them. ### Preflight (apply at the start of any non-trivial task) - Identify project type from detector output (node, go, wails, tauri, rust, python, php, postgres, docker, ...). - Summarize merged policy expectations (commandPolicy, workflow, artifactPolicy) before risky action. - List installed prompts/extensions/skills you intend to rely on before installing new tools. - Choose the smallest safe next step. ### Diagnose (apply on errors, unexpected output, or repeated failures) - Identify project type, classify command risk via the policy gate categories (SAFE_READ / BUILD / TEST / LINT / INSTALL / DESTRUCTIVE / DEPLOY / DATABASE / NETWORK / UNKNOWN). - Inspect raw evidence (logs, output, file contents) directly. Do not summarize from memory. - Stay on the current workstream. Surface drift instead of switching silently. - End with one concrete next step. ### Risk-aware fix (apply when proposing or applying a fix) - Make the smallest safe change. Touch only what is needed. - Respect commandPolicy and policy rules. If a category requires approval/reason, surface that before action. - Verify with the smallest relevant proof. - Do not start a new workstream without explicit user request. ### Verify (apply after meaningful changes) - Run the smallest relevant verification (build/test/lint scripts, artifact check, read-back). - Check artifact expectations from policy (verifyOutputExists / verifyNonEmptyOutputDir / verifyOutputTimestamp). - Report verified vs unverified clearly using the labels in Output Enforcement. - End with one concrete next step. ### Policy review (apply when the user asks about policy, rules, or workflow gates) - Inspect merged policy across global → workspace → project → task layers. - Cite relevant profiles, command categories, drift boundaries. - Recommend the smallest safe adjustment, not a rewrite. ### Promote rule (apply when a user correction or repeated failure looks like a durable rule) - Extract a candidate rule (id, condition, action, message). - Require explicit user approval before writing to .pi/minimax-policy.json. - Keep enforcement in policy. Do not rely on memory alone. ### Skill use - Skill routing already auto-injects skill instructions when prompt keywords match (web/browser/incident/research/design/multimodal/iterate). Follow the listed skills first when present. `; if (event.systemPrompt.includes("## Global Agent Contract (always-on)")) return; // Decomposition enforcement: detect complex prompts and force step-by-step let decompositionDirective = ""; const promptWords = promptText.split(/\s+/).filter(Boolean).length; const actionKeywords = (promptText.match(/\b(and then|also|plus|additionally|kemudian|lalu|setelah itu|dan juga|terus)\b/gi) || []).length; const isComplex = promptWords > 40 || actionKeywords >= 2; if (isComplex) { decompositionDirective = `\n\n## Decomposition (always-on — complex prompt detected) This prompt contains multiple actions or is long (${promptWords} words, ${actionKeywords} sequential keywords). You MUST: 1. Break it into numbered steps. 2. Execute ONLY step 1 now. 3. After step 1 is verified, report what remains. 4. Wait for user confirmation before proceeding to step 2. Do NOT attempt all steps in one response. `; } return { systemPrompt: event.systemPrompt + `\n\n---\n\n## Global Agent Contract (always-on)\n\n${contract}\n${ENFORCEMENT}${AUTO_WORKFLOW}${skillDirective}${validationDirective}${advisorDirective}${decompositionDirective}`, }; }); // === message_end — auto-grind with debounce + status report === pi.on("message_end", async (event, ctx) => { const msg: any = event.message; if (!msg || msg.role !== "assistant") return; if (msg.stopReason && msg.stopReason !== "stop") return; // Decomposition enforcement: if agent wrote 3+ files in one response, force stop and steer next step const writtenThisTurn = turnState.ops.filter((o) => o.toolName === "write" || o.toolName === "edit"); const uniqueWrittenFiles = [...new Set(writtenThisTurn.map((o) => o.path).filter(Boolean))]; if (uniqueWrittenFiles.length >= 3) { const decompMsg = [ { type: "text", text: `You wrote ${uniqueWrittenFiles.length} files in one turn. This is too many changes without verification.` }, { type: "text", text: "STOP. Verify what you just wrote (read-back + run tests if available), then report status. Do NOT continue to the next step until verified." }, ]; try { await pi.sendUserMessage(decompMsg, { deliverAs: "steer" }); } catch {} } const grindEnabled = pi.getFlag("minimax-grind") !== false; const grindMaxRaw = String(pi.getFlag("minimax-grind-max") ?? "2"); const grindMax = Number.isFinite(Number(grindMaxRaw)) ? Math.max(0, Number(grindMaxRaw)) : 2; const state = await ensureState(ctx.cwd); const validationCmds = detectValidationCommands(ctx.cwd, state.sessionChangedPaths); let grindWillFire = false; if ( grindEnabled && !grindFiredThisSession && state.sessionChangedPaths.length > 0 && validationCmds.length > 0 && !state.sessionGrindIteration && !grindIterationExceeded(state, grindMax) ) { const freshPaths = state.sessionChangedPaths.filter((p) => !isGrindDebounced(state, p)); if (freshPaths.length > 0) { nextGrindIteration(state); grindFiredThisSession = true; grindWillFire = true; const now = new Date().toISOString(); for (const p of freshPaths) { const existing = getGrindRecord(state, p); const rec: GrinderRecord = { path: p, lastGrindSent: now, grindCount: (existing?.grindCount ?? 0) + 1, lastGrindIteration: state.sessionGrindIteration, }; upsertGrindRecord(state, rec); } await flushState(); const grindMsg = [ { type: "text", text: "Run verification commands now and report results." }, { type: "text", text: "Commands (run in order; stop at first failure):\n" + validationCmds.map((c) => `- ${c}`).join("\n") }, { type: "text", text: "If any command fails: read the failure output, apply ONE fix, then rerun the failing command." }, ]; try { await pi.sendUserMessage(grindMsg, { deliverAs: "steer" }); } catch (_e) { // Steer queueing is the documented Pi path; if it fails, surface but do not crash. ctx.ui?.notify?.("Auto-grind queue failed; manual verification needed.", "warning"); } } } // Attach auto Status Report const report = buildAutoStatusReport(turnState); if (!report) return; // Force verification if Status Report has unverified items, but only when grind did // not already trigger a verification turn this message_end (avoids double-fire). if (!grindWillFire && report.includes("unverified") && !state.sessionGrindIteration) { const NL = String.fromCharCode(10); const unverifiedPaths = report.split(NL).filter((l: string) => l.includes("unverified: wrote")).map((l: string) => { const m = l.match(/wrote (\S+)/); return m ? m[1] : null; }).filter(Boolean) as string[]; if (unverifiedPaths.length !== 0) { const verifyMsg = [ { type: "text", text: "You have UNVERIFIED writes. Read back these files NOW and confirm content:" }, { type: "text", text: unverifiedPaths.map((p) => "- read " + p).join(NL) }, ]; try { await pi.sendUserMessage(verifyMsg, { deliverAs: "steer" }); } catch (_e) { ctx.ui?.notify?.("Force-verify queue failed; please re-read written files manually.", "warning"); } } } return { message: appendTextToMessage(msg, report) }; }); }