/** * BrainBank — LLM Pruner * * LLM-based noise filter using Anthropic models (Haiku 4.5 or Sonnet 4.6). * Binary classification: for each search result, the model decides * "relevant" or "noise" based on filePath, metadata, and full * file content (capped at ~8K chars per item by prune.ts). * * Latency: ~300-600ms (Haiku), ~500-1200ms (Sonnet). */ import type { Pruner, PrunerItem } from '@/types.ts'; import * as fs from 'node:fs'; const DEFAULT_MODEL = 'claude-haiku-4-5-20251001'; /** Budget map keyed by model prefix. Sonnet 4.6 has a larger context window. */ const MODEL_BUDGETS: Record = { 'claude-haiku': 700_000, // ~180K tokens ≈ ~720K chars 'claude-sonnet': 900_000, // ~200K tokens ≈ ~900K chars (Sonnet 4.6) }; function _getBudget(model: string): number { for (const [prefix, budget] of Object.entries(MODEL_BUDGETS)) { if (model.startsWith(prefix)) return budget; } return 700_000; // Default: conservative budget } const _debug = !!process.env.BRAINBANK_DEBUG; function dbg(msg: string): void { if (_debug) console.error(msg); } export interface HaikuPrunerOptions { /** Anthropic API key. Falls back to ANTHROPIC_API_KEY env var. */ apiKey?: string; /** Model to use. Default: claude-haiku-4-5-20251001 */ model?: string; } export class HaikuPruner implements Pruner { private readonly _apiKey: string; private readonly _model: string; constructor(options: HaikuPrunerOptions = {}) { this._apiKey = options.apiKey ?? process.env.ANTHROPIC_API_KEY ?? ''; this._model = options.model ?? DEFAULT_MODEL; if (!this._apiKey) { throw new Error( 'HaikuPruner: No API key provided. Set ANTHROPIC_API_KEY env var or pass apiKey option.', ); } } async prune(query: string, items: PrunerItem[], context?: string): Promise { if (items.length === 0) return []; if (items.length === 1) return [items[0].id]; // ── Adaptive preview truncation ── // Estimate total chars and truncate previews if we'd exceed token budget const budget = _getBudget(this._model); let totalChars = _estimatePromptChars(query, items); if (totalChars > budget && items.length > 1) { // Calculate max preview chars per item to fit budget const overheadPerItem = 200; // filePath + metadata + separators const promptOverhead = 1500 + (context ? context.length : 0); // system prompt + rules + context const availableForPreviews = budget - promptOverhead - (items.length * overheadPerItem); const maxPreviewPerItem = Math.max(1000, Math.floor(availableForPreviews / items.length)); dbg(`[Pruner:${this._model}] Budget overflow: ${totalChars} chars for ${items.length} items (budget: ${budget}). Truncating previews to ${maxPreviewPerItem} chars each.`); for (const item of items) { if (item.preview.length > maxPreviewPerItem) { item.preview = _truncatePreview(item.preview, maxPreviewPerItem); } } totalChars = _estimatePromptChars(query, items); dbg(`[Pruner:${this._model}] After truncation: ${totalChars} chars`); } const itemLines = items.map(item => { // Only show useful metadata fields (skip raw scores, IDs, large arrays) const SKIP_KEYS = new Set(['id', 'chunkIds', 'rrfScore', 'filePath']); const meta = Object.entries(item.metadata) .filter(([k, v]) => v !== undefined && v !== null && !SKIP_KEYS.has(k)) .map(([k, v]) => `${k}=${v}`) .join(' | '); return `#${item.id} ${item.filePath} | ${meta}\n${item.preview}`; }).join('\n---\n'); // Build context section if provided const contextSection = context ? `\nTask Context:\n"""\n${context.slice(0, 2000)}\n"""\n\n` : ''; const prompt = `Query: "${query}"\n${contextSection}Search results (full file content):\n${itemLines}\n\n` + `You are a precision search filter. Remove noise but NEVER drop core implementation files.\n` + `Return a JSON array of #IDs to KEEP, ordered by relevance (most relevant FIRST).\n` + `${context ? 'Use the Task Context to understand EXACTLY what the user needs — keep files that would be needed to implement the described change.\n' : ''}` + `\nKEEP (any ONE of these is enough):\n` + `- File contains the method/handler/function that performs the queried action.\n` + `- File defines the entity, types, or state machine for the queried system.\n` + `- File is the controller/route that exposes the queried feature.\n` + `- File orchestrates or composes the queried workflow (even if it also handles other workflows).\n` + `- Service files that handle the queried action alongside other actions → KEEP (they contain the implementation).\n\n` + `DROP (any ONE of these triggers a drop):\n` + `- File is from a DIFFERENT domain that shares vocabulary with the query.\n` + ` Example: query "job encounter time" → DROP priority.entity.ts, notification.worker.ts\n` + `- Database seeders, migrations, test fixtures — unless query asks about seeding/migration.\n` + `- Generic infrastructure: loggers, config factories, decorators — unless query targets that infra.\n` + `- File where the queried concept appears only as an import, foreign key, or one-line reference.\n` + `- Permission configs — unless query is about permissions/auth.\n` + `- DTOs that only mirror entity fields without adding logic.\n\n` + `Target: 25-50% keep rate. Drop infrastructure/boilerplate aggressively, but NEVER drop a service that implements the queried action.\n` + `Order: core implementation → entity/types → controller/service → peripheral.\n\n` + `Respond with ONLY the JSON array. Example: [3, 0, 5]`; try { const response = await fetch('https://api.anthropic.com/v1/messages', { method: 'POST', headers: { 'Content-Type': 'application/json', 'x-api-key': this._apiKey, 'anthropic-version': '2023-06-01', }, body: JSON.stringify({ model: this._model, max_tokens: 512, messages: [{ role: 'user', content: prompt, }], }), }); if (!response.ok) { const body = await response.text().catch(() => ''); // API error → fail-open, return all — but LOG IT console.error(`[Pruner:${this._model}] API error: ${response.status} ${response.statusText} (${items.length} items, ~${Math.round(totalChars / 1000)}K chars) — keeping all. ${body.slice(0, 200)}`); return items.map(i => i.id); } const data = await response.json() as { content: { type: string; text: string }[]; usage?: { input_tokens: number; output_tokens: number; cache_creation_input_tokens?: number; cache_read_input_tokens?: number; }; }; // ── Cost tracking (authoritative: these are Anthropic's exact billing counts) ── const usage = data.usage; if (usage) { const cost = _estimateCost(this._model, usage); const parts = [ `${usage.input_tokens} in`, `${usage.output_tokens} out`, ]; if (usage.cache_creation_input_tokens) parts.push(`${usage.cache_creation_input_tokens} cache_write`); if (usage.cache_read_input_tokens) parts.push(`${usage.cache_read_input_tokens} cache_read`); const costLine = `[Pruner] ${this._model} — ${parts.join(' + ')} = $${cost.toFixed(4)}`; dbg(costLine); // Append to brainbank.log (daemon stdio is 'ignore' so console.error is lost) try { fs.appendFileSync('/tmp/brainbank.log', costLine + '\n'); } catch { /* best effort */ } } const text = data.content?.[0]?.text ?? ''; dbg(`[Pruner:${this._model}] Raw response: ${text}`); // Haiku may wrap in ```json ... ``` — extract any JSON array const match = text.match(/\[[\d\s,]+\]/); if (!match) { console.error(`[Pruner:${this._model}] No JSON array found in response — keeping all ${items.length} items. Raw: ${text.slice(0, 200)}`); return items.map(i => i.id); } const keepIds = JSON.parse(match[0]) as number[]; const validIds = new Set(items.map(i => i.id)); const filtered = keepIds.filter(id => validIds.has(id)); dbg(`[HaikuPruner] Keep IDs: [${filtered.join(', ')}] (${items.length - filtered.length} dropped)`); return filtered; } catch (err) { // Network error → fail-open, return all — but LOG IT console.error(`[Pruner:${this._model}] Error: ${err instanceof Error ? err.message : String(err)} (${items.length} items) — keeping all`); return items.map(i => i.id); } } async close(): Promise { // No resources to release (stateless HTTP) } } // ── Helpers ────────────────────────────────────────── /** Per-million-token pricing for supported models (May 2026). */ const MODEL_PRICING: Record = { 'claude-haiku': { input: 1.00, output: 5.00 }, // Haiku 4.5 'claude-sonnet': { input: 3.00, output: 15.00 }, // Sonnet 4.6 }; interface UsageInfo { input_tokens: number; output_tokens: number; cache_creation_input_tokens?: number; cache_read_input_tokens?: number; } /** * Calculate USD cost from the API's authoritative usage counts. * Cache pricing: writes cost 1.25x input, reads cost 0.1x input. */ function _estimateCost(model: string, usage: UsageInfo): number { let pricing = { input: 3.00, output: 15.00 }; // Default to Sonnet for (const [prefix, p] of Object.entries(MODEL_PRICING)) { if (model.startsWith(prefix)) { pricing = p; break; } } const inputCost = usage.input_tokens * pricing.input; const outputCost = usage.output_tokens * pricing.output; const cacheWriteCost = (usage.cache_creation_input_tokens ?? 0) * pricing.input * 1.25; const cacheReadCost = (usage.cache_read_input_tokens ?? 0) * pricing.input * 0.1; return (inputCost + outputCost + cacheWriteCost + cacheReadCost) / 1_000_000; } /** Estimate total prompt chars for budget checking. */ function _estimatePromptChars(query: string, items: PrunerItem[]): number { const promptOverhead = 1500; let total = promptOverhead + query.length; for (const item of items) { total += item.filePath.length + item.preview.length + 200; // 200 for metadata + separators } return total; } /** * Truncate a preview to fit within a character budget. * Keeps top 60% + bottom 25% of lines (same strategy as prune.ts _buildPreview). */ function _truncatePreview(content: string, maxChars: number): string { if (content.length <= maxChars) return content; const lines = content.split('\n'); const totalLines = lines.length; // Keep ~60% from top, ~25% from bottom const topCount = Math.floor(totalLines * 0.6); const bottomCount = Math.floor(totalLines * 0.25); const omitted = totalLines - topCount - bottomCount; let result = lines.slice(0, topCount).join('\n') + `\n\n// [... ${omitted} lines omitted ...]\n\n` + lines.slice(totalLines - bottomCount).join('\n'); // If still too long, hard-truncate if (result.length > maxChars) { result = result.slice(0, maxChars) + '\n// [truncated]'; } return result; }