import type { ScanContext } from "../types.js"; export type JudgeVerdict = { /** * The judge's call: * - `malicious` — confident injection / jailbreak attempt * - `suspicious` — instruction-shaped but ambiguous * - `benign` — no manipulation detected * - `error` — backend failed or timed out (fail-open: do not block on this) */ verdict: "malicious" | "suspicious" | "benign" | "error"; /** 0..1 confidence parsed from the judge, best-effort. */ confidence: number; /** Short rationale the judge gave, if any. */ rationale?: string; /** Judge round-trip latency in ms. */ durationMs: number; /** Raw model text, for audit / debugging. */ raw?: string; }; /** Structured backend. Implement `complete()` to call your judge model. */ export interface JudgeBackend { complete(prompt: string): Promise; } /** Either a structured backend or a bare completion function. */ export type JudgeBackendLike = JudgeBackend | ((prompt: string) => Promise); export interface AsyncJudgeConfig { /** Your judge-model caller. Use a small, fast model (e.g. Haiku, a 22M * DeBERTa-class classifier, or a local model). */ backend: JudgeBackendLike; /** * Override the prompt sent to the judge. Receives the (truncated) input * and the scan context. Must instruct the model to answer in the * `VERDICT: … / CONFIDENCE: … / REASON: …` shape the default parser reads, * or supply your own `parse`. */ promptTemplate?: (input: string, context?: ScanContext) => string; /** Custom parser for the judge's raw response. */ parse?: (raw: string) => Omit; /** Max input chars sent to the judge (cost guard). Default 4000. */ maxInputChars?: number; /** Judge-call timeout in ms; on timeout the verdict is `"error"`. Default 8000. */ timeoutMs?: number; /** Invoked with every verdict — wire this to your audit log. */ onVerdict?: (verdict: JudgeVerdict, input: string, context?: ScanContext) => void; } export interface AsyncJudge { /** * Evaluate one input. Resolves with a verdict; never rejects (errors map * to `verdict: "error"`). Fire it in a parallel lane — do NOT await it on * the critical path: * * ```ts * const [syncResult] = await Promise.all([ * shield.scan(input), // deterministic, fast — gates the request * judge.evaluate(input), // semantic, slow — lands in the audit log * ]); * ``` */ evaluate(input: string, context?: ScanContext): Promise; } /** * Build an async LLM judge. The returned `evaluate()` never throws — * backend failures and timeouts resolve to `verdict: "error"`. * * @example * ```ts * import { createAsyncJudge } from "ai-shield-core"; * import Anthropic from "@anthropic-ai/sdk"; * * const client = new Anthropic(); * const judge = createAsyncJudge({ * async backend(prompt) { * const r = await client.messages.create({ * model: "claude-haiku-4-5", * max_tokens: 128, * messages: [{ role: "user", content: prompt }], * }); * return r.content[0]?.type === "text" ? r.content[0].text : ""; * }, * onVerdict: (v, input) => auditLog.record({ judge: v, input }), * }); * ``` */ export declare function createAsyncJudge(config: AsyncJudgeConfig): AsyncJudge; //# sourceMappingURL=async-judge.d.ts.map