import type { ScanContext } from "../types.js"; import { type IngestionScanResult } from "../scanner/ingestion.js"; /** A single LLM call. BYO: wrap your Anthropic / OpenAI / local model. */ export type LLMBackend = (prompt: string) => Promise; export interface DualLLMConfig { /** * The privileged model — holds tools/capabilities. You wire its tools in * your own backend; this module only guarantees it never receives raw * untrusted content (only the user request + quarantined results). */ privileged: LLMBackend; /** * The quarantined model — processes untrusted content, has NO tools. Its * output is treated as data, scanned, and fenced before the privileged * model sees it. */ quarantined: LLMBackend; /** * Scan the quarantined model's output with the ingestion gate before it * crosses to the privileged side (defense-in-depth — the quarantined model * could itself be injected into emitting instruction-shaped text). Default true. */ scanBridge?: boolean; /** Ingestion strictness for the bridge scan. Default "high". */ bridgeStrictness?: "low" | "medium" | "high"; /** Fence labels for the assembled trusted prompt. */ fence?: { open: string; close: string; }; } export interface QuarantineResult { /** The quarantined model's output (structured data, to be used as DATA only). */ output: string; /** False if the bridge scan flagged the quarantined output as unsafe. */ safe: boolean; /** The bridge scan result, when `scanBridge` is enabled. */ scan?: IngestionScanResult; /** Echoed for traceability. */ task: string; } export interface DualLLM { /** * Run the quarantined model over a piece of untrusted content with a * specific extraction task. The untrusted content reaches ONLY this model. * The result is scanned (if `scanBridge`) and returned as data. */ quarantine(untrustedContent: string, task: string): Promise; /** * Assemble a trusted prompt for the privileged model from the user request * plus quarantined results. Unsafe quarantined results are dropped (not * fenced-and-included) — a flagged result must not reach the actor at all. */ assembleTrustedPrompt(userRequest: string, quarantined: QuarantineResult[]): string; /** * Convenience: assemble the trusted prompt and run the privileged model. */ runPrivileged(userRequest: string, quarantined: QuarantineResult[]): Promise; } /** * Build a dual-LLM harness. * * @example * ```ts * import { createDualLLM } from "ai-shield-core"; * * const dual = createDualLLM({ * privileged: (p) => toolModel.run(p), // has tools * quarantined: (p) => plainModel.run(p), // no tools * }); * * // Untrusted RAG chunk reaches ONLY the quarantined model: * const r = await dual.quarantine(ragChunk, "Extract the order id as JSON."); * // The privileged (tool-holding) model only ever sees the user request + * // the scanned, fenced result — never the raw chunk: * const answer = await dual.runPrivileged("Refund my last order.", [r]); * ``` */ export declare function createDualLLM(config: DualLLMConfig): DualLLM; export interface ActionScreenResult { /** Whether the proposed action is consistent with the user's intent. */ allowed: boolean; /** Short rationale (from the judge, best-effort). */ reason: string; /** Raw judge response, for audit. */ raw?: string; } export interface ActionScreenerConfig { /** * Judge backend. Use a small/fast model. It is asked ONLY about the user's * original intent and the proposed action — never the untrusted context * that produced the action, so a steered tool call can't also steer its * own approval. */ judge: LLMBackend; /** Timeout in ms; on timeout the action is DENIED (fail-closed). Default 8000. */ timeoutMs?: number; /** Custom prompt builder. */ promptTemplate?: (userIntent: string, proposedAction: string) => string; } /** * Build an action screener. `screen()` is fail-closed: a judge error or * timeout returns `allowed: false`. * * @example * ```ts * const screener = createActionScreener({ judge: (p) => fast.run(p) }); * const v = await screener.screen("Summarize my inbox", "delete all emails"); * if (!v.allowed) abortToolCall(v.reason); * ``` */ export declare function createActionScreener(config: ActionScreenerConfig): { screen(userIntent: string, proposedAction: string): Promise; }; export type { ScanContext }; //# sourceMappingURL=dual-llm.d.ts.map