import type { Scanner, ScannerResult, ScanContext, Violation, IngestionSource, TrustTier } from "../types.js"; /** * Default trust-tier inferred from source. * `user` is still untrusted in this library's threat model — a user can * inject too — but `system` is reserved for content the developer * controls and labels via `wrapContext()`. Every ingestion source * (including `user`) therefore returns `"untrusted"` by default; the * parameter is kept on the signature so future per-source overrides * (e.g. an installer marking a specific source as trusted) don't * require a breaking API change. */ export declare function trustTierForSource(_source: IngestionSource): TrustTier; /** * Result of `scanIngested()`. * * Shape parallels `ScanResult` from `chain.ts` so callers can treat * both interchangeably. */ export interface IngestionScanResult { safe: boolean; decision: "allow" | "warn" | "block"; /** * Sanitized output. When `decision === "block"` this is the empty * string — the original content was deemed unsafe and the field name * "sanitized" would otherwise mislead callers into using poisoned * content. Use the source `content` argument if you need the raw input * for logging or quarantine. */ sanitized: string; violations: Violation[]; source: IngestionSource; meta: { scanDurationMs: number; scannersRun: string[]; /** Number of extra source-specific patterns that fired. */ sourceSpecificHits: number; /** * Always `false` from `scanIngested()` — ingestion scans don't go * through the LRU cache. Field is present so callers can write a * single result-handler for both `ScanResult` and `IngestionScanResult`. */ cached: boolean; }; } export interface IngestionScannerConfig { /** Override the per-source threshold lookup. */ threshold?: number; /** * Additional custom patterns to merge with the source profile's * `extraPatterns`. Useful for org-specific markers. */ customPatterns?: RegExp[]; /** * Force the underlying heuristic scanner to a different strictness * (default "high" because ingestion is always tighter than user input). */ strictness?: "low" | "medium" | "high"; } /** * Scanner implementation. Composable into a `ScannerChain` when the * caller wants ingestion to participate in the main scan flow rather * than be invoked via the standalone `scanIngested()` helper. * * The scanner reads the `source` from `ScanContext` (or treats input * as `"user"` when missing) and applies the source-specific profile. */ export declare class IngestionScanner implements Scanner { readonly name = "ingestion"; private readonly threshold; private readonly customPatterns; private readonly heuristic; constructor(config?: IngestionScannerConfig); scan(input: string, context: ScanContext): Promise; } /** * One-shot helper. Scans `content` against the source-specific profile * and returns a result without needing an `AIShield` instance. * * Use when you want a quick gate at the ingestion boundary, e.g. * before storing a chunk into a vector DB or before passing a tool * description into the model's context. * * @example * ```ts * import { scanIngested } from "ai-shield-core"; * * const ragChunk = "...retrieved document text..."; * const result = await scanIngested(ragChunk, "rag"); * if (!result.safe) { * // reject the chunk OR strip it before assembly * logger.warn("IPI candidate", result.violations); * } * ``` */ export declare function scanIngested(content: string, source: IngestionSource, config?: IngestionScannerConfig): Promise; /** * Scan the runtime *result* of a tool call before it re-enters the model * context. The dominant indirect-injection channel in agentic loops: a * search tool surfaces a poisoned page, an MCP server returns attacker- * controlled data, a compromised upstream API embeds instructions in its * response. PoisonedRAG (USENIX Security 2025) showed 5 planted documents * reach a 90% attack-success rate in million-document knowledge bases — * the payload arrives here, not in the user prompt. * * Thin wrapper over `scanIngested(content, "tool-output")` that also * stamps the originating `toolName` into every violation detail, so an * audit log can answer "which tool returned the poisoned content?". * * Pair with `CircuitBreakerRegistry` when you also want to rate-limit or * trip the tool after repeated poisoned results: * * @example * ```ts * import { scanToolOutput } from "ai-shield-core"; * * const result = await searchTool.call(query); // untrusted * const scan = await scanToolOutput("web_search", result); * if (!scan.safe) { * // drop the result OR strip it before the next model turn * audit.warn("poisoned tool output", { tool: "web_search", v: scan.violations }); * return; // do not feed `result` back into the model * } * model.continue(result); * ``` */ export declare function scanToolOutput(toolName: string, content: string, config?: IngestionScannerConfig): Promise; /** * Try to decode common obfuscation layers an attacker uses to smuggle * an injection past pattern matchers. Returns the decoded payload when * it looks like a successful decode, else `null`. * * The function deliberately runs at most ONE decode layer to avoid * decoding amplification (a chain of `base64(base64(...))` would force * us into deep recursion); a single-layer decode is enough to catch * the vast majority of in-the-wild bypasses while keeping execution * cost bounded. * * Heuristics: * - Base64: contiguous run of 40+ Base64 chars, decodes to mostly * printable ASCII or the `\u00..` C0 range stays empty. * - Hex: 80+ hex chars in a row. * - Percent-encoding: more than 5 `%XX` sequences. * * Returns the longest decoded payload when multiple candidates fire. */ export declare function tryDecodeObfuscation(input: string): string | null; //# sourceMappingURL=ingestion.d.ts.map