/** * DOM-free HTML sanitizer backend. * * `sanitizeHtml()` used to require `document` and `DOMParser`, so it threw on * every server runtime — in a framework that advertises runtime-agnostic SSR * and sanitizes DOM writes by default, the sanitizer was the one piece that * needed a browser (#229). * * This backend parses and re-serializes HTML with a small scanner instead, so * it runs anywhere. It shares every policy decision with the DOM backend via * `sanitize-policy.ts`; only the parsing and serialization differ. * * Two notes on the approach, both deliberate: * * - **It is a scanner, not an HTML5 tokenizer.** Where a browser would apply * error recovery, this backend prefers to drop or escape. That can differ * from the DOM backend on malformed input, and the difference always lands * on the safe side: output is escaped rather than guessed at. * - **Output is re-parsed and compared**, exactly as the DOM backend does, so * markup that changes shape on a second pass (mXSS) falls back to escaped * text rather than being emitted. * * `src/ssr/html-parser.ts` solves a similar problem, but `security` is a leaf * module that `ssr` itself imports, so reusing it would invert the dependency * direction. The policy is shared; the scanner is not. * * @module bquery/security * @internal */ import type { SanitizeOptions } from './types.cjs'; /** * Decode character references the way an HTML parser does. * * Decoding matters for safety, not convenience: the policy checks run on * decoded values, so `href="javascript:alert(1)"` is compared against the * dangerous-protocol list as `javascript:alert(1)` rather than slipping past * as an unrecognized string. Everything is re-escaped on the way out. * * It follows the tokenizer's rules rather than approximating them, because * the DOM backend gets those rules from the browser and the two must agree: * * - names are case-sensitive (`É` is not `é`); * - only the legacy names decode without a `;`, and they match as a prefix, * longest first (`¬it;` is `¬it;`); * - `inAttribute` applies the rule that keeps query strings intact: a legacy * name without `;` followed by `=` or an alphanumeric stays literal, so * `href="?a=1©=2"` is not rewritten to `?a=1©=2`. * * Names outside `ENTITY_RUNS` stay literal — see `entities.ts` for why the * table is a subset and why that can only fall short of a browser, never * contradict it. * @internal */ export declare const decodeEntities: (input: string, inAttribute?: boolean) => string; interface Attribute { name: string; value: string; } interface OpenTag { kind: 'open'; tag: string; attributes: Attribute[]; selfClosing: boolean; } interface CloseTag { kind: 'close'; tag: string; } interface TextToken { kind: 'text'; value: string; } type Token = OpenTag | CloseTag | TextToken; /** * Tokenize HTML into open tags, close tags and text. * * Anything that is not recognizable markup — a bare `<`, an unterminated tag, * a comment, a doctype, a processing instruction — becomes text or is * discarded. Nothing is passed through as raw markup. * @internal */ export declare const tokenize: (html: string) => Token[]; /** * Sanitize HTML with no DOM available. * * Mirrors the DOM backend's contract, including its mutation-XSS guard: the * output is sanitized a second time and, if the two passes disagree, the * escaped text content is returned instead of markup whose meaning changes * when it is re-parsed. * @internal */ export declare const sanitizeHtmlString: (html: string, options?: SanitizeOptions) => string; /** Plain-text extraction with no DOM. @internal */ export declare const stripTagsString: (html: string) => string; export {}; //# sourceMappingURL=sanitize-string.d.ts.map