/** * DOM-free HTML sanitizer backend. * * `sanitizeHtml()` used to require `document` and `DOMParser`, so it threw on * every server runtime — in a framework that advertises runtime-agnostic SSR * and sanitizes DOM writes by default, the sanitizer was the one piece that * needed a browser (#229). * * This backend parses and re-serializes HTML with a small scanner instead, so * it runs anywhere. It shares every policy decision with the DOM backend via * `sanitize-policy.ts`; only the parsing and serialization differ. * * Two notes on the approach, both deliberate: * * - **It is a scanner, not an HTML5 tokenizer.** Where a browser would apply * error recovery, this backend prefers to drop or escape. That can differ * from the DOM backend on malformed input, and the difference always lands * on the safe side: output is escaped rather than guessed at. * - **Output is re-parsed and compared**, exactly as the DOM backend does, so * markup that changes shape on a second pass (mXSS) falls back to escaped * text rather than being emitted. * * `src/ssr/html-parser.ts` solves a similar problem, but `security` is a leaf * module that `ssr` itself imports, so reusing it would invert the dependency * direction. The policy is shared; the scanner is not. * * @module bquery/security * @internal */ import { escapeHtmlText, isAllowedTag, isAttributeAllowed, isValidAttributeName, relForAnchor, resolvePolicy, suppressesTextContent, type SanitizePolicy, } from './sanitize-policy'; import type { SanitizeOptions } from './types'; import { loadEntities } from './entities'; /** Elements that never have a closing tag. */ const VOID_ELEMENTS = new Set([ 'area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr', ]); /** * Elements whose content is text, not markup. Their content must be consumed * verbatim to the matching close tag — otherwise `"` * and friends re-enter markup parsing in the wrong place. */ const RAW_TEXT_ELEMENTS = new Set(['script', 'style', 'textarea', 'title', 'xmp']); /** * Elements whose start tag may close itself with a trailing slash. * * Only these. HTML ignores the `/` in `` is its raw text — so honouring it for any * element let `` escape raw-text consumption and * surface `alert(1)` as ordinary text. `svg` and `math` are the exception: * they switch the tree builder into foreign content, where every self-closing * start tag is acknowledged — the roots' descendants (``) included. * Only the roots are listed because both are always-dropped tags: their whole * subtree is removed, so how the descendants nest never reaches the output. */ const SELF_CLOSING_ELEMENTS = new Set([...VOID_ELEMENTS, 'svg', 'math']); const TAG_NAME = /^[a-zA-Z][a-zA-Z0-9:-]*/; /** * Document-structure elements a real HTML parser absorbs rather than nests. * * The DOM backend parses into a document and lifts `body`'s children into the * fragment, so `` and `` never appear as elements to remove — but * their contents survive. Suppressing them here like any other disallowed tag * would discard an entire server-rendered page. `` is deliberately not * in this set: its contents do not reach the body, so it is dropped whole. */ const TRANSPARENT_ELEMENTS = new Set(['html', 'body']); /** * Elements whose close tag HTML lets you omit, and what opening them implies * should close first. `` means two siblings, not nesting. */ const IMPLIED_END_TAGS: Record> = { li: new Set(['li']), dt: new Set(['dt', 'dd']), dd: new Set(['dt', 'dd']), p: new Set(['p']), option: new Set(['option']), optgroup: new Set(['optgroup', 'option']), td: new Set(['td', 'th']), th: new Set(['td', 'th']), tr: new Set(['tr', 'td', 'th']), tbody: new Set(['thead', 'tbody', 'tfoot', 'tr', 'td', 'th']), tfoot: new Set(['thead', 'tbody', 'tfoot', 'tr', 'td', 'th']), thead: new Set(['thead', 'tbody', 'tfoot', 'tr', 'td', 'th']), }; /** The longest legacy name (`frac12`, `Ccedil`, …), which bounds the prefix search. */ const LEGACY_MAX_LENGTH = 6; /** * What a browser substitutes for `€`–`Ÿ`: those are C1 controls in * Unicode, but the spec reads them as windows-1252, as legacy pages meant. */ const C1_REPLACEMENTS: Readonly> = { 0x80: 0x20ac, 0x82: 0x201a, 0x83: 0x0192, 0x84: 0x201e, 0x85: 0x2026, 0x86: 0x2020, 0x87: 0x2021, 0x88: 0x02c6, 0x89: 0x2030, 0x8a: 0x0160, 0x8b: 0x2039, 0x8c: 0x0152, 0x8e: 0x017d, 0x91: 0x2018, 0x92: 0x2019, 0x93: 0x201c, 0x94: 0x201d, 0x95: 0x2022, 0x96: 0x2013, 0x97: 0x2014, 0x98: 0x02dc, 0x99: 0x2122, 0x9a: 0x0161, 0x9b: 0x203a, 0x9c: 0x0153, 0x9e: 0x017e, 0x9f: 0x0178, }; const decodeNumeric = (digits: string, radix: number): string => { const num = Number.parseInt(digits, radix); if (num === 0 || num > 0x10ffff || (num >= 0xd800 && num <= 0xdfff)) return '\ufffd'; return String.fromCodePoint(C1_REPLACEMENTS[num] ?? num); }; const ATTRIBUTE_BLOCKER = /[=a-zA-Z0-9]/; /** * Decode character references the way an HTML parser does. * * Decoding matters for safety, not convenience: the policy checks run on * decoded values, so `href="javascript:alert(1)"` is compared against the * dangerous-protocol list as `javascript:alert(1)` rather than slipping past * as an unrecognized string. Everything is re-escaped on the way out. * * It follows the tokenizer's rules rather than approximating them, because * the DOM backend gets those rules from the browser and the two must agree: * * - names are case-sensitive (`É` is not `é`); * - only the legacy names decode without a `;`, and they match as a prefix, * longest first (`¬it;` is `¬it;`); * - `inAttribute` applies the rule that keeps query strings intact: a legacy * name without `;` followed by `=` or an alphanumeric stays literal, so * `href="?a=1©=2"` is not rewritten to `?a=1©=2`. * * Names outside `ENTITY_RUNS` stay literal — see `entities.ts` for why the * table is a subset and why that can only fall short of a browser, never * contradict it. * @internal */ export const decodeEntities = (input: string, inAttribute = false): string => { if (!input.includes('&')) return input; const { named, legacy } = loadEntities(); return input.replace( /&(?:#(?:[xX]([0-9a-fA-F]+)|([0-9]+));?|([a-zA-Z0-9]+)(;?))/g, (match, hex: string, dec: string, name: string, semi: string, offset: number) => { if (hex !== undefined) return decodeNumeric(hex, 16); if (dec !== undefined) return decodeNumeric(dec, 10); if (semi && named.has(name)) return named.get(name) as string; for (let length = Math.min(name.length, LEGACY_MAX_LENGTH); length >= 2; length--) { const prefix = name.slice(0, length); if (!legacy.has(prefix)) continue; // `name` is a maximal alphanumeric run, so when the prefix is all of it // the next character is whatever follows the match (`;` was ruled out // above: every legacy name also exists with one). const next = length < name.length ? name[length] : input[offset + match.length]; if (inAttribute && next !== undefined && ATTRIBUTE_BLOCKER.test(next)) return match; return (named.get(prefix) as string) + name.slice(length) + semi; } return match; } ); }; /** Escape a value for use inside a double-quoted attribute. */ const escapeAttribute = (value: string): string => value.replace(/&/g, '&').replace(/"/g, '"').replace(//g, '>'); /** Escape text content. `&` first, so escapes are not double-escaped. */ const escapeText = (value: string): string => value.replace(/&/g, '&').replace(//g, '>'); interface Attribute { name: string; value: string; } interface OpenTag { kind: 'open'; tag: string; attributes: Attribute[]; selfClosing: boolean; } interface CloseTag { kind: 'close'; tag: string; } interface TextToken { kind: 'text'; value: string; } type Token = OpenTag | CloseTag | TextToken; const isWhitespace = (char: string): boolean => char === ' ' || char === '\t' || char === '\n' || char === '\r' || char === '\f'; /** * Read the attributes of an open tag, starting just after the tag name. * Returns the attributes and the index just past the closing `>`. */ const readAttributes = ( source: string, start: number ): { attributes: Attribute[]; end: number; selfClosing: boolean } => { const attributes: Attribute[] = []; let pos = start; let selfClosing = false; while (pos < source.length) { while (pos < source.length && isWhitespace(source[pos])) pos++; if (pos >= source.length) break; if (source[pos] === '>') { pos++; break; } if (source[pos] === '/' && source[pos + 1] === '>') { selfClosing = true; pos += 2; break; } // A stray `/` inside the tag — skip it rather than reading it as a name. if (source[pos] === '/') { pos++; continue; } const nameStart = pos; while ( pos < source.length && !isWhitespace(source[pos]) && source[pos] !== '=' && source[pos] !== '>' && source[pos] !== '/' ) { pos++; } const name = source.slice(nameStart, pos); if (name.length === 0) { pos++; continue; } while (pos < source.length && isWhitespace(source[pos])) pos++; let value = ''; if (source[pos] === '=') { pos++; while (pos < source.length && isWhitespace(source[pos])) pos++; const quote = source[pos]; if (quote === '"' || quote === "'") { pos++; const valueStart = pos; const closing = source.indexOf(quote, pos); if (closing === -1) { value = source.slice(valueStart); pos = source.length; } else { value = source.slice(valueStart, closing); pos = closing + 1; } } else { const valueStart = pos; while (pos < source.length && !isWhitespace(source[pos]) && source[pos] !== '>') pos++; value = source.slice(valueStart, pos); } } attributes.push({ name, value: decodeEntities(value, true) }); } return { attributes, end: pos, selfClosing }; }; /** * Tokenize HTML into open tags, close tags and text. * * Anything that is not recognizable markup — a bare `<`, an unterminated tag, * a comment, a doctype, a processing instruction — becomes text or is * discarded. Nothing is passed through as raw markup. * @internal */ export const tokenize = (html: string): Token[] => { const tokens: Token[] = []; let pos = 0; let textStart = 0; // Lowercased once, not per raw-text element: recomputing it inside the loop // made tokenization O(raw-text elements x input length), so a megabyte of // `