/**
* DOM-free HTML sanitizer backend.
*
* `sanitizeHtml()` used to require `document` and `DOMParser`, so it threw on
* every server runtime — in a framework that advertises runtime-agnostic SSR
* and sanitizes DOM writes by default, the sanitizer was the one piece that
* needed a browser (#229).
*
* This backend parses and re-serializes HTML with a small scanner instead, so
* it runs anywhere. It shares every policy decision with the DOM backend via
* `sanitize-policy.ts`; only the parsing and serialization differ.
*
* Two notes on the approach, both deliberate:
*
* - **It is a scanner, not an HTML5 tokenizer.** Where a browser would apply
* error recovery, this backend prefers to drop or escape. That can differ
* from the DOM backend on malformed input, and the difference always lands
* on the safe side: output is escaped rather than guessed at.
* - **Output is re-parsed and compared**, exactly as the DOM backend does, so
* markup that changes shape on a second pass (mXSS) falls back to escaped
* text rather than being emitted.
*
* `src/ssr/html-parser.ts` solves a similar problem, but `security` is a leaf
* module that `ssr` itself imports, so reusing it would invert the dependency
* direction. The policy is shared; the scanner is not.
*
* @module bquery/security
* @internal
*/
import {
escapeHtmlText,
isAllowedTag,
isAttributeAllowed,
isValidAttributeName,
relForAnchor,
resolvePolicy,
suppressesTextContent,
type SanitizePolicy,
} from './sanitize-policy';
import type { SanitizeOptions } from './types';
import { loadEntities } from './entities';
/** Elements that never have a closing tag. */
const VOID_ELEMENTS = new Set([
'area',
'base',
'br',
'col',
'embed',
'hr',
'img',
'input',
'link',
'meta',
'param',
'source',
'track',
'wbr',
]);
/**
* Elements whose content is text, not markup. Their content must be consumed
* verbatim to the matching close tag — otherwise `"`
* and friends re-enter markup parsing in the wrong place.
*/
const RAW_TEXT_ELEMENTS = new Set(['script', 'style', 'textarea', 'title', 'xmp']);
/**
* Elements whose start tag may close itself with a trailing slash.
*
* Only these. HTML ignores the `/` in `` — the element still opens,
* and everything up to `` is its raw text — so honouring it for any
* element let `alert(1)` escape raw-text consumption and
* surface `alert(1)` as ordinary text. `svg` and `math` are the exception:
* they switch the tree builder into foreign content, where every self-closing
* start tag is acknowledged — the roots' descendants (``) included.
* Only the roots are listed because both are always-dropped tags: their whole
* subtree is removed, so how the descendants nest never reaches the output.
*/
const SELF_CLOSING_ELEMENTS = new Set([...VOID_ELEMENTS, 'svg', 'math']);
const TAG_NAME = /^[a-zA-Z][a-zA-Z0-9:-]*/;
/**
* Document-structure elements a real HTML parser absorbs rather than nests.
*
* The DOM backend parses into a document and lifts `body`'s children into the
* fragment, so `` and `
` never appear as elements to remove — but
* their contents survive. Suppressing them here like any other disallowed tag
* would discard an entire server-rendered page. `` is deliberately not
* in this set: its contents do not reach the body, so it is dropped whole.
*/
const TRANSPARENT_ELEMENTS = new Set(['html', 'body']);
/**
* Elements whose close tag HTML lets you omit, and what opening them implies
* should close first. `` means two siblings, not nesting.
*/
const IMPLIED_END_TAGS: Record> = {
li: new Set(['li']),
dt: new Set(['dt', 'dd']),
dd: new Set(['dt', 'dd']),
p: new Set(['p']),
option: new Set(['option']),
optgroup: new Set(['optgroup', 'option']),
td: new Set(['td', 'th']),
th: new Set(['td', 'th']),
tr: new Set(['tr', 'td', 'th']),
tbody: new Set(['thead', 'tbody', 'tfoot', 'tr', 'td', 'th']),
tfoot: new Set(['thead', 'tbody', 'tfoot', 'tr', 'td', 'th']),
thead: new Set(['thead', 'tbody', 'tfoot', 'tr', 'td', 'th']),
};
/** The longest legacy name (`frac12`, `Ccedil`, …), which bounds the prefix search. */
const LEGACY_MAX_LENGTH = 6;
/**
* What a browser substitutes for ``–``: those are C1 controls in
* Unicode, but the spec reads them as windows-1252, as legacy pages meant.
*/
const C1_REPLACEMENTS: Readonly> = {
0x80: 0x20ac,
0x82: 0x201a,
0x83: 0x0192,
0x84: 0x201e,
0x85: 0x2026,
0x86: 0x2020,
0x87: 0x2021,
0x88: 0x02c6,
0x89: 0x2030,
0x8a: 0x0160,
0x8b: 0x2039,
0x8c: 0x0152,
0x8e: 0x017d,
0x91: 0x2018,
0x92: 0x2019,
0x93: 0x201c,
0x94: 0x201d,
0x95: 0x2022,
0x96: 0x2013,
0x97: 0x2014,
0x98: 0x02dc,
0x99: 0x2122,
0x9a: 0x0161,
0x9b: 0x203a,
0x9c: 0x0153,
0x9e: 0x017e,
0x9f: 0x0178,
};
const decodeNumeric = (digits: string, radix: number): string => {
const num = Number.parseInt(digits, radix);
if (num === 0 || num > 0x10ffff || (num >= 0xd800 && num <= 0xdfff)) return '\ufffd';
return String.fromCodePoint(C1_REPLACEMENTS[num] ?? num);
};
const ATTRIBUTE_BLOCKER = /[=a-zA-Z0-9]/;
/**
* Decode character references the way an HTML parser does.
*
* Decoding matters for safety, not convenience: the policy checks run on
* decoded values, so `href="javascript:alert(1)"` is compared against the
* dangerous-protocol list as `javascript:alert(1)` rather than slipping past
* as an unrecognized string. Everything is re-escaped on the way out.
*
* It follows the tokenizer's rules rather than approximating them, because
* the DOM backend gets those rules from the browser and the two must agree:
*
* - names are case-sensitive (`É` is not `é`);
* - only the legacy names decode without a `;`, and they match as a prefix,
* longest first (`¬it;` is `¬it;`);
* - `inAttribute` applies the rule that keeps query strings intact: a legacy
* name without `;` followed by `=` or an alphanumeric stays literal, so
* `href="?a=1©=2"` is not rewritten to `?a=1©=2`.
*
* Names outside `ENTITY_RUNS` stay literal — see `entities.ts` for why the
* table is a subset and why that can only fall short of a browser, never
* contradict it.
* @internal
*/
export const decodeEntities = (input: string, inAttribute = false): string => {
if (!input.includes('&')) return input;
const { named, legacy } = loadEntities();
return input.replace(
/&(?:#(?:[xX]([0-9a-fA-F]+)|([0-9]+));?|([a-zA-Z0-9]+)(;?))/g,
(match, hex: string, dec: string, name: string, semi: string, offset: number) => {
if (hex !== undefined) return decodeNumeric(hex, 16);
if (dec !== undefined) return decodeNumeric(dec, 10);
if (semi && named.has(name)) return named.get(name) as string;
for (let length = Math.min(name.length, LEGACY_MAX_LENGTH); length >= 2; length--) {
const prefix = name.slice(0, length);
if (!legacy.has(prefix)) continue;
// `name` is a maximal alphanumeric run, so when the prefix is all of it
// the next character is whatever follows the match (`;` was ruled out
// above: every legacy name also exists with one).
const next = length < name.length ? name[length] : input[offset + match.length];
if (inAttribute && next !== undefined && ATTRIBUTE_BLOCKER.test(next)) return match;
return (named.get(prefix) as string) + name.slice(length) + semi;
}
return match;
}
);
};
/** Escape a value for use inside a double-quoted attribute. */
const escapeAttribute = (value: string): string =>
value.replace(/&/g, '&').replace(/"/g, '"').replace(//g, '>');
/** Escape text content. `&` first, so escapes are not double-escaped. */
const escapeText = (value: string): string =>
value.replace(/&/g, '&').replace(//g, '>');
interface Attribute {
name: string;
value: string;
}
interface OpenTag {
kind: 'open';
tag: string;
attributes: Attribute[];
selfClosing: boolean;
}
interface CloseTag {
kind: 'close';
tag: string;
}
interface TextToken {
kind: 'text';
value: string;
}
type Token = OpenTag | CloseTag | TextToken;
const isWhitespace = (char: string): boolean =>
char === ' ' || char === '\t' || char === '\n' || char === '\r' || char === '\f';
/**
* Read the attributes of an open tag, starting just after the tag name.
* Returns the attributes and the index just past the closing `>`.
*/
const readAttributes = (
source: string,
start: number
): { attributes: Attribute[]; end: number; selfClosing: boolean } => {
const attributes: Attribute[] = [];
let pos = start;
let selfClosing = false;
while (pos < source.length) {
while (pos < source.length && isWhitespace(source[pos])) pos++;
if (pos >= source.length) break;
if (source[pos] === '>') {
pos++;
break;
}
if (source[pos] === '/' && source[pos + 1] === '>') {
selfClosing = true;
pos += 2;
break;
}
// A stray `/` inside the tag — skip it rather than reading it as a name.
if (source[pos] === '/') {
pos++;
continue;
}
const nameStart = pos;
while (
pos < source.length &&
!isWhitespace(source[pos]) &&
source[pos] !== '=' &&
source[pos] !== '>' &&
source[pos] !== '/'
) {
pos++;
}
const name = source.slice(nameStart, pos);
if (name.length === 0) {
pos++;
continue;
}
while (pos < source.length && isWhitespace(source[pos])) pos++;
let value = '';
if (source[pos] === '=') {
pos++;
while (pos < source.length && isWhitespace(source[pos])) pos++;
const quote = source[pos];
if (quote === '"' || quote === "'") {
pos++;
const valueStart = pos;
const closing = source.indexOf(quote, pos);
if (closing === -1) {
value = source.slice(valueStart);
pos = source.length;
} else {
value = source.slice(valueStart, closing);
pos = closing + 1;
}
} else {
const valueStart = pos;
while (pos < source.length && !isWhitespace(source[pos]) && source[pos] !== '>') pos++;
value = source.slice(valueStart, pos);
}
}
attributes.push({ name, value: decodeEntities(value, true) });
}
return { attributes, end: pos, selfClosing };
};
/**
* Tokenize HTML into open tags, close tags and text.
*
* Anything that is not recognizable markup — a bare `<`, an unterminated tag,
* a comment, a doctype, a processing instruction — becomes text or is
* discarded. Nothing is passed through as raw markup.
* @internal
*/
export const tokenize = (html: string): Token[] => {
const tokens: Token[] = [];
let pos = 0;
let textStart = 0;
// Lowercased once, not per raw-text element: recomputing it inside the loop
// made tokenization O(raw-text elements x input length), so a megabyte of
// `