/** * DOM-backed HTML sanitizer backend. * * Parses with `DOMParser` into an inert document, walks the tree applying the * shared policy from `sanitize-policy.ts`, and re-serializes. Used wherever a * DOM exists; `sanitize-string.ts` covers the runtimes that have none (#229). * * @module bquery/security * @internal */ import { DANGEROUS_TAGS } from './constants'; import { escapeHtmlText, isAttributeAllowed, relForAnchor, resolvePolicy, suppressesTextContent, } from './sanitize-policy'; import { decodeEntities } from './sanitize-string'; import { trustedPreparedHtmlForSink } from './trusted-types'; import type { SanitizeOptions } from './types'; /** * Parse an HTML string into a Document using DOMParser. * This helper is intentionally separated to make the control-flow around HTML parsing * explicit for static analysis tools. It should ONLY be called when the input is * known to contain HTML syntax (angle brackets). * * DOMParser creates an inert document where scripts don't execute, making it safe * for parsing untrusted HTML that will subsequently be sanitized. * * @param htmlContent - A string that is known to contain HTML markup (has < or >) * @returns The parsed Document * @internal */ const parseHtmlDocument = (htmlContent: string): Document => { const parser = new DOMParser(); // Parse as a full HTML document in an inert context; scripts won't execute. // // CodeQL reports `js/xss-through-dom` ("DOM text reinterpreted as HTML") // against this call. The flow it traces is the mutation-XSS guard in // `sanitizeHtmlDom`: that reads the sanitized fragment back out through // `innerHTML` and re-parses it here to check the markup is stable across a // second parse. That round trip is the point of the guard — the string a // caller assigns to an HTML sink has to be the string we verified — so the // flow cannot be removed without deleting the check. // // It is safe: `DOMParser.parseFromString` builds an inert document, so // nothing executes and no resource is fetched, and every node then goes // through the allow lists in `sanitize-policy.ts` before anything reaches a // caller. // // Do not try to silence it with a `// codeql[js/xss-through-dom]` comment. // One was here and did not work: GitHub code scanning does not honour // inline suppression comments, so the only effect was to suggest the alert // was handled when it was not. It is resolved by dismissing the alert in // the code-scanning UI. // // Under an enforced `require-trusted-types-for 'script'` CSP, // `parseFromString` is itself a Trusted Types sink and throws on a plain // string — which broke every sanitizer call, and with it the policy's own // `createHTML`. The input is wrapped by the same `bquery-sanitizer` policy // without a sanitizer pass (that pass is what is running here), so no extra // policy name has to be allowed in the CSP. return parser.parseFromString(trustedPreparedHtmlForSink(htmlContent), 'text/html'); }; /** * Safely parse HTML string into a DocumentFragment using DOMParser. * DOMParser is preferred over innerHTML for security as it creates an inert document * where scripts don't execute and provides better static analysis recognition. * * This function includes input normalization to satisfy static analysis tools: * - Coerces input to string and trims whitespace * - For plain text (no HTML tags), creates a Text node directly without parsing * - Only invokes DOMParser for actual HTML-like content via parseHtmlDocument * * The separation between plain text handling and HTML parsing is intentional: * DOM text that contains no HTML syntax is never fed into an HTML parser, * preventing "DOM text reinterpreted as HTML" issues. * * @internal */ const parseHtmlSafely = (html: string): DocumentFragment => { // Step 1: Normalize input - coerce to string and trim // This defensive check handles edge cases even though TypeScript says it's a string const normalizedHtml = (typeof html === 'string' ? html : String(html ?? '')).trim(); // Step 2: Create the fragment that will hold our result const fragment = document.createDocumentFragment(); // Step 3: Early return for empty input if (normalizedHtml.length === 0) { return fragment; } // Step 4: If input contains no angle brackets, it's plain text - no HTML parsing needed. // Plain text is handled as a Text node, never passed to an HTML parser. // This explicitly prevents "DOM text reinterpreted as HTML" for purely textual inputs. const containsHtmlSyntax = normalizedHtml.includes('<') || normalizedHtml.includes('>'); if (!containsHtmlSyntax) { // Decoded, because a Text node built from the raw string keeps entities as // literal characters: `Tom & Jerry` came back out of `stripTags()` with // the `&` intact, and serialization escaped it a second time, so // `sanitizeHtml()` returned `Tom &amp; Jerry`. The string backend // decodes here, which is why the two disagreed on input this branch was // added to handle. // // `decodeEntities` is a pure string transform — no parser is involved, so // the property this branch exists for still holds: text with no HTML // syntax never reaches `DOMParser`. The decoded value goes into a Text // node, where markup cannot come alive, and is re-escaped on the way out. // It is also the string backend's decoder, so this branch and that backend // agree by construction; where either stops short of `DOMParser` (a name // outside the table in `entities.ts`), the reference stays literal. fragment.appendChild(document.createTextNode(decodeEntities(normalizedHtml))); return fragment; } // Step 5: Input contains HTML syntax - parse it via the dedicated HTML parsing helper. // This separation makes the data-flow explicit: only strings with HTML syntax // are passed to DOMParser, satisfying static analysis requirements. const doc = parseHtmlDocument(normalizedHtml); // Move all children from the document body into the fragment. // This avoids interpolating untrusted HTML into an outer wrapper string. const body = doc.body; if (!body) { return fragment; } while (body.firstChild) { fragment.appendChild(body.firstChild); } return fragment; }; const TEXT_NODE = 3; const ELEMENT_NODE = 1; /** * The text of a subtree, minus the elements whose content is not prose. * * `Node.textContent` would do this in one property read, but it includes the * body of every `