/**
* DOM-free HTML sanitizer backend.
*
* `sanitizeHtml()` used to require `document` and `DOMParser`, so it threw on
* every server runtime — in a framework that advertises runtime-agnostic SSR
* and sanitizes DOM writes by default, the sanitizer was the one piece that
* needed a browser (#229).
*
* This backend parses and re-serializes HTML with a small scanner instead, so
* it runs anywhere. It shares every policy decision with the DOM backend via
* `sanitize-policy.ts`; only the parsing and serialization differ.
*
* Two notes on the approach, both deliberate:
*
* - **It is a scanner, not an HTML5 tokenizer.** Where a browser would apply
* error recovery, this backend prefers to drop or escape. That can differ
* from the DOM backend on malformed input, and the difference always lands
* on the safe side: output is escaped rather than guessed at.
* - **Output is re-parsed and compared**, exactly as the DOM backend does, so
* markup that changes shape on a second pass (mXSS) falls back to escaped
* text rather than being emitted.
*
* `src/ssr/html-parser.ts` solves a similar problem, but `security` is a leaf
* module that `ssr` itself imports, so reusing it would invert the dependency
* direction. The policy is shared; the scanner is not.
*
* @module bquery/security
* @internal
*/
import type { SanitizeOptions } from './types.cjs';
/**
* Decode character references the way an HTML parser does.
*
* Decoding matters for safety, not convenience: the policy checks run on
* decoded values, so `href="javascript:alert(1)"` is compared against the
* dangerous-protocol list as `javascript:alert(1)` rather than slipping past
* as an unrecognized string. Everything is re-escaped on the way out.
*
* It follows the tokenizer's rules rather than approximating them, because
* the DOM backend gets those rules from the browser and the two must agree:
*
* - names are case-sensitive (`É` is not `é`);
* - only the legacy names decode without a `;`, and they match as a prefix,
* longest first (`¬it;` is `¬it;`);
* - `inAttribute` applies the rule that keeps query strings intact: a legacy
* name without `;` followed by `=` or an alphanumeric stays literal, so
* `href="?a=1©=2"` is not rewritten to `?a=1©=2`.
*
* Names outside `ENTITY_RUNS` stay literal — see `entities.ts` for why the
* table is a subset and why that can only fall short of a browser, never
* contradict it.
* @internal
*/
export declare const decodeEntities: (input: string, inAttribute?: boolean) => string;
interface Attribute {
name: string;
value: string;
}
interface OpenTag {
kind: 'open';
tag: string;
attributes: Attribute[];
selfClosing: boolean;
}
interface CloseTag {
kind: 'close';
tag: string;
}
interface TextToken {
kind: 'text';
value: string;
}
type Token = OpenTag | CloseTag | TextToken;
/**
* Tokenize HTML into open tags, close tags and text.
*
* Anything that is not recognizable markup — a bare `<`, an unterminated tag,
* a comment, a doctype, a processing instruction — becomes text or is
* discarded. Nothing is passed through as raw markup.
* @internal
*/
export declare const tokenize: (html: string) => Token[];
/**
* Sanitize HTML with no DOM available.
*
* Mirrors the DOM backend's contract, including its mutation-XSS guard: the
* output is sanitized a second time and, if the two passes disagree, the
* escaped text content is returned instead of markup whose meaning changes
* when it is re-parsed.
* @internal
*/
export declare const sanitizeHtmlString: (html: string, options?: SanitizeOptions) => string;
/** Plain-text extraction with no DOM. @internal */
export declare const stripTagsString: (html: string) => string;
export {};
//# sourceMappingURL=sanitize-string.d.ts.map