/** * Strip ANSI escape sequences, remove control characters / lone surrogates, * and normalize line endings. * * Bun-native implementation of the former native `sanitizeText` (see * `crates/pi-natives/src/text.rs::sanitize_text`). JavaScript strings are * already UTF-16 code-unit arrays. `toWellFormed()` handles the uncommon * malformed path; when it changes the input, replacement characters are * dropped and the normalized result goes through the well-formed sanitizer. * * Fast path: well-formed input with no controls or ANSI returns the original * string after the control probe. */ const ESC_CHAR = "\x1b"; // Well-formed strings only need control/ANSI detection: C0 (excl. \t \n), // CR, DEL, and C1. ESC (0x1B) is in \x0B-\x1F. const CONTROL_RE = /[\x00-\x08\x0B-\x1F\x7F-\x9F]/g; const REPLACEMENT_CHAR = "\ufffd"; export function sanitizeText(text: string): string { const wellFormed = text.toWellFormed(); if (wellFormed !== text) { return sanitizeWellFormedText(wellFormed.replaceAll(REPLACEMENT_CHAR, "")); } return sanitizeWellFormedText(text); } function sanitizeWellFormedText(text: string): string { CONTROL_RE.lastIndex = 0; if (CONTROL_RE.exec(text) === null) return text; const stripped = text.indexOf(ESC_CHAR) === -1 ? text : Bun.stripANSI(text); CONTROL_RE.lastIndex = 0; return stripped.replace(CONTROL_RE, ""); } /** * Escape the three XML-significant characters (`&`, `<`, `>`) in text destined * for an XML/markup element body. Allocation-conscious: returns the input * unchanged (same reference) when nothing needs escaping. Quotes are left as-is * — use it for element text, not attribute values. */ export function escapeXmlText(input: string): string { let firstEscapable = -1; for (let index = 0; index < input.length; index++) { const char = input.charCodeAt(index); if (char === 38 || char === 60 || char === 62) { firstEscapable = index; break; } } if (firstEscapable === -1) return input; let output = input.slice(0, firstEscapable); for (let index = firstEscapable; index < input.length; index++) { const char = input[index]; if (char === "&") output += "&"; else if (char === "<") output += "<"; else if (char === ">") output += ">"; else output += char; } return output; } /** * Escape XML-significant characters for an attribute VALUE: the three body * characters (`&`, `<`, `>`) plus the double quote (`"` → `"`) that would * otherwise close the attribute. Allocation-conscious: returns the input * unchanged (same reference) when nothing needs escaping. Use it for attribute * values; {@link escapeXmlText} is for element bodies and leaves `"` intact. */ export function escapeXmlAttribute(input: string): string { let firstEscapable = -1; for (let index = 0; index < input.length; index++) { const char = input.charCodeAt(index); if (char === 38 || char === 60 || char === 62 || char === 34) { firstEscapable = index; break; } } if (firstEscapable === -1) return input; let output = input.slice(0, firstEscapable); for (let index = firstEscapable; index < input.length; index++) { const char = input[index]; if (char === "&") output += "&"; else if (char === "<") output += "<"; else if (char === ">") output += ">"; else if (char === '"') output += """; else output += char; } return output; }