/** * Strip ANSI escape sequences to a fixed point. Removing one sequence can * reconstitute another around it (a lone ESC left of `ESC[32m[0m` gains the * trailing `[0m` once the inner sequence is removed, forming a brand-new valid * sequence the single pass would miss), so iterate — but only a fixed few * times. Bounding by the input's ESC count is quadratic on attacker-controlled * text: `("\x1b[").repeat(n) + "m".repeat(n)` reconstitutes exactly ONE * sequence per pass, so n full O(n) scans run, and 48 KB already costs seconds. * The passes are not what makes Layer 1 safe — applyLayer1's residual * CONTROL_INTRODUCER_RE sweep is, and it removes every ESC/C1 byte whatever * survives here. Past the bound a reconstituted sequence therefore degrades to * VISIBLE text rather than a hidden control, which is the fail-open direction. * @param {string} input * @param {Set} [kinds] see {@link stripAnsiOnce}; accumulates across passes * @returns {string} */ export function stripAnsiFully(input: string, kinds?: Set): string; /** * True when the ANSI a Layer-1 strip removed was INERT: every removed sequence * was either a display-only SGR colour token or a LONE 7-bit `ESC` that opened * nothing at all (a stray byte in a file, a truncated write, a log fragment cut * mid-escape). * * The two other orphan kinds are deliberately NOT inert. A raw C1 orphan * (TOKEN_KIND.ORPHAN_C1): legit UTF-8 text does not carry raw C1 bytes, so an * unrecognized one means a terminal may still act on what follows it. An * incomplete CSI (TOKEN_KIND.ORPHAN_CSI) * for the same reason at 7 bits: the CSI parser keeps consuming until a final * byte, so `hello ESC[12 world` hides ` w` from the human while the model reads * the whole prompt. * * This draws a severity line, not a presence line: the bytes are stripped * either way, so all that rides on the answer is whether the operator sees a * WARNING or a terse note. An orphan introducer cannot move the cursor, erase * the screen, relabel a window, or open an OSC string — every one of those needs * a COMPLETE token, which {@link scanAnsi} classifies as CSI or OSC and this * rejects. Warning on a lone `ESC` is the false positive that costs the most: one * pre-existing `ESC` in a markdown file, echoed back in an Edit result, raises * the same alarm as a cursor-spoofing payload, and an alarm that fires on inert * bytes is the one operators learn to scroll past. * * It takes the kinds the STRIP recorded, never a fresh scan of the raw text, * and that is the whole point: a scan of the raw text answers about sequences * that have not been reconstituted yet, so `ESC` + `ESC[m` + `[2J` (a bare ESC, * an SGR, then plain text) reads as orphan-only there while the strip's second * pass actually removes a CSI erase. Recording what each pass removed reports * the sequences that really existed at Layer 1's fixed point. * @param {readonly string[] | Set} kinds {@link TOKEN_KIND} values removed * @returns {boolean} */ export function isBenignAnsiKinds(kinds: readonly string[] | Set): boolean; /** * True when the SGR sequences in `text` do more than style visible text — the * question a consumer must answer before PRESERVING escapes rather than * stripping them. Two arms, both model-sees/human-sees divergences: * * 1. CONCEAL is on while text that would otherwise render goes by, or the text * ENDS concealed (hiding its own tail, and everything the terminal prints * afterwards). Only SGR moves that state (see {@link sgrConcealState}) — a * cursor move or an erase leaves a terminal exactly as concealed as it was. * 2. {@link SGR_RUN_THRESHOLD} sequences with nothing that renders between * them: the run puts no glyph on the screen, so it carries data rather than * styling. * * What it does NOT claim: a message spread thinly through legitimately coloured * text — one sequence per line, say — is indistinguishable from styling, and * this answers no for it. That is the same bargain the invisible layer's * thresholds strike, and the same direction: an unprovable payload is left alone * rather than costing every colourized file its colour. * * Scans the RAW text in one pass, which is what a terminal does with these * bytes — no reconstitution across passes, because a preserved byte is never * removed to complete a sequence around it. * @param {string} text * @returns {boolean} */ export function sgrCarriesPayload(text: string): boolean; /** * {@link isBenignAnsiKinds} for callers that hold only the text — it runs the * full Layer-1 composition to get the fixed-point view. Callers that already * ran {@link applyLayer1} must read its `ansiKinds` instead of paying for a * second strip. * @param {string} text * @returns {boolean} */ export function isBenignAnsi(text: string): boolean; /** * Layer 1: ANSI + invisible-char strip with a result guaranteed free of every * raw ANSI control introducer (7-bit ESC U+001B and the whole 8-bit C1 control * block U+0080–U+009F: CSI, the DCS/SOS/OSC/PM/APC string introducers, and ST). * * The two passes FEED each other in both directions, so neither ordering is * enough on its own and the composition is iterated to a fixed point instead: * removing an invisible char reconstitutes an escape its split hid * (`ESC``[32m` → `ESC[32m`), and removing an escape makes two invisibles * ADJACENT that were not (`م ZWJ ESC[m ZWJ م` → a joiner run the invisible pass * classifies as a payload channel rather than linguistic). A pipeline with a * fixed number of alternations always leaves one of those unanswered — this used * to re-strip ANSI after the invisible pass and stop, so `applyLayer1` was not * idempotent and the rehydrator's "re-cleaning reproduces the view" assumption * did not hold. * * The residual sweep runs only once the composition is STABLE, because sweeping * an introducer early destroys the sequence a later ANSI pass would have removed * whole, promoting a hidden control to visible text. A final UNCONDITIONAL sweep * follows the loop so the no-raw-introducer guarantee does not depend on the * pass bound. * * `deAnsi` is the ANSI strip of the ORIGINAL text (invisible runs intact), the * scope a LONG_RUN payload check needs — not an intermediate of the loop. * * `ansiKinds` is the {@link TOKEN_KIND} of every ANSI sequence the composition * removed, deduped — the severity detail `found`'s single ANSI category cannot * carry (see {@link isBenignAnsiKinds}). It is also what DERIVES that category: * a kind is recorded exactly when bytes were removed, so "we reported ANSI" and * "here is what the ANSI was" can no longer disagree. * @param {string} text * @returns {{ cleaned: string, deAnsi: string, found: string[], ansiKinds: string[] }} */ export function applyLayer1(text: string): { cleaned: string; deAnsi: string; found: string[]; ansiKinds: string[]; }; /** * Map every lone UTF-16 surrogate to U+FFFD. Load-bearing on ANY path that feeds * text to an injected redactor: a secret split by an interposed lone surrogate * reads as adjacent to a model rendering its own UTF-16 but as broken to a * redactor (Node maps the lone surrogate to U+FFFD en route), so a secret * reconstituted across the surrogate survives redaction unless the text is * normalized first. It also keeps an HTML tokenizer from throwing on a stray * code unit. The substitution is same-LENGTH — one UTF-16 unit for one — so * offsets computed against the un-normalized string stay valid against this one. * @param {string} text * @returns {string} */ export function normalizeLoneSurrogates(text: string): string; /** * {@link applyLayer1} followed by {@link normalizeLoneSurrogates} — the exact * composition `output.mjs`'s processLayer1 runs before Layers 2+ see the text, * so this is the string a model is actually shown. * * Take this one, not `applyLayer1`, whenever the result feeds a redactor, an * offset calculation or a view map: those consumers compare their own string to * the model-facing view, and an unpaired surrogate is exactly where the two * spellings diverge. Take `applyLayer1` when you want Layer 1's removals alone * and intend to hand a possibly ill-formed string onward unchanged. * * `found` gains {@link CATEGORY.LONE_SURROGATES} exactly when the normalization * changed the text, matching what the pipeline reports for the same input. * `deAnsi` is untouched: it is the ANSI strip of the ORIGINAL text, the scope * the long-run payload check needs. * @param {string} text * @returns {{ cleaned: string, deAnsi: string, found: string[], ansiKinds: string[] }} */ export function applyLayer1WellFormed(text: string): { cleaned: string; deAnsi: string; found: string[]; ansiKinds: string[]; }; export const LONE_SURROGATE_RE: RegExp; /** * What a reader is told when the ONLY thing a strip removed was inert ANSI (see * {@link isBenignAnsiKinds}). * * It lives here, beside the predicate that decides it, because every entry point * that can reach that verdict must say the same thing: the tool-output pipeline, * the prompt gate's pass-with-note, and any host wiring its own. The wording is * deliberately not `describeStripped`'s — "Stripped: ANSI escapes" names a * category that reads like an attack, when the honest report is "these were * colour codes, and here is how to look at the raw bytes". */ export const INERT_ANSI_NOTE: string; /** * How many SGR sequences may sit back to back, with nothing that RENDERS between * them, before the run is read as a covert channel rather than styling. * * A styling emitter puts colour AROUND text: `ls --color`, chalk and pygments * chain at most a handful of attributes before the glyphs they style. A run of * escape sequences that puts no glyph on the screen renders as literally nothing * to a human while a model reads every byte, so past some length the run is not * styling any more — it is a message, in the same shape (and for the same * reason) as the invisible layer's LONG_RUN_THRESHOLD. Ten matches that * threshold and sits well above what a real emitter chains. */ export const SGR_RUN_THRESHOLD: 10;