/** * True for either orphan kind — the tokens {@link scanAnsi} emits for an * introducer that completes no sequence, which the stripper must leave in place * for the residual sweep rather than splice (see stripAnsiOnce). * @param {string} kind one of {@link TOKEN_KIND} * @returns {boolean} */ export function isOrphanKind(kind: string): boolean; /** * The orphan kind for the introducer character `ch`, given the character `next` * that follows it: a raw C1 byte, an `ESC` that opened an incomplete CSI, or a * lone `ESC`. The one place that split is decided, shared by the tokenizer and * by Layer 1's residual sweep (which sees bare characters, not tokens). * * `[` is the only lookahead that matters. The five string introducers — `ESC ]`, * `ESC P`, `ESC X`, `ESC ^`, `ESC _` — are consumed whole by * {@link scanControlString}, to the end of input if unterminated, so they never * reach here; every other second byte (`ESC (`, `ESC #`) bounds what a terminal * swallows to a byte or two rather than running until a final byte arrives. * @param {string} ch * @param {string} [next] the following character, or undefined at end of input * @returns {string} */ export function orphanKindFor(ch: string, next?: string): string; /** * Tokenize every raw control introducer in `text`. * * Every introducer yields exactly one token — an orphan kind (see * {@link orphanKindFor}) when it starts nothing the grammar recognizes — so * "which introducers are in this text" and "which * sequences are in this text" are answered by the same scan. That is what lets * the stripper (splice every non-orphan token, then sweep) and the SGR-only * predicate (every token is SGR) agree by construction. * * Tokens are disjoint and ordered by `start`; each `end` is strictly greater * than its `start`, so the scan always advances. * @param {string} text * @returns {AnsiToken[]} */ export function scanAnsi(text: string): AnsiToken[]; /** * The CONCEAL state after the SGR token `token` is applied to a terminal already * in state `concealed` — the state in which a terminal renders what follows as * blank while its bytes stay readable to anything reading the file. * * A TRANSITION, not a property of the token: conceal is terminal state that * outlives the sequence that set it, so `ESC[8m` followed by `ESC[31m` is still * concealed — a per-token predicate would read the second token as "no conceal * here" and lose the state. Only `8` (set), `28` (reveal) and `0` (reset all) * move it, and the token's parameters are applied in order, so a later reveal or * reset in the SAME token cancels an earlier `8`: `ESC[8;28;31m` renders red and * visible. * * The parameters are read the way a terminal reads them rather than scanned for * the digit. `38`/`48`/`58` take their colour arguments from the parameters that * FOLLOW them in the semicolon form, so `ESC[38;5;8m` is bright-black foreground * and its `8` is a palette index; a parameter carrying its arguments as ITU T.416 * sub-parameters (`38:5:8`) is self-contained, so only its head counts and * nothing after it is consumed. A colour-space selector T.416 gives no argument * count for is malformed, and a terminal that ignores the colour form still * applies what follows it, so the scan CONTINUES from the next parameter rather * than consuming arguments it cannot size — reading one parameter too many costs * a token shape no emitter produces, while stopping there would hand * `ESC[38;9;8m` a pass. * @param {string} token a token {@link scanAnsi} classified {@link TOKEN_KIND.SGR} * @param {boolean} concealed the state before this token * @returns {boolean} */ export function sgrConcealState(token: string, concealed: boolean): boolean; /** * The ONE ANSI grammar: the raw control-introducer charset and the tokenizer * every consumer scans with. * * Two modules need this grammar and they cannot import each other — * `layer1.mjs` imports `invisible.mjs`, so `invisible.mjs` (which owns the * public `isSgrOnly` / `SGR_RE`) must not import back. Before this module the * grammar was therefore written out twice with DIFFERENT param rules * (`invisible.mjs`'s SGR regex accepted any digit run, `layer1.mjs`'s CSI * branch capped each parameter at four digits), and the introducer charset * three times. The looser copy suppressed the operator warning for a sequence * the stripper could not match: `ESC[12345m` read as "display-only colour" * while `[12345m` was spliced into the model's view as visible text. One * tokenizer, one charset, consumed by both — the disagreement cannot recur. * * Same precedent (and same reason) as `cf-charset.mjs`: a dependency-free leaf * module both layers read from. */ export const CONTROL_INTRODUCER_CODEPOINTS: readonly number[]; export const CONTROL_INTRODUCER_SOURCE: string; /** * Public alias kept for compatibility (re-exported by `invisible.mjs` and the * package root). It is now DERIVED: {@link scanAnsi} classifies a token as SGR * by testing the token's own text against this exact source, so the predicate * and the regex can no longer describe different languages. */ export const SGR_RE: RegExp; /** * The same grammar {@link scanAnsi} implements, as a REGEX SOURCE — the shipped * artifact for a consumer that cannot run this module. * * The scanner below is AUTHORITATIVE and this is derived from its own constants, * never the other way round: the scanner emits token KINDS a regex cannot, and * it is linear by construction where the regex form has to carry an explicit * guard to stay linear (see the CSI arm's lookahead). What a regex CAN be is data — * a stdlib-only Python filter on an uncontrolled host, with no install path for * this package, can read a pattern string but cannot import a tokenizer. So the * generator pins this into `data/invisible-charset.json` beside the introducer * set, `agent_sanitizer.textstrip` compiles it, and the two ports stop being two * hand-written spellings of one grammar. * * Every construct here is common to JS and Python `re` with NO flags — * `\uXXXX`, `(?:)`, `(?=)`, `(?!)`, and `(?![\s\S])` for end-of-input (Python's * `$` also matches before a trailing newline, JS's does not; `\Z` is Python-only) * — so ONE pattern string is what both engines read. * `test/ansi-pattern-parity.test.mjs` runs it against the scanner over a fuzz * corpus; `tests/test_textstrip.py` asserts it compiles under plain `re`. */ export const ESCAPE_SEQUENCE_SOURCE: string; /** The seven things an introducer can turn out to be. */ export const TOKEN_KIND: Readonly<{ /** A display-only `ESC[…m` / `U+009B…m` colour sequence. */ SGR: "sgr"; /** Any other complete CSI / two-byte escape (cursor move, erase, charset). */ CSI: "csi"; /** An OSC string: introducer, body and terminator as one unit. */ OSC: "osc"; /** * One of the other four ECMA-48 control strings — DCS, SOS, PM or APC — * introducer, body and terminator as one unit, exactly like * {@link TOKEN_KIND.OSC}. Split from it only so a warning can name what it * found; both are payload-carrying strings and neither is benign. */ CONTROL_STRING: "control-string"; /** * A 7-bit `ESC` that starts no sequence the grammar recognizes — a truncated * write, a log fragment cut mid-escape, a stray byte living in a file. */ ORPHAN: "orphan-introducer"; /** * A 7-bit `ESC` that OPENS a CSI (`ESC [`) it never completes. Split from * {@link TOKEN_KIND.ORPHAN} because a terminal's CSI parser is STATEFUL: it * keeps consuming what follows as parameters and intermediates until a final * byte (0x40-0x7E) arrives, so `hello ESC[12 world` renders as `hello orld` * — the ` w` is eaten as the sequence's intermediate and final. That is the * model-sees/human-sees divergence the gate exists for, so consumers that * downgrade an inert strip to a note must keep warning on this one; only a * lone `ESC` that opens nothing is inert. */ ORPHAN_CSI: "orphan-csi-introducer"; /** * A RAW C1 byte (U+0080-U+009F) that starts no sequence the grammar * recognizes. Split from {@link TOKEN_KIND.ORPHAN} because the two carry very * different weight: a lone `ESC` is ordinary debris in terminal output, while * a raw C1 byte is not something legitimate UTF-8 text produces. The five * string introducers in the block open a {@link TOKEN_KIND.OSC} or * {@link TOKEN_KIND.CONTROL_STRING} token instead, so a byte that reaches * here is one the grammar recognizes no sequence for at all — and a terminal * may still act on it. Consumers that downgrade an inert strip to a note (see * `isBenignAnsiKinds` in ./layer1.mjs) must keep warning on this one. */ ORPHAN_C1: "orphan-c1-introducer"; }>; export type AnsiToken = { /** * Index of the introducer. */ start: number; /** * Index one past the last character of the token. */ end: number; /** * One of {@link TOKEN_KIND}. */ kind: string; };