/** * Wrapper interpretation for a bash command unit: what kind of wrapper it is, * what it actually runs, and whether its floor still has a reason to hold. * * Pure and word-based; the AST walk that produces the words lives in * `command-enumeration.ts`. The three questions live here together * deliberately: the shape that floors a unit to `ask`, the shape that names its * inner command, and the shape that exempts it must agree, and separate * classifiers over the same vocabulary would drift. */ import type { FloorExemption } from "#src/types"; import { proveCommandEffect } from "./command-effects"; import type { ArgWord } from "./node-text"; /** * One word of a command unit: its source text and its offset into the unit's * text, beside the value the program receives once the shell removes quotes. * * The source text is what unwrapping and slicing read; the value is what a * capability proof reads, since `'-o'` reaches `sort` as `-o`. */ export interface CommandWord extends ArgWord { readonly text: string; readonly offset: number; } /** * Why a command unit's decision is floored to at least `ask`. * `"opaque-payload"` — an inline-shell payload (`bash -c`/`eval`) whose inner * program is not re-parsed (#481). * `"indirection"` — a prefix/exec wrapper (`sudo`/`env`/`xargs`/`find -exec`/…) * whose inner command is a visible argument but is not gated on its own (#490). * The kind selects the audit sentinel; both floor identically. */ export type WrapperKind = "opaque-payload" | "indirection"; /** * Classify a command unit's words as a floored wrapper, or `undefined` for an * ordinary command. `words[0]` is the command name; a leading * `variable_assignment` prefix is already stripped by the caller. The command * name is matched on its basename, so `/bin/bash -c …` counts. * * `"opaque-payload"`: `eval`, or a shell (`bash`/`sh`/`dash`/`zsh`/`ksh`) with a * `-c` short-flag cluster (`-c`, `-ec`, `-xc`) — the inner program is a quoted * argument the enumerator does not re-parse (#481). * * `"indirection"`: an always-invoking prefix/exec wrapper * ({@link INDIRECTION_WRAPPER_NAMES}), or a search tool * ({@link EXEC_CONDITIONAL_WRAPPERS}, `find`/`fd`) carrying a per-result exec * flag — the inner command is a visible argument that a ` *` rule would * otherwise never match (#490). A bare `find`/`fd` search runs no subcommand and * is not flagged. */ export function classifyWrapperWords( words: readonly CommandWord[], ): WrapperKind | undefined { const commandName = wrapperName(words); if (commandName === undefined) return undefined; const args = words.slice(1).map((word) => word.text); if (commandName === "eval") return "opaque-payload"; if (SHELL_WRAPPER_NAMES.has(commandName) && hasShortFlagC(args)) { return "opaque-payload"; } if (INDIRECTION_WRAPPER_NAMES.has(commandName)) return "indirection"; if (execFlagIndex(commandName, args) !== -1) return "indirection"; return undefined; } /** * Index within `words` of the inline-shell payload — the inner program a shell or * `eval` runs — or `-1` when the unit carries none. * * Indirection layers are peeled first, so `sudo bash -c '…'` answers the * payload's index rather than `-1`. That is not a convenience: * {@link executedUnitOf} peels them too, so a consumer that did not would mask a * secret under `executedUnit` and write it verbatim under `command` — the * inconsistency #923 reports, reintroduced one wrapper layer up. * * The index names the payload's *position*, which a vacant one still has * (`bash -c`), so each caller decides for itself what reading past the end of * `words` is worth. */ export function inlineShellPayloadIndex(words: readonly CommandWord[]): number { let base = 0; let current = words; for (let depth = 0; depth < MAX_UNWRAP_DEPTH; depth++) { const direct = directPayloadIndex(current); if (direct !== -1) return base + direct; if (classifyWrapperWords(current) !== "indirection") return -1; const start = innerCommandIndex(current); if (start === -1 || start >= current.length) return -1; const end = execTerminatorIndex(current, start); base += start; current = current.slice(start, end); } return -1; } /** * The payload index of a unit that is *already* the shell or `eval` running it. * * `eval` takes its program as the first argument (no `-c`, so the flag scan * answers -1 and the index falls out as 1); a shell takes it after the `-c` * cluster. Any other command name carries no inline program at all, which is * what keeps an interpreter (`python3 -c`, `node -e`) out: its payload is * another language, not shell. */ function directPayloadIndex(words: readonly CommandWord[]): number { const commandName = wrapperName(words); if (commandName === undefined) return -1; const isShell = SHELL_WRAPPER_NAMES.has(commandName); if (commandName !== "eval" && !isShell) return -1; const flagIndex = shortFlagCIndex(words.slice(1).map((word) => word.text)); if (isShell && flagIndex === -1) return -1; return flagIndex + 2; } // ── Wrapper vocabulary ─────────────────────────────────────────────────────── /** * The command a wrapper unit actually runs, or `null` when it cannot be * established or adds nothing over the unit itself. * * Display-only (ADR 0011 §3.5, #713): the result names what runs, including * the payload of an inline shell, so it deliberately looks *past* a `sh -c` * layer that the gate must not look past. Because it is shown on a decision * surface, the rule is to fail to `null` rather than to a guess — an * unrecognized option shape yields nothing rather than a remainder that might * name the wrong command. * * Nested wrappers unwrap to the innermost command (`sudo timeout 5 xargs grep * foo` → `grep foo`), bounded by {@link MAX_UNWRAP_DEPTH}. */ export function executedUnitOf( unitText: string, words: readonly CommandWord[], ): string | null { const unwrapped = unwrapIndirection(words, unitText); const text = unwrapped.kind === "opaque" ? unwrapped.payload : unwrapped.text; return nothingNew(text, unitText); } /** * Why a wrapper unit's floor has no reason left to hold, or `undefined` when it * still does. * * `"core-reader"`: the command it runs is in the pure-reader core, so its * *direction* is provable however unknown its argument feed is (ADR 0013 §11, * #803). * * The floor exists because a wrapper hides the command that should be gated, * and the unknowability it guards is unknowability of scope — which stays the * projection's and the path surfaces' job, for wrapped and unwrapped commands * alike. Argument-independence is the core's admission bar, so there are no * arguments that make `grep` write a file. * * Four things must hold for it, and each is a way the reason could still hold: * * 1. The unit is an indirection wrapper — an ordinary command has no floor. * 2. Unwrapping reached the inner command without passing through an opaque * payload. `sh -c '…'` carries an unparsed program whose first word says * nothing about the rest of it, which is why this cannot consult * {@link executedUnitOf}'s string. * 3. The inner command *proves* a read — bare-basename core word, retraction * guards applied, so `xargs sort -o /tmp/x` and `xargs find . -delete` are * not transparent. * 4. The unit writes no file through a redirect, which the caller reads off * the parse tree and this module never sees. * * `"execution-modifier"`: every wrapper layer only changes *how* the visible * inner command runs, so the unit resolves by that command's own rule whatever * it does (see {@link onlyModifiesExecution}). The first two conditions above * hold for it too; the core-reader reason is preferred when both apply. */ export function floorExemptionOf( words: readonly CommandWord[], statement: { readonly writesViaRedirect: boolean }, ): FloorExemption | undefined { if (classifyWrapperWords(words) !== "indirection") return undefined; // Only the peeled words matter here, so the walk is handed no source span to // cut from — the text slice is `executedUnitOf`'s product, not this one's. const unwrapped = unwrapIndirection(words, ""); if (unwrapped.kind === "opaque" || unwrapped.peeled.length === 0) { return undefined; } const head = unwrapped.words.at(0)?.text ?? ""; const provesRead = proveCommandEffect(head, unwrapped.words.slice(1)).effect === "read"; if (provesRead && !statement.writesViaRedirect) return "core-reader"; return onlyModifiesExecution(unwrapped.peeled, unwrapped.words) ? "execution-modifier" : undefined; } /** * True when every peeled layer only changes *how* the inner command runs, and * that command is a literal name the gate can resolve by its own rule. * * The floor's reason (a wrapper hides the command that should be gated) is * false for an execution modifier whatever the inner command does: every * operand is on the command line, and the wrapper adds no privilege, * environment, or argument feed. So the unit inherits the inner verdict rather * than being classified a read, which is why no redirect refusal applies — * the destination is gated by the path surfaces exactly as for the bare * command. * * Each condition is a way the inherited verdict could name the wrong command: * * 1. Every layer is a modifier. The peel looks through `sudo` and `env` too, * so `time sudo rm` would otherwise inherit `rm`'s verdict. * 2. Every option in every layer is on that modifier's allowlist. The real * tools accept abbreviations (`timeout --sig KILL 5 rm`) the inner-command * search does not know, and one of them misplaces where the command starts. * Each option, value, and operand must also be literal: the shell splits * `timeout $D pnpm` or expands `timeout {5,sudo} rm` into extra words before * the modifier runs, and one of those may be a wrapper or the real command. * 3. The peel ended at an ordinary command, not a wrapper it could not see * past. * 4. The inner head is a literal command name. The grammar has no `time` * keyword, so `time { …; }` and `time ( … )` reach here with shell syntax * where the command name should be. */ function onlyModifiesExecution( peeled: readonly (readonly CommandWord[])[], inner: readonly CommandWord[], ): boolean { return ( peeled.every(isAdmittedModifierLayer) && classifyWrapperWords(inner) === undefined && isLiteralCommandName(inner.at(0)?.text ?? "") ); } /** * True when a peeled layer is an execution modifier whose every word before * the inner command is one it admits. * * Walks the words with the same value-taking table {@link innerCommandIndex} * skips by, so an admitted option's value can never be taken for the command. */ function isAdmittedModifierLayer(layer: readonly CommandWord[]): boolean { const name = wrapperName(layer); const flags = name === undefined ? undefined : EXECUTION_MODIFIER_FLAGS.get(name); if (name === undefined || flags === undefined) return false; const valueTaking = admittedValueTaking(name); for (let index = 1; index < layer.length; index++) { const word = layer[index].text; if (isEnvironmentAssignment(word)) continue; // A word the shell rewrites (`$D`, `{5,sudo}`, `*`) may become several, one // of them a wrapper or the real command, so only a literal word is admitted. if (layer[index].computed) return false; if (word === "--") continue; if (!word.startsWith("-")) continue; if (flags.has(word)) continue; if (valueTaking.has(word)) { index++; if (layer[index]?.computed) return false; continue; } if (!hasAttachedValue(word, valueTaking)) return false; } return true; } /** A modifier's value-taking options, less any that write a file. */ function admittedValueTaking(name: string): ReadonlySet { const valueTaking = VALUE_TAKING_FLAGS.get(name) ?? EMPTY_FLAGS; const writing = WRITING_OPTIONS.get(name) ?? EMPTY_FLAGS; return new Set([...valueTaking].filter((flag) => !writing.has(flag))); } /** `--long=value`, or a short option with its value attached (`-sKILL`). */ function hasAttachedValue( word: string, valueTaking: ReadonlySet, ): boolean { if (word.startsWith("--")) { const equals = word.indexOf("="); return equals !== -1 && valueTaking.has(word.slice(0, equals)); } return word.length > 2 && valueTaking.has(word.slice(0, 2)); } /** A command name spelled literally: no quoting, expansion, or shell syntax. */ function isLiteralCommandName(text: string): boolean { return LITERAL_COMMAND_NAME.test(text) && !RESERVED_WORDS.has(text); } // ── Unwrapping ─────────────────────────────────────────────────────────────── /** * How an unwrap ended: at the innermost command reachable by peeling * indirection layers, or at an inline-shell payload. * * The two are kept apart because they are different kinds of answer. A peeled * result is a slice of this command line, with words the caller can inspect; * a payload is an inner *program* the enumerator never parsed, so it has text * and nothing else. Collapsing them is how a core-looking first word inside * `sh -c '…'` would come to stand for the whole payload (#803). */ type UnwrapResult = | { readonly kind: "opaque"; readonly payload: string | null } | { readonly kind: "peeled"; readonly text: string; readonly words: readonly CommandWord[]; /** * Each peeled layer's words before its inner command: the wrapper name, * its options, and any leading operand. Empty when none came off. */ readonly peeled: readonly (readonly CommandWord[])[]; }; /** * Peel indirection layers off a command unit until an ordinary command, an * opaque payload, or an unrecognized option shape stops the walk. * * Stopping early is not an error: the words peeled so far are returned, and * each caller decides what an incomplete peel is worth — `executedUnitOf` * shows it, {@link floorExemptionOf} declines it because the head word it * would judge is the wrapper's own. * * `unitText` is the span the peeled `text` is cut from; a caller that wants * only the words passes the empty string and ignores it. */ function unwrapIndirection( words: readonly CommandWord[], unitText: string, ): UnwrapResult { let text = unitText; let current = words; const peeled: (readonly CommandWord[])[] = []; for (let depth = 0; depth < MAX_UNWRAP_DEPTH; depth++) { const kind = classifyWrapperWords(current); if (kind === undefined) break; if (kind === "opaque-payload") { // The payload is an inner *program*, not a slice of this command line, so // it is unquoted and terminal — unwrapping it further would need a parse. return { kind: "opaque", payload: opaquePayload(current) }; } const start = innerCommandIndex(current); if (start === -1 || start >= current.length) break; const end = execTerminatorIndex(current, start); text = sliceWords(text, current, start, end).trimEnd(); peeled.push(current.slice(0, start)); current = rebase(current, start, end); } return { kind: "peeled", text, words: current, peeled }; } /** How many wrapper layers to unwrap before giving up. */ const MAX_UNWRAP_DEPTH = 4; /** * The extracted text, or `null` when it establishes nothing new — it is absent * or empty, it still begins with an option (so the inner command was never * reached), or it simply repeats the unit. */ function nothingNew(text: string | null, unitText: string): string | null { if (text === null || text === "" || text === unitText) return null; return text.startsWith("-") ? null : text; } /** * The inline-shell payload argument, unquoted; `null` when absent. * * Reads the **direct** index: {@link unwrapIndirection} has already peeled every * wrapper layer by the time it reaches its opaque branch, so peeling again would * look past a shell that is itself an outer wrapper's payload. */ function opaquePayload(words: readonly CommandWord[]): string | null { const index = directPayloadIndex(words); const payload = index === -1 ? undefined : words.at(index); return payload === undefined ? null : unquote(payload.text); } /** Strip one matching pair of surrounding quotes. */ function unquote(text: string): string { const first = text.at(0); const quoted = (first === "'" || first === '"') && text.length >= 2 && text.endsWith(first); return quoted ? text.slice(1, -1) : text; } /** * Index of the word beginning the inner command, or `-1` when the wrapper's own * options run out first. * * Skips the wrapper name, environment assignments, options (consuming a * following value for the options in {@link VALUE_TAKING_FLAGS}), and a leading * operand for the wrappers that take one. An exec-conditional wrapper instead * starts immediately after its exec flag. */ function innerCommandIndex(words: readonly CommandWord[]): number { const name = wrapperName(words); if (name === undefined) return -1; const argTexts = words.slice(1).map((word) => word.text); const execFlag = execFlagIndex(name, argTexts); if (execFlag !== -1) return execFlag + 2; const valueTaking = VALUE_TAKING_FLAGS.get(name) ?? EMPTY_FLAGS; let operandPending = LEADING_OPERAND_WRAPPERS.has(name); let index = 1; while (index < words.length) { const word = words[index].text; if (word === "--") { // The end of options, not of operands: `timeout -- 5 cmd` still takes // its duration before the command. return operandPending ? index + 2 : index + 1; } if (isEnvironmentAssignment(word)) { index++; continue; } if (word.startsWith("-")) { index += valueTaking.has(word) ? 2 : 1; continue; } if (operandPending) { operandPending = false; index++; continue; } return index; } return -1; } /** * Index of an exec wrapper's `;`/`+` terminator, or `words.length` — the * terminator belongs to `find`, not to the command it runs. */ function execTerminatorIndex( words: readonly CommandWord[], start: number, ): number { const terminator = words.findIndex( (word, index) => index >= start && EXEC_TERMINATORS.has(word.text.replace(/^\\/, "")), ); return terminator === -1 ? words.length : terminator; } /** The unit text spanned by `words[start..end)`. */ function sliceWords( unitText: string, words: readonly CommandWord[], start: number, end: number, ): string { const from = words[start].offset; return end < words.length ? unitText.slice(from, words[end].offset) : unitText.slice(from); } /** `words[start..end)` with offsets rebased onto the sliced text. */ function rebase( words: readonly CommandWord[], start: number, end: number, ): CommandWord[] { const origin = words[start].offset; return words .slice(start, end) .map((word) => ({ ...word, offset: word.offset - origin })); } /** True for a `NAME=value` environment prefix. */ function isEnvironmentAssignment(word: string): boolean { return /^[A-Za-z_][A-Za-z0-9_]*=/.test(word); } /** * Shell command names whose `-c` flag introduces an opaque inline program. */ const SHELL_WRAPPER_NAMES = new Set(["bash", "sh", "dash", "zsh", "ksh"]); /** * Indirection wrappers that always invoke a following command, so the wrapper * (not the inner command) is what a bash rule matches. Floored by command-name * basename alone. Extend this set to cover another always-invoking wrapper. */ const INDIRECTION_WRAPPER_NAMES = new Set([ "sudo", "env", "xargs", "time", "nohup", "timeout", "nice", // Exec-capable rewrites and prefix wrappers surveyed in #575: parallelizers // (parallel/rust-parallel/rush), a sudo rewrite (doas), and prefix wrappers // (setsid/stdbuf/watch/flock) that all always invoke a following command. "parallel", "rust-parallel", "rush", "doas", "setsid", "stdbuf", "watch", "flock", ]); /** * Search tools that invoke a command per result only when an exec flag is * present; a bare search runs no subcommand. Floored only when an argument * exactly matches one of the tool's exec flags. Extend by adding a tool with * its exec-flag set. */ const EXEC_CONDITIONAL_WRAPPERS = new Map>([ ["find", new Set(["-exec", "-execdir", "-ok", "-okdir"])], ["fd", new Set(["-x", "--exec", "-X", "--exec-batch"])], ]); /** * Curated per-wrapper options that consume the following word, so skipping a * wrapper's own arguments does not mistake an option's value for the inner * command. Attached forms (`-I{}`, `--user=root`) need no entry — they are one * word. Only the display-side extraction reads this, and a missing or wrong * entry yields `null` (see {@link executedUnitOf}), never a weaker gate. */ const VALUE_TAKING_FLAGS = new Map>([ ["sudo", new Set(["-u", "-g", "-p", "-C", "-h", "-U", "-r", "-t"])], ["doas", new Set(["-u", "-C"])], ["env", new Set(["-u", "-C", "--unset", "--chdir"])], [ "xargs", new Set(["-n", "-P", "-I", "-i", "-d", "-E", "-L", "-l", "-s", "-a"]), ], ["timeout", new Set(["-s", "-k", "--signal", "--kill-after"])], ["nice", new Set(["-n", "--adjustment"])], ["time", new Set(["-o", "-f", "--output", "--format"])], ["stdbuf", new Set(["-i", "-o", "-e", "--input", "--output", "--error"])], ["watch", new Set(["-n", "--interval"])], ["flock", new Set(["-w", "-E", "--timeout", "--conflict-exit-code"])], ]); const EMPTY_FLAGS: ReadonlySet = new Set(); /** * Wrappers that change only *how* the same visible command runs — timing, kill * deadline, scheduling, buffering, session — each with the flags (options * taking no value) it admits. Value-taking options are admitted from * {@link VALUE_TAKING_FLAGS}, less {@link WRITING_OPTIONS}, so the two tables * cannot disagree about where the inner command starts. Any other option * refuses the exemption. */ const EXECUTION_MODIFIER_FLAGS = new Map>([ // BSD `man 1 time`: `time [-al] [-h | -p] [-o file]`; `-p` is also the bash // keyword's only option. `-a` appends to the `-o` file, so it is not listed. ["time", new Set(["-p", "-l", "-h"])], // GNU coreutils `timeout --help`. [ "timeout", new Set([ "-f", "--foreground", "-p", "--preserve-status", "-v", "--verbose", ]), ], // `nice` and `stdbuf` take only value options; util-linux `setsid`'s flags // are unverified on this host, so none is admitted. ["nice", EMPTY_FLAGS], ["stdbuf", EMPTY_FLAGS], ["setsid", EMPTY_FLAGS], ]); /** * Each modifier's value-taking options that write a file, never admitted. * Per wrapper: `time -o` names an output file, while `stdbuf -o` sets a mode. */ const WRITING_OPTIONS = new Map>([ ["time", new Set(["-o", "--output"])], ]); /** * A command name with no quoting, expansion, glob, or grouping character, and * not an option: a name led by `-` is one {@link executedUnitOf} declines to * name, so it could never be resolved. */ const LITERAL_COMMAND_NAME = /^[A-Za-z0-9_./+@%,:][A-Za-z0-9_./+@%,:-]*$/; /** Bash reserved words, which open syntax rather than name a command. */ const RESERVED_WORDS: ReadonlySet = new Set([ "!", "{", "}", "[[", "]]", "case", "coproc", "do", "done", "elif", "else", "esac", "fi", "for", "function", "if", "in", "select", "then", "time", "until", "while", ]); /** * Wrappers whose first bare word is an operand (a duration, a lock file) rather * than the start of the inner command. */ const LEADING_OPERAND_WRAPPERS = new Set(["timeout", "flock"]); /** Words ending a `find -exec` clause; they belong to `find`, not its command. */ const EXEC_TERMINATORS = new Set([";", "+"]); // ── Shared helpers ─────────────────────────────────────────────────────────── /** The wrapper's command-name basename, or `undefined` for an empty unit. */ function wrapperName(words: readonly CommandWord[]): string | undefined { return words.length === 0 ? undefined : basename(words[0].text); } /** * True when an argument list has a short-flag cluster containing `c` before any * `--` end-of-options marker (`-c`, `-ec`, `-xc`) — the inline-shell payload * flag for `bash`/`sh`/`dash`/`zsh`/`ksh`. */ function hasShortFlagC(args: readonly string[]): boolean { return shortFlagCIndex(args) !== -1; } /** Index within `args` of the `-c` short-flag cluster, or `-1`. */ function shortFlagCIndex(args: readonly string[]): number { for (const [index, arg] of args.entries()) { if (arg === "--") return -1; if (arg.startsWith("-") && !arg.startsWith("--") && arg.includes("c")) { return index; } } return -1; } /** Index within `args` of a matched per-result exec flag, or `-1`. */ function execFlagIndex(commandName: string, args: readonly string[]): number { const execFlags = EXEC_CONDITIONAL_WRAPPERS.get(commandName); if (!execFlags) return -1; return args.findIndex((arg) => execFlags.has(arg)); } /** The final path segment of a command name (`/bin/bash` → `bash`). */ function basename(name: string): string { const slash = name.lastIndexOf("/"); return slash === -1 ? name : name.slice(slash + 1); }