/** * 手写状态机 GDScript tokenizer(2026-09-11 P6 批,整文件移植自 Erodenn-godot-mcp-runtime * src/utils/gdscript-scanner.ts,经真读验证;适配点仅头注释与尾部追加的 classifyFirstArgument)。 * * 消费方:gdscript-executor.ts 的 scanGdscriptSandbox Phase 3(非字面量 load/preload 拦截)。 * tokenizer 职责:剥注释与字符串内容(策略绝不匹配 "OS.execute" 或 # OS.execute 里的文本), * 并把成员访问链(OS.execute / Foo.bar.baz)合并为单个 memberChain token(chain 数组供规则 * 前缀匹配)。 * * ⚠️ P6 架构判断(为什么没有移植 Erodenn 的三级 tier 规则表):enhanced 的安全架构是 * "工具级确认 + 内容级硬拦"——execute_gdscript 的 actionRisk='process' 意味着每次调用 * 都经确认令牌 out-of-band gate(AI 不能自确认)。Erodenn 的 Tier 2 elicit_required 是 * 为"默认无确认"的 run_script 设计的分级确认,在 enhanced 里是重复建设;且把文件写/网络 * 类从硬拦降为"问了再跑"是安全回退(用户确认可能被注入内容欺骗,确认后的硬拦恰是纵深)。 * 本文件只取 tokenizer 的结构化能力(classifyFirstArgument 的括号深度感知——正则做不到)。 * * Not a full GDScript parser — we only need enough to: * - Recognize comments (`#` to EOL). * - Skip string-literal contents in all GDScript forms (`"..."`, `'...'`, * `"""..."""`, `'''...'''`). * - Skip node-path literals (`$Foo/Bar`, `^"..."`) — their contents are * Godot scene paths, not GDScript code. * - Emit identifiers, member chains, parentheses, commas, and a small set * of other punctuation. Everything else (operators, numbers) collapses to * an `other` token the policy ignores. * - Track line numbers and the rough start column of each token so policy * findings can name the offending line. * * Line continuation (`\` at end of line) is handled by treating the next line * as a continuation of the current logical line for member-chain coalescing * purposes. * * This tokenizer is a best-effort accident guard, not a sound static * analysis — see `run-script-policy.ts` and `docs/security.md` for the full * doctrine. One structural blind spot worth stating plainly here, since it's * inherent to token-level scanning and not a gap the next feature closes: * identifier aliasing / dataflow is invisible. `var f = OS; f.execute(...)` * tokenizes as two unrelated identifiers — the tokenizer has no notion of * "what does this variable refer to," so a rule keyed on `OS.execute` never * fires. Do not mistake this for a TODO; closing it would require a dataflow * analysis, which is out of scope for a hand-written tokenizer by design. */ export type TokenKind = 'identifier' | 'memberChain' | 'string' | 'number' | 'punct' | 'newline' | 'other'; export interface Token { kind: TokenKind; text: string; /** For memberChain, the dotted segments in order: `OS.execute` → `['OS','execute']`. */ chain?: string[]; line: number; column: number; } /** * Tokens emitted by `tokenize`. Comments and string-literal contents are NOT * present — they are consumed silently. String literals as a whole are emitted * as a single `string` token so the policy can recognize "literal first * argument" patterns (e.g. `load("res://foo.tscn")`) without seeing the * characters inside. */ export declare function tokenize(source: string): Token[]; /** * Convenience: return only the non-newline, non-whitespace tokens. Useful for * policy rules that don't care about line structure. */ export declare function tokenizeStripped(source: string): Token[]; export type ArgumentClassification = 'none' | 'literal' | 'nonliteral'; /** * 分类调用表达式的**整个**首参:从 `(` 起收集 token 到顶层 `,` 或 `)`(括号深度感知, * 嵌套调用不误判)。'literal' = 孤立 string token(如 load("res://foo"));'nonliteral' = * 其他一切(标识符/表达式/多 token——"a" + b 正确判 nonliteral 而非被前导字面量骗过); * 'none' = 无参。这是正则扫描做不到的结构化判断(load(p) 变量形式正则不可见)。 */ export declare function classifyFirstArgument(tokens: readonly Token[], openParenIndex: number): ArgumentClassification;