/** * Normalized per-symbol content hashes (change: add-symbol-content-hashes). * * A symbol's hash is taken over the parse tree its extractor already built, never the raw bytes: * the pre-order stream of node types, leaf token texts, and open/close markers for every node the * symbol's span fully contains. Comments are left out, and whitespace between tokens never enters * the stream, so a re-indent, a rewrapped argument list, or a rewritten comment hashes identically. * The open/close markers keep the tree SHAPE in the stream, which is what makes this sound for * layout-significant languages: moving a Python statement out of an `if` block leaves the tokens * alone but changes the nesting, so it changes the hash. * * The same walk yields a RESIDUAL hash over everything no symbol span contains — imports, * module-level statements, class fields, decorators outside a span — and, separately, a LAYOUT: the * sequence of symbol spans and residual runs as they occur in the file. The two are kept apart on * purpose. If the residual carried a marker per span, adding or deleting one function would move it * and every other symbol in the file would read as changed; with the layout beside it, an added * symbol is just a new entry, while moving module-level code across a symbol (`main()` before * versus after a definition — a real difference in every language that executes a module top to * bottom) still shows up. Every non-comment token of the file lands in exactly one of the two, so two * revisions whose symbol hashes and residual hash all agree have the same non-comment tree. That is * the property a symbol-level changed-set rests on: a change can only hide from the per-symbol * hashes by showing up in the residual. * * Text a node owns but no child covers (a template literal's raw text in some grammars) is hashed * too: verbatim inside string-like nodes, where whitespace is content, and whitespace-collapsed * elsewhere. Text in comment syntax that changes how the file is built, parsed, or run — a shebang, * a Go pragma, a Ruby `frozen_string_literal` magic comment, an encoding cookie, `@ts-expect-error`, * `@jsx`, a lint or coverage pragma — is kept as a token rather than dropped; see * {@link DIRECTIVE_COMMENT} for the closed list and the limit it names. * * Hashing discipline matches `decisions/anchor.ts` `hashSpan` (sha256, first 16 hex characters), * but the hash is a different one: `hashSpan` is deliberately unnormalized and stays the freshness * baseline. Equality is the only comparison made on these hashes. There is no score or threshold. * * Computed only when a caller asks for it (see {@link withContentHashes}); a normal analyze never * pays for the walk. */ /** The minimal parse-tree view the hash walk needs. Real tree-sitter nodes satisfy it. */ export interface HashTreeNode { type: string; startIndex: number; endIndex: number; childCount?: number; child?(i: number): HashTreeNode | null; children?: HashTreeNode[]; } /** A symbol span to hash: an extracted function node's id and character range. */ export interface HashSpan { id: string; startIndex: number; endIndex: number; } /** * One top-level import statement, hashed on its own rather than into the residual. An import that is * purely ADDED — it takes a name that no other import bound — cannot change what the file's existing * symbols do, so hashing imports apart lets an ordinary "new import plus an edited function" diff * stay symbol-exact instead of collapsing the whole file. A removed or rewritten import DOES rebind * a name the existing symbols may use, and is a module-level change like any other. */ export interface ImportStatementHash { hash: string; /** Identifier texts inside the statement — an over-approximation of what it binds. */ names: string[]; /** * False when the statement binds nothing nameable, so it cannot be treated as purely additive: a * bare side-effect import (`import './polyfill'`), a Go blank import (`import _ "x"`), or a * WILDCARD (`from x import *`, `use x::*`, `import java.util.*`, `using namespace x`) — a wildcard * binds names this walk cannot enumerate, so it may shadow what the file's other symbols resolve. */ binds: boolean; } /** Why a file's residual hash could not be computed. */ export type ResidualUnavailableReason = 'invalid-span' | 'span-not-contiguous'; /** Per-file result: one hash per symbol (in input order) plus the residual. */ export interface FileContentHashes { /** One entry per input span, same order. Ids may repeat when an extractor emits a twin. */ symbols: Array<{ id: string; hash: string; }>; /** * Hash of every token outside every symbol span: imports, module-level statements, class bodies. * It carries no marker for the spans themselves, so adding or removing a symbol leaves it alone. */ residual?: string; /** Top-level import statements, hashed individually and excluded from the residual and layout. */ imports: ImportStatementHash[]; /** * Identifiers named by module-level code OUTSIDE the import statements — the names the file's * module level could be binding or handing around (`const h = get;`, a handler table). An * import's own names are in {@link ImportStatementHash.names} instead: the statement that binds a * name is not evidence that other module-level code uses it. Collected from the walk, so comments * and layout never enter it, in every language the walk covers. A string literal whose whole content is an identifier counts: a handler table keyed * by name is exactly the binding this evidence exists to catch. */ residualNames: string[]; /** * The file's shape: `T:` for a run of `n` residual tokens, `S:` for a run of one span's * tokens, `I:` for one import statement, in file order. Comparing two revisions' layouts projected onto the symbols they share * (dropping the other spans and summing the runs that then adjoin) is what detects a reordering, * or module-level code moving across a symbol — `main()` before a definition versus after it, * which no token or residual hash can see because the tokens themselves are identical. */ layout: string[]; /** Set when `residual` is absent. */ residualUnavailable?: ResidualUnavailableReason; } /** * Run `fn` with content hashing switched on for every extraction it awaits. Scoped with * `AsyncLocalStorage` rather than a module flag, so a concurrent full build in the same process * (the MCP daemon) never starts hashing, and the worker-pool lane never sees it at all. */ export declare function withContentHashes(fn: () => Promise): Promise; /** True inside {@link withContentHashes}. The extractors check this before walking. */ export declare function contentHashesRequested(): boolean; /** * Hash every span over `root`, and the residual. Iterative (never recursive): a deeply nested * expression must not overflow the stack. Deterministic: the same tree yields the same hashes. */ export declare function computeFileContentHashes(root: HashTreeNode, spans: readonly HashSpan[], content: string, /** Only {@link PATH_BOUND_IMPORT_LANGUAGES} is read: those bind an import by its module path. */ language?: string): FileContentHashes; //# sourceMappingURL=symbol-content-hash.d.ts.map