/** * jev-find-source — the source units `jev_find` asks Jev about and returns verbatim. * * A unit is a line range of one file. JavaScript and TypeScript files are split with the * TypeScript parser into top-level statements (consecutive imports merged, and a class or * interface longer than FIND_CHUNK_LINES split into its members, each keeping the header line), * each with the comments directly above it. Every other file, and a script file when the parser * is not installed, is split into blank-line-separated chunks of at most FIND_CHUNK_LINES lines. * Line numbers are 1-based and always the file's own: excerpts are cut from the file, never * reconstructed. */ import { extname } from "node:path"; import type * as TypeScript from "typescript"; type TypeScriptApi = typeof TypeScript; /** ponytail: 40-line chunks for non-script files; raise once Jev is measured on longer units. */ export const FIND_CHUNK_LINES = 40; const UNIT_NAME_CHARS = 40; export interface SourceUnit { /** What the unit is: a declaration name, `Class.member`, `imports`, or the chunk's first line. */ name: string; /** First line, 1-based, inclusive. */ start: number; /** Last line, 1-based, inclusive. */ end: number; /** Header line of the enclosing class or interface, kept with a member unit. */ header?: number; } let typeScript: Promise | undefined; /** * The TypeScript compiler, or undefined when it is not installed: it is a devDependency, so a * published install parses nothing and every file is chunked instead. */ export function loadTypeScript(): Promise { typeScript ??= import("typescript").then( (module: unknown) => { const api = ((module as { default?: TypeScriptApi }).default ?? module) as TypeScriptApi; return typeof api.createSourceFile === "function" ? api : undefined; }, () => undefined, ); return typeScript; } const SCRIPT_KINDS: Record = { ".ts": "TS", ".tsx": "TSX", ".js": "JS", ".jsx": "JSX", ".mjs": "JS", ".cjs": "JS", }; /** Whether `path` is split with the TypeScript parser (when it is installed). */ export function isScriptPath(path: string): boolean { return extname(path).toLowerCase() in SCRIPT_KINDS; } /** A file's source units in file order; `ts` undefined chunks every file. */ export function splitSourceUnits(path: string, text: string, ts: TypeScriptApi | undefined): SourceUnit[] { const kind = SCRIPT_KINDS[extname(path).toLowerCase()]; if (ts && kind) { try { const units = scriptUnits(ts, path, text, ts.ScriptKind[kind]); if (units.length > 0) return units; } catch { // A parser crash on odd input falls back to chunks; the file is still readable. } } return chunkUnits(text); } function firstLine(text: string): string { const line = text.split("\n").find((row) => row.trim() !== "") ?? ""; return line.trim().slice(0, UNIT_NAME_CHARS); } function scriptUnits(ts: TypeScriptApi, path: string, text: string, kind: TypeScript.ScriptKind): SourceUnit[] { const source = ts.createSourceFile(path, text, ts.ScriptTarget.Latest, true, kind); const lineOf = (pos: number) => source.getLineAndCharacterOfPosition(pos).line + 1; /** * A node's first line, counting the comments directly above it: a comment separated by a blank * line (a section banner) or sitting on the previous node's last line (`a(); // note`) is not its own. */ const startOf = (node: TypeScript.Node, floor: number): number => { let start = lineOf(node.getStart(source)); const comments = ts.getLeadingCommentRanges(text, node.pos) ?? []; for (let i = comments.length - 1; i >= 0; i -= 1) { const comment = comments[i]; const from = lineOf(comment.pos); if (from <= floor || lineOf(comment.end) < start - 1) break; start = from; } return start; }; const nameOf = (node: TypeScript.Node): string => { const named = (node as { name?: TypeScript.Node }).name; if (named) return named.getText(source); if (ts.isVariableStatement(node)) { return node.declarationList.declarations.map((declaration) => declaration.name.getText(source)).join(", "); } if (ts.isConstructorDeclaration(node)) return "constructor"; if (ts.isExportAssignment(node)) return "export default"; return firstLine(node.getText(source)); }; const units: SourceUnit[] = []; let floor = 0; for (const statement of source.statements) { const start = startOf(statement, floor); const end = lineOf(statement.end); floor = end; if (ts.isImportDeclaration(statement) || ts.isImportEqualsDeclaration(statement)) { const last = units.at(-1); if (last?.name === "imports" && last.end >= start - 1) last.end = end; else units.push({ name: "imports", start, end }); continue; } const container = ts.isClassDeclaration(statement) || ts.isInterfaceDeclaration(statement) ? statement : undefined; const members = container?.members; if (container && members && members.length > 0 && end - start + 1 > FIND_CHUNK_LINES) { const owner = nameOf(container); const header = lineOf((container.name ?? container).getStart(source)); let memberFloor = header; for (const member of members) { const memberStart = startOf(member, memberFloor); const memberEnd = lineOf(member.end); memberFloor = memberEnd; units.push({ name: `${owner}.${nameOf(member)}`, start: memberStart, end: memberEnd, ...(memberStart > header ? { header } : {}), }); } continue; } units.push({ name: nameOf(statement), start, end }); } return units; } /** Blank-line-separated paragraphs packed into chunks of at most FIND_CHUNK_LINES lines. */ export function chunkUnits(text: string): SourceUnit[] { const rows = text.split("\n"); const paragraphs: Array<{ start: number; end: number }> = []; rows.forEach((row, index) => { if (row.trim() === "") return; const last = paragraphs.at(-1); if (last && last.end === index) last.end = index + 1; else paragraphs.push({ start: index + 1, end: index + 1 }); }); const ranges: Array<{ start: number; end: number }> = []; for (const paragraph of paragraphs) { const current = ranges.at(-1); if (current && paragraph.end - current.start + 1 <= FIND_CHUNK_LINES) { current.end = paragraph.end; continue; } // A paragraph longer than a chunk is cut into chunk-sized pieces. for (let start = paragraph.start; start <= paragraph.end; start += FIND_CHUNK_LINES) { ranges.push({ start, end: Math.min(paragraph.end, start + FIND_CHUNK_LINES - 1) }); } } return ranges.map((range) => ({ ...range, name: firstLine(rows[range.start - 1]) })); } /** * Line windows around the best keyword lines (±`context` lines), merged where they overlap: the * source returned when Jev could not judge a file's units. With no keyword line, the file's head. */ export function keywordWindows(lineCount: number, hits: readonly number[], context = 3, maxWindows = 3): SourceUnit[] { if (lineCount <= 0) return []; const anchors = [...new Set(hits.filter((line) => line >= 1 && line <= lineCount))].slice(0, maxWindows); if (anchors.length === 0) return [{ name: "file head", start: 1, end: Math.min(lineCount, 2 * context + 1) }]; const windows: SourceUnit[] = []; for (const line of anchors.sort((a, b) => a - b)) { const start = Math.max(1, line - context); const end = Math.min(lineCount, line + context); const last = windows.at(-1); if (last && start <= last.end + 1) last.end = Math.max(last.end, end); else windows.push({ name: "keyword window", start, end }); } return windows; }