#!/usr/bin/env node /** * 需求表/需求文本解析工具 —— 把一个需求文件拆成「多条独立需求」,供 hmos-spec-generate * skill 逐条生成 spec。一个输入文件包含多条需求,一条需求 = 一份 spec。 * * 支持两种输入类型,拆分后结构完全一致: * - Excel 类型 (.xlsx / .xlsm / .csv):一行 = 一条需求 * 第 1 列 = UI 路径(如「设置-歌词-悬浮窗状态栏歌词」) * 第 2 列 = 功能描述(可含换行与缩进的多行文本) * 第 3 列及之后的列会按行追加到功能描述末尾,不会被丢弃 * - 文本类型 (.txt):空行分隔的一段 = 一条需求 * 段落首行 = UI 路径,其余行 = 功能描述 * 参见 template/REQ.xlsx 与 template/REQ.txt,两者内容等价。 * * 零第三方依赖:xlsx 本质是 ZIP + XML,这里用 node:zlib 手工解压 + 正则解析。 * 运行环境:Node ≥ 22.18(原生剥离类型,可直接 `node parse_requirements.ts`)。 * * 用法: * node parse_requirements.ts --input REQ.xlsx --json # 逐条需求(推荐,供 skill 迭代) * node parse_requirements.ts --input REQ.xlsx # 归一化文本 → stdout * node parse_requirements.ts --input REQ.xlsx --out a.txt # 归一化文本 → 文件 * node parse_requirements.ts --input REQ.xlsx --split-dir ./out # 每条需求一个 .txt * node parse_requirements.ts --input REQ.xlsx --sheet 需求 # 指定工作表(名称或 1-based 序号) * node parse_requirements.ts --input REQ.xlsx --list-sheets * node parse_requirements.ts --input legacy.txt --single # 整个文件当作单条需求 */ import * as fs from 'node:fs'; import * as path from 'node:path'; import * as zlib from 'node:zlib'; type Row = string[]; interface RequirementItem { /** 1-based 序号,即该需求在输入文件中的出现顺序 */ index: number; /** 供下游命名 spec 文件用的确定性名称(不含扩展名) */ name: string; uiPath: string; description: string; } interface SheetRef { name: string; entry: string; } interface Options { input: string; out?: string; json: boolean; splitDir?: string; sheet?: string; listSheets: boolean; header: 'auto' | 'yes' | 'no'; single: boolean; prefix?: string; } // ---------------------------------------------------------------- ZIP 容器 function findEndOfCentralDirectory(buf: Buffer): number { const start = Math.max(0, buf.length - (22 + 0xffff)); for (let i = buf.length - 22; i >= start; i--) { if (buf.readUInt32LE(i) === 0x06054b50) return i; } return -1; } /** 读出 zip 中所有条目(名称 → 解压后的内容)。仅支持 stored / deflate,够覆盖 xlsx。 */ function readZipEntries(buf: Buffer): Map { const eocd = findEndOfCentralDirectory(buf); if (eocd < 0) throw new Error('不是合法的 xlsx/zip 文件:找不到 ZIP 中央目录结束记录'); let entryCount = buf.readUInt16LE(eocd + 10); let cdOffset = buf.readUInt32LE(eocd + 16); // ZIP64:条目数或中央目录偏移溢出时改读 ZIP64 结束记录 if (entryCount === 0xffff || cdOffset === 0xffffffff) { const locator = eocd - 20; if (locator >= 0 && buf.readUInt32LE(locator) === 0x07064b50) { const z64 = Number(buf.readBigUInt64LE(locator + 8)); if (z64 >= 0 && z64 + 56 <= buf.length && buf.readUInt32LE(z64) === 0x06064b50) { entryCount = Number(buf.readBigUInt64LE(z64 + 32)); cdOffset = Number(buf.readBigUInt64LE(z64 + 48)); } } } const files = new Map(); let p = cdOffset; for (let i = 0; i < entryCount; i++) { if (p + 46 > buf.length || buf.readUInt32LE(p) !== 0x02014b50) break; const method = buf.readUInt16LE(p + 10); const compressedSize = buf.readUInt32LE(p + 20); const nameLen = buf.readUInt16LE(p + 28); const extraLen = buf.readUInt16LE(p + 30); const commentLen = buf.readUInt16LE(p + 32); const localOffset = buf.readUInt32LE(p + 42); const name = buf.toString('utf8', p + 46, p + 46 + nameLen); p += 46 + nameLen + extraLen + commentLen; if (localOffset + 30 > buf.length || buf.readUInt32LE(localOffset) !== 0x04034b50) continue; const lNameLen = buf.readUInt16LE(localOffset + 26); const lExtraLen = buf.readUInt16LE(localOffset + 28); const dataStart = localOffset + 30 + lNameLen + lExtraLen; const raw = buf.subarray(dataStart, dataStart + compressedSize); if (method === 0) files.set(name, Buffer.from(raw)); else if (method === 8) files.set(name, zlib.inflateRawSync(raw)); else throw new Error(`xlsx 内部条目 ${name} 使用了不支持的压缩方式 ${method}`); } return files; } // ---------------------------------------------------------------- XML 解析 function decodeXmlText(s: string): string { return s .replace(/_x000D_/g, '') .replace(/&#x([0-9a-fA-F]+);/g, (_, h: string) => String.fromCodePoint(parseInt(h, 16))) .replace(/&#(\d+);/g, (_, d: string) => String.fromCodePoint(Number(d))) .replace(/</g, '<') .replace(/>/g, '>') .replace(/"/g, '"') .replace(/'/g, "'") .replace(/&/g, '&'); } function attr(tag: string, name: string): string | undefined { const m = new RegExp(`\\b${name.replace(/:/g, '\\:')}\\s*=\\s*"([^"]*)"`).exec(tag); return m ? decodeXmlText(m[1]) : undefined; } /** 取出片段内所有 文本并拼接(富文本多个 run 会被合并;忽略日文注音 )。 */ function collectTextNodes(xml: string): string { const cleaned = xml.replace(//g, ''); let text = ''; const tRe = /]*\/>|]*>([\s\S]*?)<\/t>/g; let m: RegExpExecArray | null; while ((m = tRe.exec(cleaned)) !== null) text += decodeXmlText(m[1] ?? ''); return text; } function parseSharedStrings(xml: string): string[] { const out: string[] = []; const siRe = /]*\/>|]*>([\s\S]*?)<\/si>/g; let m: RegExpExecArray | null; while ((m = siRe.exec(xml)) !== null) out.push(collectTextNodes(m[1] ?? '')); return out; } /** "AB12" → 27(0-based 列号) */ function columnIndexFromRef(ref: string): number { const letters = /^([A-Z]+)/.exec(ref.toUpperCase()); if (!letters) return -1; let n = 0; for (const ch of letters[1]) n = n * 26 + (ch.charCodeAt(0) - 64); return n - 1; } function parseSheet(xml: string, shared: string[]): Row[] { const rows: Row[] = []; const rowRe = /]*\/>|]*>([\s\S]*?)<\/row>/g; let rm: RegExpExecArray | null; while ((rm = rowRe.exec(xml)) !== null) { const inner = rm[1] ?? ''; const cells: Row = []; let auto = 0; const cellRe = /]*)\/>|]*)>([\s\S]*?)<\/c>/g; let cm: RegExpExecArray | null; while ((cm = cellRe.exec(inner)) !== null) { const attrs = cm[1] ?? cm[2] ?? ''; const body = cm[3] ?? ''; const ref = attr(attrs, 'r'); const col = ref ? columnIndexFromRef(ref) : -1; const index = col >= 0 ? col : auto; auto = index + 1; const type = attr(attrs, 't') ?? 'n'; let value = ''; if (type === 'inlineStr') { value = collectTextNodes(body); } else { const v = /]*>([\s\S]*?)<\/v>/.exec(body); const rawValue = v ? decodeXmlText(v[1]) : ''; if (type === 's') { const i = Number(rawValue); value = Number.isInteger(i) && i >= 0 && i < shared.length ? shared[i] : ''; } else if (type === 'b') { value = rawValue === '1' ? 'TRUE' : rawValue === '0' ? 'FALSE' : rawValue; } else { value = rawValue; } } while (cells.length < index) cells.push(''); cells[index] = value; } rows.push(cells); } return rows; } function resolveSheets(zip: Map): SheetRef[] { const workbook = zip.get('xl/workbook.xml'); if (!workbook) throw new Error('xlsx 缺少 xl/workbook.xml,文件可能已损坏'); const relsBuf = zip.get('xl/_rels/workbook.xml.rels'); const rels = new Map(); if (relsBuf) { const relRe = /]*\/>|]*>[\s\S]*?<\/Relationship>/g; let m: RegExpExecArray | null; const relXml = relsBuf.toString('utf8'); while ((m = relRe.exec(relXml)) !== null) { const id = attr(m[0], 'Id'); const target = attr(m[0], 'Target'); if (id && target) rels.set(id, target); } } const sheets: SheetRef[] = []; const sheetRe = /]*\/>|]*>[\s\S]*?<\/sheet>/g; let m: RegExpExecArray | null; const wbXml = workbook.toString('utf8'); let fallbackIndex = 0; while ((m = sheetRe.exec(wbXml)) !== null) { fallbackIndex++; const name = attr(m[0], 'name') ?? `Sheet${fallbackIndex}`; const rid = attr(m[0], 'r:id') ?? attr(m[0], 'relationshipId'); let target = rid ? rels.get(rid) : undefined; if (!target) target = `worksheets/sheet${fallbackIndex}.xml`; const entry = target.startsWith('/') ? target.slice(1) : `xl/${target.replace(/^\.\//, '')}`; sheets.push({ name, entry }); } if (sheets.length === 0) throw new Error('xlsx 中没有找到任何工作表'); return sheets; } function readXlsx(file: string, sheetSelector?: string): { rows: Row[]; sheets: SheetRef[]; picked: SheetRef } { const zip = readZipEntries(fs.readFileSync(file)); const sheets = resolveSheets(zip); let picked: SheetRef | undefined; if (sheetSelector) { const asIndex = Number(sheetSelector); if (Number.isInteger(asIndex) && asIndex >= 1 && asIndex <= sheets.length) picked = sheets[asIndex - 1]; else picked = sheets.find((s) => s.name === sheetSelector); if (!picked) { throw new Error( `找不到工作表 "${sheetSelector}"。可用工作表:${sheets.map((s, i) => `${i + 1}. ${s.name}`).join(',')}`, ); } } else { picked = sheets[0]; } const sheetBuf = zip.get(picked.entry); if (!sheetBuf) throw new Error(`xlsx 内部缺少工作表数据 ${picked.entry}`); const sharedBuf = zip.get('xl/sharedStrings.xml'); const shared = sharedBuf ? parseSharedStrings(sharedBuf.toString('utf8')) : []; return { rows: parseSheet(sheetBuf.toString('utf8'), shared), sheets, picked }; } // ---------------------------------------------------------------- CSV function parseCsv(text: string): Row[] { const body = text.replace(/^/, ''); const firstLine = body.split(/\r?\n/, 1)[0] ?? ''; const delimiter = firstLine.includes('\t') ? '\t' : firstLine.includes(';') && !firstLine.includes(',') ? ';' : ','; const rows: Row[] = []; let row: Row = []; let cell = ''; let quoted = false; for (let i = 0; i < body.length; i++) { const ch = body[i]; if (quoted) { if (ch === '"') { if (body[i + 1] === '"') { cell += '"'; i++; } else quoted = false; } else cell += ch; continue; } if (ch === '"') { quoted = true; continue; } if (ch === delimiter) { row.push(cell); cell = ''; continue; } if (ch === '\r') continue; if (ch === '\n') { row.push(cell); rows.push(row); row = []; cell = ''; continue; } cell += ch; } if (cell !== '' || row.length > 0) { row.push(cell); rows.push(row); } return rows; } // ---------------------------------------------------------------- 拆分为需求条目 interface RawItem { uiPath: string; description: string; } function isBlankRow(row: Row): boolean { return row.every((c) => (c ?? '').trim() === ''); } function looksLikeHeader(row: Row): boolean { const first = (row[0] ?? '').trim(); const second = (row[1] ?? '').trim(); if (!first) return false; if (/^(ui\s*路径|ui\s*path|路径|界面路径|菜单路径|设置路径|功能路径)$/i.test(first)) return true; return /(路径|path)/i.test(first) && /(描述|说明|功能|需求|desc)/i.test(second); } /** 表格:一行 = 一条需求。去掉末尾空白行/空白列后逐行映射。 */ function rowsToRawItems(rows: Row[], header: Options['header']): RawItem[] { const data = rows.filter((r) => !isBlankRow(r)); if (data.length === 0) return []; // auto 模式下,若跳过表头会把唯一一行数据也吃掉,则宁可不跳——单行需求表比表头更常见 const autoSkip = header === 'auto' && looksLikeHeader(data[0]) && data.length > 1; const skipFirst = header === 'yes' || autoSkip; const body = skipFirst ? data.slice(1) : data; return body .map((row) => { const cells = [...row]; while (cells.length > 0 && (cells[cells.length - 1] ?? '').trim() === '') cells.pop(); const uiPath = (cells[0] ?? '').trim(); const description = cells .slice(1) .map((c) => (c ?? '').replace(/\s+$/, '')) .filter((c) => c.trim() !== '') .join('\n') .trim(); return { uiPath, description }; }) .filter((item) => item.uiPath !== '' || item.description !== ''); } /** 文本:空行分隔的一段 = 一条需求,段首行 = UI 路径,其余行 = 功能描述。 */ function textToRawItems(text: string, single: boolean): RawItem[] { const body = text.replace(/^/, '').replace(/\r\n?/g, '\n'); const blocks = single ? [body] : body.split(/\n[ \t]*\n+/); return blocks .map((b) => b.replace(/\s+$/, '')) .filter((b) => b.trim() !== '') .map((block) => { const lines = block.split('\n'); return { uiPath: (lines[0] ?? '').trim(), description: lines.slice(1).join('\n').replace(/\s+$/, ''), }; }); } function sanitizeFileName(name: string): string { const cleaned = name.replace(/[\\/:*?"<>|]+/g, '_').replace(/\s+/g, '_').replace(/^_+|_+$/g, ''); return (cleaned || 'requirement').slice(0, 60); } /** * 给每条需求编号并生成确定性的输出名:`<前缀>-<序号>-`。 * 只有一条需求时退化为 `<前缀>`,保持"一个文件一条需求"的旧命名不变。 */ function withNames(items: RawItem[], prefix: string): RequirementItem[] { const width = String(items.length).length; return items.map((item, i) => { const index = i + 1; const name = items.length === 1 ? sanitizeFileName(prefix) : `${sanitizeFileName(prefix)}-${String(index).padStart(width, '0')}-${sanitizeFileName( item.uiPath || `requirement_${index}`, )}`; return { index, name, uiPath: item.uiPath, description: item.description }; }); } /** 单条需求归一化为纯文本:UI 路径一行 + 功能描述若干行。 */ function rawToText(item: RawItem): string { return [item.uiPath, item.description].filter((s) => s !== '').join('\n'); } /** 多条需求合并为纯文本:需求之间空一行(与等价的 .txt 逐字节一致)。 */ function rawListToText(items: RawItem[]): string { return items.map(rawToText).join('\n\n'); } /** --single:把整份表格压成一条需求(UI 路径留空,描述是全表归一化文本)。 */ function collapseToSingle(items: RawItem[]): RawItem[] { if (items.length <= 1) return items; return [{ uiPath: '', description: rawListToText(items) }]; } /** 整个文件归一化为纯文本,结尾补一个换行。 */ function toNormalizedText(items: RequirementItem[]): string { return items.map(rawToText).join('\n\n') + '\n'; } // ---------------------------------------------------------------- CLI function parseArgs(argv: string[]): Options { const opts: Options = { input: '', json: false, listSheets: false, header: 'auto', single: false }; for (let i = 0; i < argv.length; i++) { const a = argv[i]; const next = (): string => { const v = argv[++i]; if (v === undefined) throw new Error(`参数 ${a} 缺少取值`); return v; }; switch (a) { case '--input': case '-i': opts.input = next(); break; case '--out': case '-o': opts.out = next(); break; case '--json': opts.json = true; break; case '--split-dir': opts.splitDir = next(); break; case '--sheet': opts.sheet = next(); break; case '--list-sheets': opts.listSheets = true; break; case '--single': opts.single = true; break; case '--prefix': opts.prefix = next(); break; case '--header': { const v = next(); if (v !== 'auto' && v !== 'yes' && v !== 'no') throw new Error('--header 只能是 auto | yes | no'); opts.header = v; break; } case '--help': case '-h': console.log( [ '用法: node parse_requirements.ts --input [选项]', '', '把一个需求文件拆成多条独立需求(一条需求 = 一份 spec)。', '', ' --input, -i 需求文件(.xlsx / .xlsm / .csv / .txt)', ' --json 输出 [{index, name, uiPath, description}](推荐)', ' --out, -o 归一化文本写入文件(默认写 stdout)', ' --split-dir 每条需求单独写一个 .txt 到该目录', ' --prefix 覆盖 name 前缀(默认取输入文件名)', ' --sheet 指定工作表,名称或 1-based 序号(默认第一个)', ' --list-sheets 只打印工作表列表', ' --header auto|yes|no 表格是否跳过表头行(默认 auto,自动识别)', ' --single 整个文件只算一条需求(不按空行/行拆分)', ].join('\n'), ); process.exit(0); // eslint-disable-next-line no-fallthrough default: if (!opts.input && !a.startsWith('-')) opts.input = a; else throw new Error(`无法识别的参数:${a}`); } } if (!opts.input) throw new Error('缺少必需参数 --input <需求文件>'); return opts; } function main(): void { const opts = parseArgs(process.argv.slice(2)); const input = path.resolve(opts.input); if (!fs.existsSync(input)) throw new Error(`输入文件不存在:${input}`); const ext = path.extname(input).toLowerCase(); if (ext === '.xls') { throw new Error('不支持旧版二进制 .xls (BIFF) 格式,请在 Excel/WPS 中另存为 .xlsx 或 .csv 后重试'); } let raw: RawItem[]; if (ext === '.txt' || ext === '.md') { if (opts.listSheets) { console.log('1\t(纯文本无工作表)'); return; } raw = textToRawItems(fs.readFileSync(input, 'utf8'), opts.single); } else if (ext === '.csv' || ext === '.tsv') { if (opts.listSheets) { console.log('1\t(CSV 无工作表)'); return; } const rows = rowsToRawItems(parseCsv(fs.readFileSync(input, 'utf8')), opts.header); raw = opts.single ? collapseToSingle(rows) : rows; } else { const parsed = readXlsx(input, opts.sheet); if (opts.listSheets) { console.log(parsed.sheets.map((s, i) => `${i + 1}\t${s.name}`).join('\n')); return; } const rows = rowsToRawItems(parsed.rows, opts.header); raw = opts.single ? collapseToSingle(rows) : rows; } const prefix = opts.prefix ?? path.basename(input, path.extname(input)); const items = withNames(raw, prefix); if (items.length === 0) throw new Error(`需求文件 ${input} 中没有解析到任何需求`); if (opts.splitDir) { const dir = path.resolve(opts.splitDir); fs.mkdirSync(dir, { recursive: true }); const written: string[] = []; for (const item of items) { const file = path.join(dir, `${item.name}.txt`); fs.writeFileSync(file, rawToText(item) + '\n', 'utf8'); written.push(file); } console.log(written.join('\n')); return; } const output = opts.json ? JSON.stringify(items, null, 2) + '\n' : toNormalizedText(items); if (opts.out) { const file = path.resolve(opts.out); fs.mkdirSync(path.dirname(file), { recursive: true }); fs.writeFileSync(file, output, 'utf8'); console.error(`[parse_requirements] 已解析 ${items.length} 条需求 → ${file}`); } else { process.stdout.write(output); } } try { main(); } catch (err) { console.error(`[parse_requirements] 错误:${err instanceof Error ? err.message : String(err)}`); process.exit(1); }