/** * cli:sources-ingest — sources.ts (the IO half, testable outside index.ts) * * Format detection + content reading + DOCX/XLSX text extraction. This file is * the ONLY place in the whole skills tree that touches a binary-document * parser (mammoth, exceljs) — the shared lib stays parser-free by design. * * Extraction is fail-closed: an extractor error returns a typed * `{ ok:false, reason }`, never a throw — the caller registers the document * `blocked/unreadable` instead of crashing the pipeline. */ import { readFileSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { extname, join } from 'node:path' import type { SourceFormat } from '../../../../lib/ba-sources.js' import type { ExtractionInfo } from './types.js' const EXT_TO_FORMAT: Record = { '.pdf': 'pdf', '.md': 'md', '.markdown': 'md', '.txt': 'txt', '.csv': 'csv', '.tsv': 'csv', '.json': 'json', '.eml': 'eml', '.png': 'image', '.jpg': 'image', '.jpeg': 'image', '.gif': 'image', '.webp': 'image', '.bmp': 'image', '.docx': 'docx', '.doc': 'docx', '.xlsx': 'xlsx', '.xls': 'xlsx', '.msg': 'msg', } export function detectFormat(kind: 'file' | 'web', originPath?: string): SourceFormat { if (kind === 'web') return 'web' const ext = extname(originPath ?? '').toLowerCase() return EXT_TO_FORMAT[ext] ?? 'other' } /** Formats whose fingerprint is computed on NORMALIZED TEXT (EOL-stable). */ const TEXT_FORMATS: readonly SourceFormat[] = ['md', 'txt', 'csv', 'json', 'eml'] export function isTextFormat(format: SourceFormat): boolean { return TEXT_FORMATS.includes(format) } export interface OriginContent { buffer: Buffer /** utf8 decode for text formats — the fingerprint basis. */ text: string | null bytes: number } export function readOriginFile(path: string, format: SourceFormat): OriginContent { const buffer = readFileSync(path) return { buffer, text: isTextFormat(format) ? buffer.toString('utf8') : null, bytes: buffer.length, } } // --------------------------------------------------------------------------- // DOCX / XLSX extraction — the ONLY binary parsers of the skills tree // --------------------------------------------------------------------------- export type ExtractOutcome = { ok: true; text: string } | { ok: false; reason: string } export async function extractDocx(buffer: Buffer): Promise { try { const mammoth = (await import('mammoth')).default const result = await mammoth.extractRawText({ buffer }) const text = result.value.trim() if (text === '') return { ok: false, reason: 'mammoth extracted an empty document' } return { ok: true, text } } catch (e) { return { ok: false, reason: `mammoth failed: ${e instanceof Error ? e.message : String(e)}` } } } export async function extractXlsx(buffer: Buffer): Promise { try { const ExcelJS = (await import('exceljs')).default const wb = new ExcelJS.Workbook() await wb.xlsx.load(buffer as unknown as ArrayBuffer) const parts: string[] = [] wb.eachSheet((sheet) => { parts.push(`## Feuille : ${sheet.name}`) sheet.eachRow((row) => { const cells = (row.values as unknown[]) .slice(1) // exceljs row.values is 1-based .map((v) => (v === null || v === undefined ? '' : cellText(v))) parts.push(cells.join('\t')) }) parts.push('') }) const text = parts.join('\n').trim() if (text === '') return { ok: false, reason: 'exceljs extracted an empty workbook' } return { ok: true, text } } catch (e) { return { ok: false, reason: `exceljs failed: ${e instanceof Error ? e.message : String(e)}` } } } function cellText(v: unknown): string { if (typeof v === 'object' && v !== null) { const o = v as Record if (typeof o.text === 'string') return o.text if (typeof o.result !== 'undefined') return String(o.result) if (o.richText && Array.isArray(o.richText)) { return (o.richText as Array<{ text?: string }>).map((r) => r.text ?? '').join('') } } return String(v) } /** Envelope cap — beyond this the full text lands in a temp FILE, said out * loud, never truncated silently. */ export const EXTRACT_INLINE_MAX = 60_000 /** * Resolve how the model gets the content (mode=plan). * pdf/image/text → the model reads the ORIGIN itself (Read tool); * docx/xlsx → extracted here; msg/other → unavailable (register-blocked). */ export async function extractionFor( format: SourceFormat, content: OriginContent | null, fingerprint: string, ): Promise { if (format === 'web') { return { status: 'model-read', detail: 'Source web — le contenu est ce que la recherche a retenu (extraits verbatim).' } } if (format === 'pdf' || format === 'image' || isTextFormat(format)) { return { status: 'model-read', detail: 'Lis le document original directement avec l’outil Read.' } } if (format === 'docx' || format === 'xlsx') { if (!content) return { status: 'unavailable', detail: 'Contenu introuvable.' } const out = format === 'docx' ? await extractDocx(content.buffer) : await extractXlsx(content.buffer) if (!out.ok) { return { status: 'unavailable', detail: `Extraction impossible (${out.reason}) — enregistre en mode=register-blocked (blocked/unreadable).` } } if (out.text.length <= EXTRACT_INLINE_MAX) { return { status: 'inline', chars: out.text.length, text: out.text, detail: 'Texte extrait complet ci-dessous.' } } const file = join(tmpdir(), `ba-sources-extract-${fingerprint}.txt`) writeFileSync(file, out.text, 'utf8') return { status: 'file', chars: out.text.length, file, detail: `Extraction de ${out.text.length} caractères — trop longue pour l'enveloppe, texte COMPLET écrit dans ce fichier (Read-le).`, } } return { status: 'unavailable', detail: `Aucun extracteur pour le format ${format} — enregistre en mode=register-blocked (blocked/needs-export) et demande un export PDF/MD/CSV.`, } }