/** * cli:sources-ingest — execute.ts (the PURE half, filesystem-free) * * Allocation, rendering and guards. Everything here is deterministic on its * inputs and individually exported for the tests; index.ts does the IO. * * Guards (refusals, not warnings): * - a web source with ZERO verbatim extract — « enregistrée seulement si on * en a extrait de l'information » is enforced here, not in prose; * - a rendered source.md over the hard cap — the registry is read by agent * contexts (the 394M-token incident), a source that cannot be summarized * under the cap must be SPLIT, not swallowed. */ import { sourceCodeOf, SOURCE_CODE_RE, type SourceEntry, type SourceFormat, type SourceKind, type SourceOrigin, type SourcesIndex, type SourceStatus, } from '../../../../lib/ba-sources.js' import { scrubSecrets } from '../../../../lib/support-report.js' import type { IngestAnalysis, RenderedIngest } from './types.js' /** Hard cap on one rendered source.md — beyond it the ingest REFUSES. */ export const RENDER_HARD_CAP = 40_000 /** Soft cap — a warning that the summary is drifting toward a dump. */ export const RENDER_WARN_CAP = 15_000 /** raw/ copy cap — beyond it the origin path alone is kept (rawOmitted). */ export const RAW_COPY_MAX_BYTES = 10 * 1024 * 1024 // --------------------------------------------------------------------------- // Allocation — 1 fingerprint = 1 code, never renumbered, never reused // --------------------------------------------------------------------------- export interface Allocation { kind: 'new' | 'existing' | 'as' code: string error?: string } export function resolveAllocation(index: SourcesIndex | null, fingerprint: string, asCode?: string): Allocation { if (asCode !== undefined) { if (!SOURCE_CODE_RE.test(asCode)) return { kind: 'as', code: asCode, error: `${asCode} is not a SRC-NNN code` } if (!index || !(asCode in index.sources)) { return { kind: 'as', code: asCode, error: `as=${asCode} — this code does not exist in the registry; \`as\` attaches to an EXISTING code, it never allocates.` } } return { kind: 'as', code: asCode } } if (index) { const hit = Object.values(index.sources).find((e) => e.fingerprint === fingerprint) if (hit) return { kind: 'existing', code: hit.code } } return { kind: 'new', code: sourceCodeOf(index?.nextSeq ?? 1) } } // --------------------------------------------------------------------------- // Rendering — the normalized citable source.md // --------------------------------------------------------------------------- function asBlockquote(body: string): string { return body .split('\n') .map((l) => (/^\s*>/.test(l) || l.trim() === '' ? l : `> ${l}`)) .join('\n') } function originLine(origin: SourceOrigin, rawFile?: string): string { if ('path' in origin) { const copy = rawFile ? ` (copie : \`${rawFile}\`)` : '' return `- **Origine** : \`${origin.path}\`${copy}` } return `- **Origine** : ${origin.url} — récupérée le ${origin.fetchedAt}` } export interface RenderMeta { code: string kind: SourceKind format: SourceFormat fingerprint: string status: SourceStatus origin: SourceOrigin /** Injected by index.ts — keeps this half deterministic. */ today: string rawFile?: string } export function renderSourceDoc(meta: RenderMeta, title: string, analysis: IngestAnalysis | null, reason?: string): string { const lines: string[] = [] lines.push(``) lines.push(`# ${meta.code} — ${title}`) lines.push('') lines.push('## Métadonnées') lines.push(originLine(meta.origin, meta.rawFile)) lines.push(`- **Format** : ${meta.format} · **Ingéré le** : ${meta.today}`) if (analysis) { lines.push(`- **Tags** : ${analysis.tags.join(', ')}`) if (analysis.scopes && analysis.scopes.length > 0) { lines.push(`- **Portée pressentie** : ${analysis.scopes.join(', ')}`) } } lines.push('') lines.push('## Résumé') if (analysis) { lines.push(analysis.summary.trim()) } else { lines.push(`Document NON LU (${meta.status}) : ${reason ?? 'raison non précisée'}.`) } lines.push('') if (analysis) { lines.push('## Points saillants') analysis.sections.forEach((s, i) => { const verbatim = s.verbatim === true || /^\s*>/m.test(s.body) const tags = `[${s.tags.join(', ')}]` const where = s.where ? ` (${s.where})` : '' const marker = verbatim ? ' — extrait verbatim' : '' lines.push(`### §${i + 1} — ${s.title} ${tags}${where}${marker}`) lines.push(verbatim ? asBlockquote(s.body.trim()) : s.body.trim()) }) lines.push('') } lines.push('## Ce que cette source ne couvre PAS') const notes = analysis?.scopeNotes && analysis.scopeNotes.length > 0 ? analysis.scopeNotes : analysis ? ['Portée non précisée par l’analyse — à compléter si une phase bute dessus.'] : [`Contenu à rattacher via \`ingest\` avec \`"as":"${meta.code}"\` une fois le document lisible.`] for (const n of notes) lines.push(`- ${n}`) lines.push('') return scrubSecrets(lines.join('\n')) } // --------------------------------------------------------------------------- // Build — entry + doc + rebuilt index, or a typed refusal // --------------------------------------------------------------------------- export interface BuildIngestInput { mode: 'write' | 'register-blocked' kind: SourceKind format: SourceFormat fingerprint: string origin: SourceOrigin /** register-blocked: which typed status (default blocked/needs-export). */ blockedStatus?: Extract reason?: string analysis?: IngestAnalysis /** Fallback title when no analysis (register-blocked): basename or url. */ fallbackTitle: string today: string index: SourcesIndex | null asCode?: string /** mode=write: existing code this new source REPLACES. */ supersedes?: string /** raw/ copy plan (kind=file): basename + size, resolved by index.ts. */ raw?: { basename: string; bytes: number; copyRaw: boolean } } export type BuildResult = { ok: true; rendered: RenderedIngest; allocation: Allocation } | { ok: false; errors: string[] } export function buildIngest(input: BuildIngestInput): BuildResult { const warnings: string[] = [] const allocation = resolveAllocation(input.index, input.fingerprint, input.asCode) if (allocation.error) return { ok: false, errors: [allocation.error] } const code = allocation.code const existing = input.index?.sources[code] // --- guards --------------------------------------------------------------- if (input.mode === 'write') { if (!input.analysis) return { ok: false, errors: ['mode=write without analysis (validate should have refused).'] } if (input.kind === 'web') { const hasVerbatim = input.analysis.sections.some((s) => s.verbatim === true || /^\s*>/m.test(s.body)) if (!hasVerbatim) { return { ok: false, errors: [ 'Source web sans extrait verbatim retenu — NON enregistrée. Une recherche web n’entre au registre que si de l’information en a été extraite (≥1 section verbatim:true).', ], } } } } // --- status + raw --------------------------------------------------------- const status: SourceStatus = input.mode === 'write' ? 'ingested' : (input.blockedStatus ?? 'blocked/needs-export') let rawFile: string | undefined let rawOmitted: boolean | undefined if (input.kind === 'file' && input.raw && input.raw.copyRaw) { if (input.raw.bytes <= RAW_COPY_MAX_BYTES) { rawFile = `raw/${input.raw.basename}` } else { rawOmitted = true warnings.push( `raw/ non copié : ${input.raw.bytes} octets > cap ${RAW_COPY_MAX_BYTES} — le chemin d'origine reste la seule trace du fichier brut.`, ) } } // --- render --------------------------------------------------------------- const title = input.analysis?.title ?? existing?.title ?? input.fallbackTitle const meta: RenderMeta = { code, kind: input.kind, format: existing && allocation.kind === 'as' ? existing.format : input.format, fingerprint: input.fingerprint, status, origin: allocation.kind === 'as' && existing ? existing.origin : input.origin, today: input.today, rawFile, } const sourceMd = renderSourceDoc(meta, title, input.analysis ?? null, input.reason) if (sourceMd.length > RENDER_HARD_CAP) { return { ok: false, errors: [ `source.md rendu = ${sourceMd.length} caractères > cap dur ${RENDER_HARD_CAP} — le registre est lu par des contextes d'agent : SCINDE la source (plusieurs SRC ciblés) ou résume plus court.`, ], } } if (sourceMd.length > RENDER_WARN_CAP) { warnings.push(`source.md rendu = ${sourceMd.length} caractères > ${RENDER_WARN_CAP} — le résumé dérive vers un dump ; resserre.`) } if (allocation.kind === 'as') { warnings.push(`Contenu rattaché à ${code} (statut précédent : ${existing?.status ?? '?'}) — l'origine d'ORIGINE est conservée, l'empreinte est celle du nouveau contenu.`) } // --- entry + rebuilt index ------------------------------------------------ const extracts = input.analysis ? input.analysis.sections.filter((s) => s.verbatim === true || /^\s*>/m.test(s.body)).length : 0 const summaryLine = input.analysis ? scrubSecrets((input.analysis.summaryLine ?? input.analysis.summary.trim().split('\n')[0]!).slice(0, 160)) : scrubSecrets(`Document non lu (${status}) : ${(input.reason ?? '').slice(0, 120)}`) const entry: SourceEntry = { code, kind: meta.kind, title, fingerprint: input.fingerprint, origin: meta.origin, format: meta.format, ingestedAt: existing?.ingestedAt ?? input.today, updatedAt: input.today, status, ...(existing?.supersededBy && status === 'superseded' ? { supersededBy: existing.supersededBy } : {}), tags: input.analysis?.tags ?? [], scopes: input.analysis?.scopes ?? [], summary: summaryLine, sections: input.analysis?.sections.length ?? 0, extracts, ...(rawFile ? { rawFile } : {}), ...(rawOmitted ? { rawOmitted } : {}), } const prevSources = input.index?.sources ?? {} if (input.supersedes !== undefined) { if (input.supersedes === code) { return { ok: false, errors: [`supersedes=${code} désigne le code alloué lui-même — une source ne se remplace pas.`] } } if (!(input.supersedes in prevSources)) { return { ok: false, errors: [`supersedes=${input.supersedes} — ce code n'existe pas dans le registre.`] } } } const sources: Record = {} for (const k of [...new Set([...Object.keys(prevSources), code])].sort()) { if (k === code) { sources[k] = entry } else if (k === input.supersedes) { sources[k] = { ...prevSources[k]!, status: 'superseded', supersededBy: code, updatedAt: input.today } } else { sources[k] = prevSources[k]! } } const prevNext = input.index?.nextSeq ?? 1 const index: SourcesIndex = { version: 1, nextSeq: allocation.kind === 'new' ? Math.max(prevNext, Number(code.slice(4))) + 1 : prevNext, sources, } return { ok: true, rendered: { entry, sourceMd, index, warnings }, allocation } }