/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * Build a parquet file from the DeepSeek-generated kryptonite jsonl and emit the corpus-v0.4.0 * manifest. corpus-v0.4.0 is a pure adapter-addition revision: it points at every parquet file from * v0.3.0 plus the new kryptonite file(s). No v0.3.0 bytes are touched or re-shuffled. * * See docs/records/engineering/CORPUS_V0_4_0_GENERATION.mdx for the why. that doc also pins the * DeepSeek model version + prompt versions used to produce the jsonl. * * Invoke via `mailwoman corpus slice kryptonite \ * --jsonl /data/corpus/versioned/v0.4.0/kryptonite/canonical-kryptonite.jsonl \ * --base-manifest /data/corpus/versioned/v0.3.0/corpus-v0.3.0/manifest.json \ * --out-dir /data/corpus/versioned/v0.4.0` */ import { delimitedSource } from "@mailwoman/core/fs/delimited" import { pathExists, readLocalJSONFile } from "@mailwoman/core/fs/readers" import { writeLocalJSONFile, writeLocalTextFile, makeDirectories } from "@mailwoman/core/fs/writers" import { PathBuilder, type PathBuilderLike } from "path-ts" import { JSONSpliterator } from "spliterator" import { corpusDirectoryName } from "#directory" import { PARQUET_COLUMNS, PARQUET_COMPRESSION, ROW_GROUP_SIZE } from "#parquet/schema" import { type ParquetManifest, writeParquetSplits } from "#parquet/writers" import { type CanonicalRow, type LabeledRow, requireSurface } from "#types" import { alignRow } from "#utils" export interface KryptoniteOverlayOptions { jsonl: string baseManifest: string outDir: PathBuilderLike /** * Default `"0.4.0"`. */ corpusVersion?: string /** * Default `"deepseek-kryptonite"`. */ source?: string } async function* canonicalRows(jsonl: string, corpusVersion: string): AsyncIterable { for await (const raw of JSONSpliterator.fromAsync>(delimitedSource(jsonl))) { // Strip sidecar underscore-prefixed fields the generator left behind for debugging. const components = raw["components"] as Record yield { raw: raw["raw"] as string, components, country: (raw["country"] as string) ?? "US", locale: (raw["locale"] as string) ?? undefined, source: (raw["source"] as string) ?? "deepseek-kryptonite", source_id: raw["source_id"] as string, corpus_version: corpusVersion, license: (raw["license"] as string) ?? "Synthetic (DeepSeek-v4-flash, AGPL-compatible)", recipe: raw["recipe"] as CanonicalRow["recipe"], register: (raw["register"] as string | null) ?? null, surface: requireSurface(raw, "deepseek-kryptonite"), } } } async function* labeledRows(jsonl: string, corpusVersion: string, quarantineLog: string[]): AsyncIterable { for await (const row of canonicalRows(jsonl, corpusVersion)) { const result = alignRow(row) if (result.kind === "labeled") { yield result.row } else { quarantineLog.push(`${row.source_id}\t${result.row.reason}`) } } } export async function buildKryptoniteOverlay( options: KryptoniteOverlayOptions, report?: (line: string) => void ): Promise { const corpusVersion = options.corpusVersion ?? "0.4.0" const source = options.source ?? "deepseek-kryptonite" if (!(await pathExists(options.jsonl))) throw new Error(`jsonl not found: ${options.jsonl}`) if (!(await pathExists(options.baseManifest))) throw new Error(`base-manifest not found: ${options.baseManifest}`) const outDir = PathBuilder.from(options.outDir) const corpusDir = outDir(corpusDirectoryName(corpusVersion)) await makeDirectories(outDir) const quarantine: string[] = [] const newManifest = await writeParquetSplits( { train: labeledRows(options.jsonl, corpusVersion, quarantine) }, { outputDir: outDir, corpusVersion } ) report?.( `wrote ${newManifest.total_rows} rows into ${newManifest.slices.length} parquet file(s); ` + `quarantined ${quarantine.length}` ) if (quarantine.length) { const qPath = corpusDir("quarantine-kryptonite.tsv") await writeLocalTextFile(quarantine, qPath) report?.(`quarantine log → ${qPath}`) } // Stamp the new file's source field for audit.ts // (which prefers the descriptor's `source` over first_source_id-prefix inference). // Without this, deepseek-kryptonite IDs would have to match a prefix in KNOWN_SOURCE_PREFIXES. // We add it there too as a belt-and-braces. for (const file of newManifest.slices) { file.source = source } // Compose the final corpus-v0.4.0 manifest: every parquet file from base + the new file(s). const base = await readLocalJSONFile(options.baseManifest) const combined: ParquetManifest = { corpus_version: corpusVersion, schema: PARQUET_COLUMNS, rows_per_slice: base.rows_per_slice, row_group_size: base.row_group_size ?? ROW_GROUP_SIZE, slices: [...base.slices, ...newManifest.slices], counts: { train: base.counts.train + (newManifest.counts.train ?? 0), val: base.counts.val, test: base.counts.test, }, total_rows: base.total_rows + newManifest.total_rows, } // The v0.3.0 files have no `source`: they mix sources, so audit.ts falls back to its // first_source_id-prefix inference for them (it re-derives on its own when `source` is absent). const combinedPath = corpusDir("MANIFEST.json") await writeLocalJSONFile(combined, combinedPath) report?.(`wrote combined manifest → ${combinedPath}`) report?.(` total_rows=${combined.total_rows} (base=${base.total_rows}, added=${newManifest.total_rows})`) report?.(` files=${combined.slices.length} (base=${base.slices.length}, added=${newManifest.slices.length})`) report?.(` compression=${PARQUET_COMPRESSION}`) }