/** * Chunked parallel templater for GB-scale POC pulls. * * Splits the input events into N chunks, runs N concurrent tenx * processes via @apps/mcp-file (each with its own LOG10X_MCP_RUNTIME_NAME * so output directories don't clash), then merges the three artifacts * (templates.json, encoded.log, aggregated.csv) by their stable keys. * * Merge semantics: * * templates.json — dedupe by templateHash. The engine's templateHash * is deterministic over template content, so the same template * surfaced in chunks A and B produces the same hash in both * templates.json files. Keep one entry per hash. * * encoded.log — concat. Each encoded line is self-contained * (~hash + slot values + pattern= + patternHash=) and references * its templateHash by string; merging is line-wise. * * aggregated.csv — sum by (severity, message_pattern, tenx_hash). * The aggregator emits one row per unique enrichment tuple per * chunk; we sum summaryVolume + summaryBytes and concat * summaryTotals across rows with the same key. Header copied from * the first chunk's file. * * Returns the same {templatesJson, encodedLog, aggregatedCsv, * wallTimeMs, cliVersion, configPath, tempDir} shape that * runDevCliFileOutput returns for a single chunk — drop-in for * extractPatterns' merged-parsing step. */ export interface ChunkedTemplaterOptions { /** Parallelism level. Default: min(cpus - 1, 8). */ parallelism?: number; /** * Target chunk size in bytes of input text. When the input total * exceeds this, the chunker splits to keep each tenx process * working on roughly this much input. Default: 32 MB per chunk. */ chunkTargetBytes?: number; } export interface ChunkedTemplaterResult { templatesJson: string; encodedLog: string; aggregatedCsv: string; wallTimeMs: number; /** * Which engine build ran the chunks. Same field the single-shot runners * return, so callers can record engine identity without branching on * which path produced the result. */ cliVersion?: string; /** Per-chunk timings + sizes for telemetry. */ chunkStats: Array<{ chunkIndex: number; eventCount: number; bytes: number; wallTimeMs: number; templatesCount: number; encodedEventCount: number; aggregatedRowCount: number; }>; } /** * Run the templater on a single string of newline-joined events, * possibly split across multiple parallel tenx invocations. * * For inputs below 2 × chunkTargetBytes, runs single-process (no * chunking) — the parallelism overhead isn't worth it for small * inputs. Above that threshold, splits into chunks of approximately * chunkTargetBytes each and runs up to `parallelism` in parallel. */ export declare function runChunkedTemplater(rawLogText: string, opts?: ChunkedTemplaterOptions): Promise;