/** * Indexing pipeline orchestrator. * * Coordinates the full indexing pipeline: * 1. Process all configured sources to generate chunks * 2. Generate embeddings for all chunks * 3. Upload to Vectorize (unless dry-run) * 4. Invalidate search cache (after successful upload) * * This replaces the mixed logic previously in index.ts, providing * cleaner separation between orchestration and source-specific processing. */ import type { IndexerConfig } from './config/types.js'; import type { Chunk } from './chunking/types.js'; import type { SourceStats } from './sources/types.js'; /** * Options for the indexing pipeline. */ export interface IndexOptions { /** Preview chunks without uploading */ dryRun?: boolean; /** Show detailed progress */ verbose?: boolean; /** Write chunks to JSON file instead of uploading */ outputPath?: string; /** Maximum pages/files to process per source (for preview runs) */ maxPages?: number; /** Skip cache invalidation after upload */ skipCacheInvalidation?: boolean; /** API URL for cache invalidation (defaults to production) */ apiUrl?: string; /** Cache invalidation token (from env if not provided) */ cacheInvalidateToken?: string; /** * Continue processing remaining sources if one source fails. * When enabled, errors are collected and reported at the end instead of * stopping the entire indexing run. * @default false */ continueOnError?: boolean; /** * Process chunks in batches of this size to reduce memory usage. * When set, chunks are embedded and uploaded in batches rather than * accumulating all chunks in memory before processing. * Recommended for large indexing runs (>10k chunks). * @default undefined (process all chunks at once) */ chunkBatchSize?: number; } /** * Statistics for a single source. */ export interface SourceStatistics extends SourceStats { /** Source name */ source: string; /** Error message if source processing failed */ error?: string; } /** * Result from running the indexing pipeline. */ export interface IndexResult { /** Total chunks generated across all sources */ chunks: Chunk[]; /** Whether chunks were uploaded to Vectorize */ uploaded: boolean; /** Processing statistics by source */ stats: SourceStatistics[]; /** Errors from sources that failed (when continueOnError is enabled) */ errors?: Array<{ source: string; error: string; }>; } /** * Run the indexing pipeline. * * Processes all configured sources, generates embeddings, and uploads * to Vectorize. Supports dry-run mode for previewing chunks. * * @param config - Indexer configuration * @param options - Pipeline options * @returns Index result with chunks and statistics * * @example * ```typescript * const config = await loadConfig({ config: './config.json' }); * const result = await runIndex(config, { verbose: true }); * console.log(`Indexed ${result.chunks.length} chunks`); * ``` */ export declare function runIndex(config: IndexerConfig, options?: IndexOptions): Promise; //# sourceMappingURL=orchestrator.d.ts.map