/** * Content preprocessor for web scraping. * * Cleans and normalizes extracted text content before embedding. * Focuses on removing noise, normalizing text, and ensuring clean data * for high-quality vector embeddings. */ import type { PreprocessorOptions } from './types.js'; /** * Default preprocessing options. */ export declare const DEFAULT_PREPROCESSOR_OPTIONS: PreprocessorOptions; /** * Preprocess content for cleaner embeddings. * * @param content - Raw extracted content (markdown format) * @param options - Preprocessing options * @returns Cleaned content */ export declare function preprocessContent(content: string, options?: Partial): string; /** * Remove duplicate/repeated content blocks. * * Useful when the same content appears in multiple selectors. */ export declare function deduplicateContent(content: string): string; /** * Remove boilerplate sections that appear on multiple pages. * * Call this with content from multiple pages to identify common blocks. */ export declare function identifyBoilerplate(contentSamples: string[], threshold?: number): string[]; /** * Check if content chunk has sufficient information density * for meaningful embeddings. */ export declare function isChunkMeaningful(content: string, options?: { minWords?: number; minUniqueWords?: number; maxRepetitionRatio?: number; minWordLength?: number; }): boolean; //# sourceMappingURL=preprocessor.d.ts.map