/**
* HTML content extractor.
*
* Extracts clean text content from HTML pages and converts to Markdown.
* Includes Divi-specific handling for WordPress + Divi page builder sites.
*/
import type { ExtractedPage, ExtractorOptions, PreprocessorOptions } from './types.js';
/**
* Extended extractor options including preprocessing.
*/
export interface ExtendedExtractorOptions extends ExtractorOptions {
/** Preprocessing options for content cleaning */
preprocessing?: PreprocessorOptions;
/** Enable deduplication of repeated content blocks */
deduplicate?: boolean;
/** Remove elements hidden via inline styles */
removeHiddenElements?: boolean;
}
/**
* Default extractor options (Divi-optimized).
*/
export declare const DEFAULT_EXTRACTOR_OPTIONS: ExtendedExtractorOptions;
/**
* Extract content from HTML page and convert to Markdown.
*
* @param html - Raw HTML content
* @param url - Page URL (used for title extraction and metadata)
* @param options - Extractor options
* @returns Extracted page with title and Markdown content
*/
export declare function extractContent(html: string, url: string, options?: Partial): ExtractedPage;
/**
* Check if extracted content is meaningful (not just boilerplate or empty).
*/
export declare function isContentMeaningful(content: string): boolean;
//# sourceMappingURL=extractor.d.ts.map