/** * HTML content extractor. * * Extracts clean text content from HTML pages and converts to Markdown. * Includes Divi-specific handling for WordPress + Divi page builder sites. */ import type { ExtractedPage, ExtractorOptions, PreprocessorOptions } from './types.js'; /** * Extended extractor options including preprocessing. */ export interface ExtendedExtractorOptions extends ExtractorOptions { /** Preprocessing options for content cleaning */ preprocessing?: PreprocessorOptions; /** Enable deduplication of repeated content blocks */ deduplicate?: boolean; /** Remove elements hidden via inline styles */ removeHiddenElements?: boolean; } /** * Default extractor options (Divi-optimized). */ export declare const DEFAULT_EXTRACTOR_OPTIONS: ExtendedExtractorOptions; /** * Extract content from HTML page and convert to Markdown. * * @param html - Raw HTML content * @param url - Page URL (used for title extraction and metadata) * @param options - Extractor options * @returns Extracted page with title and Markdown content */ export declare function extractContent(html: string, url: string, options?: Partial): ExtractedPage; /** * Check if extracted content is meaningful (not just boilerplate or empty). */ export declare function isContentMeaningful(content: string): boolean; //# sourceMappingURL=extractor.d.ts.map