/** * Type definitions for web source scraping. * * Provides types for sitemap parsing, page fetching, content extraction, * and preprocessing operations. */ /** * A single entry from a sitemap. */ export interface SitemapEntry { /** Page URL */ url: string; /** Last modification date (if provided) */ lastmod?: string; } /** * Result from fetching a page. */ export interface FetchResult { /** Original URL */ url: string; /** Raw HTML content */ html: string; /** HTTP status code */ status: number; } /** * Options for fetching pages. */ export interface FetcherOptions { /** Requests per second (default: 2) */ rateLimit: number; /** Request timeout in ms (default: 30000) */ timeout: number; /** Number of retries (default: 2) */ retries: number; } /** * Extracted page content ready for chunking. */ export interface ExtractedPage { /** Original URL */ url: string; /** Page title */ title: string; /** Extracted content in Markdown format */ content: string; /** Keywords extracted from meta tags, JSON-LD, or content */ keywords: string[]; } /** * Preprocessing options for content cleaning. */ export interface PreprocessorOptions { /** Remove common CTA phrases like "Learn More", "Get Started" */ removeCTAs?: boolean; /** Remove phone numbers */ removePhoneNumbers?: boolean; /** Normalize Unicode characters (curly quotes, dashes, etc.) */ normalizeUnicode?: boolean; /** Remove orphaned markdown formatting */ cleanMarkdown?: boolean; /** Custom patterns to remove (regex strings) */ customPatterns?: string[]; /** Minimum word count for a line to be kept (0 = keep all) */ minLineWords?: number; } /** * Options for content extraction. */ export interface ExtractorOptions { /** CSS selectors for main content areas */ contentSelectors: string[]; /** CSS selectors for elements to remove */ excludeSelectors: string[]; } /** * Options for web source processing. */ export interface WebSourceOptions { /** Sitemap index URL */ sitemapUrl: string; /** Only process these child sitemaps */ includeSitemaps?: string[]; /** URL patterns to include (if specified, only matching URLs are processed) */ includePatterns?: string[]; /** URL patterns to exclude */ excludePatterns?: string[]; /** CSS selectors for main content */ contentSelectors: string[]; /** CSS selectors to remove */ excludeSelectors: string[]; /** Category mappings (URL pattern -> category) */ categories: Record; /** Requests per second */ rateLimit: number; /** Content preprocessing options */ preprocessing?: PreprocessorOptions; /** Remove elements hidden via inline styles */ removeHiddenElements?: boolean; /** Deduplicate repeated content blocks */ deduplicate?: boolean; } /** * Default Divi content selectors. * * Divi is a WordPress page builder. These selectors target * the main content areas in Divi-built pages. */ export declare const DIVI_CONTENT_SELECTORS: string[]; /** * Default Divi exclude selectors. * * These elements are removed before content extraction to reduce noise * in the embedding data. Add selectors here for elements that appear * across many pages but don't contain useful content. */ export declare const DIVI_EXCLUDE_SELECTORS: string[]; /** * Docusaurus content selectors. * * Docusaurus uses semantic HTML and class names for content areas. * The main content is in the markdown container within an article. */ export declare const DOCUSAURUS_CONTENT_SELECTORS: string[]; /** * Docusaurus exclude selectors. * * Elements that should be removed before content extraction. * Docusaurus has consistent class naming conventions. */ export declare const DOCUSAURUS_EXCLUDE_SELECTORS: string[]; //# sourceMappingURL=types.d.ts.map