/** * LLM-based content processor for web pages. * * Uses Cloudflare Workers AI to extract clean, semantic content from messy web pages. * Produces search-optimized content that matches how users ask questions. */ export interface LLMProcessorOptions { /** Cloudflare account ID */ accountId: string; /** Cloudflare API token */ apiToken: string; /** Enable verbose logging */ verbose?: boolean; /** Request timeout in ms (default: 30000) */ timeout?: number; /** Cloudflare Workers AI model to use (default: @cf/meta/llama-4-scout-17b-16e-instruct) */ model?: string; } export interface PageContent { /** Page URL */ url: string; /** Page title */ title: string; /** Raw extracted content (markdown) */ rawContent: string; /** Page category hint */ category?: string; } export interface ProcessedPage { /** Page URL */ url: string; /** Clean page title */ title: string; /** Page type (floor-plan, community, general, etc.) */ pageType: string; /** Brief summary of what this page is about (1-2 sentences) */ summary: string; /** Clean content sections ready for embedding */ sections: ProcessedSection[]; /** Structured data extracted from the page */ structuredData?: Record; } export interface ProcessedSection { /** Section heading */ heading: string; /** Clean content for this section */ content: string; } /** * Detect page type from URL and content. */ export declare function detectPageType(url: string, content: string): string; /** * Process a single page with LLM. */ export declare function processPageWithLLM(page: PageContent, options: LLMProcessorOptions): Promise; /** * Process multiple pages with LLM (with rate limiting). */ export declare function processPages(pages: PageContent[], options: LLMProcessorOptions & { concurrency?: number; }): Promise; //# sourceMappingURL=llm-processor.d.ts.map