/** * Markdown/MDX content preprocessor * * Cleans and normalizes markdown content before chunking and embedding. * Handles MDX components, Docusaurus-specific syntax, and general * markdown cleanup for high-quality embeddings. */ /** * Markdown preprocessing options */ export interface MarkdownPreprocessorOptions { /** Remove import statements (default: true) */ removeImports?: boolean; /** Remove JSX/MDX components, keeping inner text content (default: true) */ removeJsxComponents?: boolean; /** Remove code blocks (default: false - code can be useful context) */ removeCodeBlocks?: boolean; /** Remove inline code (default: false) */ removeInlineCode?: boolean; /** Convert Docusaurus admonitions to plain text (default: true) */ convertAdmonitions?: boolean; /** Remove HTML comments (default: true) */ removeHtmlComments?: boolean; /** Remove link/image reference definitions (default: true) */ removeReferenceDefinitions?: boolean; /** Normalize Unicode characters (default: true) */ normalizeUnicode?: boolean; /** Remove excessive whitespace (default: true) */ normalizeWhitespace?: boolean; /** Custom patterns to remove (regex strings) */ customRemovePatterns?: string[]; } /** * Default preprocessing options */ export declare const DEFAULT_MARKDOWN_PREPROCESSOR_OPTIONS: MarkdownPreprocessorOptions; /** * Preprocess markdown/MDX content for cleaner embeddings * * @param content - Raw markdown/MDX content * @param options - Preprocessing options * @returns Cleaned markdown content */ export declare function preprocessMarkdown(content: string, options?: Partial): string; /** * Extract plain text from markdown for analysis * Useful for content quality checks */ export declare function extractPlainText(content: string): string; /** * Check if markdown content has meaningful information */ export declare function isMarkdownMeaningful(content: string, options?: { minWords?: number; minUniqueWords?: number; }): boolean; //# sourceMappingURL=preprocessor.d.ts.map