/** * Abstract base class for content sources. * * Provides shared utilities that all source implementations can use, * reducing code duplication across different source types. * * @example * ```typescript * class MySource extends BaseContentSource { * readonly type = 'my-source'; * readonly name = 'My Custom Source'; * * async process(config, context) { * // Use inherited utilities * this.log(context, 'Processing...'); * const id = this.generateChunkId(path, context.site, 'section', 0); * // ... * } * } * ``` */ import type { ContentSource, BaseSourceConfig, SourceContext, SourceResult } from './types.js'; import type { ChunkMetadata, ChunkFilterOptions } from '../chunking/types.js'; import { type Section } from '../chunking/markdown.js'; /** * Abstract base class providing shared utilities for content sources. * * Extend this class to implement new source types. Subclasses must * implement the abstract `process` and `validateConfig` methods. * * @typeParam TConfig - Source-specific configuration type */ export declare abstract class BaseContentSource implements ContentSource { /** * Unique identifier for this source type. * Must be overridden by subclasses. */ abstract readonly type: string; /** * Human-readable name for logging. * Must be overridden by subclasses. */ abstract readonly name: string; /** * Process the source and return chunks. * Must be implemented by subclasses. */ abstract process(config: TConfig, context: SourceContext): Promise; /** * Validate configuration. * Must be implemented by subclasses. */ abstract validateConfig(config: unknown): TConfig; /** * Generate a unique, deterministic chunk ID. * * IDs are based on content path and section for consistency across * re-indexing runs. Limited to 64 bytes for Vectorize compatibility. * * @param path - Content path or URL * @param site - Site identifier * @param section - Section heading or identifier * @param index - Chunk index within the document * @returns Unique chunk ID (max 64 characters) */ protected generateChunkId(path: string, site: string, section: string | null, index: number): string; /** * Generate URL path from file path. * * @param filePath - Relative file path * @param routeBasePath - Base path for routing (e.g., "warranty") * @param version - Content version * @param isLatest - Whether this is the latest version (omits version from path) * @returns URL path (e.g., "/warranty/standards/plumbing") */ protected generateUrlPath(filePath: string, routeBasePath: string, version: string, isLatest: boolean): string; /** * Generate full URL from path. * * @param path - URL path * @param siteUrl - Site base URL * @returns Full URL (e.g., "https://example.com/warranty/intro") */ protected generateFullUrl(path: string, siteUrl: string): string; /** * Default category patterns imported from config/defaults.ts. * See DEFAULT_CATEGORY_PATTERNS for the full list of patterns. */ protected static readonly DEFAULT_CATEGORY_PATTERNS: Record; /** * Detect category from path using custom mappings and defaults. * * Checks path against custom category patterns first, then falls back * to default patterns for common documentation structures. * * @param path - Content path to check * @param categories - Map of path patterns to category names (custom) * @returns Category name or 'general' if no match */ protected detectCategory(path: string, categories?: Record): string; /** * Create base metadata object for a chunk. * * Provides common metadata fields that all chunks need. * Callers should add `section`, `anchor`, and `content` fields. * * @param context - Source context * @param config - Source configuration * @param path - Content path * @param title - Document title * @returns Partial chunk metadata (missing section, anchor, content) */ protected createBaseMetadata(context: SourceContext, config: TConfig, path: string, title: string): Omit; /** * Convert heading text to URL anchor slug. * * @param heading - Heading text * @returns Anchor slug (e.g., "my-heading") */ protected headingToAnchor(heading: string | null): string | null; /** * Log a message if verbose mode is enabled. * * @param context - Source context * @param message - Message to log */ protected log(context: SourceContext, message: string): void; /** * Log a warning message. * * @param message - Warning message */ protected warn(message: string): void; /** * Clean numeric prefixes from path segments. * * Docusaurus uses numeric prefixes for ordering (e.g., "01-intro.md"). * This removes them for cleaner URLs. * * @param path - Path with potential numeric prefixes * @returns Cleaned path */ protected cleanNumericPrefixes(path: string): string; /** * Extract title from markdown frontmatter or first heading. * * @param content - Markdown content * @param fallback - Fallback title if none found * @returns Extracted or fallback title */ protected extractTitle(content: string, fallback: string): string; /** * Split markdown content by heading level. * * Splits content into sections based on headings of the specified level. * Content before the first heading is assigned a null heading. * * @param content - Markdown content to split * @param level - Heading level to split on (e.g., 2 for ##) * @returns Array of sections with heading and content */ protected splitByHeading(content: string, level: number): Section[]; /** * Cap section size so no chunk exceeds the embedder's useful window. * Heading-only / H1-only documents otherwise collapse into one giant section * that BGE-M3 silently truncates at embed time. */ protected splitOversizedSections(sections: Section[]): Section[]; /** * Build URL path for a file. * * Converts a relative file path to a URL path, handling: * - File extension removal * - Index file handling * - Numeric prefix removal * - Version segments * * @param relativePath - Relative path to the file * @param routeBasePath - Base path for routing (e.g., "warranty") * @param version - Content version * @param useCleanUrl - Whether to omit version from URL (for latest/current) * @returns URL path (e.g., "/warranty/standards/plumbing") */ protected buildUrlPathFromFile(relativePath: string, routeBasePath: string, version: string, useCleanUrl: boolean): string; /** * Parse keywords from frontmatter. * * Extracts an array of keywords from frontmatter, handling various formats. * * @param keywords - Raw keywords value from frontmatter * @returns Array of keyword strings */ protected parseKeywords(keywords: unknown): string[]; /** * Generate chunk ID with version information. * * The ID must be unique across everything indexed into one Vectorize index * and stable across runs, because upload is an upsert: two chunks that * produce the same ID silently overwrite each other, and a chunk whose ID * changes leaves its old vector orphaned. * * Two properties the key depends on: * * - `guide` is part of the key. `relativePath` is relative to each guide's * own content root, so `definitions.md` exists under several guides and is * NOT unique on its own. Omitting the guide collapsed every guide's landing * page and shared filenames into a single vector. * - `ordinal` counts repeats of THIS heading within the file, not the * section's position in it. A positional index re-IDs every chunk below an * inserted section, orphaning the old vectors under the same version where * version-based pruning cannot see them. The ordinal only moves when a * duplicate of the same heading is added, and still separates same-named * sections (including the "(continued)" parts of an oversized section). * * @returns Unique chunk ID (max 64 characters) */ protected generateChunkIdWithVersion(opts: { /** Path relative to the guide's content root */ relativePath: string; /** Site identifier */ site: string; /** Content version */ version: string; /** Guide the content belongs to (scopes relativePath) */ guide: string; /** Section heading, or null for the pre-heading chunk */ heading: string | null; /** 0-based count of earlier chunks in this file with the same heading */ ordinal: number; }): string; /** * Get merged chunk filter options with defaults. * * @param options - Custom filter options (partial) * @returns Complete filter options with defaults applied */ protected getChunkFilterOptions(options?: Partial): Required; /** * Check if a section should be skipped based on its heading. * * @param heading - Section heading (can be null for intro sections) * @param skipSections - List of section names to skip (case-insensitive) * @returns True if the section should be skipped */ protected shouldSkipSection(heading: string | null, skipSections: string[]): boolean; /** * Calculate the ratio of markdown link content to total content. * * Links are detected by the pattern [text](url). * * @param content - Content to analyze * @returns Ratio from 0 to 1 (1 = all links) */ protected calculateLinkRatio(content: string): number; /** * Strip leading H1 heading from content. * * Removes patterns like "# Title\n\n" from the start of content. * * @param content - Content to process * @returns Content with leading H1 removed */ protected stripLeadingH1(content: string): string; /** * Filter and process sections based on filter options. * * Applies all filtering rules: * - Skips sections by heading name * - Removes chunks below minimum length * - Removes link-heavy chunks * - Strips leading H1 from first section * * @param sections - Raw sections from splitByHeading * @param options - Filter options * @returns Filtered and processed sections */ protected filterSections(sections: Array<{ heading: string | null; content: string; }>, options?: Partial): Array<{ heading: string | null; content: string; }>; } //# sourceMappingURL=base.d.ts.map