/** * Markdown content chunking */ import type { Chunk, ChunkContext } from './types.js'; import { type MarkdownPreprocessorOptions } from './preprocessor.js'; /** * Section represents a split portion of markdown content */ export interface Section { heading: string | null; content: string; } /** * Extended chunk context with preprocessing options */ export interface ChunkContextWithPreprocessing extends ChunkContext { /** Markdown preprocessing options */ preprocessing?: MarkdownPreprocessorOptions; } /** * Chunk a markdown file into indexable pieces */ export declare function chunkMarkdownFile(filePath: string, content: string, context: ChunkContextWithPreprocessing): Chunk[]; /** * Split markdown content by heading level. * * Splits content into sections based on headings of the specified level. * Content before the first heading is assigned a null heading. * * @param content - Markdown content to split * @param level - Heading level to split on (e.g., 2 for ##) * @returns Array of sections with heading and content */ export declare function splitByHeading(content: string, level: number): Section[]; /** * Default max characters per chunk before paragraph-splitting kicks in. * * BGE-M3 truncates long inputs at embed time, silently dropping context. A * heading-only or H1-only document collapses into one giant section, so we cap * section size here as a second axis of splitting beyond headings. */ export declare const MAX_SECTION_CHARS = 4000; /** * Split an oversized section into multiple sections along paragraph (then * sentence) boundaries. Sections at or under `maxSize` pass through untouched. * Continuation parts keep the heading, tagged "(continued)". */ export declare function splitOversizedSection(section: Section, maxSize?: number): Section[]; //# sourceMappingURL=markdown.d.ts.map