/** * Ingest-depth step extensions: the unified `extract.text` format dispatch (PDF text layer, * DOCX via mammoth, XLSX via ExcelJS, ODT/ODS via ODF XML, PPTX via slide XML, images via the * OCR engine) and format-aware `extract.metadata`. * * @remarks * Internal sibling of the `@nhtio/adk/batteries/media` entry. Ported from the source server's * four extractors, collapsed into one verb that routes on the resolved media's format family * (the single biggest order-out-of-chaos win — the model says `extract text` and doesn't care * what the container is). OCR runs through the configured `ocr` engine when the input is an * image, or when a PDF has no text layer with `ocr=auto`, or always with `ocr=force`. */ import type { StepImpl } from "../runtime"; /** * `extract.text` — full format dispatch. Routes by MIME, with OCR engine fallback for images * and force-OCR for any input via `ocr=force`. */ export declare const extractTextDeepStep: StepImpl; /** `extract.metadata` — format-aware metadata (page counts, sheet/slide inventories). */ export declare const extractMetadataDeepStep: StepImpl; /** The ingest step registry fragment (overrides the Phase 0 native-text-only versions). */ export declare const INGEST_STEPS: ReadonlyArray<[ string, StepImpl ]>;