/** * Options for the OCR recognition request. */ export type RecognizeOptions = { /** * URI of the document to recognize. * Supports file:// and absolute paths. */ uri: string; /** * Type of the document. * - "auto": Automatically detect based on file extension (default). * Supports: pdf, jpg, png, heic, docx, txt, epub * - "pdf": Treat as PDF document * - "image": Treat as image (jpg, png, heic) * - "docx": Treat as DOCX document * - "txt": Treat as plain text file * - "epub": Treat as EPUB document */ type?: "auto" | "pdf" | "image" | "docx" | "txt" | "epub"; /** * Recognition languages in order of preference. * Uses BCP 47 language tags (e.g., ["en-US", "zh-Hans"]). * If not specified, defaults to device language settings. */ language?: string[]; /** * Recognition mode. * - "fast": Prioritize speed over accuracy * - "accurate": Prioritize accuracy over speed (default) */ mode?: "fast" | "accurate"; /** * Automatically detect the language of the text. * When true, Vision will attempt to identify the language automatically. * Requires iOS 16+. On older versions, this option is ignored. * @default true when language is not specified */ automaticallyDetectsLanguage?: boolean; /** * Use language correction during recognition. * When true, Vision applies language-specific corrections to improve accuracy. * @default true */ usesLanguageCorrection?: boolean; /** * Maximum number of PDF pages recognized at the same time. * Only applies to PDF documents. Use 1 for serial processing. * @default 2 */ maxConcurrentPages?: 1 | 2; }; /** * OCR result for a single page. */ export type OcrPageResult = { /** * Page number (1-indexed). */ page: number; /** * Recognized text content from this page. */ text: string; }; /** * Result of the OCR recognition. */ export type OcrResult = { /** * Full concatenated text from all pages. */ text: string; /** * Per-page results. Only present for multi-page documents (PDFs). */ pages?: OcrPageResult[]; /** * Source of the text extraction. * - "vision": Text was extracted using Apple Vision OCR * - "pdf-text": Text was extracted directly from PDF text layer * - "docx-xml": Text was extracted from DOCX XML content * - "txt": Text was read directly from plain text file * - "epub-html": Text was extracted from EPUB HTML/XHTML content */ source: "vision" | "pdf-text" | "docx-xml" | "txt" | "epub-html"; }; /** * Alias for OcrResult for API consistency with recognize(). */ export type RecognizeResult = OcrResult; /** Metadata describing how pages in a PDF will be processed. */ export type PdfInfo = { pageCount: number; textPageCount: number; scannedPageCount: number; hasTextLayer: boolean; }; export type PdfPageStatus = "success" | "blank" | "failed" | "cancelled"; export type PdfPageError = { code: string; message: string; }; /** Result for one page in a PDF range request. */ export type PdfPageResult = { page: number; text: string; status: PdfPageStatus; source: "pdf-text" | "vision"; error?: PdfPageError; }; /** Options for recognizing an inclusive, 1-based PDF page range. */ export type RecognizePdfPagesOptions = { uri: string; startPage: number; endPage: number; language?: string[]; mode?: "fast" | "accurate"; automaticallyDetectsLanguage?: boolean; usesLanguageCorrection?: boolean; /** Maximum concurrent Vision OCR page requests. Defaults to 2. */ maxConcurrentPages?: 1 | 2; }; export type PdfPageRangeResult = { pageCount: number; startPage: number; endPage: number; pages: PdfPageResult[]; cancelled?: boolean; }; /** A native PDF session that keeps its document open across page ranges. */ export type PdfOcrSession = { sessionId: string; info: PdfInfo; }; export type RecognizePdfSessionPagesOptions = Omit & { sessionId: string; }; //# sourceMappingURL=types.d.ts.map