// Generated by dts-bundle-generator v9.5.1 /** * Standard error types for OfficeParser. * Use these to identify the kind of error being reported. */ export declare enum OfficeErrorType { /** Unsupported file extension */ EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED", /** Unsupported output generator format */ FORMAT_UNSUPPORTED = "FORMAT_UNSUPPORTED", /** File appears to be corrupted or malformed */ FILE_CORRUPTED = "FILE_CORRUPTED", /** File could not be found at the specified path */ FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST", /** Specified location/directory is not reachable or is a directory */ LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND", /** Arguments passed to the function are missing or invalid */ IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS", /** Error occurred while reading or processing file buffers */ IMPROPER_BUFFERS = "IMPROPER_BUFFERS", /** Input type is not a supported type (string, Buffer, ArrayBuffer, Uint8Array) */ INVALID_INPUT = "INVALID_INPUT", /** PDF worker source is missing (required in browser) */ PDF_WORKER_MISSING = "PDF_WORKER_MISSING", /** Attempted to use Node.js-only features in a browser environment */ FEATURE_NOT_SUPPORTED_IN_BROWSER = "FEATURE_NOT_SUPPORTED_IN_BROWSER", /** Style mapping string is malformed */ INVALID_STYLE_MAPPING = "INVALID_STYLE_MAPPING", /** Selector in style mapping is invalid */ INVALID_SELECTOR = "INVALID_SELECTOR", /** Output mapping in style mapping is invalid */ INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING", /** Semantic chunking strategy is selected but no embedding function is provided */ MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION", /** The operation was aborted */ OPERATION_ABORTED = "OPERATION_ABORTED", /** ZIP entry count exceeds limit */ ZIP_ENTRY_COUNT_LIMIT_EXCEEDED = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED", /** ZIP entry missing a valid declared size */ ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE", /** ZIP uncompressed size limit exceeded */ ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED", /** ZIP data yielded no readable entries (corrupt, truncated, or not a ZIP archive) */ ZIP_NO_ENTRIES_FOUND = "ZIP_NO_ENTRIES_FOUND", /** ZIP data is truncated: the End of Central Directory record is absent */ ZIP_TRUNCATED = "ZIP_TRUNCATED", /** A readable ZIP archive is missing the part its document format requires */ REQUIRED_PART_MISSING = "REQUIRED_PART_MISSING", /** Document element/structure nesting exceeded the safe recursion depth */ MAX_NESTING_DEPTH_EXCEEDED = "MAX_NESTING_DEPTH_EXCEEDED", /** Embedding call timed out */ EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT" } /** * Standard warning types for OfficeParser. * Use these for reporting non-fatal issues or performance tips. */ export declare enum OfficeWarningType { /** Performance advice (e.g., Rosetta translation on Mac) */ PERFORMANCE_TIP = "PERFORMANCE_TIP", /** OCR processing failed for an attachment */ OCR_FAILED = "OCR_FAILED", /** Extraction of structured chart data failed */ CHART_DATA_EXTRACTION_FAILED = "CHART_DATA_EXTRACTION_FAILED", /** Automatic worker path failed, falling back to CDN */ PDF_WORKER_FALLBACK = "PDF_WORKER_FALLBACK", /** General attachment extraction failure */ ATTACHMENT_EXTRACTION_FAILED = "ATTACHMENT_EXTRACTION_FAILED", /** Failed to load a specific page in a multi-page document */ PAGE_LOAD_FAILED = "PAGE_LOAD_FAILED", /** Failed to load a required dynamic dependency */ DEPENDENCY_LOAD_FAILED = "DEPENDENCY_LOAD_FAILED", /** Failed to extract images from a source */ IMAGE_EXTRACTION_FAILED = "IMAGE_EXTRACTION_FAILED", /** Failed to extract annotations from a document */ ANNOTATION_EXTRACTION_FAILED = "ANNOTATION_EXTRACTION_FAILED", /** Failed to process an extracted image bitmap */ IMAGE_PROCESSING_FAILED = "IMAGE_PROCESSING_FAILED", /** Warning about limitations of browser-based generation */ BROWSER_GENERATION_LIMITATION = "BROWSER_GENERATION_LIMITATION", /** Specified sheet range in Excel/ODS export was not found */ SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND", /** Buffer content type does not match the provided or expected file extension */ BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH", /** Failed to detect file type from buffer due to library error or incompatibility */ FILE_TYPE_DETECTION_FAILED = "FILE_TYPE_DETECTION_FAILED", /** No chunks were generated for the document given the current strategy */ EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED", /** A node was skipped because it only contained whitespace */ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED", /** The HTML generator containerWidth option is invalid */ INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH", /** A document's repeated-cell expansion hit the configured cell limit and was truncated */ TABLE_CELL_LIMIT_EXCEEDED = "TABLE_CELL_LIMIT_EXCEEDED", /** A metadata override could not be represented in the destination format's vocabulary */ METADATA_NOT_REPRESENTABLE = "METADATA_NOT_REPRESENTABLE", /** A styleMap output.tag was not an allowed element name and was ignored */ INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG", /** A workbook archive contains no worksheet parts (chartsheet-only workbooks are legitimate) */ NO_WORKSHEETS_FOUND = "NO_WORKSHEETS_FOUND", /** A presentation archive contains no slides (a zero-slide presentation is legitimate) */ NO_SLIDES_FOUND = "NO_SLIDES_FOUND" } /** * Consolidated timeout settings for OCR operations. * Preferred over the individual flat timeout properties on {@link OcrConfig}, * which are now deprecated. * * If a key is present here, it takes priority over the corresponding deprecated * flat property (e.g. `timeout.autoTerminate` wins over `autoTerminateTimeout`). * Set any value to `0` to disable that specific timeout. */ export interface OcrTimeoutConfig { /** * Timeout in milliseconds of inactivity before the OCR worker pool is * automatically terminated and freed. * * The timer resets every time a new OCR job is enqueued. When the last * job completes and this duration passes without a new one, the entire * worker pool is torn down so that no background threads keep the Node.js * process alive unnecessarily. * * Set to `0` to keep workers alive indefinitely (useful when you want to * call {@link terminateOcr} manually at shutdown time). * Default is 10,000 ms (10 seconds). */ autoTerminate?: number; /** * Timeout in milliseconds for initializing a Tesseract worker * (loading the JS runtime, downloading or loading the `.traineddata` * language file) or for re-initializing an existing worker with a * different language. * * Multi-language combinations (e.g. `'por+eng+spa'`) must download a * separate `.traineddata` file for each language and are therefore * particularly susceptible to slow networks. Tune this value upward if * your OCR environment has high network latency or if you are loading * languages from disk in a large container image. * * When the timeout fires, the failed job is rejected with a non-fatal * {@link OfficeWarningType.OCR_FAILED} warning and parsing continues * without OCR output for that image. The stalled worker is terminated * and removed from the pool to prevent thread leaks. * * Set to `0` to wait indefinitely (not recommended for production; a hung * network request will block the entire OCR queue for that language). * Default is 60,000 ms (60 seconds). */ workerLoad?: number; /** * Timeout in milliseconds for the actual OCR text-recognition call * (`worker.recognize(image)`) on an already-initialized Tesseract worker. * * Recognition time scales with image resolution and the number of active * languages. Very high-resolution scans or unusual character sets can * exceed the default. If this timeout fires, the job is rejected with a * non-fatal {@link OfficeWarningType.OCR_FAILED} warning; the worker is * terminated and evicted from the pool because its internal state after a * mid-recognition timeout is undefined. * * Set to `0` to wait indefinitely. * Default is 30,000 ms (30 seconds). */ recognition?: number; } /** * Configuration options for OCR. */ export interface OcrConfig { /** * Language for OCR. * Default is 'eng'. * * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French). * The OCR engine will then attempt to recognize text in any of the specified languages. * * See the list of supported languages and their codes here: * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016 */ language?: string; /** * Path to the Tesseract worker script. * Primarily used for offline/air-gapped environments. * Default is ''. */ workerPath?: string; /** * Path to the Tesseract core script. * Primarily used for offline/air-gapped environments. * Default is ''. */ corePath?: string; /** * Path for Tesseract language files (traineddata). * Primarily used for offline/air-gapped environments. * Default is ''. */ langPath?: string; /** * Consolidated timeout settings for all OCR operations. * * Prefer this over the deprecated flat timeout properties. * If `timeout.autoTerminate` is set, it takes priority over the deprecated `autoTerminateTimeout`. */ timeout?: OcrTimeoutConfig; /** * @deprecated Use `timeout.autoTerminate` instead. * * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated. * Set to 0 to disable auto-termination. * Default is 10,000 (10 seconds). * * If `timeout.autoTerminate` is also set, that value takes priority over this one. */ autoTerminateTimeout?: number; /** * An optional AbortSignal propagated from the main parser configuration to abort active OCR jobs. * If the signal is aborted: * 1. Any pending OCR jobs in the scheduler queue are rejected immediately. * 2. Any active OCR job running on a Tesseract worker will reject, the worker will be * terminated, and it will be removed from the pool to avoid hanging worker threads. * * Developers should prefer passing this at the top level of `parseOffice` (as `config.abortSignal`), * which automatically propagates here. */ abortSignal?: AbortSignal | null; } /** * Configuration options shared across every input format. */ export interface CommonOfficeParserConfig { /** * @deprecated Use `onWarning` instead. * Flag to show all the logs to console in case of an error irrespective of your own handling. * Default is false. */ outputErrorToConsole?: boolean; /** * Callback for warnings or non-fatal errors encountered during parsing. * Allows you to capture issues like OCR failures or attachment extraction errors * without stopping the parsing process. */ onWarning?: (issue: OfficeIssue) => void; /** * The delimiter used for every new line in places that allow multiline text like word. * Default is \n. */ newlineDelimiter?: string; /** * Flag to ignore notes from parsing in files like powerpoint. * Default is false. It includes notes in the parsed text by default. */ ignoreNotes?: boolean; /** * Flag to ignore comments from parsing. * Default is false. */ ignoreComments?: boolean; /** * Flag to ignore headers and footers from parsing. * Default is false. */ ignoreHeadersAndFooters?: boolean; /** * Flag to ignore slide masters from parsing in PowerPoint. * Default is false. */ ignoreSlideMasters?: boolean; /** * @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`. * This option is now completely ignored by all parsers. */ putNotesAtLast?: boolean; /** * Flag to extract attachments like images, charts, etc. * Default is false. */ extractAttachments?: boolean; /** * Flag to include raw content (XML for XML-based formats, RTF for RTF) in the AST. * Default is false. */ includeRawContent?: boolean; /** * Flag to enable OCR for images. * Default is false. */ ocr?: boolean; /** * @deprecated Use `ocrConfig.language` instead. * Language for OCR. * Default is 'eng'. * * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French). * The OCR engine will then attempt to recognize text in any of the specified languages. * * See the list of supported languages and their codes here: * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016 */ ocrLanguage?: string; /** * Shared OCR configuration for worker pooling and offline support. * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`. */ ocrConfig?: OcrConfig; /** * An optional AbortSignal to cancel the parsing operation. * When aborted, the parser immediately rejects with a standard AbortError (DOMException). * * ### Format-Specific Abort Behavior: * - **PDF**: Checked between page loads and before individual image OCR operations. * - **RTF**: Checked before parsing/traversal and before running OCR on image attachments. * - **DOCX/XLSX/PPTX/ODF**: Checked during zip decompression before loading and parsing XML files. * - **CSV/MD/HTML**: Checked at the start of the parsing phase. * * Note: If an OCR operation is currently running on a Tesseract worker when aborted, * the worker will be terminated and removed from the worker pool automatically to prevent leaks. */ abortSignal?: AbortSignal | null; /** * Flag to serialize raw content (XML) as clean, formatted strings. * Only relevant when `includeRawContent` is true. * Default is true. * * If false, the parser will attempt to extract the original raw substring from the * source document instead of re-serializing the DOM node. */ serializeRawContent?: boolean; /** * Flag to preserve original XML whitespace and line endings when serializing. * Only relevant when `includeRawContent` is true and `serializeRawContent` is true. * Default is false. */ preserveXmlWhitespace?: boolean; /** * The URL/path to the PDF.js worker script. * * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors. * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs`. * You can override this with your own local path or a different CDN link. */ pdfWorkerSrc?: string; /** * Flag to include break nodes in the AST. * This is currently only supported for Word documents. (w:br nodes) * * Default is false */ includeBreakNodes?: boolean; /** * Flag to ignore all internal (anchor) links during parsing. * When true, all bookmarks, cross-references, and internal document jumps are stripped * from the AST. Only external URLs will be preserved. * * Use this if you want a "flat" document without any internal interactivity. * * Default is false. */ ignoreInternalLinks?: boolean; /** * Optional hint for the file format. * When a Buffer or ArrayBuffer is passed, the parser relies on magic bytes to detect the file type. * Text-based formats like 'md', 'html', and 'csv' lack reliable magic bytes. * If you are parsing these formats from a Buffer, you must provide this fileType hint. * * This is authoritative and is used to determine the file type, so it should be accurate. * If provided, this bypasses the magic bytes detection and the file extension-based detection either way. * * Default is null. */ fileType?: SupportedFileType | null; /** * Custom delimiter for CSV files. * Defaults to ',' but can be overridden (e.g., ';', '\t'). */ csvDelimiter?: string; /** * Limits and checks applied during ZIP extraction to protect against excessive * memory and resource usage. */ decompressionLimits?: DecompressionLimits; } /** * Format-specific options for HTML (and XHTML/EPUB, which parse through the same code path). * * Note there is deliberately no `MdParserConfig`: the Markdown parser populates its * dialect-provenance metadata (e.g. `AdmonitionMetadata.sourceSyntax`) unconditionally because * doing so costs nothing and changes no existing field's value, so it has nothing to configure. * An empty placeholder interface would be worse than useless here - `interface X {}` accepts any * non-nullish value in TypeScript, so `mdParserConfig: 5` would type-check. */ export interface HtmlParserConfig { /** * Preserve source HTML attributes that no typed metadata field consumed, on * `OfficeContentNode.htmlAttributes`, so they can be replayed on generation. * * Off by default: with it off nothing is populated, so the AST is byte-identical to previous * releases, and the attribute-replay surface stays something a consumer opts into rather than * something switched on for every existing caller. Captured values are sanitized on the way in * *and* on the way out - see `BaseContentNode.htmlAttributes`. * * Defaults to false. */ preserveAttributes?: boolean; /** * Preserve `