/** * Text extraction for the two document formats a CV or a receipt comes in * (issue #263). A 5 MiB PDF costs ~6.6 MB of base64 in the model's context for * a content that fits in a few KB of text; extracting server-side is what * makes reading resumes in batches affordable. * * - PDF: `unpdf` (a serverless build of pdf.js, no native binary, ~2 MB). * - DOCX: the file is a zip whose `word/document.xml` holds the text; the zip * is read here with `node:zlib` only (STORED and DEFLATE entries) — no * dependency for a 60-line format. */ export interface ExtractedText { text: string; /** Page count for PDFs; absent for DOCX. */ pages?: number; } export declare const IMAGE_MIMES: Set; export declare function isImageMime(mime: string): boolean; export declare function isPdfMime(mime: string): boolean; /** * Is this a DOCX? BoondManager serves `.docx` resumes as `application/msword` * (observed live on 2026-09-26, issue #311) — the legacy Word mime — so the * mime alone misses them. Accept the real mime, a `.docx` extension on any * generic mime, or the zip signature (`PK\x03\x04`) on a Word / octet-stream * mime: an OLE `.doc` never starts with it, a `.docx` always does. */ export declare function isDocxMime(mime: string, filename?: string, data?: Buffer): boolean; export declare function extractPdfText(data: Buffer): Promise; export declare function extractDocxText(data: Buffer): ExtractedText; /** The bytes of one entry of a zip archive, or undefined when the name is absent. */ export declare function readZipEntry(zip: Buffer, name: string): Buffer | undefined; //# sourceMappingURL=document-text.d.ts.map