export interface Block { blockName: string; startCode: number; endCode: number; } /** Character is a raw representation of a character. */ export interface Character { /** numeric character code (a non-negative integer) */ code: number; /** character name (ASCII only) */ name: string; /** general category */ cat: string; /** canonical combining class (missing if == 0 "not reordered") */ comb?: number; /** bidirectional category (missing if == 'L' "Letter") */ bidi?: string; /** decomposition type and mapping */ decompType?: string; decomp?: number[]; /** numeric value of character (may be a fraction, so it not unevaluated) */ num?: string; /** true if character is mirrored in bidirectional text (missing otherwise) */ bidiMirror?: boolean; /** Unicode 1.0 name, if it differs from the current name */ oldName?: string; /** simple uppercase mapping */ upper?: number; /** simple lowercase mapping */ lower?: number; /** simple titlecase mapping */ title?: number; } export declare function getBlocks(): Block[]; export declare function getCharacters(): Character[]; /** evaluateNum(num: string): number Take a Character.num string value and return a native Javascript number. 1000000000000 is the biggest numeric value defined for any Unicode character (U+16B61 "PAHAWH HMONG NUMBER TRILLIONS"), so what we have to worry about are the fractions (which can have negative signs, though U+0F33 "TIBETAN DIGIT HALF ZERO" is the only one of those) */ export declare function evaluateNum(num: any): number; /** Snippet from `Blocks.txt`: 0000..007F; Basic Latin 0080..00FF; Latin-1 Supplement 0100..017F; Latin Extended-A 0180..024F; Latin Extended-B 0250..02AF; IPA Extensions */ export declare function parseBlocks(Blocks_txt: string): Block[]; /** Snippet from `UnicodeData.txt`: 00A0;NO-BREAK SPACE;Zs;0;CS; 0020;;;;N;NON-BREAKING SPACE;;;; 00A1;INVERTED EXCLAMATION MARK;Po;0;ON;;;;;N;;;;; 00A2;CENT SIGN;Sc;0;ET;;;;;N;;;;; 00A3;POUND SIGN;Sc;0;ET;;;;;N;;;;; 00A4;CURRENCY SIGN;Sc;0;ET;;;;;N;;;;; 00A5;YEN SIGN;Sc;0;ET;;;;;N;;;;; 00A6;BROKEN BAR;So;0;ON;;;;;N;BROKEN VERTICAL BAR;;;; If there were a header of column names, it might look like this: Code;Name;Cat;Comb;BidiC;Decomp;Num1;Num2;Num3;BidiM;Unicode_1_Name;ISO_Comment;Upper;Lower;Title There are 14 ;'s per line, and so there are 15 fields per UnicodeDatum: 0. Code 1. Name 2. General_Category 3. Canonical_Combining_Class 4. Bidi_Class 5. Decomposition_Mapping 6. Numeric Value if decimal 7. Numeric Value if only digit 8. Numeric Value otherwise 9. Bidi_Mirrored 10. Unicode_1_Name 11. ISO_Comment (always empty) 12. Simple_Uppercase_Mapping 13. Simple_Lowercase_Mapping 14. Simple_Titlecase_Mapping */ export declare function parseUnicodeData(UnicodeData_txt: string): Character[];