export declare const normalizeLabel: (label: string) => readonly string[]; export declare const headTokens: (tokens: readonly string[]) => readonly string[]; export declare const similarity: (left: string, right: string) => number; export interface IdentitySubject { readonly id: string; readonly kind: string; readonly labels: readonly string[]; readonly owner?: string; readonly neighbours: ReadonlySet; readonly distinctFrom: ReadonlySet; } export interface NearDuplicatePair { readonly left: string; readonly right: string; readonly score: number; readonly corroboration: 'lexical' | 'owner' | 'neighbourhood'; } export declare const strongLexicalThreshold = 0.95; export declare const moderateLexicalThreshold = 0.8; export declare const lexicalScore: (left: IdentitySubject, right: IdentitySubject) => number; /** * Every candidate near-duplicate pair, sorted by subject id. Only subjects * of the same qualified kind are compared: kind is the cheapest structural * disagreement there is, and it also bounds the pairwise cost to the square * of the largest kind bucket rather than of the whole model. */ export declare const findNearDuplicates: (subjects: readonly IdentitySubject[]) => readonly NearDuplicatePair[]; /** * Both members of every candidate pair, mapped to their counterparts. The * question attaches to both sides deliberately: `ask --advise` filters open * questions to the subjects in a slice, so a finding recorded only against * the lexically-first member would vanish from a slice seeded on the other. */ export declare const nearDuplicateIndex: (subjects: readonly IdentitySubject[]) => ReadonlyMap;