/** Letters without their accents, so a match survives the editor's typing. */ const fold = (text: string): string => text.normalize('NFD').replace(/[̀-ͯ]/g, ''); const normalize = (text: string): string => fold(text) .toLowerCase() .replace(/[‘’“”]/g, "'") .replace(/\s+/g, ' '); /** * The spans of the HTML that are plain text outside every tag, and outside * the links, headings and block comments a new link has no business in. */ const textSpans = (html: string): { start: number; end: number }[] => { const spans: { start: number; end: number }[] = []; const skip = /||<(script|style)\b[\s\S]*?<\/\2>||<[^>]+>/gi; let at = 0; for (const match of html.matchAll(skip)) { if (match.index !== undefined && match.index > at) { spans.push({ start: at, end: match.index }); } at = (match.index ?? 0) + match[0].length; } if (at < html.length) { spans.push({ start: at, end: html.length }); } return spans; }; /** * Wrap the first plain-text occurrence of `anchor` in a link to `url`. * * The match ignores case, accents and how the whitespace fell, and skips * text that already sits in a link or a heading, so a suggestion never * nests one link in another or rewrites a subheading. * * @returns The content with the link, or null when the anchor is not in it. */ export const linkAnchor = ( html: string, anchor: string, url: string ): string | null => { const needle = normalize(anchor).trim(); if (needle === '' || url.trim() === '') { return null; } // The anchor may have been typed with other whitespace; match its words in order. const pattern = new RegExp( needle .split(' ') .map(word => word.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')) .join('[\\s\\u00a0]+'), 'i' ); for (const span of textSpans(html)) { const text = html.slice(span.start, span.end); const match = pattern.exec(fold(text)); if (!match) { continue; } const start = span.start + match.index; const end = start + match[0].length; const shown = html.slice(start, end); return `${html.slice(0, start)}${shown}${html.slice(end)}`; } return null; };