/** * How a focus keyword is found in a text: all its words in one sentence, in * any order and in any of their inflected forms ("hanoracul" for "hanorac", * "mărimii potrivite" for "mărimea potrivită", "hoodies" for "hoodie"), * the small words of the language ("din", "for") optional. Romanian and * English have their endings; other languages match word for word. The * agent's analyzer reads a keyword the same way, and so does the check for * a keyword another page already targets (helpers/Keyphrase.php). */ /** A language with rules of its own; empty for the others. */ export type KeyphraseLanguage = 'ro' | 'en' | ''; /** Letters NFD does not split into a base and an accent. */ const LETTERS: Record = { ß: 'ss', æ: 'ae', œ: 'oe', ø: 'o', ł: 'l', đ: 'd', ı: 'i', }; /** * Letters without their accents, so "drumetie" typed on a keyboard without * diacritics finds "drumeție" (either ș or ş) and a keyword finds its * ASCII slug. */ export const fold = (text: string): string => text .normalize('NFD') .replace(/[̀-ͯ]/g, '') .replace(/[ßæœøłđı]/g, letter => LETTERS[letter] ?? letter); /** Lower case, accents folded, punctuation and hyphens as spaces. */ export const normalize = (text: string): string => fold(text.toLowerCase()) .replace(/[‘’“”]/g, "'") .replace(/[^\p{L}\p{N}'\s]/gu, ' ') .replace(/\s+/g, ' ') .trim(); /** The words of a text, normalized; an apostrophe around a word goes. */ export const words = (text: string): string[] => normalize(text) .split(' ') .map(word => word.replace(/^'+|'+$/g, '')) .filter(word => word.length > 0); /** * The sentences of a text or of some HTML: a block (paragraph, heading, list * item, table cell) ends one, and so do . ! ? ; followed by a space, and the * separators of a title (|, –, », a spaced hyphen). */ export const sentences = (text: string): string[] => text .replace(//g, ' ') .replace(/<(script|style)[\s\S]*?<\/\1>/gi, ' ') .replace( /<\/(p|h[1-6]|li|dt|dd|td|th|tr|div|figcaption|blockquote|summary|caption|pre|section|article)>|/gi, '\n' ) .replace(/<[^>]+>/g, ' ') .replace(/ /g, ' ') .replace(/&[a-z#0-9]+;/gi, ' ') .replace(/([.!?;…])\s/g, '$1\n') .replace(/\s[|•·–—»«]\s|\s-\s/g, '\n') .split('\n') .map(part => part.trim()) .filter(Boolean); /** Words a keyword may leave out, folded: articles, prepositions, conjunctions, pronouns. */ const FUNCTION_WORDS: Record<'ro' | 'en', ReadonlySet> = { ro: new Set( `a al ale ai alor o un una unei unui unor niste cel cea cei cele celui celei celor de din dintre la in intr intru pe cu pentru prin spre despre dupa intre sub peste pana catre ca si sau ori dar iar ci nici daca care ce cine cum cand unde cat cata cati cate sa se isi iti imi ma te ne va il ii le lui lor mai foarte acest aceasta acesti aceste acel acea acei acele asta este e esti sunt era fi fost vor ar am are au eu tu el ea noi voi ei ele meu mea mei mele tau ta tai tale nostru noastra vostru voastra`.split( ' ' ) ), en: new Set( `a an the of for to in on at by with from and or nor as into onto about over per via vs versus your our my his her its their this that these those is are be was were been how what why when where which who whom do does can it you we i me us them than so if`.split( ' ' ) ), }; /** Words only one of the two languages uses, to tell a page's language from its text. */ const MARKERS: Record<'ro' | 'en', ReadonlySet> = { ro: new Set( `si sau este sunt pentru cu din pe care mai sa cum unde cand dupa intre prin despre iar dar nu fi fost poti esti iti acest aceasta aceste`.split( ' ' ) ), en: new Set( `the and of to is for with that this you your it on be from by at how what which was were will can has have an`.split( ' ' ) ), }; /** Letters only Romanian writes: ă, ș, ț (either comma or cedilla). */ const ROMANIAN_LETTERS = /[ăĂșȘşŞțȚţŢ]/g; /** * The language a text is written in, as far as its small words and letters * tell: Romanian and English, or empty when it says too little or neither. * * @param text - The page's title and text. * @return The language. */ export const detectLanguage = (text: string): KeyphraseLanguage => { const counts = { ro: 0, en: 0 }; words(text) .slice(0, 3000) .forEach(word => { if (MARKERS.ro.has(word)) counts.ro += 1; if (MARKERS.en.has(word)) counts.en += 1; }); const letters = (text.match(ROMANIAN_LETTERS) ?? []).length; if ( (letters >= 2 && counts.ro >= counts.en) || (counts.ro >= 4 && counts.ro >= counts.en * 3) ) { return 'ro'; } if (counts.en >= 3 && counts.en >= counts.ro * 2) { return 'en'; } return ''; }; /** * The language whose rules a page is read by: the one a multilingual plugin * gives the post, else the one its text is written in, else the site's. * * @param given - The post's language from Polylang or WPML, empty without one. * @param text - The page's title and text. * @param site - The site's language. * @return The language. */ export const pageLanguage = ( given: string | undefined, text: string, site: string | undefined ): KeyphraseLanguage => { const ruled = (code: string | undefined): KeyphraseLanguage => { const base = (code ?? '').toLowerCase().slice(0, 2); return base === 'ro' || base === 'en' ? base : ''; }; if (given) { return ruled(given); } return detectLanguage(text) || ruled(site); }; /** Romanian endings of case, number and the definite article, the longest first. */ const ROMANIAN_ENDINGS = [ 'urilor', 'urile', 'ilor', 'elor', 'ului', 'uri', 'ele', 'ile', 'lor', 'ul', 'le', 'ii', 'ei', 'ea', 'a', 'e', 'i', 'u', ]; /** * A Romanian word without its ending, the same for all its forms: * hanorac, hanoracul, hanoracului, hanorace, hanoracele → hanorac; * mărime, mărimea, mărimii, mărimi, mărimile → marim; tigaie, tigaia, * tigăi, tigăile → tiga; neagră, negru, negri → negr. Rule-based, no * dictionary, and never below three letters. * * @param word - A folded word. * @return Its stem. */ export const romanianStem = (word: string): string => { let stem = word; const ending = ROMANIAN_ENDINGS.find( suffix => stem.endsWith(suffix) && stem.length - suffix.length >= 3 ); if (ending) { stem = stem.slice(0, -ending.length); } // tigai → tiga, rochi → roch: the i the plural and the article leave. if (stem.endsWith('i') && stem.length > 3) { stem = stem.slice(0, -1); } // floare/flori, geantă/genți, seară/seri: the diphthong of the singular. return stem.replace(/oa/g, 'o').replace(/ea/g, 'e'); }; /** * The forms an English word may stand for: itself, without the possessive, * and its singulars (hoodies → hoodie, cities → city, boxes → box). * * @param word - A folded word. * @return Its forms. */ export const englishForms = (word: string): string[] => { const forms = new Set([word]); const base = word.replace(/'s$/, ''); forms.add(base); if (base.length > 3 && base.endsWith('s') && !/(ss|us|is)$/.test(base)) { forms.add(base.slice(0, -1)); if (base.endsWith('ies')) { forms.add(`${base.slice(0, -3)}y`); } if (/(s|x|z|ch|sh|o)es$/.test(base)) { forms.add(base.slice(0, -2)); } } return [...forms]; }; /** Forms already worked out: a page repeats its words, the analysis runs on every keystroke. */ const known = new Map(); /** The forms two words must share to be one word. */ const formsOf = (word: string, language: KeyphraseLanguage): string[] => { if (!language) { return [word]; } const key = `${language}:${word}`; let forms = known.get(key); if (!forms) { forms = language === 'ro' ? [romanianStem(word)] : englishForms(word); if (known.size > 20000) { known.clear(); } known.set(key, forms); } return forms; }; /** A focus keyword ready to be looked for: the forms of each word it needs. */ export interface Keyphrase { /** One entry per word that has to be found, with the forms it may take. */ words: string[][]; language: KeyphraseLanguage; } /** * The words of a keyword that have to be found, with their forms: the small * words of the language are left out, unless the keyword is nothing else. * * @param keyword - The focus keyword as typed. * @param language - The page's language. * @return The keyphrase. */ export const keyphrase = ( keyword: string, language: KeyphraseLanguage ): Keyphrase => { const all = words(keyword); const small = language ? FUNCTION_WORDS[language] : undefined; const needed = small ? all.filter(word => !small.has(word)) : all; return { words: (needed.length > 0 ? needed : all).map(word => formsOf(word, language) ), language, }; }; const sameWord = (forms: string[], other: string[]): boolean => forms.some(form => other.includes(form)); /** * How many times each keyphrase stands in one sentence, the one with the * most words first: a word serves one mention only, so "hoodie size" and * "hoodie" in "the hoodie size chart" are one mention, not two. * * @param sentence - The words of a sentence. * @param phrases - The keyphrases. * @return The mentions. */ export const mentions = (sentence: string[], phrases: Keyphrase[]): number => { if (phrases.length === 0 || sentence.length === 0) { return 0; } const language = phrases[0].language; const forms = sentence.map(word => formsOf(word, language)); const used = forms.map(() => false); let count = 0; [...phrases] .filter(phrase => phrase.words.length > 0) .sort((a, b) => b.words.length - a.words.length) .forEach(phrase => { for (;;) { const picked: number[] = []; const complete = phrase.words.every(needed => { const at = forms.findIndex( (word, index) => !used[index] && !picked.includes(index) && sameWord(needed, word) ); if (at < 0) { return false; } picked.push(at); return true; }); if (!complete) { return; } picked.forEach(index => { used[index] = true; }); count += 1; } }); return count; }; /** * Whether the keyphrase stands in one of the sentences of a text or HTML. * * @param text - The text. * @param phrase - The keyphrase. * @return Whether it is there. */ export const hasKeyphrase = (text: string, phrase: Keyphrase): boolean => phrase.words.length > 0 && sentences(text).some(sentence => mentions(words(sentence), [phrase]) > 0); /** * Mentions of every keyphrase in a text, sentence by sentence. * * @param text - The text or HTML. * @param phrases - The keyphrases. * @return The mentions. */ export const countKeyphrases = (text: string, phrases: Keyphrase[]): number => sentences(text).reduce( (sum, sentence) => sum + mentions(words(sentence), phrases), 0 );