/** * Ideas for the focus keyword, read from the page itself: the phrases its * title and headings use and its text keeps coming back to. Nothing leaves * the editor and nothing is charged. */ /** * Words no keyword starts or ends with, in English and Romanian: the small * words, and verbs and adjectives that wait for their noun ("choose the * right", "ești între două"). */ const STOP = new Set( `a about above after again against all also am an and any are as at be because been before being below between both but by can could did do does doing down during each even every few first for from further get gets good had has have having he her here hers him his how i if in into is it its just like lot made make makes many may me might more most much must my need needs new no nor not now of off on once one only or other our out over own per really same second she should so some still such take takes than that the their them then there these they thing things this those three through to too two under until up use used very was way we well were what when where which while who whom why will with within without would you your yours across along come comes ever rather right choose choosing pick find want know see look keep go put less several various whether acea aceasta această acel acela acest acesta al ale alt altă am ar are asta au aici ai astfel avea ca care cat cât ce cea cei cel cele cu cum da dacă dar de deci deja din doi după e ea el ei ele era este eu fac face fi fie fost iar ii îi in în la le lor lui mai mult nici noi nu o ori pe pentru prima primul prin sa să se si şi și sau sub sunt tot toate toti toţi toți trei un una unei unor unui va vor voi ești esti te îți iti îmi imi își isi mă ma ne vă îl il li mi ți ti între intre două doua dintre printre spre despre până pana către catre fără fara lângă langa acum apoi acolo unde când cand câte cate câți cati ceea ceva cineva fiecare orice oricare prea foarte doar încă inca însă insa deoarece fiindcă fiind poate pot poți poti trebuie suntem sunteți avem aveți aș ați eram erau meu mea mei mele tău tau ta tăi tai tale său săi sale nostru noastră noastra noștri noastre vostru voastră alegi alege găsești gasesti găsi gasi vrei faci`.split( /\s+/ ) ); const clean = (text: string): string => text .toLowerCase() .replace(/[‘’]/g, "'") .replace(/[^\p{L}\p{N}'\s-]/gu, ' ') .replace(/\s+/g, ' ') .trim(); /** The text of some HTML, one line per block, without its tags. */ const plain = (html: string): string => html .replace(/<(script|style)[^>]*>[\s\S]*?<\/\1>/gi, ' ') .replace(/<\/(p|h[1-6]|li|div|td|th|blockquote|figcaption)>/gi, '\n') .replace(//gi, '\n') .replace(/<[^>]+>/g, ' ') .replace(/ /g, ' ') .replace(/&/g, '&'); /** Sentences and lines, the stretches a phrase may not cross. */ const stretches = (text: string): string[] => text .split(/[.!?;:\n•·|()[\]"“”]+/) .map(clean) .filter(Boolean); /** Every run of two or three words that starts and ends on a word that means something. */ const phrasesOf = (stretch: string): string[] => { const words = stretch.split(' '); const found: string[] = []; for (const size of [2, 3]) { for (let at = 0; at + size <= words.length; at++) { const run = words.slice(at, at + size); const first = run[0]; const last = run[run.length - 1]; if ( STOP.has(first) || STOP.has(last) || run.some(word => word.length < 2 || /^\d+$/.test(word)) ) { continue; } found.push(run.join(' ')); } } return found; }; const contains = (outer: string, inner: string): boolean => ` ${outer} `.includes(` ${inner} `); /** * The phrases worth trying as the focus keyword, the strongest first: a * phrase in the title counts three times, in a heading twice, and one the * text uses once only is left out unless the title has it. * * @param page - The page as the editor has it. * @param page.title - Its title. * @param page.html - Its content. * @param existing - Keywords already set, left out. * @param limit - How many ideas. * @return Ideas, in lower case. */ export const keywordIdeas = ( page: { title: string; html: string }, existing: string[] = [], limit = 5 ): string[] => { const text = plain(page.html); const headings = [ ...page.html.matchAll(/]*>([\s\S]*?)<\/h[1-4]>/gi), ].map(match => plain(match[1])); const score = new Map(); const count = new Map(); const add = (source: string, weight: number, counts: boolean): void => { stretches(source).forEach(stretch => phrasesOf(stretch).forEach(phrase => { score.set(phrase, (score.get(phrase) ?? 0) + weight); if (counts) count.set(phrase, (count.get(phrase) ?? 0) + 1); }) ); }; add(text, 1, true); add(page.title, 3, false); headings.forEach(heading => add(heading, 2, false)); const inTitle = new Set(stretches(page.title).flatMap(phrasesOf)); const inHeadings = new Set( headings.flatMap(heading => stretches(heading).flatMap(phrasesOf)) ); const taken = existing.map(clean); // A long text repeats a phrase by chance; it has to come back more often. const often = text.split(/\s+/).length > 600 ? 3 : 2; const candidates = [...score.entries()].filter(([phrase]) => { const shown = inTitle.has(phrase) || inHeadings.has(phrase); const words = phrase.split(' '); // "list of measurable" is a piece of a longer phrase; "tigaie din // fontă" in a title is a keyword. if (words.length === 3 && STOP.has(words[1]) && !shown) { return false; } return (count.get(phrase) ?? 0) >= often || inTitle.has(phrase); }); // "cast iron" gives way to "cast iron pan" when the longer one carries // most of its weight: the fuller phrase is the one people type. const fuller = candidates.filter( ([phrase, weight]) => !candidates.some( ([other, otherWeight]) => other.length > phrase.length && contains(other, phrase) && otherWeight >= weight * 0.6 ) ); const picked: string[] = []; fuller .sort((a, b) => b[1] - a[1] || b[0].length - a[0].length) .forEach(([phrase]) => { if ( picked.length >= limit || taken.some( keyword => contains(keyword, phrase) || contains(phrase, keyword) ) || picked.some(other => contains(other, phrase) || contains(phrase, other)) ) { return; } picked.push(phrase); }); return picked; };