/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * Fetch GeoNames per-country gazetteer dumps. These are the 19-column `.txt` files under * `https://download.geonames.org/export/dump/`, the export directory rather than the postal export * at `export/zip/` that `geonames-postal.ts` handles. The dumps contain feature classes and codes * (column 8: `pplc` national capital, `ppla` first-order administrative seat), which is what the * capitals reference build consumes. * * `countryInfo.txt` supplies the country catalog. It lists each country GeoNames publishes with one * row per ISO alpha-2 code. Each row also lists the capital. The capitals build checks it against * its `pplc` extraction. The tool fetches this file first and derives the country set from it. A dump * absent from disk then counts as a gap in the source's own catalog. * The dump directory may contain files this tool never fetched. The tool keeps present files unchanged. * It checks each present `.txt` for the 19-column gazetteer format. GeoNames postal exports share * the basename, so a postal export receives `wrong_format_present` and does not count as coverage. */ import { pathExists, readFileHead, readLocalBuffer, readLocalTextFile } from "@mailwoman/core/fs/readers" import { makeDirectories, removePathIfPresent } from "@mailwoman/core/fs/writers" import { extractZipEntry } from "@mailwoman/core/fs/zip" import { sha256File } from "@mailwoman/core/hash" import { TSVSpliterator } from "spliterator" import type { BaseFetchOptions, FetchSummary } from "#tools/fetch/download" import { downloadToFile, HTTPStatusError, writeManifest } from "#tools/fetch/download" /** * The one status that means "the source does not publish this country" rather than "the transfer failed". */ const HTTP_NOT_FOUND = 404 const SLUG = "geonames-dump" /** * GeoNames' gazetteer-dump directory — one zip per ISO alpha-2 code holding `.txt`, * plus `countryInfo.txt` as a bare text file. */ const BASE_URL = "https://download.geonames.org/export/dump" export interface FetchGeonamesDumpOptions extends BaseFetchOptions { /** * ISO alpha-2 codes, any casing. * * Absent → every country `countryInfo.txt` enumerates. */ countries?: readonly string[] /** * Dump directory to read from. * * Defaults to GeoNames' own. * Exists so the 404 and coverage behavior can be exercised against a local server. */ baseURL?: string /** * Refetch a dump whose `.txt` already exists. * * Default false — the tool fills gaps. */ force?: boolean } interface GeonamesDumpFileEntry { country: string filename: string source_url: string sha256: string bytes: number } export interface GeonamesDumpManifest { source: string base_url: string license: string attribution: string downloaded_at: string files: GeonamesDumpFileEntry[] /** * `.txt` files already on disk and leaves unchanged — the hand-fetched population this tool extends. */ skipped_present: string[] /** * Countries in the source catalog that the source's dump directory nonetheless 404s. * * A fact about the source, recorded so a later reader does not spend the fetch to rediscover it. */ unavailable: string[] /** * Present `.txt` files that are not 19-column gazetteer dumps. * GeoNames' postal exports share the same basename. * * Left in place (this tool never overwrites data it did not fetch). * The fix is to move the file to its own home and rerun. */ wrong_format_present: string[] } /** * Column count of a gazetteer dump row. * * GeoNames postal exports have 12 columns and share the `.txt` basename. */ const GAZETTEER_DUMP_COLUMNS = 19 /** * True when the first non-empty line contains the gazetteer dump's 19 tab-separated columns. * * Accepts a partial head read. * The first line is the whole question, so callers need not hand it a resident 350 MB dump. * * Walk the string directly instead of constructing a spliterator. * The capitals builder already holds each country dump as a string. * * A byte-oriented spliterator would UTF-8 encode that input and allocate up to * another 350 MB just to inspect the first non-empty line. */ export function looksLikeGazetteerDump(text: string): boolean { let start = 0 while (start < text.length) { const end = text.indexOf("\n", start) const line = end === -1 ? text.slice(start) : text.slice(start, end) if (line.trim()) { let tabs = 0 for (let i = line.indexOf("\t"); i !== -1; i = line.indexOf("\t", i + 1)) { tabs++ } return tabs === GAZETTEER_DUMP_COLUMNS - 1 } if (end === -1) break start = end + 1 } return false } /** * Parse the ISO codes (column 1) and capital names (column 6) out of `countryInfo.txt`. * `#`-prefixed lines are the file's own commentary. */ export function parseCountryInfo(text: string): Array<{ country: string; capital: string }> { const rows: Array<{ country: string; capital: string }> = [] for (const cols of TSVSpliterator.from(text, { header: false, enableQuoteHandling: false })) { const country = cols[0]?.toUpperCase() if (!country || country.startsWith("#")) continue if (country.length === 2) { rows.push({ country, capital: cols[5] ?? "" }) } } return rows } /** * Bytes read to classify a present file: enough to cover a first dump row whose alternate-names * column runs long (they reach several KB), a fraction of the largest dumps (US.txt is ~350 MB). */ const FORMAT_SNIFF_BYTES = 65_536 /** * Download `countryInfo.txt` and each missing `.zip`. * * Extract each dump to `/.txt` beside the hand-fetched files. * The `manifest.json` lists fetched countries. * * It also lists countries skipped because their files were already present * and countries unavailable from the source. */ export async function fetchGeonamesDumps( options: FetchGeonamesDumpOptions, report?: (line: string) => void ): Promise { await makeDirectories(options.outRoot) const baseURL = options.baseURL ?? BASE_URL const countryInfoDest = options.outRoot("countryInfo.txt") await downloadToFile({ url: `${baseURL}/countryInfo.txt`, dest: countryInfoDest, timeoutMs: 120_000, retries: 2, report, }) const catalog = parseCountryInfo(await readLocalTextFile(countryInfoDest)) const countries = options.countries?.map((code) => code.trim().toUpperCase()) ?? catalog.map((row) => row.country) const entries: GeonamesDumpFileEntry[] = [] const failedCodes: string[] = [] const unavailable: string[] = [] const skippedPresent: string[] = [] const wrongFormatPresent: string[] = [] let fetched = 0 for (const country of countries) { const txtDest = options.outRoot(`${country}.txt`) if (!options.force && (await pathExists(txtDest))) { if (looksLikeGazetteerDump(await readFileHead(txtDest, FORMAT_SNIFF_BYTES))) { skippedPresent.push(country) } else { report?.(`✗ ${country}.txt is present but is not a 19-column gazetteer dump — move it aside and rerun`) wrongFormatPresent.push(country) } continue } const filename = `${country}.zip` const url = `${baseURL}/${filename}` const zipDest = options.outRoot(filename) report?.(`=== ${SLUG} / ${country}`) try { await downloadToFile({ url, dest: zipDest, timeoutMs: 300_000, retries: 2, report }) await extractZipEntry(zipDest, `${country}.txt`, txtDest) await removePathIfPresent(zipDest) entries.push({ country, filename: `${country}.txt`, source_url: url, sha256: await sha256File(txtDest), bytes: (await readLocalBuffer(txtDest)).byteLength, }) fetched++ } catch (error) { await removePathIfPresent(zipDest) const message = error instanceof Error ? error.message : String(error) // Branch on the typed status. // The error message contains a URL. // URLs can contain any substring. // Branch on the typed status instead of matching message text. if (error instanceof HTTPStatusError && error.status === HTTP_NOT_FOUND) { report?.(`✗ ${country}: GeoNames publishes no gazetteer dump for this country`) unavailable.push(country) } else { report?.(`✗ ${country}: ${message}`) } failedCodes.push(country) } } const manifest: GeonamesDumpManifest = { source: SLUG, base_url: baseURL, license: "CC-BY-4.0", attribution: "GeoNames", downloaded_at: new Date().toISOString(), files: entries, skipped_present: skippedPresent.toSorted(), unavailable, wrong_format_present: wrongFormatPresent.toSorted(), } await writeManifest(options.outRoot("MANIFEST.json"), manifest) return { fetched, skipped: skippedPresent.length, failed: failedCodes.length, failedCodes, skippedPresent } }