/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * Fetch the full French BAN (Base Adresse Nationale) — all metropolitan départements (01-95, 2A, * 2B) plus 5 overseas DOM/TOM (971-976 excl. 975). * * Source: https://adresse.data.gouv.fr/data/ban/adresses/latest/csv/ * Licence: Licence Ouverte 2.0 (attribution required — Tier B). * * Files already present with matching sha256 are skipped (re-runnable). Downloads `.csv.gz`, * decompresses to `.csv`, deletes the `.gz` artifact. One shared `MANIFEST.json` at * `/ban/MANIFEST.json` covers all codes. * * Invoke via `mailwoman corpus fetch ban --out-root `. Built-in `fetch` with gzip/brotli * decompression replaces curl; native `node:zlib` gunzip replaces the `gunzip` subprocess; no * Python. */ import { gunzip } from "@mailwoman/core/fs/compression" import { BYTES_PER_KIB, ByteFormatter } from "@mailwoman/core/fs/formatters" import { isFile, readLocalBuffer, tryStat } from "@mailwoman/core/fs/readers" import { makeDirectories, removePath, writeLocalFile } from "@mailwoman/core/fs/writers" import { sha256File } from "@mailwoman/core/hash" import { sleep } from "@mailwoman/core/utils/sleep" import { join } from "path-ts" import type { BaseFetchOptions, FetchSummary } from "#tools/fetch/download/index" import { downloadToFile, loadManifestEntries, writeManifest } from "#tools/fetch/download/index" /** * Bytes per KiB — the divisor for human-readable sizes, and the floor below which a "download" is an error page rather * than data. */ const BASE_URL = "https://adresse.data.gouv.fr/data/ban/adresses/latest/csv" /** * All département codes — metropolitan 01-95 (with 2A/2B for Corsica instead of 20) plus overseas DOM/TOM. Codes do not * change. */ const DEPT_CODES = [ "01", "02", "03", "04", "05", "06", "07", "08", "09", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "21", "22", "23", "24", "25", "26", "27", "28", "29", "2A", "2B", "30", "31", "32", "33", "34", "35", "36", "37", "38", "39", "40", "41", "42", "43", "44", "45", "46", "47", "48", "49", "50", "51", "52", "53", "54", "55", "56", "57", "58", "59", "60", "61", "62", "63", "64", "65", "66", "67", "68", "69", "70", "71", "72", "73", "74", "75", "76", "77", "78", "79", "80", "81", "82", "83", "84", "85", "86", "87", "88", "89", "90", "91", "92", "93", "94", "95", "971", "972", "973", "974", "976", ] export interface BanManifestEntry { dept_code: string filename: string source_url: string downloaded_at: string sha256: string bytes: number } export type FetchBanOptions = BaseFetchOptions export async function fetchBan(options: FetchBanOptions, report?: (line: string) => void): Promise { const banDir = join(options.outRoot, "ban") const manifestPath = join(banDir, "MANIFEST.json") await makeDirectories(banDir) // Load existing entries (code -> entry): skip detection + preservation of untouched codes. const entries = await loadManifestEntries(manifestPath, (entry) => entry.dept_code) let fetched = 0 let skipped = 0 let failed = 0 const failedCodes: string[] = [] for (const code of DEPT_CODES) { const filename = `adresses-${code}.csv` const gzFile = join(banDir, `${filename}.gz`) const csvFile = join(banDir, filename) const url = `${BASE_URL}/adresses-${code}.csv.gz` report?.(`=== dept ${code}`) // If the CSV already exists, compare its sha256 against the manifest. if (await isFile(csvFile)) { const existingSha = await sha256File(csvFile) const recordedSha = entries.get(code)?.sha256 if (recordedSha && existingSha === recordedSha) { report?.(` → already present + sha matches — skipping`) skipped++ continue } report?.(` → present but sha mismatch or no manifest entry — re-fetching`) await removePath(csvFile) } // Download the gzipped CSV. try { await downloadToFile({ url, dest: gzFile, timeoutMs: 600_000, headers: { "Accept-Encoding": "gzip, br" }, report, }) } catch (error) { report?.(` ✗ download failed: ${url} (${(error as Error).message})`) failed++ failedCodes.push(code) continue } // Guard against truncated 404/error pages. const gzSize = (await tryStat(gzFile))?.size ?? 0 if (gzSize < BYTES_PER_KIB) { report?.(` ✗ response too small (${gzSize} bytes) — probable 404 / error page`) await removePath(gzFile) failed++ failedCodes.push(code) continue } // Decompress in-place; delete the .gz. try { await writeLocalFile(await gunzip(await readLocalBuffer(gzFile)), csvFile) } catch (error) { report?.(` ✗ decompress failed: ${(error as Error).message}`) await removePath(gzFile) failed++ failedCodes.push(code) continue } await removePath(gzFile) if (!(await isFile(csvFile))) { report?.(` ✗ decompressed file not found at ${csvFile}`) failed++ failedCodes.push(code) continue } const bytes = (await tryStat(csvFile))?.size ?? 0 const sha = await sha256File(csvFile) entries.set(code, { dept_code: code, filename, source_url: url, downloaded_at: new Date().toISOString(), sha256: sha, bytes, }) report?.(` ✓ ${ByteFormatter.formatIEC(bytes)} sha256=${sha}`) fetched++ // Be a polite citizen — short pause between requests. await sleep(200) } // Write the consolidated MANIFEST.json (entries sorted by dept_code, codepoint order). const sorted = [...entries.values()].toSorted((a, b) => a.dept_code < b.dept_code ? -1 : a.dept_code > b.dept_code ? 1 : 0 ) await writeManifest(manifestPath, sorted) report?.(`Wrote ${manifestPath} with ${sorted.length} entries.`) report?.(`=== summary ===`) report?.(`fetched: ${fetched}`) report?.(`skipped: ${skipped} (already present + sha matched)`) report?.(`failed: ${failed}`) if (failedCodes.length) { report?.(`failed codes: ${failedCodes.join(" ")}`) } return { fetched, skipped, failed, failedCodes } }