/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * Re-fetch the NPPES (National Plan and Provider Enumeration System) full monthly data * dissemination file. ~7M provider rows with venue+address data. Source for the `usgov-nppes` * adapter. US Public Domain. * * The file is published monthly by CMS. This module discovers the current filename by scraping the * NPI_Files.html index, then downloads the ZIP and extracts only the main registry CSV * (npidata_pfile_*.csv). The smaller endpoint/othername/pl files stay zipped — we don't need them. * * Uses Node's built-in fetch (gzip/brotli) to parse the HTML index and download the ZIP, and * streaming sha256 instead of sha256sum. The ZIP is unpacked with the `unzip` binary via * `node:child_process` (no clean Node equivalent for member listing + selective extraction). NOTE: * the old bash fetcher used `curl --continue-at -` to resume a partial download; native fetch has * no resume, so a partial run re-downloads from the start. * * Invoke via `mailwoman corpus fetch nppes --out-root `. Idempotent: if dest CSV exists and * sha256 matches MANIFEST, skips download. */ /* oxlint-disable sister-software/prefer-region-over-marks -- these markers label steps inside one procedure, not sections of declarations. A region there folds nothing a reader wants folded. */ import { APIClient, pluckResponseData } from "@mailwoman/core/api" import { statPath, pathExists } from "@mailwoman/core/fs/readers" import { makeDirectories, removePathIfPresent } from "@mailwoman/core/fs/writers" import { extractZipEntry, listZipEntries } from "@mailwoman/core/fs/zip" import { sha256File } from "@mailwoman/core/hash" import { basename, join } from "path-ts" import type { BaseFetchOptions, FetchSummary, SourceManifest } from "#tools/fetch/download/index" import { downloadToFile, readManifest, writeManifest } from "#tools/fetch/download/index" const INDEX_URL = "https://download.cms.gov/nppes/NPI_Files.html" const BASE_URL = "https://download.cms.gov/nppes" const SLUG = "usgov-nppes" export type FetchNPPESOptions = BaseFetchOptions /** * Scrape the NPI_Files.html index for the latest full monthly ZIP. Full-replacement files match * `NPPES_Data_Dissemination__*.zip`; weekly files carry a `MMDDYY_MMDDYY` date range, which we exclude. */ async function discoverLatestZip(): Promise { // `responseType: "text"` — the index is HTML, scraped by regex below. const html = await new APIClient({ displayName: "nppes-index", retry: true, axios: { headers: { "Accept-Encoding": "gzip, br" } }, }) .fetch({ url: INDEX_URL, responseType: "text", timeout: 60_000 }) .then(pluckResponseData) for (const match of html.matchAll(/NPPES_Data_Dissemination_[A-Za-z]+_\d{4}[^"]*\.zip/g)) { const name = match[0] if (name && !/\d{6}_\d{6}/.test(name)) return name } return undefined } /** * The main registry CSV (npidata_pfile_*.csv), which the archive also carries alongside a header file and a per-month * change file. */ async function findNpidataCSV(zipPath: string): Promise { const entries = await listZipEntries(zipPath) return entries.find((entry) => /npidata_pfile\S+\.csv/i.test(entry.name))?.name } export async function fetchNPPES(options: FetchNPPESOptions, report?: (line: string) => void): Promise { const destDir = join(options.outRoot, SLUG) await makeDirectories(destDir) const manifestPath = join(destDir, "MANIFEST.json") report?.(`=== ${SLUG}`) report?.(` Discovering latest full-replacement ZIP from ${INDEX_URL} ...`) const zipFilename = await discoverLatestZip() if (!zipFilename) { report?.(` ✗ Could not discover ZIP filename from ${INDEX_URL}`) return { fetched: 0, skipped: 0, failed: 1, failedCodes: [SLUG] } } const zipURL = `${BASE_URL}/${zipFilename}` const zipDest = join(destDir, zipFilename) report?.(` Latest full file: ${zipFilename}`) // Idempotency check: if the main CSV already exists and sha matches, // skip re-download. const recorded = await readManifest>(manifestPath) if (recorded?.sha256 && recorded.filename) { const recordedPath = join(destDir, recorded.filename) if ((await pathExists(recordedPath)) && (await sha256File(recordedPath)) === recorded.sha256) { report?.(" ✓ Already current (sha256 matches MANIFEST) — skipping download.") return { fetched: 0, skipped: 1, failed: 0, failedCodes: [] } } } // MARK: Download ZIP (large; 60-minute timeout) report?.(` Downloading ${zipURL} ...`) const { bytes: zipSize } = await downloadToFile({ url: zipURL, dest: zipDest, timeoutMs: 3_600_000, headers: { "Accept-Encoding": "gzip, br" }, report, }) report?.(` Downloaded: ${(zipSize / 1024 / 1024).toFixed(1)} MB`) // MARK: Extract only the main registry CSV (npidata_pfile_*.csv) report?.(" Extracting npidata_pfile CSV from ZIP ...") const csvName = await findNpidataCSV(zipDest) if (!csvName) { report?.(" ✗ Could not find npidata_pfile CSV inside ZIP") return { fetched: 0, skipped: 0, failed: 1, failedCodes: [SLUG] } } report?.(` Extracting: ${csvName}`) const csvDest = join(destDir, basename(csvName)) await extractZipEntry(zipDest, csvName, csvDest) const csvSize = (await statPath(csvDest)).size const csvSha = await sha256File(csvDest) report?.(` CSV size: ${(csvSize / 1024 / 1024).toFixed(1)} MB`) // MARK: Remove the ZIP to reclaim ~1 GB await removePathIfPresent(zipDest) report?.(" Removed ZIP (CSV kept)") // MARK: Write MANIFEST (records the extracted CSV, not the ZIP) const manifest: SourceManifest = { source_url: zipURL, downloaded_at: new Date().toISOString(), filename: csvName, sha256: csvSha, bytes: csvSize, } await writeManifest(manifestPath, manifest) report?.(` ✓ ${(csvSize / 1024 / 1024).toFixed(1)} MB sha256=${csvSha}`) report?.(` MANIFEST written to ${manifestPath}`) return { fetched: 1, skipped: 0, failed: 0, failedCodes: [] } }