/** * @copyright Sister Software * @license AGPL-3.0 * @author Teffen Ellis, et al. * * The paging, manifest and resumption machinery a paged WFS harvest shares. * * `./inspire-addresses.ts` owns what an INSPIRE Addresses service exposes — its capabilities, its * feature-type names and its GeoJSON pages. This module owns what a harvest of any paged WFS does * with those pages: it appends each one to the single file an adapter's `inputPath` points at, * records the page in a manifest beside it, and resumes from that manifest after an interrupted run. * * One file rather than one file per page, because the adapters read one. `#ee/adapters/ads/adapter` * opens `opts.inputPath` as newline-delimited JSON and `#pl/adapters/emuia/adapter` opens it as one * markup stream, so a directory of pages reaches neither. The manifest therefore records each page's * byte range inside that file: `offset`, `bytes` and the sha256 of exactly those bytes. A range is * what makes a page verifiable after it has been concatenated into its neighbors, and what lets an * interrupted append be truncated back to a page boundary rather than discarding the harvest. * * Three properties of a WFS make this harder than reading pages until they run out, and each is * handled here rather than by each country's module: * * 1. A service may ignore `startIndex` and answer every page with the first. `countWFSFeaturesByPaging` * in `@mailwoman/core/api` refuses such a service by comparing two pages' leading feature. A * harvest sees the same defect as two consecutive pages with equal bytes, which this refuses. * 2. A service's own `numberMatched` may be a per-request cap rather than a count. The caller states * what its count is and where it came from, and {@linkcode PagedWFSHarvestOptions.featureCount} * carries both. No count is written into this module, and a harvest with no usable count ends on * the first page that returns no features. * 3. A page may return fewer features than `count` asks for. The next `startIndex` therefore advances * by the features the service returned rather than by the page size, so a service that caps a page * below the requested size is read whole instead of being read in strides that skip features. */ import { APIClient, type CheckedWFSFeatureCount } from "@mailwoman/core/api"; import { type PathBuilderLike } from "path-ts"; import { type FetchSummary } from "#tools/fetch/download"; /** * The WFS version this repository's harvesters speak. * * 2.0.0 defines `startIndex` paging and the `hits` result type, and it is what * every service measured for these modules advertises. */ export declare const WFS_VERSION = "2.0.0"; /** * The file name a harvest writes its manifest under, beside the data file. */ export declare const HARVEST_MANIFEST_FILE = "MANIFEST.json"; /** * The first element of a response, as the service wrote it. */ export interface MarkupRoot { /** * The opening tag verbatim, ``. */ tag: string; /** * The element's qualified name, prefix included. */ name: string; /** * The text between the name and the tag's terminator, which holds the attributes. */ attributes: string; /** * The offset just past the opening tag. */ contentStart: number; } /** * The count an opening tag states, or `null` where it states none or declines to state one. * * WFS 2.0 permits `numberMatched="unknown"`, which declines to count rather than counting none. */ export declare function rootCount(root: MarkupRoot, attribute: string): number | null; /** * The document's first element tag. * * @throws When the body holds no element tag, so a response that is not markup * reads as unreadable rather than as a page of no features. */ export declare function readMarkupRoot(body: string, context: string): MarkupRoot; /** * The content between a document's root tags, which for a `wfs:FeatureCollection` is its members. * * @throws When the root element is not closed, so a truncated transfer reads as * unreadable rather than as a page of no features. */ export declare function rootContent(body: string, root: MarkupRoot, context: string): string; /** * A page's root tag with its namespace declarations kept and its per-page attributes * dropped, so it can open a file holding every page. */ export declare function collectionOpeningTag(root: MarkupRoot): string; /** * One `GetFeature` response in a markup format, with what its root states about itself. */ export interface WFSMarkupPage { body: string; root: MarkupRoot; /** * The `numberReturned` the root states, or `null` where it states none. */ numberReturned: number | null; /** * The `numberMatched` the root states, or `null` where it answers `unknown`. */ numberMatched: number | null; /** * The `timeStamp` the root states, or `null` where it states none. */ timeStamp: string | null; } /** * Issues one `GetFeature` and reads the counts off the root element. * * @throws When the service answers an OGC exception report, so an HTTP 400 carrying * `ows:ExceptionCode` raises rather than being written into the harvest as a page. */ export declare function readWFSMarkupPage(client: Pick, options: { wfsURL: string; context: string; params: Readonly>; }): Promise; /** * What a harvest asked the service for. * * Recorded in the manifest and compared against a resumed run's request: a harvest whose page * size, ordering or format changed cannot be continued from pages taken under the old one. */ export interface WFSHarvestRequest { wfsURL: string; typeName: string; outputFormat: string; /** * The property the service was asked to order by, or `null` where it grants no ordering. * * Without one, a resumed harvest rests on the service returning the same features * in the same order as the earlier run, which no WFS guarantees. */ sortBy: string | null; } /** * A feature count and how it was obtained. * * `count` is `null` where the service states no usable count. * The harvest then ends on the first page that returns no features, and * `because` records what the service said instead. */ export interface FeatureCountStatement { count: number | null; because: string; } /** * The count a service states, or no count and the reason there is none. * * `readCheckedWFSFeatureCount` proves a reported `numberMatched` against a page of the * same type, which catches a service whose count is smaller than one of its own pages. * It cannot catch a count that equals the service's page cap: no page can return more * features than the cap, so the page it asks for agrees with the cap every time. * * Poland's service is that case. * Measured 2026-10-02, `resultType=hits` at `startIndex=1` reports `numberMatched="1000"`, its * capabilities document advertises `CountDefault` as 1000, and a page returns at most 1,000 features. * * The reported number is the cap, and a harvest that read it as a count would store 1,000 * of the 8.6 million features the service holds and record the type as read whole. * A reported count equal to the cap is therefore carried as no count. * * @param pageCap The largest page the service will serve, from its advertised `CountDefault`. */ export declare function statedFeatureCount(checked: CheckedWFSFeatureCount, pageCap: number | null): FeatureCountStatement; /** * One page's contribution to the harvest file. */ export interface HarvestedPayload { /** * The text appended to the harvest file for this page. */ payload: string; /** * Text written once, before the first page's payload — a document prolog * and root tag, where the format needs one. * * Read from the first page a run writes and recorded in the manifest. */ header?: string; /** * Text written once the harvest reaches the end of the feature type, such as a closing root tag. */ footer?: string; /** * The features the service returned, which is what the next `startIndex` advances by. * * `null` where the service stated none. * The harvest refuses that rather than reading it as zero. */ numberReturned: number | null; numberMatched: number | null; retrievedAt: string | null; } /** * One page as the manifest records it. * * `offset` and `bytes` locate the page inside the harvest file and `sha256` is the digest of * exactly those bytes, so a page stays verifiable after the pages around it were appended. */ export interface WFSHarvestPageRecord { start_index: number; /** * The page size requested, which a service may answer with fewer features. */ count: number; offset: number; bytes: number; sha256: string; number_returned: number; /** * The `numberMatched` this page stated, where the service states a real one. */ number_matched: number | null; retrieved_at: string | null; } /** * The manifest a paged harvest writes beside its data file. * * Written after every page, so an interrupted run resumes from the last page that reached disk. */ export interface WFSHarvestManifest { source: string; source_url: string; wfs_version: string; type_name: string; output_format: string; sort_by: string | null; license: string; attribution: string; downloaded_at: string; /** * The file an adapter's `inputPath` points at, beside this manifest. */ filename: string; page_size: number; /** * The service's own count where it states a usable one, and `null` where it does not. */ feature_count: number | null; feature_count_source: string; features_written: number; header_bytes: number; /** * The text that closes the data file, appended once the harvest reaches the end of the type. */ footer: string; /** * The data file's byte length as this manifest describes it, footer included once complete. */ bytes: number; /** * The whole data file's digest, once the harvest is complete. * * `null` on a partial harvest, where the file is still being appended to * and a digest of it would describe a prefix rather than the harvest. */ sha256: string | null; complete: boolean; pages: WFSHarvestPageRecord[]; } export interface PagedWFSHarvestOptions { /** * Names the service in every refusal. */ context: string; /** * The adapter slug this harvest feeds, stamped into the manifest. */ source: string; /** * Where the harvest is written. * * The adapter reads {@linkcode PagedWFSHarvestOptions.filename} inside this directory. */ outputDir: PathBuilderLike; filename: string; request: WFSHarvestRequest; /** * The license the address-source register elected for this publisher. */ license: string; /** * The attribution the elected terms require, or an empty string where they require none. */ attribution: string; featureCount: FeatureCountStatement; pageSize: number; /** * Stop once the manifest holds this many pages, for a probe rather than a full harvest. * * A cap on the manifest rather than on the run, so a second run under the same cap makes no request. */ maxPages?: number; signal?: AbortSignal; /** * Reads one page and states the bytes the harvest file receives for it. */ readPage: (request: { startIndex: number; count: number; }) => Promise; report?: (line: string) => void; } /** * What a caller knows about the request it is about to make before it has asked the service. * * The type name and the output format come from a capabilities document, so they are not here: * the point of this check is to answer whether a harvest is already current without sending a request. */ export interface HarvestCurrencyExpectation { wfsURL: string; sortBy: string | null; pageSize: number; } /** * The manifest of a harvest that needs no further request, or `null` where one does. * * A caller reads this before it reads a capabilities document, so a current harvest costs no request at all. * Two harvests need none: * * - A complete one whose data file still hashes to the digest the manifest records. * - A partial one that already holds `maxPages` pages, which is what a bounded probe asks for. * * @throws When a page's recorded byte range no longer hashes to its recorded digest, * so a file that was altered under its manifest is reported rather than used. */ export declare function readCurrentHarvest(options: { outputDir: PathBuilderLike; filename: string; context: string; expect: HarvestCurrencyExpectation; maxPages?: number; }): Promise; /** * Harvests one WFS feature type into a single file, recording each page in a manifest beside it. * * Re-runnable in three ways, each of which makes no request: * * - A complete harvest whose file still hashes to the manifest's `sha256`. * - A partial harvest that already holds {@linkcode PagedWFSHarvestOptions.maxPages} pages. * - A partial harvest resumes at the page after the last one on disk, * so the pages already taken are not taken again. * * @returns The manifest as written. */ export declare function harvestPagedWFS(options: PagedWFSHarvestOptions): Promise; /** * What a registered `mailwoman corpus fetch` entry needs to run one service's harvest. */ export interface RunWFSHarvestOptions { /** * The slug the harvest is written under and reported by, which matches the adapter's own id. */ slug: string; /** * Where the harvest directory sits, as the fetch registry hands it over. */ outRoot: (segment: string) => PathBuilderLike; /** * The data file inside the harvest directory, which is also the adapter's `inputPath`. */ filename: string; /** * What a stored manifest has to agree with before its harvest counts as current. */ expect: HarvestCurrencyExpectation; /** * The name the client reports itself under. */ displayName: string; /** * The floor between two requests to one government host. */ minRequestIntervalMs: number; /** * Runs the harvest against a client the caller does not own. */ harvest: (client: Pick) => Promise; /** * A client to use rather than constructing one, which a test supplies. */ client?: Pick; /** * A page cap, which makes an incomplete harvest a success. */ maxPages?: number; } /** * Runs one service's harvest and reports it as a fetch registry entry does. * * Estonia and Poland wrote this sequence twice: ask whether the harvest on disk is * already current, construct a paced client unless the caller supplied one, run, * report, and decide whether an incomplete harvest is a failure. * A third WFS source would have written it a third time. * * The currency check runs before any request, so a complete harvest or a bounded * probe that already holds its pages makes no request at all. * * A bounded run stops short by instruction rather than by failure, so a page cap the caller * asked for reads as a success while an incomplete harvest without one reads as a failure. */ export declare function runWFSHarvest(options: RunWFSHarvestOptions, report?: (line: string) => void): Promise; //# sourceMappingURL=wfs-harvest.d.ts.map