/** * @fileoverview PubChem API client with rate limiting, retry, and response parsing. * Wraps both PUG REST and PUG View APIs behind a shared rate limiter. * @module services/pubchem/pubchem-client */ import type { BioactivityRow, CompoundClassification, InteractionsResult, SafetyLookup } from './types.js'; export declare class PubChemClient { private readonly pugBase; private readonly viewBase; private readonly sdqBase; private readonly rateLimiter; /** Shared HTTP core: rate-limit, 30s timeout, retry once on 5xx (except a rejected search * query) and once on a transient network error, and surface a clean timeout message. Returns * the ok Response; callers extract the body (JSON, bytes, or text). Centralizing this keeps * every fetch variant on one resilience contract — the divergence that left fetchBinary * without retry or a clean timeout message (#16) cannot recur. * * `init.signal` is the caller's cancellation, combined with the per-attempt timeout. Once it * aborts — queued for a slot, mid-fetch, or in a retry backoff — the request ends with * `RequestCancelled`: no retry, and no failed-request log, since nothing failed upstream. */ private fetchResponse; private fetchJson; private fetchBinary; /** Fetch a text/plain body (e.g. an SDF record). Non-2xx is classified and thrown. */ private fetchText; /** Fetch CID list, with automatic ListKey polling for async searches. * * `maxRecords` carries a bounded search's cap into the ListKey retrieval. A search that * answers synchronously is already bounded by the `MaxRecords` on its own URL, but that * bound lives in the request, not the ListKey — so an async answer would otherwise return * the full match set and the caller would read PubChem's ceiling as a real count. Omitted * for identifier lookups, which are unbounded by design. */ private fetchCids; /** Poll a PubChem ListKey until results are ready. * * `listkey_count` asks PubChem to trim the page; the slice enforces the same bound * locally, so the caller's saturation test holds whether or not the parameter is honored. * A cancellation ends the wait before the next poll. */ private pollListKey; /** Resolve one caller-supplied identifier to its CIDs; [] when PubChem has no match. * * The identifier is the only variable in these requests, so an HTTP 400 is PubChem rejecting * that identifier — for SMILES, a string it cannot standardize into a structure, including * well-formed SMILES it cannot process such as a "*" wildcard atom. It is re-thrown as a * ValidationError tagged `identifier_rejected`, so a batch caller can set that one input * aside. Every other failure — 5xx, rate limit, timeout — keeps its own classification. */ private lookupIdentifier; searchByName(name: string, signal?: AbortSignal): Promise; searchBySmiles(smiles: string, signal?: AbortSignal): Promise; searchByInchiKey(inchikey: string, signal?: AbortSignal): Promise; /** Formula search, bounded server-side by `maxRecords`. * * `MaxRecords` is mandatory rather than optional because an unbounded fast search moves * PubChem's entire match set: a benzoic-acid substructure query returns 1,000,000 CIDs in * 16.3 MB, which is PubChem's own ceiling rather than a real count. Callers ask for the * window they will actually use. * * A response shorter than `maxRecords` is the complete match set; a response of exactly * `maxRecords` is saturated, and the true total is not recoverable — PubChem returns the * CID list alone, with no match count beside it. Callers that need to distinguish the two * compare the returned length against the cap they passed. */ searchByFormula(formula: string, allowOther: boolean, maxRecords: number, signal?: AbortSignal): Promise; /** Substructure, superstructure, and 2D-similarity search, bounded server-side by * `maxRecords`. Same cap contract as {@link searchByFormula}; a query PubChem cannot search * on throws `search_query_rejected` the same way. */ searchByStructure(mode: 'substructure' | 'superstructure' | 'similarity', query: string, queryType: 'smiles' | 'cid', threshold: number | undefined, maxRecords: number, signal?: AbortSignal): Promise; getProperties(cids: number[], properties: string[], signal?: AbortSignal): Promise & { CID: number; }>>; getSynonyms(cid: number, signal?: AbortSignal): Promise; getImage(cid: number, size?: 'small' | 'large', signal?: AbortSignal): Promise; getXrefs(cid: number, xrefType: string, signal?: AbortSignal): Promise<(string | number)[]>; getDescription(cid: number, signal?: AbortSignal): Promise>; /** Fetch a compound's GHS classification. * * Reports "this CID has no PubChem record" separately from "this compound has no deposited * GHS classification" — a nonexistent CID and a real compound without safety data are * different answers that call for different follow-up, and both used to surface as an * absent result. */ getSafetyData(cid: number, signal?: AbortSignal): Promise; getClassification(cid: number, signal?: AbortSignal): Promise; getAssaySummary(cid: number, signal?: AbortSignal): Promise; private parseAssayTable; searchAssaysByTarget(targetType: string, query: string, signal?: AbortSignal): Promise; getEntitySummary(entityType: string, identifier: string | number, signal?: AbortSignal): Promise | null>; /** Fetch drug-drug, drug-food, and/or chemical-target interactions for a compound. * Drug-drug and target data live in PubChem SDQ external tables (drugbankddi, bioactivity); * drug-food is inline PUG View text. Each kind reads a window of `maxEntries` entries * starting at `offset` records into that kind's own stream; absent data for a kind * contributes an empty page rather than erroring. A cancellation fails the whole call. */ getInteractions(cid: number, kinds: Array<'drug-drug' | 'drug-food' | 'target'>, maxEntries: number, offset: number, signal?: AbortSignal): Promise; private getInteractionsForKind; private getDrugDrugInteractions; /** Chemical-target binding/activity from PubChem's `bioactivity` SDQ collection (BindingDB, * ChEMBL, and others). That collection is cid-keyed and scopes correctly to the requested * compound; the `consolidatedcompoundtarget` collection used previously is gene-indexed, so * its `cid` filter was silently ignored — every CID returned the same default compound (#20). * Only rows naming a molecular target are returned; untargeted assay outcomes are the domain * of pubchem_get_bioactivity. * * Because most rows are dropped, the page position is an index into the activity records, not * into the entries returned — the walk records exactly how many rows it read so the next page * resumes at the first unread one. */ private getTargetInteractions; private getDrugFoodInteractions; /** Query a PubChem SDQ external table for a single CID, projecting only the columns the * caller maps. Projection keeps payloads small and excludes the free-text `citations` column * whose unescaped quotes make PubChem emit invalid JSON (#20). * * Uses SDQ's `select` projection rather than `download`: the `SDQOutputSet` envelope it * answers with carries `totalCount` — the records matching the query across all pages — * beside the requested window, which is the only way to tell a full page from the last one. * `start` is 1-based upstream; `offset` here is zero-based like every other paging input. * * Returns an empty page on not-found. A rejected query is reported, never read as "no data": * SDQ answers a malformed query with a 5xx that the shared HTTP core already classifies and * throws, and the `status.error` check covers the same rejection arriving inside a 2xx body. * An unparseable body throws with the collection and a body snippet attached, so per-kind * isolation can name what failed. */ private fetchSdq; /** Resolve a kind's record total for the page just fetched. * * SDQ reports `totalCount` alongside every non-empty page, but a `start` past the last record * answers `totalCount` 0 — so an offset that overshoots would otherwise report "no records" * for a compound that has thousands. A one-row probe from the top recovers the real bound, * and only runs on that case. */ private resolveSdqTotal; /** Fetch the default 3D conformer as raw V2000 SDF text. Throws a typed not-found when * PubChem has no computed 3D coordinates (large molecules, mixtures, undefined salts). */ getSdf3d(cid: number, signal?: AbortSignal): Promise; /** List the conformer IDs PubChem has computed for a compound. Returns [] on not-found. */ getConformerIds(cid: number, signal?: AbortSignal): Promise; } export declare function initPubChemClient(): void; export declare function getPubChemClient(): PubChemClient; //# sourceMappingURL=pubchem-client.d.ts.map