/** * Factories for configured Scrapper web-extraction tools (article + links). * * @module @nhtio/adk/batteries/tools/scrapper * * @remarks * [Scrapper](https://github.com/amerkurev/scrapper) is a self-hosted service that loads a page in a * real headless browser and extracts either the readable article (`/api/article`) or the page's * links (`/api/links`). It gives an agent browser-grade reading power — JS-rendered pages a * renderless fetcher can't see — but as a **stateless** HTTP call: each request runs in a fresh * incognito context, stores no session or credentials, and shares nothing with any other call. * * Like the SearXNG battery, this exports **factories** (not ready-made `Tool` constants), because a * scrape tool needs per-deployment config (instance URL + custom auth headers). Two verbs, each with * an async factory ({@link createScrapperArticleTool} / {@link createScrapperLinksTool}, accepting a * dynamic-import `artifact` resolver) and a sync variant ({@link createScrapperArticleToolSync} / * {@link createScrapperLinksToolSync}). Because these are factories, they MUST NOT be bulk-registered * via `Object.values(batteries)` — call one, then register the returned tool. * * @see https://github.com/amerkurev/scrapper */ import { type ScrapperBaseConfig } from "./shared"; import type { Tool } from "../../../forge"; import type { ArtifactResolver, SyncArtifactResolver } from "../_shared/index"; export { E_INVALID_SCRAPPER_CONFIG } from "./exceptions"; export type { ScrapperRequestContext, ScrapperResponseContext, ScrapperInputMiddlewareFn, ScrapperOutputMiddlewareFn, } from "./shared"; /** Model-facing params common to both verbs (snake_case; mapped to kebab on the wire). */ export interface ScrapperCommonParams { /** Return a cached result when available instead of re-scraping. */ cache?: boolean; /** Capture a screenshot; the result carries a `screenshotUri`. */ screenshot?: boolean; /** Run in an incognito browser context (no persisted browsing data). Default true upstream. */ incognito?: boolean; /** Browser navigation timeout in ms (`0` disables). Distinct from the tool's own fetch timeout. */ timeout?: number; /** When navigation is considered finished. */ wait_until?: 'load' | 'domcontentloaded' | 'networkidle' | 'commit'; /** Wait this many ms after load before parsing. */ sleep?: number; /** Scroll down N pixels for lazy-loading pages. Requires a positive `sleep`. */ scroll_down?: number; /** Emulated device, e.g. `Desktop Chrome`. Overrides individual viewport/UA settings. */ device?: string; /** Explicit user-agent (prefer `device`). */ user_agent?: string; /** Extra headers the SCRAPER's browser sends to the TARGET site, `K:v;K2:v2` (NOT instance auth). */ extra_http_headers?: string; /** Upstream proxy, e.g. `http://host:3128` or `socks5://host:1080`. */ proxy_server?: string; } /** Model-facing params for `/api/article`. */ export interface ScrapperArticleParams extends ScrapperCommonParams { /** Populate `fullContent` with the page's full HTML. */ full_content?: boolean; } /** Model-facing params for `/api/links`. */ export interface ScrapperLinksParams extends ScrapperCommonParams { /** Median link-text length threshold for the link parser. */ text_len_threshold?: number; /** Median words-per-link threshold for the link parser. */ words_threshold?: number; } /** A normalised Scrapper article (loose/nullable upstream). */ export interface ScrapperArticle { /** The page URL the article was extracted from. */ url?: string; /** Article title. */ title?: string; /** Author / byline metadata. */ byline?: string; /** Short excerpt or description of the article. */ excerpt?: string; /** Name of the site the article came from. */ siteName?: string; /** Detected content language. */ lang?: string; /** Character count of the extracted article text. */ length?: number; /** Publication time, when the page exposed one. */ publishedTime?: string; /** Scrapper's own date field for the result. */ date?: string; /** Article text with HTML stripped. */ textContent?: string; /** Processed article HTML; present when the caller requested it. */ content?: string; /** Full page HTML; present only when `full_content` was set. */ fullContent?: string; /** Screenshot URI; present only when `screenshot` was set. */ screenshotUri?: string; } /** A single link from `/api/links` (verified live: `{ url, text }`). */ export interface ScrapperLink { /** The link's target URL. */ url?: string; /** The link's anchor text. */ text?: string; } /** A normalised Scrapper links payload. */ export interface ScrapperLinks { /** The page URL the links were collected from. */ url?: string; /** The page title. */ title?: string; /** The page's domain. */ domain?: string; /** Scrapper's own date field for the result. */ date?: string; /** The collected links, each `{ url, text }`. */ links: ScrapperLink[]; /** Screenshot URI; present only when `screenshot` was set. */ screenshotUri?: string; } export type { ScrapperBaseConfig } from "./shared"; /** Async-factory config for `/api/article` (full `artifact` resolver, incl. dynamic import). */ export type ScrapperArticleConfig = ScrapperBaseConfig; /** Sync-factory config for `/api/article` (`artifact` narrowed to the sync subset). */ export type ScrapperArticleConfigSync = ScrapperBaseConfig; /** Async-factory config for `/api/links`. */ export type ScrapperLinksConfig = ScrapperBaseConfig; /** Sync-factory config for `/api/links`. */ export type ScrapperLinksConfigSync = ScrapperBaseConfig; /** * Create a configured Scrapper **article** {@link Tool} (async — accepts a dynamic-import `artifact`). * * @remarks * Async because `artifact` may be an async / dynamic-import resolver, which must resolve to the sync * `() => Ctor` `Tool.artifactConstructor` requires before the tool is built. For the common case, * use {@link createScrapperArticleToolSync} and skip the `await`. * * @warning * Two distinct "headers": `config.headers` authenticates to the Scrapper *instance*; the * `extra_http_headers` *parameter* is what the scraper's browser sends to the *target site* — do not * conflate them. Also note `scroll_down` requires a positive `sleep`, and `resultUri`/`screenshotUri` * are instance-relative and may come back `http://` even over HTTPS — do not assume they match * `instanceUrl`. * * @param config - Instance URL, instance-auth headers, output policy, `artifact` resolver, * per-parameter disposition (`fixed`/`defaults`/`fixedQuery`), and middleware pipelines. * @returns A promise of a `Tool` ready to register in a `ToolRegistry`. * @throws {@link E_INVALID_SCRAPPER_CONFIG} when `instanceUrl` or `artifact` is invalid. */ export declare const createScrapperArticleTool: (config: ScrapperArticleConfig) => Promise; /** * Synchronous {@link createScrapperArticleTool} — `artifact` narrowed to the sync subset. * * @param config - Same as {@link createScrapperArticleTool}, with a sync-only `artifact`. * @returns A `Tool` ready to register in a `ToolRegistry`. * @throws {@link E_INVALID_SCRAPPER_CONFIG} when `instanceUrl` or `artifact` is invalid (incl. an async resolver). */ export declare const createScrapperArticleToolSync: (config: ScrapperArticleConfigSync) => Tool; /** * Create a configured Scrapper **links** {@link Tool} (async — accepts a dynamic-import `artifact`). * * @remarks * See {@link createScrapperArticleTool} for the two-headers caveat and the async rationale. Each * `links` item is `{ url, text }`. * * @param config - Instance URL, instance-auth headers, output policy, `artifact` resolver, * per-parameter disposition, and middleware pipelines. * @returns A promise of a `Tool` ready to register in a `ToolRegistry`. * @throws {@link E_INVALID_SCRAPPER_CONFIG} when `instanceUrl` or `artifact` is invalid. */ export declare const createScrapperLinksTool: (config: ScrapperLinksConfig) => Promise; /** * Synchronous {@link createScrapperLinksTool} — `artifact` narrowed to the sync subset. * * @param config - Same as {@link createScrapperLinksTool}, with a sync-only `artifact`. * @returns A `Tool` ready to register in a `ToolRegistry`. * @throws {@link E_INVALID_SCRAPPER_CONFIG} when `instanceUrl` or `artifact` is invalid (incl. an async resolver). */ export declare const createScrapperLinksToolSync: (config: ScrapperLinksConfigSync) => Tool;