import { describe, expect, test } from "bun:test"; import { parseRssFeed, parseRssFeedDocument, type RssFeedConfig } from "./parser"; const DEFAULT_CONFIG: RssFeedConfig = { id: "test-feed", url: "https://example.com/feed", name: "Test Feed", authority: 60, enabled: true, }; const RSS2_FIXTURE = ` Test Feed Fed holds rates steady https://example.com/fed-rates Thu, 10 Apr 2026 14:30:00 GMT <p>The Federal Reserve held rates steady at 4.25%.</p> Economy <![CDATA[NVIDIA beats Q1 estimates]]> https://example.com/nvda-q1 Thu, 10 Apr 2026 10:00:00 GMT NVIDIA reported strong earnings. Oil surges on OPEC cuts https://example.com/oil-opec Thu, 10 Apr 2026 08:00:00 GMT `; const ATOM_FIXTURE = ` Atom Feed Markets rally on trade deal 2026-04-10T12:00:00Z Global markets surged on news of a trade agreement. `; describe("parseRssFeed", () => { test("parses RSS items with normalized text, categories, source, dates, and stable ids", () => { const items = parseRssFeed(RSS2_FIXTURE, DEFAULT_CONFIG); const again = parseRssFeed(RSS2_FIXTURE, DEFAULT_CONFIG); expect(items).toHaveLength(3); expect(items[0]).toMatchObject({ title: "Fed holds rates steady", url: "https://example.com/fed-rates", source: "Test Feed", categories: ["Economy"], }); expect(items[0]!.publishedAt).toBeInstanceOf(Date); expect(items[0]!.publishedAt.getFullYear()).toBe(2026); expect(items[0]!.summary).toContain("Federal Reserve"); expect(items[0]!.summary).not.toContain("

"); expect(items[0]!.summary).not.toContain("<"); expect(items[1]!.title).toBe("NVIDIA beats Q1 estimates"); expect(items[2]).toMatchObject({ title: "Oil surges on OPEC cuts", summary: undefined, }); expect(items[0]!.id).toBe(again[0]!.id); }); test("parses Atom entries", () => { const items = parseRssFeed(ATOM_FIXTURE, DEFAULT_CONFIG); expect(items).toHaveLength(1); expect(items[0]).toMatchObject({ title: "Markets rally on trade deal", url: "https://example.com/trade-deal", }); expect(items[0]!.publishedAt).toBeInstanceOf(Date); expect(items[0]!.publishedAt.getMonth()).toBe(3); expect(items[0]!.summary).toContain("trade agreement"); }); test("ignores empty or invalid input", () => { expect(parseRssFeed("", DEFAULT_CONFIG)).toHaveLength(0); expect(parseRssFeed("not xml at all <<<", DEFAULT_CONFIG)).toHaveLength(0); expect(parseRssFeed(" \n\t ", DEFAULT_CONFIG)).toHaveLength(0); }); test("truncates long summaries and accepts title-only items", () => { const longDesc = "x".repeat(400); const xml = ` Long item https://example.com/long Thu, 10 Apr 2026 08:00:00 GMT ${longDesc} Titleonly item `; const items = parseRssFeed(xml, DEFAULT_CONFIG); expect(items).toHaveLength(2); expect(items[0]!.summary!.length).toBeLessThanOrEqual(301); expect(items[0]!.summary).toContain("…"); expect(items[1]!.title).toBe("Titleonly item"); }); test("uses config category when the item has none", () => { const items = parseRssFeed(RSS2_FIXTURE, { ...DEFAULT_CONFIG, category: "markets" }); expect(items[1]!.categories).toContain("markets"); }); }); test("Atom chooses direct alternate links with namespaces, bases, quotes and XML entities", () => { const xml = ` Wrong feed titlewrong-id Cash offerurn:Case:422026-09-12T13:00:00Z `; expect(parseRssFeedDocument(xml, DEFAULT_CONFIG)).toMatchObject([{ id: "atom:urn:Case:42", title: "Cash offer", url: "https://example.com/research/deals/42?a=1&b=2", }]); expect(parseRssFeed(xml.replace("urn:Case:42", "urn:case:42"), DEFAULT_CONFIG)[0]!.id) .not.toBe(parseRssFeed(xml, DEFAULT_CONFIG)[0]!.id); }); test("Atom correction identity survives title and URL changes and uses publisher update time", () => { const xml = (title: string, url: string, time: string) => ` urn:issuer:42${title} 2026-09-11T08:00:00Z${time}`; const first = parseRssFeed(xml("Offer agreed", "https://example.com/deal", "2026-09-12T12:00:00Z"), DEFAULT_CONFIG)[0]!; const next = parseRssFeed(xml("Offer withdrawn", "https://example.com/withdrawal", "2026-09-12T13:00:00Z"), DEFAULT_CONFIG)[0]!; expect(next.id).toBe(first.id); expect(next.publishedAt.toISOString()).toBe("2026-09-12T13:00:00.000Z"); }); test("Atom text, HTML, XHTML and CDATA retain readable text without nested fake stories", () => { const xml = `Revenue < $10m & cash > $2m

<p>Oil &amp; gas</p> <div xmlns="http://www.w3.org/1999/xhtml">Cash <b>offer</b> withdrawn</div> Literal content only]]>`; const items = parseRssFeedDocument(xml, DEFAULT_CONFIG); expect(items).toHaveLength(2); expect(items[0]).toMatchObject({ title: "Revenue < $10m & cash > $2m", summary: "Oil & gas" }); expect(items[1]!.title).toBe("Cash offer withdrawn"); }); test("feed document failures differ from valid emptiness, and external DTDs are not resolved", () => { expect(parseRssFeedDocument('', DEFAULT_CONFIG)).toEqual([]); expect(parseRssFeedDocument('', DEFAULT_CONFIG)).toEqual([]); for (const xml of ["", "Proxy error", "Truncated", 'Foreign']) { expect(() => parseRssFeedDocument(xml, DEFAULT_CONFIG)).toThrow("Invalid or unsupported news feed"); } expect(parseRssFeedDocument('' + RSS2_FIXTURE.replace('', ''), DEFAULT_CONFIG)).toHaveLength(3); }); test("RSS items with attributes and RDF RSS retain the previously readable article fields", () => { expect(parseRssFeedDocument(RSS2_FIXTURE.replace('', ''), DEFAULT_CONFIG)).toHaveLength(3); const rdf = 'FeedIssuer filinghttps://example.com/item'; expect(parseRssFeedDocument(rdf, DEFAULT_CONFIG)).toMatchObject([{ title: "Issuer filing", url: "https://example.com/item" }]); }); test("RSS comment and CDATA markup cannot create additional stories or replace item metadata", () => { const xml = ` Example markup]]> Actual issuer updatehttps://example.com/update`; expect(parseRssFeedDocument(xml, DEFAULT_CONFIG)).toMatchObject([{ title: "Actual issuer update", url: "https://example.com/update" }]); expect(parseRssFeedDocument(xml, DEFAULT_CONFIG)).toHaveLength(1); });