import { describe, expect, it } from "vitest"; import { capDiscovered, classifyUrl, extractLinks, parseFeed, parseSitemap, pickSitemapsFromRobots } from "./discovery.js"; import { parseConnectorConfig } from "./config.js"; describe("parseSitemap", () => { it("parses a urlset with lastmod", () => { const r = parseSitemap(` https://x.com/a 2026-09-01https://x.com/b`); expect(r.urls).toEqual([{ loc: "https://x.com/a", lastmod: "2026-09-01" }, { loc: "https://x.com/b", lastmod: null }]); expect(r.sitemaps).toEqual([]); }); it("parses a sitemap index and a single-entry index", () => { expect(parseSitemap(`https://x.com/s1.xmlhttps://x.com/s2.xml`).sitemaps).toEqual(["https://x.com/s1.xml", "https://x.com/s2.xml"]); expect(parseSitemap(`https://x.com/only.xml`).sitemaps).toEqual(["https://x.com/only.xml"]); }); it("accepts plain-text sitemaps and ignores garbage", () => { expect(parseSitemap("https://x.com/a\nnot a url\nhttps://x.com/b\n").urls.map((u) => u.loc)).toEqual(["https://x.com/a", "https://x.com/b"]); expect(parseSitemap("<<<").urls).toEqual([]); }); }); describe("parseFeed", () => { it("RSS 2.0 items with categories, guid fallback and content:encoded", () => { const items = parseFeed(`t Ahttps://x.com/aMon, 01 Sep 2026 10:00:00 GMTSummary ANewsPowera-1 Bhttps://x.com/bBody B

]]>
no link
`); expect(items).toHaveLength(2); expect(items[0]).toMatchObject({ title: "A", link: "https://x.com/a", published: "Mon, 01 Sep 2026 10:00:00 GMT", summary: "Summary A", id: "a-1", categories: ["News", "Power"] }); expect(items[1]).toMatchObject({ title: "B", link: "https://x.com/b", summary: "

Body B

" }); }); it("Atom entries: alternate link, published/updated, term categories", () => { const items = parseFeed(`E2026-09-02T00:00:00ZSurn:e`); expect(items).toEqual([{ title: "E", link: "https://x.com/e", published: "2026-09-02T00:00:00Z", summary: "S", id: "urn:e", categories: ["dc"] }]); }); it("returns [] on malformed XML", () => { expect(parseFeed("")).toEqual([]); }); }); describe("extractLinks", () => { const html = `abcmjtfdup`; it("resolves, dedupes, strips fragments and keeps same-site links only (www-insensitive)", () => { expect(extractLinks(html, "https://x.com/page")).toEqual(["https://x.com/a", "https://www.x.com/b"]); }); it("can keep external links", () => { expect(extractLinks(html, "https://x.com/page", false)).toContain("https://other.com/c"); }); }); describe("classifyUrl / capDiscovered / pickSitemapsFromRobots", () => { const cfg = parseConnectorConfig(`id: t\nname: T\ndomain: x.com\nkind: operator\ndiscovery:\n include: ["/data-centers/"]\n exclude: ["\\\\?"]\n classify:\n - { pattern: "/data-centers/[a-z0-9-]+/$", group: facility_pages, pageType: facility_page, priority: 60 }\n`); it("applies exclude, include and classify rules", () => { expect(classifyUrl(cfg, "https://x.com/data-centers/abc/?x=1")).toBeNull(); expect(classifyUrl(cfg, "https://x.com/blog/abc/")).toBeNull(); expect(classifyUrl(cfg, "https://x.com/data-centers/abc/")).toMatchObject({ group: "facility_pages", pageType: "facility_page", priority: 60 }); expect(classifyUrl(cfg, "https://x.com/data-centers/")).toMatchObject({ group: "default", priority: 30 }); }); it("keeps the highest-priority urls when capping", () => { const urls = [{ url: "a", group: "default", priority: 30 }, { url: "b", group: "facility_pages", priority: 60 }, { url: "c", group: "seed", priority: 80 }, { url: "d", group: "default", priority: 30 }]; expect(capDiscovered(urls, 2).map((u) => u.url)).toEqual(["c", "b"]); expect(capDiscovered(urls, 10)).toHaveLength(4); }); it("keeps only robots sitemaps on the connector's domain", () => { expect(pickSitemapsFromRobots(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml", "https://y.com/s.xml", "junk"], "www.x.com")).toEqual(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml"]); }); });