import { describe, expect, it } from "vitest";
import { capDiscovered, classifyUrl, extractLinks, parseFeed, parseSitemap, pickSitemapsFromRobots } from "./discovery.js";
import { parseConnectorConfig } from "./config.js";
describe("parseSitemap", () => {
it("parses a urlset with lastmod", () => {
const r = parseSitemap(` https://x.com/a 2026-09-01https://x.com/b`);
expect(r.urls).toEqual([{ loc: "https://x.com/a", lastmod: "2026-09-01" }, { loc: "https://x.com/b", lastmod: null }]);
expect(r.sitemaps).toEqual([]);
});
it("parses a sitemap index and a single-entry index", () => {
expect(parseSitemap(`https://x.com/s1.xmlhttps://x.com/s2.xml`).sitemaps).toEqual(["https://x.com/s1.xml", "https://x.com/s2.xml"]);
expect(parseSitemap(`https://x.com/only.xml`).sitemaps).toEqual(["https://x.com/only.xml"]);
});
it("accepts plain-text sitemaps and ignores garbage", () => {
expect(parseSitemap("https://x.com/a\nnot a url\nhttps://x.com/b\n").urls.map((u) => u.loc)).toEqual(["https://x.com/a", "https://x.com/b"]);
expect(parseSitemap("<<<").urls).toEqual([]);
});
});
describe("parseFeed", () => {
it("RSS 2.0 items with categories, guid fallback and content:encoded", () => {
const items = parseFeed(`t
- Ahttps://x.com/aMon, 01 Sep 2026 10:00:00 GMTSummary ANewsPowera-1
- Bhttps://x.com/bBody B
]]>
- no link
`);
expect(items).toHaveLength(2);
expect(items[0]).toMatchObject({ title: "A", link: "https://x.com/a", published: "Mon, 01 Sep 2026 10:00:00 GMT", summary: "Summary A", id: "a-1", categories: ["News", "Power"] });
expect(items[1]).toMatchObject({ title: "B", link: "https://x.com/b", summary: "Body B
" });
});
it("Atom entries: alternate link, published/updated, term categories", () => {
const items = parseFeed(`E2026-09-02T00:00:00ZSurn:e`);
expect(items).toEqual([{ title: "E", link: "https://x.com/e", published: "2026-09-02T00:00:00Z", summary: "S", id: "urn:e", categories: ["dc"] }]);
});
it("returns [] on malformed XML", () => { expect(parseFeed("- ")).toEqual([]); });
});
describe("extractLinks", () => {
const html = `abcmjtfdup`;
it("resolves, dedupes, strips fragments and keeps same-site links only (www-insensitive)", () => {
expect(extractLinks(html, "https://x.com/page")).toEqual(["https://x.com/a", "https://www.x.com/b"]);
});
it("can keep external links", () => {
expect(extractLinks(html, "https://x.com/page", false)).toContain("https://other.com/c");
});
});
describe("classifyUrl / capDiscovered / pickSitemapsFromRobots", () => {
const cfg = parseConnectorConfig(`id: t\nname: T\ndomain: x.com\nkind: operator\ndiscovery:\n include: ["/data-centers/"]\n exclude: ["\\\\?"]\n classify:\n - { pattern: "/data-centers/[a-z0-9-]+/$", group: facility_pages, pageType: facility_page, priority: 60 }\n`);
it("applies exclude, include and classify rules", () => {
expect(classifyUrl(cfg, "https://x.com/data-centers/abc/?x=1")).toBeNull();
expect(classifyUrl(cfg, "https://x.com/blog/abc/")).toBeNull();
expect(classifyUrl(cfg, "https://x.com/data-centers/abc/")).toMatchObject({ group: "facility_pages", pageType: "facility_page", priority: 60 });
expect(classifyUrl(cfg, "https://x.com/data-centers/")).toMatchObject({ group: "default", priority: 30 });
});
it("keeps the highest-priority urls when capping", () => {
const urls = [{ url: "a", group: "default", priority: 30 }, { url: "b", group: "facility_pages", priority: 60 }, { url: "c", group: "seed", priority: 80 }, { url: "d", group: "default", priority: 30 }];
expect(capDiscovered(urls, 2).map((u) => u.url)).toEqual(["c", "b"]);
expect(capDiscovered(urls, 10)).toHaveLength(4);
});
it("keeps only robots sitemaps on the connector's domain", () => {
expect(pickSitemapsFromRobots(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml", "https://y.com/s.xml", "junk"], "www.x.com")).toEqual(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml"]);
});
});