SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
3 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
5.1 KB · 66 lines typescript
Raw Blame History
1import { describe, expect, it } from "vitest";2import { capDiscovered, classifyUrl, extractLinks, parseFeed, parseSitemap, pickSitemapsFromRobots } from "./discovery.js";3import { parseConnectorConfig } from "./config.js";45describe("parseSitemap", () => {6  it("parses a urlset with lastmod", () => {7    const r = parseSitemap(`<?xml version="1.0"?><urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"><url><loc> https://x.com/a </loc><lastmod>2026-09-01</lastmod></url><url><loc>https://x.com/b</loc></url></urlset>`);8    expect(r.urls).toEqual([{ loc: "https://x.com/a", lastmod: "2026-09-01" }, { loc: "https://x.com/b", lastmod: null }]);9    expect(r.sitemaps).toEqual([]);10  });11  it("parses a sitemap index and a single-entry index", () => {12    expect(parseSitemap(`<sitemapindex><sitemap><loc>https://x.com/s1.xml</loc></sitemap><sitemap><loc>https://x.com/s2.xml</loc></sitemap></sitemapindex>`).sitemaps).toEqual(["https://x.com/s1.xml", "https://x.com/s2.xml"]);13    expect(parseSitemap(`<sitemapindex><sitemap><loc>https://x.com/only.xml</loc></sitemap></sitemapindex>`).sitemaps).toEqual(["https://x.com/only.xml"]);14  });15  it("accepts plain-text sitemaps and ignores garbage", () => {16    expect(parseSitemap("https://x.com/a\nnot a url\nhttps://x.com/b\n").urls.map((u) => u.loc)).toEqual(["https://x.com/a", "https://x.com/b"]);17    expect(parseSitemap("<<<").urls).toEqual([]);18  });19});2021describe("parseFeed", () => {22  it("RSS 2.0 items with categories, guid fallback and content:encoded", () => {23    const items = parseFeed(`<rss version="2.0" xmlns:content="http://purl.org/rss/1.0/modules/content/"><channel><title>t</title>24      <item><title>A</title><link>https://x.com/a</link><pubDate>Mon, 01 Sep 2026 10:00:00 GMT</pubDate><description>Summary A</description><category>News</category><category>Power</category><guid>a-1</guid></item>25      <item><title>B</title><guid>https://x.com/b</guid><content:encoded><![CDATA[<p>Body B</p>]]></content:encoded></item>26      <item><title>no link</title></item>27    </channel></rss>`);28    expect(items).toHaveLength(2);29    expect(items[0]).toMatchObject({ title: "A", link: "https://x.com/a", published: "Mon, 01 Sep 2026 10:00:00 GMT", summary: "Summary A", id: "a-1", categories: ["News", "Power"] });30    expect(items[1]).toMatchObject({ title: "B", link: "https://x.com/b", summary: "<p>Body B</p>" });31  });32  it("Atom entries: alternate link, published/updated, term categories", () => {33    const items = parseFeed(`<feed xmlns="http://www.w3.org/2005/Atom"><entry><title>E</title><link rel="self" href="https://x.com/self"/><link rel="alternate" href="https://x.com/e"/><updated>2026-09-02T00:00:00Z</updated><summary>S</summary><id>urn:e</id><category term="dc"/></entry></feed>`);34    expect(items).toEqual([{ title: "E", link: "https://x.com/e", published: "2026-09-02T00:00:00Z", summary: "S", id: "urn:e", categories: ["dc"] }]);35  });36  it("returns [] on malformed XML", () => { expect(parseFeed("<rss><channel><item>")).toEqual([]); });37});3839describe("extractLinks", () => {40  const html = `<a href="/a">a</a><a href="https://www.x.com/b#frag">b</a><a href="https://other.com/c">c</a><a href="mailto:x@y">m</a><a href="javascript:void(0)">j</a><a href="tel:1">t</a><a href="ftp://x.com/f">f</a><a href="/a">dup</a>`;41  it("resolves, dedupes, strips fragments and keeps same-site links only (www-insensitive)", () => {42    expect(extractLinks(html, "https://x.com/page")).toEqual(["https://x.com/a", "https://www.x.com/b"]);43  });44  it("can keep external links", () => {45    expect(extractLinks(html, "https://x.com/page", false)).toContain("https://other.com/c");46  });47});4849describe("classifyUrl / capDiscovered / pickSitemapsFromRobots", () => {50  const cfg = parseConnectorConfig(`id: t\nname: T\ndomain: x.com\nkind: operator\ndiscovery:\n  include: ["/data-centers/"]\n  exclude: ["\\\\?"]\n  classify:\n    - { pattern: "/data-centers/[a-z0-9-]+/$", group: facility_pages, pageType: facility_page, priority: 60 }\n`);51  it("applies exclude, include and classify rules", () => {52    expect(classifyUrl(cfg, "https://x.com/data-centers/abc/?x=1")).toBeNull();53    expect(classifyUrl(cfg, "https://x.com/blog/abc/")).toBeNull();54    expect(classifyUrl(cfg, "https://x.com/data-centers/abc/")).toMatchObject({ group: "facility_pages", pageType: "facility_page", priority: 60 });55    expect(classifyUrl(cfg, "https://x.com/data-centers/")).toMatchObject({ group: "default", priority: 30 });56  });57  it("keeps the highest-priority urls when capping", () => {58    const urls = [{ url: "a", group: "default", priority: 30 }, { url: "b", group: "facility_pages", priority: 60 }, { url: "c", group: "seed", priority: 80 }, { url: "d", group: "default", priority: 30 }];59    expect(capDiscovered(urls, 2).map((u) => u.url)).toEqual(["c", "b"]);60    expect(capDiscovered(urls, 10)).toHaveLength(4);61  });62  it("keeps only robots sitemaps on the connector's domain", () => {63    expect(pickSitemapsFromRobots(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml", "https://y.com/s.xml", "junk"], "www.x.com")).toEqual(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml"]);64  });65});66