spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import { describe, expect, it } from "vitest";2import { capDiscovered, classifyUrl, extractLinks, parseFeed, parseSitemap, pickSitemapsFromRobots } from "./discovery.js";3import { parseConnectorConfig } from "./config.js";45describe("parseSitemap", () => {6 it("parses a urlset with lastmod", () => {7 const r = parseSitemap(`<?xml version="1.0"?><urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"><url><loc> https://x.com/a </loc><lastmod>2026-09-01</lastmod></url><url><loc>https://x.com/b</loc></url></urlset>`);8 expect(r.urls).toEqual([{ loc: "https://x.com/a", lastmod: "2026-09-01" }, { loc: "https://x.com/b", lastmod: null }]);9 expect(r.sitemaps).toEqual([]);10 });11 it("parses a sitemap index and a single-entry index", () => {12 expect(parseSitemap(`<sitemapindex><sitemap><loc>https://x.com/s1.xml</loc></sitemap><sitemap><loc>https://x.com/s2.xml</loc></sitemap></sitemapindex>`).sitemaps).toEqual(["https://x.com/s1.xml", "https://x.com/s2.xml"]);13 expect(parseSitemap(`<sitemapindex><sitemap><loc>https://x.com/only.xml</loc></sitemap></sitemapindex>`).sitemaps).toEqual(["https://x.com/only.xml"]);14 });15 it("accepts plain-text sitemaps and ignores garbage", () => {16 expect(parseSitemap("https://x.com/a\nnot a url\nhttps://x.com/b\n").urls.map((u) => u.loc)).toEqual(["https://x.com/a", "https://x.com/b"]);17 expect(parseSitemap("<<<").urls).toEqual([]);18 });19});2021describe("parseFeed", () => {22 it("RSS 2.0 items with categories, guid fallback and content:encoded", () => {23 const items = parseFeed(`<rss version="2.0" xmlns:content="http://purl.org/rss/1.0/modules/content/"><channel><title>t</title>24 <item><title>A</title><link>https://x.com/a</link><pubDate>Mon, 01 Sep 2026 10:00:00 GMT</pubDate><description>Summary A</description><category>News</category><category>Power</category><guid>a-1</guid></item>25 <item><title>B</title><guid>https://x.com/b</guid><content:encoded><![CDATA[<p>Body B</p>]]></content:encoded></item>26 <item><title>no link</title></item>27 </channel></rss>`);28 expect(items).toHaveLength(2);29 expect(items[0]).toMatchObject({ title: "A", link: "https://x.com/a", published: "Mon, 01 Sep 2026 10:00:00 GMT", summary: "Summary A", id: "a-1", categories: ["News", "Power"] });30 expect(items[1]).toMatchObject({ title: "B", link: "https://x.com/b", summary: "<p>Body B</p>" });31 });32 it("Atom entries: alternate link, published/updated, term categories", () => {33 const items = parseFeed(`<feed xmlns="http://www.w3.org/2005/Atom"><entry><title>E</title><link rel="self" href="https://x.com/self"/><link rel="alternate" href="https://x.com/e"/><updated>2026-09-02T00:00:00Z</updated><summary>S</summary><id>urn:e</id><category term="dc"/></entry></feed>`);34 expect(items).toEqual([{ title: "E", link: "https://x.com/e", published: "2026-09-02T00:00:00Z", summary: "S", id: "urn:e", categories: ["dc"] }]);35 });36 it("returns [] on malformed XML", () => { expect(parseFeed("<rss><channel><item>")).toEqual([]); });37});3839describe("extractLinks", () => {40 const html = `<a href="/a">a</a><a href="https://www.x.com/b#frag">b</a><a href="https://other.com/c">c</a><a href="mailto:x@y">m</a><a href="javascript:void(0)">j</a><a href="tel:1">t</a><a href="ftp://x.com/f">f</a><a href="/a">dup</a>`;41 it("resolves, dedupes, strips fragments and keeps same-site links only (www-insensitive)", () => {42 expect(extractLinks(html, "https://x.com/page")).toEqual(["https://x.com/a", "https://www.x.com/b"]);43 });44 it("can keep external links", () => {45 expect(extractLinks(html, "https://x.com/page", false)).toContain("https://other.com/c");46 });47});4849describe("classifyUrl / capDiscovered / pickSitemapsFromRobots", () => {50 const cfg = parseConnectorConfig(`id: t\nname: T\ndomain: x.com\nkind: operator\ndiscovery:\n include: ["/data-centers/"]\n exclude: ["\\\\?"]\n classify:\n - { pattern: "/data-centers/[a-z0-9-]+/$", group: facility_pages, pageType: facility_page, priority: 60 }\n`);51 it("applies exclude, include and classify rules", () => {52 expect(classifyUrl(cfg, "https://x.com/data-centers/abc/?x=1")).toBeNull();53 expect(classifyUrl(cfg, "https://x.com/blog/abc/")).toBeNull();54 expect(classifyUrl(cfg, "https://x.com/data-centers/abc/")).toMatchObject({ group: "facility_pages", pageType: "facility_page", priority: 60 });55 expect(classifyUrl(cfg, "https://x.com/data-centers/")).toMatchObject({ group: "default", priority: 30 });56 });57 it("keeps the highest-priority urls when capping", () => {58 const urls = [{ url: "a", group: "default", priority: 30 }, { url: "b", group: "facility_pages", priority: 60 }, { url: "c", group: "seed", priority: 80 }, { url: "d", group: "default", priority: 30 }];59 expect(capDiscovered(urls, 2).map((u) => u.url)).toEqual(["c", "b"]);60 expect(capDiscovered(urls, 10)).toHaveLength(4);61 });62 it("keeps only robots sitemaps on the connector's domain", () => {63 expect(pickSitemapsFromRobots(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml", "https://y.com/s.xml", "junk"], "www.x.com")).toEqual(["https://www.x.com/s.xml", "https://cdn.x.com/s.xml"]);64 });65});66