import { afterEach, describe, expect, it } from "vitest"; import { botToken, getRobots, groupAllows, isAllowedByRobots, matchLen, parseRobots, resetRobotsCache, robotsAllows, robotsCacheSize, ROBOTS_CACHE_MAX } from "./robots.js"; const ORIGIN = "https://example.com"; afterEach(() => resetRobotsCache()); describe("parseRobots", () => { it("picks the group addressed to our token and the * group separately", () => { const r = parseRobots(`User-agent: *\nDisallow: /private\nCrawl-delay: 2\n\nUser-agent: DataCenterIndexBot\nAllow: /private/public-report\nDisallow: /tmp\nCrawl-delay: 5\nSitemap: https://example.com/sitemap.xml`, "datacenterindexbot"); expect(r.agent).toEqual({ allow: ["/private/public-report"], disallow: ["/tmp"], crawlDelay: 5 }); expect(r.star).toEqual({ allow: [], disallow: ["/private"], crawlDelay: 2 }); expect(r.crawlDelay).toBe(5); // our group's delay wins expect(r.sitemaps).toEqual(["https://example.com/sitemap.xml"]); }); it("falls back to * when no group names us; consecutive User-agent lines share one group; comments stripped", () => { const r = parseRobots(`# comment\nUser-agent: Googlebot\nUser-agent: Bingbot\nDisallow: /g\n\nUser-agent: *\nDisallow: /x # trailing\nAllow: /x/y`); expect(r.agent).toBeNull(); expect(r.star).toEqual({ allow: ["/x/y"], disallow: ["/x"], crawlDelay: null }); expect(r.disallow).toEqual(["/x"]); }); it("matches a partial token (product name without version)", () => { const r = parseRobots(`User-agent: datacenterindex\nDisallow: /a\n\nUser-agent: *\nDisallow: /b`, "datacenterindexbot"); expect(r.agent?.disallow).toEqual(["/a"]); }); it("derives the bot token from the User-Agent string", () => { expect(botToken("DataCenterIndexBot/0.1 (+https://www.datacenterindex.io/bot)")).toBe("datacenterindexbot"); expect(botToken("mybot")).toBe("mybot"); }); it("ignores invalid crawl-delays", () => { expect(parseRobots("User-agent: *\nCrawl-delay: fast").crawlDelay).toBeNull(); expect(parseRobots("User-agent: *\nCrawl-delay: -3").crawlDelay).toBeNull(); expect(parseRobots("User-agent: *\nCrawl-delay: 0.5").crawlDelay).toBe(0.5); }); }); describe("matchLen / groupAllows", () => { it("prefix match, * wildcard and $ end anchor", () => { expect(matchLen("/a", "/abc")).toBe(2); expect(matchLen("/a*c", "/abc")).toBe(4); expect(matchLen("/a*c", "/abd")).toBe(-1); expect(matchLen("/*.pdf$", "/docs/x.pdf")).toBe(7); expect(matchLen("/*.pdf$", "/docs/x.pdf?dl=1")).toBe(-1); // anchored: must end there expect(matchLen("/docs$", "/docs")).toBe(6); expect(matchLen("/docs$", "/docs/")).toBe(-1); expect(matchLen("", "/x")).toBe(-1); }); it("percent-decodes both the pattern and the path", () => { expect(matchLen("/caf%C3%A9", "/café")).toBeGreaterThan(0); expect(matchLen("/café", "/caf%C3%A9")).toBeGreaterThan(0); expect(matchLen("/a%2Fb", "/a/b")).toBeGreaterThan(0); expect(matchLen("/%ZZ", "/%ZZ")).toBeGreaterThan(0); // malformed escapes compare raw }); it("longest match wins, allow ties beat disallow", () => { const g = { allow: ["/private/report"], disallow: ["/private"], crawlDelay: null }; expect(groupAllows(g, "/private/report.pdf")).toBe(true); expect(groupAllows(g, "/private/other")).toBe(false); expect(groupAllows(g, "/public")).toBe(true); expect(groupAllows({ allow: ["/p"], disallow: ["/p"], crawlDelay: null }, "/p")).toBe(true); expect(groupAllows(null, "/anything")).toBe(true); }); }); describe("robotsAllows", () => { it("requires both our group and the * group to allow", () => { const r = parseRobots(`User-agent: *\nDisallow: /star-only\n\nUser-agent: datacenterindexbot\nDisallow: /bot-only`); expect(robotsAllows(r, `${ORIGIN}/star-only/x`)).toBe(false); expect(robotsAllows(r, `${ORIGIN}/bot-only/x`)).toBe(false); expect(robotsAllows(r, `${ORIGIN}/ok`)).toBe(true); }); it("evaluates path + query and honours $ anchors on the query", () => { const r = parseRobots(`User-agent: *\nDisallow: /*?*sort=\nDisallow: /exact$`); expect(robotsAllows(r, `${ORIGIN}/list?sort=asc`)).toBe(false); expect(robotsAllows(r, `${ORIGIN}/list?page=2`)).toBe(true); expect(robotsAllows(r, `${ORIGIN}/exact`)).toBe(false); expect(robotsAllows(r, `${ORIGIN}/exact/child`)).toBe(true); }); it("Disallow: (empty) allows everything", () => { expect(robotsAllows(parseRobots("User-agent: *\nDisallow:"), `${ORIGIN}/x`)).toBe(true); }); }); describe("getRobots cache + fetch outcomes", () => { const fetchWith = (status: number, text = "", error?: { code: string }) => async () => ({ status, text, error: error ?? null }); it("200 → rules; 404/410 → allow all", async () => { const r = await getRobots(`${ORIGIN}/x`, fetchWith(200, "User-agent: *\nDisallow: /x")); expect(robotsAllows(r, `${ORIGIN}/x`)).toBe(false); resetRobotsCache(); const r404 = await getRobots(`${ORIGIN}/x`, fetchWith(404)); expect(r404.unknown).toBeFalsy(); expect(robotsAllows(r404, `${ORIGIN}/x`)).toBe(true); }); it("503 / timeout with nothing cached → unknown (short TTL); with a cache → previous rules kept", async () => { const u = await getRobots(`${ORIGIN}/x`, fetchWith(503)); expect(u.unknown).toBe(true); expect(u.ttlMs).toBe(3_600_000); expect((await isAllowedByRobots(`${ORIGIN}/x`)).unknown).toBe(true); resetRobotsCache(); await getRobots(`${ORIGIN}/x`, fetchWith(200, "User-agent: *\nDisallow: /x")); resetRobotsCache([[ORIGIN, { ...(await getRobots(`${ORIGIN}/x`, fetchWith(200, "User-agent: *\nDisallow: /x"))), fetchedAt: 0 }]]); // expired entry const kept = await getRobots(`${ORIGIN}/x`, fetchWith(0, "", { code: "timeout" })); expect(kept.unknown).toBeFalsy(); expect(robotsAllows(kept, `${ORIGIN}/x`)).toBe(false); }); it("caches per origin and evicts the least recently used entry beyond the cap", async () => { let calls = 0; const f = async () => { calls++; return { status: 200, text: "User-agent: *\nDisallow: /x", error: null }; }; await getRobots(`${ORIGIN}/a`, f); await getRobots(`${ORIGIN}/b`, f); expect(calls).toBe(1); resetRobotsCache(); for (let i = 0; i < ROBOTS_CACHE_MAX + 10; i++) await getRobots(`https://h${i}.example/`, f); expect(robotsCacheSize()).toBe(ROBOTS_CACHE_MAX); // the very first origin was evicted, a recent one is still cached calls = 0; await getRobots("https://h0.example/", f); expect(calls).toBe(1); calls = 0; await getRobots(`https://h${ROBOTS_CACHE_MAX + 9}.example/`, f); expect(calls).toBe(0); }); });