import { DEFAULT_UA, DirectFetcher } from "./fetchers.js"; /** * robots.txt evaluator (RFC 9309 subset): User-agent groups, Allow/Disallow longest-match with `*` wildcards and the * `$` end anchor, percent-decoded paths, Crawl-delay, Sitemap. Cached per origin (bounded, LRU-ish). * * Which group applies — the crawler identifies itself with the bot token (`DCI_USER_AGENT`, default * "datacenterindexbot") at L1. We evaluate the group addressed to that token AND the `*` group: a URL is fetched only * when both allow it (conservative: a site that disallows everybody but grants a named exception is honoured, a site * that disallows our token is honoured whatever `*` says). The L2 "browser identity" is the same direct transport with * a browser User-Agent; it is only ever used AFTER the bot token was allowed by robots — the browser UA is never the * identity robots.txt is evaluated for, and never a way around a disallow. * * Fetch outcomes: 200 → rules; 404/410 → allow everything (standard); any other status or a transport error → * keep the previously cached rules when there are some (even expired), otherwise the origin is "unknown" for one hour * and callers must restrict themselves to L1/L2 fetches (see `isAllowedByRobots().unknown`). */ export interface RobotsGroup { allow: string[]; disallow: string[]; crawlDelay: number | null } export interface Rules extends RobotsGroup { /** group addressed to our bot token (null when the file has none) */ agent: RobotsGroup | null; /** the `*` group (null when the file has none) */ star: RobotsGroup | null; sitemaps: string[]; fetchedAt: number; /** robots.txt could not be retrieved (429/5xx/timeout) and nothing was cached: only L1/L2 fetches are allowed */ unknown?: boolean; /** cache lifetime for this entry (ms) */ ttlMs?: number; } const TTL = 24 * 3_600_000; const UNKNOWN_TTL = 3_600_000; export const ROBOTS_CACHE_MAX = 5_000; const cache = new Map(); /** Product token of the crawler's User-Agent ("DataCenterIndexBot/0.1 (+…)" → "datacenterindexbot"). */ export function botToken(ua: string = DEFAULT_UA): string { const m = ua.trim().match(/^([A-Za-z0-9_.-]+)/); return (m?.[1] ?? "datacenterindexbot").toLowerCase(); } function emptyGroup(): RobotsGroup { return { allow: [], disallow: [], crawlDelay: null }; } function safeDecode(s: string): string { try { return decodeURIComponent(s); } catch { return s; } } export function parseRobots(text: string, ua: string = botToken()): Rules { const token = ua.toLowerCase(); const lines = text.split(/\r?\n/).map((l) => l.replace(/#.*/, "").trim()).filter(Boolean); const groups: Array<{ agents: string[]; rules: RobotsGroup }> = []; const sitemaps: string[] = []; let cur: (typeof groups)[number] | null = null; let lastWasAgent = false; for (const line of lines) { const i = line.indexOf(":"); if (i < 0) continue; const k = line.slice(0, i).trim().toLowerCase(); const v = line.slice(i + 1).trim(); if (k === "user-agent") { if (!cur || !lastWasAgent) { cur = { agents: [], rules: emptyGroup() }; groups.push(cur); } cur.agents.push(v.toLowerCase()); lastWasAgent = true; continue; } lastWasAgent = false; if (k === "sitemap") { sitemaps.push(v); continue; } if (!cur) continue; if (k === "allow") cur.rules.allow.push(v); else if (k === "disallow") cur.rules.disallow.push(v); else if (k === "crawl-delay") { const d = Number(v); if (Number.isFinite(d) && d >= 0) cur.rules.crawlDelay = d; } } // most specific match for our token: exact token, then a token that is a prefix of ours ("datacenterindex") or vice versa const merge = (gs: Array<{ rules: RobotsGroup }>): RobotsGroup | null => { if (!gs.length) return null; const out = emptyGroup(); for (const g of gs) { out.allow.push(...g.rules.allow); out.disallow.push(...g.rules.disallow); if (g.rules.crawlDelay != null) out.crawlDelay = out.crawlDelay == null ? g.rules.crawlDelay : Math.max(out.crawlDelay, g.rules.crawlDelay); } return out; }; const exact = groups.filter((g) => g.agents.includes(token)); const partial = exact.length ? [] : groups.filter((g) => g.agents.some((a) => a !== "*" && a.length >= 3 && (token.includes(a) || a.includes(token)))); const agent = merge(exact.length ? exact : partial); const star = merge(groups.filter((g) => g.agents.includes("*"))); const effective = agent ?? star ?? emptyGroup(); return { ...effective, agent, star, sitemaps, fetchedAt: Date.now(), ttlMs: TTL }; } /** * Length of the pattern when it matches the path (longest match wins), -1 otherwise. `*` matches any run of * characters, a trailing `$` anchors the end of the path; both sides are percent-decoded before comparison. */ export function matchLen(pattern: string, path: string): number { if (!pattern) return -1; const anchored = pattern.endsWith("$"); const body = anchored ? pattern.slice(0, -1) : pattern; const re = new RegExp("^" + body.split("*").map((s) => safeDecode(s).replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join(".*") + (anchored ? "$" : "")); return re.test(safeDecode(path)) ? pattern.length : -1; } export function groupAllows(g: RobotsGroup | null | undefined, path: string): boolean { if (!g) return true; let bestAllow = -1, bestDis = -1; for (const p of g.allow) bestAllow = Math.max(bestAllow, matchLen(p, path)); for (const p of g.disallow) bestDis = Math.max(bestDis, matchLen(p, path)); if (bestDis < 0) return true; return bestAllow >= bestDis; } /** Both the group addressed to our token (when present) and the `*` group must allow the URL. */ export function robotsAllows(rules: Rules, url: string): boolean { const u = new URL(url); const path = u.pathname + u.search; if (rules.agent || rules.star) return groupAllows(rules.agent, path) && groupAllows(rules.star, path); // legacy / hand-built rules without group detail return groupAllows({ allow: rules.allow, disallow: rules.disallow, crawlDelay: rules.crawlDelay }, path); } function cacheGet(origin: string): Rules | undefined { const hit = cache.get(origin); if (hit) { cache.delete(origin); cache.set(origin, hit); } // refresh recency return hit; } function cacheSet(origin: string, rules: Rules): void { if (cache.has(origin)) cache.delete(origin); cache.set(origin, rules); while (cache.size > ROBOTS_CACHE_MAX) { const oldest = cache.keys().next().value; if (oldest === undefined) break; cache.delete(oldest); } } /** Test hook: drop every cached robots.txt (also lets tests seed an origin). */ export function resetRobotsCache(seed?: Array<[origin: string, rules: Rules]>): void { cache.clear(); for (const [o, r] of seed ?? []) cacheSet(o, r); } export function robotsCacheSize(): number { return cache.size; } export const ALLOW_ALL: Omit = { allow: [], disallow: [], crawlDelay: null, agent: null, star: null, sitemaps: [] }; export async function getRobots(url: string, fetchImpl: (robotsUrl: string) => Promise<{ status: number; text: string; error?: { code: string } | null }> = defaultFetch): Promise { const origin = new URL(url).origin; const hit = cacheGet(origin); if (hit && Date.now() - hit.fetchedAt < (hit.ttlMs ?? TTL)) return hit; const doc = await fetchImpl(`${origin}/robots.txt`).catch(() => ({ status: 0, text: "", error: { code: "fetch_threw" } })); let rules: Rules; if (!doc.error && doc.status === 200) rules = parseRobots(doc.text); else if (!doc.error && (doc.status === 404 || doc.status === 410)) rules = { ...ALLOW_ALL, fetchedAt: Date.now(), ttlMs: TTL }; else if (hit) rules = { ...hit, fetchedAt: Date.now(), ttlMs: UNKNOWN_TTL }; // keep the last known rules a while longer else rules = { ...ALLOW_ALL, fetchedAt: Date.now(), unknown: true, ttlMs: UNKNOWN_TTL }; cacheSet(origin, rules); return rules; } async function defaultFetch(robotsUrl: string): Promise<{ status: number; text: string; error?: { code: string } | null }> { const d = await new DirectFetcher(1).fetch(robotsUrl, { timeoutMs: 15_000, maxBytes: 512 * 1024, accept: "text/plain,*/*" }); return { status: d.status, text: d.text, error: d.error ?? null }; } export interface RobotsDecision { allowed: boolean; crawlDelay: number | null; sitemaps: string[]; /** robots.txt unavailable and nothing cached: allow direct (L1/L2) fetches only */ unknown: boolean } export async function isAllowedByRobots(url: string): Promise { const rules = await getRobots(url); return { allowed: robotsAllows(rules, url), crawlDelay: rules.crawlDelay, sitemaps: rules.sitemaps, unknown: Boolean(rules.unknown) }; }