spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import { DEFAULT_UA, DirectFetcher } from "./fetchers.js";23/**4 * robots.txt evaluator (RFC 9309 subset): User-agent groups, Allow/Disallow longest-match with `*` wildcards and the5 * `$` end anchor, percent-decoded paths, Crawl-delay, Sitemap. Cached per origin (bounded, LRU-ish).6 *7 * Which group applies — the crawler identifies itself with the bot token (`DCI_USER_AGENT`, default8 * "datacenterindexbot") at L1. We evaluate the group addressed to that token AND the `*` group: a URL is fetched only9 * when both allow it (conservative: a site that disallows everybody but grants a named exception is honoured, a site10 * that disallows our token is honoured whatever `*` says). The L2 "browser identity" is the same direct transport with11 * a browser User-Agent; it is only ever used AFTER the bot token was allowed by robots — the browser UA is never the12 * identity robots.txt is evaluated for, and never a way around a disallow.13 *14 * Fetch outcomes: 200 → rules; 404/410 → allow everything (standard); any other status or a transport error →15 * keep the previously cached rules when there are some (even expired), otherwise the origin is "unknown" for one hour16 * and callers must restrict themselves to L1/L2 fetches (see `isAllowedByRobots().unknown`).17 */18export interface RobotsGroup { allow: string[]; disallow: string[]; crawlDelay: number | null }19export interface Rules extends RobotsGroup {20 /** group addressed to our bot token (null when the file has none) */21 agent: RobotsGroup | null;22 /** the `*` group (null when the file has none) */23 star: RobotsGroup | null;24 sitemaps: string[];25 fetchedAt: number;26 /** robots.txt could not be retrieved (429/5xx/timeout) and nothing was cached: only L1/L2 fetches are allowed */27 unknown?: boolean;28 /** cache lifetime for this entry (ms) */29 ttlMs?: number;30}3132const TTL = 24 * 3_600_000;33const UNKNOWN_TTL = 3_600_000;34export const ROBOTS_CACHE_MAX = 5_000;35const cache = new Map<string, Rules>();3637/** Product token of the crawler's User-Agent ("DataCenterIndexBot/0.1 (+…)" → "datacenterindexbot"). */38export function botToken(ua: string = DEFAULT_UA): string {39 const m = ua.trim().match(/^([A-Za-z0-9_.-]+)/);40 return (m?.[1] ?? "datacenterindexbot").toLowerCase();41}4243function emptyGroup(): RobotsGroup { return { allow: [], disallow: [], crawlDelay: null }; }4445function safeDecode(s: string): string {46 try { return decodeURIComponent(s); } catch { return s; }47}4849export function parseRobots(text: string, ua: string = botToken()): Rules {50 const token = ua.toLowerCase();51 const lines = text.split(/\r?\n/).map((l) => l.replace(/#.*/, "").trim()).filter(Boolean);52 const groups: Array<{ agents: string[]; rules: RobotsGroup }> = [];53 const sitemaps: string[] = [];54 let cur: (typeof groups)[number] | null = null;55 let lastWasAgent = false;56 for (const line of lines) {57 const i = line.indexOf(":");58 if (i < 0) continue;59 const k = line.slice(0, i).trim().toLowerCase();60 const v = line.slice(i + 1).trim();61 if (k === "user-agent") {62 if (!cur || !lastWasAgent) { cur = { agents: [], rules: emptyGroup() }; groups.push(cur); }63 cur.agents.push(v.toLowerCase());64 lastWasAgent = true;65 continue;66 }67 lastWasAgent = false;68 if (k === "sitemap") { sitemaps.push(v); continue; }69 if (!cur) continue;70 if (k === "allow") cur.rules.allow.push(v);71 else if (k === "disallow") cur.rules.disallow.push(v);72 else if (k === "crawl-delay") { const d = Number(v); if (Number.isFinite(d) && d >= 0) cur.rules.crawlDelay = d; }73 }74 // most specific match for our token: exact token, then a token that is a prefix of ours ("datacenterindex") or vice versa75 const merge = (gs: Array<{ rules: RobotsGroup }>): RobotsGroup | null => {76 if (!gs.length) return null;77 const out = emptyGroup();78 for (const g of gs) { out.allow.push(...g.rules.allow); out.disallow.push(...g.rules.disallow); if (g.rules.crawlDelay != null) out.crawlDelay = out.crawlDelay == null ? g.rules.crawlDelay : Math.max(out.crawlDelay, g.rules.crawlDelay); }79 return out;80 };81 const exact = groups.filter((g) => g.agents.includes(token));82 const partial = exact.length ? [] : groups.filter((g) => g.agents.some((a) => a !== "*" && a.length >= 3 && (token.includes(a) || a.includes(token))));83 const agent = merge(exact.length ? exact : partial);84 const star = merge(groups.filter((g) => g.agents.includes("*")));85 const effective = agent ?? star ?? emptyGroup();86 return { ...effective, agent, star, sitemaps, fetchedAt: Date.now(), ttlMs: TTL };87}8889/**90 * Length of the pattern when it matches the path (longest match wins), -1 otherwise. `*` matches any run of91 * characters, a trailing `$` anchors the end of the path; both sides are percent-decoded before comparison.92 */93export function matchLen(pattern: string, path: string): number {94 if (!pattern) return -1;95 const anchored = pattern.endsWith("$");96 const body = anchored ? pattern.slice(0, -1) : pattern;97 const re = new RegExp("^" + body.split("*").map((s) => safeDecode(s).replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join(".*") + (anchored ? "$" : ""));98 return re.test(safeDecode(path)) ? pattern.length : -1;99}100101export function groupAllows(g: RobotsGroup | null | undefined, path: string): boolean {102 if (!g) return true;103 let bestAllow = -1, bestDis = -1;104 for (const p of g.allow) bestAllow = Math.max(bestAllow, matchLen(p, path));105 for (const p of g.disallow) bestDis = Math.max(bestDis, matchLen(p, path));106 if (bestDis < 0) return true;107 return bestAllow >= bestDis;108}109110/** Both the group addressed to our token (when present) and the `*` group must allow the URL. */111export function robotsAllows(rules: Rules, url: string): boolean {112 const u = new URL(url);113 const path = u.pathname + u.search;114 if (rules.agent || rules.star) return groupAllows(rules.agent, path) && groupAllows(rules.star, path);115 // legacy / hand-built rules without group detail116 return groupAllows({ allow: rules.allow, disallow: rules.disallow, crawlDelay: rules.crawlDelay }, path);117}118119function cacheGet(origin: string): Rules | undefined {120 const hit = cache.get(origin);121 if (hit) { cache.delete(origin); cache.set(origin, hit); } // refresh recency122 return hit;123}124function cacheSet(origin: string, rules: Rules): void {125 if (cache.has(origin)) cache.delete(origin);126 cache.set(origin, rules);127 while (cache.size > ROBOTS_CACHE_MAX) { const oldest = cache.keys().next().value; if (oldest === undefined) break; cache.delete(oldest); }128}129/** Test hook: drop every cached robots.txt (also lets tests seed an origin). */130export function resetRobotsCache(seed?: Array<[origin: string, rules: Rules]>): void {131 cache.clear();132 for (const [o, r] of seed ?? []) cacheSet(o, r);133}134export function robotsCacheSize(): number { return cache.size; }135136export const ALLOW_ALL: Omit<Rules, "fetchedAt"> = { allow: [], disallow: [], crawlDelay: null, agent: null, star: null, sitemaps: [] };137138export async function getRobots(url: string, fetchImpl: (robotsUrl: string) => Promise<{ status: number; text: string; error?: { code: string } | null }> = defaultFetch): Promise<Rules> {139 const origin = new URL(url).origin;140 const hit = cacheGet(origin);141 if (hit && Date.now() - hit.fetchedAt < (hit.ttlMs ?? TTL)) return hit;142 const doc = await fetchImpl(`${origin}/robots.txt`).catch(() => ({ status: 0, text: "", error: { code: "fetch_threw" } }));143 let rules: Rules;144 if (!doc.error && doc.status === 200) rules = parseRobots(doc.text);145 else if (!doc.error && (doc.status === 404 || doc.status === 410)) rules = { ...ALLOW_ALL, fetchedAt: Date.now(), ttlMs: TTL };146 else if (hit) rules = { ...hit, fetchedAt: Date.now(), ttlMs: UNKNOWN_TTL }; // keep the last known rules a while longer147 else rules = { ...ALLOW_ALL, fetchedAt: Date.now(), unknown: true, ttlMs: UNKNOWN_TTL };148 cacheSet(origin, rules);149 return rules;150}151152async function defaultFetch(robotsUrl: string): Promise<{ status: number; text: string; error?: { code: string } | null }> {153 const d = await new DirectFetcher(1).fetch(robotsUrl, { timeoutMs: 15_000, maxBytes: 512 * 1024, accept: "text/plain,*/*" });154 return { status: d.status, text: d.text, error: d.error ?? null };155}156157export interface RobotsDecision { allowed: boolean; crawlDelay: number | null; sitemaps: string[]; /** robots.txt unavailable and nothing cached: allow direct (L1/L2) fetches only */ unknown: boolean }158159export async function isAllowedByRobots(url: string): Promise<RobotsDecision> {160 const rules = await getRobots(url);161 return { allowed: robotsAllows(rules, url), crawlDelay: rules.crawlDelay, sitemaps: rules.sitemaps, unknown: Boolean(rules.unknown) };162}163