SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
3 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
8.5 KB · 163 lines typescript
Raw Blame History
1import { DEFAULT_UA, DirectFetcher } from "./fetchers.js";23/**4 * robots.txt evaluator (RFC 9309 subset): User-agent groups, Allow/Disallow longest-match with `*` wildcards and the5 * `$` end anchor, percent-decoded paths, Crawl-delay, Sitemap. Cached per origin (bounded, LRU-ish).6 *7 * Which group applies — the crawler identifies itself with the bot token (`DCI_USER_AGENT`, default8 * "datacenterindexbot") at L1. We evaluate the group addressed to that token AND the `*` group: a URL is fetched only9 * when both allow it (conservative: a site that disallows everybody but grants a named exception is honoured, a site10 * that disallows our token is honoured whatever `*` says). The L2 "browser identity" is the same direct transport with11 * a browser User-Agent; it is only ever used AFTER the bot token was allowed by robots — the browser UA is never the12 * identity robots.txt is evaluated for, and never a way around a disallow.13 *14 * Fetch outcomes: 200 → rules; 404/410 → allow everything (standard); any other status or a transport error →15 * keep the previously cached rules when there are some (even expired), otherwise the origin is "unknown" for one hour16 * and callers must restrict themselves to L1/L2 fetches (see `isAllowedByRobots().unknown`).17 */18export interface RobotsGroup { allow: string[]; disallow: string[]; crawlDelay: number | null }19export interface Rules extends RobotsGroup {20  /** group addressed to our bot token (null when the file has none) */21  agent: RobotsGroup | null;22  /** the `*` group (null when the file has none) */23  star: RobotsGroup | null;24  sitemaps: string[];25  fetchedAt: number;26  /** robots.txt could not be retrieved (429/5xx/timeout) and nothing was cached: only L1/L2 fetches are allowed */27  unknown?: boolean;28  /** cache lifetime for this entry (ms) */29  ttlMs?: number;30}3132const TTL = 24 * 3_600_000;33const UNKNOWN_TTL = 3_600_000;34export const ROBOTS_CACHE_MAX = 5_000;35const cache = new Map<string, Rules>();3637/** Product token of the crawler's User-Agent ("DataCenterIndexBot/0.1 (+…)" → "datacenterindexbot"). */38export function botToken(ua: string = DEFAULT_UA): string {39  const m = ua.trim().match(/^([A-Za-z0-9_.-]+)/);40  return (m?.[1] ?? "datacenterindexbot").toLowerCase();41}4243function emptyGroup(): RobotsGroup { return { allow: [], disallow: [], crawlDelay: null }; }4445function safeDecode(s: string): string {46  try { return decodeURIComponent(s); } catch { return s; }47}4849export function parseRobots(text: string, ua: string = botToken()): Rules {50  const token = ua.toLowerCase();51  const lines = text.split(/\r?\n/).map((l) => l.replace(/#.*/, "").trim()).filter(Boolean);52  const groups: Array<{ agents: string[]; rules: RobotsGroup }> = [];53  const sitemaps: string[] = [];54  let cur: (typeof groups)[number] | null = null;55  let lastWasAgent = false;56  for (const line of lines) {57    const i = line.indexOf(":");58    if (i < 0) continue;59    const k = line.slice(0, i).trim().toLowerCase();60    const v = line.slice(i + 1).trim();61    if (k === "user-agent") {62      if (!cur || !lastWasAgent) { cur = { agents: [], rules: emptyGroup() }; groups.push(cur); }63      cur.agents.push(v.toLowerCase());64      lastWasAgent = true;65      continue;66    }67    lastWasAgent = false;68    if (k === "sitemap") { sitemaps.push(v); continue; }69    if (!cur) continue;70    if (k === "allow") cur.rules.allow.push(v);71    else if (k === "disallow") cur.rules.disallow.push(v);72    else if (k === "crawl-delay") { const d = Number(v); if (Number.isFinite(d) && d >= 0) cur.rules.crawlDelay = d; }73  }74  // most specific match for our token: exact token, then a token that is a prefix of ours ("datacenterindex") or vice versa75  const merge = (gs: Array<{ rules: RobotsGroup }>): RobotsGroup | null => {76    if (!gs.length) return null;77    const out = emptyGroup();78    for (const g of gs) { out.allow.push(...g.rules.allow); out.disallow.push(...g.rules.disallow); if (g.rules.crawlDelay != null) out.crawlDelay = out.crawlDelay == null ? g.rules.crawlDelay : Math.max(out.crawlDelay, g.rules.crawlDelay); }79    return out;80  };81  const exact = groups.filter((g) => g.agents.includes(token));82  const partial = exact.length ? [] : groups.filter((g) => g.agents.some((a) => a !== "*" && a.length >= 3 && (token.includes(a) || a.includes(token))));83  const agent = merge(exact.length ? exact : partial);84  const star = merge(groups.filter((g) => g.agents.includes("*")));85  const effective = agent ?? star ?? emptyGroup();86  return { ...effective, agent, star, sitemaps, fetchedAt: Date.now(), ttlMs: TTL };87}8889/**90 * Length of the pattern when it matches the path (longest match wins), -1 otherwise. `*` matches any run of91 * characters, a trailing `$` anchors the end of the path; both sides are percent-decoded before comparison.92 */93export function matchLen(pattern: string, path: string): number {94  if (!pattern) return -1;95  const anchored = pattern.endsWith("$");96  const body = anchored ? pattern.slice(0, -1) : pattern;97  const re = new RegExp("^" + body.split("*").map((s) => safeDecode(s).replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join(".*") + (anchored ? "$" : ""));98  return re.test(safeDecode(path)) ? pattern.length : -1;99}100101export function groupAllows(g: RobotsGroup | null | undefined, path: string): boolean {102  if (!g) return true;103  let bestAllow = -1, bestDis = -1;104  for (const p of g.allow) bestAllow = Math.max(bestAllow, matchLen(p, path));105  for (const p of g.disallow) bestDis = Math.max(bestDis, matchLen(p, path));106  if (bestDis < 0) return true;107  return bestAllow >= bestDis;108}109110/** Both the group addressed to our token (when present) and the `*` group must allow the URL. */111export function robotsAllows(rules: Rules, url: string): boolean {112  const u = new URL(url);113  const path = u.pathname + u.search;114  if (rules.agent || rules.star) return groupAllows(rules.agent, path) && groupAllows(rules.star, path);115  // legacy / hand-built rules without group detail116  return groupAllows({ allow: rules.allow, disallow: rules.disallow, crawlDelay: rules.crawlDelay }, path);117}118119function cacheGet(origin: string): Rules | undefined {120  const hit = cache.get(origin);121  if (hit) { cache.delete(origin); cache.set(origin, hit); } // refresh recency122  return hit;123}124function cacheSet(origin: string, rules: Rules): void {125  if (cache.has(origin)) cache.delete(origin);126  cache.set(origin, rules);127  while (cache.size > ROBOTS_CACHE_MAX) { const oldest = cache.keys().next().value; if (oldest === undefined) break; cache.delete(oldest); }128}129/** Test hook: drop every cached robots.txt (also lets tests seed an origin). */130export function resetRobotsCache(seed?: Array<[origin: string, rules: Rules]>): void {131  cache.clear();132  for (const [o, r] of seed ?? []) cacheSet(o, r);133}134export function robotsCacheSize(): number { return cache.size; }135136export const ALLOW_ALL: Omit<Rules, "fetchedAt"> = { allow: [], disallow: [], crawlDelay: null, agent: null, star: null, sitemaps: [] };137138export async function getRobots(url: string, fetchImpl: (robotsUrl: string) => Promise<{ status: number; text: string; error?: { code: string } | null }> = defaultFetch): Promise<Rules> {139  const origin = new URL(url).origin;140  const hit = cacheGet(origin);141  if (hit && Date.now() - hit.fetchedAt < (hit.ttlMs ?? TTL)) return hit;142  const doc = await fetchImpl(`${origin}/robots.txt`).catch(() => ({ status: 0, text: "", error: { code: "fetch_threw" } }));143  let rules: Rules;144  if (!doc.error && doc.status === 200) rules = parseRobots(doc.text);145  else if (!doc.error && (doc.status === 404 || doc.status === 410)) rules = { ...ALLOW_ALL, fetchedAt: Date.now(), ttlMs: TTL };146  else if (hit) rules = { ...hit, fetchedAt: Date.now(), ttlMs: UNKNOWN_TTL }; // keep the last known rules a while longer147  else rules = { ...ALLOW_ALL, fetchedAt: Date.now(), unknown: true, ttlMs: UNKNOWN_TTL };148  cacheSet(origin, rules);149  return rules;150}151152async function defaultFetch(robotsUrl: string): Promise<{ status: number; text: string; error?: { code: string } | null }> {153  const d = await new DirectFetcher(1).fetch(robotsUrl, { timeoutMs: 15_000, maxBytes: 512 * 1024, accept: "text/plain,*/*" });154  return { status: d.status, text: d.text, error: d.error ?? null };155}156157export interface RobotsDecision { allowed: boolean; crawlDelay: number | null; sitemaps: string[]; /** robots.txt unavailable and nothing cached: allow direct (L1/L2) fetches only */ unknown: boolean }158159export async function isAllowedByRobots(url: string): Promise<RobotsDecision> {160  const rules = await getRobots(url);161  return { allowed: robotsAllows(rules, url), crawlDelay: rules.crawlDelay, sitemaps: rules.sitemaps, unknown: Boolean(rules.unknown) };162}163