SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
3 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
10.9 KB · 171 lines typescript
Raw Blame History
1import { XMLParser } from "fast-xml-parser";2import { load } from "cheerio";3import type { PageType } from "@dci/core";4import type { ConnectorConfig } from "./config.js";5import type { ConnectorContext, DiscoveredUrl, RawDocument } from "./types.js";67const xml = new XMLParser({ ignoreAttributes: false, attributeNamePrefix: "@_", textNodeName: "#text", trimValues: true, parseTagValue: false });89/** Parse a sitemap or sitemap index. Returns child sitemaps and URL entries. */10export function parseSitemap(text: string): { sitemaps: string[]; urls: Array<{ loc: string; lastmod: string | null }> } {11  const out = { sitemaps: [] as string[], urls: [] as Array<{ loc: string; lastmod: string | null }> };12  let doc: Record<string, unknown>;13  try { doc = xml.parse(text) as Record<string, unknown>; } catch { return out; }14  const idx = (doc.sitemapindex as { sitemap?: unknown })?.sitemap;15  if (idx) for (const s of Array.isArray(idx) ? idx : [idx]) { const loc = (s as { loc?: string }).loc; if (loc) out.sitemaps.push(String(loc).trim()); }16  const set = (doc.urlset as { url?: unknown })?.url;17  if (set) for (const u of Array.isArray(set) ? set : [set]) { const e = u as { loc?: string; lastmod?: string }; if (e.loc) out.urls.push({ loc: String(e.loc).trim(), lastmod: e.lastmod ? String(e.lastmod) : null }); }18  if (!idx && !set) {19    // plain-text sitemap20    for (const line of text.split(/\r?\n/)) { const l = line.trim(); if (/^https?:\/\//.test(l)) out.urls.push({ loc: l, lastmod: null }); }21  }22  return out;23}2425export interface FeedItem { title: string; link: string; published: string | null; summary: string | null; id: string | null; categories: string[] }2627/** RSS 2.0 / Atom / RDF feed parser. */28export function parseFeed(text: string): FeedItem[] {29  let doc: Record<string, unknown>;30  try { doc = xml.parse(text) as Record<string, unknown>; } catch { return []; }31  const items: FeedItem[] = [];32  const txt = (v: unknown): string | null => { if (v == null) return null; if (typeof v === "string") return v.trim() || null; if (typeof v === "object") { const o = v as Record<string, unknown>; return txt(o["#text"] ?? o["@_href"] ?? null); } return String(v); };33  const chan = (doc.rss as { channel?: { item?: unknown } })?.channel?.item ?? (doc["rdf:RDF"] as { item?: unknown })?.item;34  if (chan) for (const it of Array.isArray(chan) ? chan : [chan]) {35    const o = it as Record<string, unknown>;36    const link = txt(o.link) ?? txt(o.guid);37    if (!link) continue;38    const cats = o.category ? (Array.isArray(o.category) ? o.category : [o.category]).map((c) => txt(c)).filter((c): c is string => Boolean(c)) : [];39    items.push({ title: txt(o.title) ?? link, link, published: txt(o.pubDate) ?? txt(o["dc:date"]) ?? null, summary: txt(o.description) ?? txt(o["content:encoded"]) ?? null, id: txt(o.guid) ?? null, categories: cats });40  }41  const entries = (doc.feed as { entry?: unknown })?.entry;42  if (entries) for (const en of Array.isArray(entries) ? entries : [entries]) {43    const o = en as Record<string, unknown>;44    const links = Array.isArray(o.link) ? o.link : o.link ? [o.link] : [];45    const alt = (links as Array<Record<string, unknown>>).find((l) => !l["@_rel"] || l["@_rel"] === "alternate") ?? (links as Array<Record<string, unknown>>)[0];46    const link = alt ? String(alt["@_href"] ?? txt(alt) ?? "") : "";47    if (!link) continue;48    const cats = o.category ? (Array.isArray(o.category) ? o.category : [o.category]).map((c) => String((c as Record<string, unknown>)["@_term"] ?? txt(c) ?? "")).filter(Boolean) : [];49    items.push({ title: txt(o.title) ?? link, link, published: txt(o.published) ?? txt(o.updated) ?? null, summary: txt(o.summary) ?? txt(o.content) ?? null, id: txt(o.id) ?? null, categories: cats });50  }51  return items;52}5354/** Absolute, same-site links from an HTML page (anchors only). */55export function extractLinks(html: string, baseUrl: string, sameHostOnly = true): string[] {56  const $ = load(html);57  const base = new URL(baseUrl);58  const out = new Set<string>();59  $("a[href]").each((_, a) => {60    const href = $(a).attr("href");61    if (!href || /^(mailto:|tel:|javascript:|#)/i.test(href)) return;62    try {63      const u = new URL(href, base);64      if (!/^https?:$/.test(u.protocol)) return;65      if (sameHostOnly && u.hostname.replace(/^www\./, "") !== base.hostname.replace(/^www\./, "")) return;66      u.hash = "";67      out.add(u.toString());68    } catch { /* ignore */ }69  });70  return [...out];71}7273function compile(patterns: string[]): RegExp[] { return patterns.map((p) => new RegExp(p, "i")); }7475/** Apply include/exclude/classify rules from the config to a URL. Returns null when excluded. */76export function classifyUrl(cfg: ConnectorConfig, url: string, discoveredFrom: string | null = null): DiscoveredUrl | null {77  const d = cfg.discovery;78  if (d.exclude.length && compile(d.exclude).some((re) => re.test(url))) return null;79  if (d.include.length && !compile(d.include).some((re) => re.test(url))) return null;80  for (const rule of d.classify) {81    if (new RegExp(rule.pattern, "i").test(url)) return { url, group: rule.group, pageType: rule.pageType as PageType | undefined, priority: rule.priority ?? 50, discoveredFrom, minLevel: rule.minLevel as DiscoveredUrl["minLevel"] };82  }83  return { url, group: "default", priority: 30, discoveredFrom };84}8586/**87 * Cap a classified URL set to `max`: every candidate is classified first, then the most valuable ones are kept —88 * higher priority first (seeds 80, facility groups typically 60, newsroom 50–70, `default` 30), ties by discovery89 * order. Truncating a sitemap by document order instead would silently drop facility pages that happen to be listed90 * after thousands of blog posts.91 */92export function capDiscovered(urls: Iterable<DiscoveredUrl>, max: number): DiscoveredUrl[] {93  const all = [...urls];94  if (!Number.isFinite(max) || max <= 0 || all.length <= max) return all;95  const rank = (u: DiscoveredUrl) => (u.priority ?? 50) + (u.group && u.group !== "default" ? 0.5 : 0);96  return all.map((u, i) => ({ u, i })).sort((a, b) => rank(b.u) - rank(a.u) || a.i - b.i).slice(0, max).map((x) => x.u);97}9899/** Generic discovery: sitemaps (auto or listed), RSS feeds, seeds, and link-following from index pages. */100export async function discoverGeneric(cfg: ConnectorConfig, ctx: ConnectorContext): Promise<DiscoveredUrl[]> {101  const found = new Map<string, DiscoveredUrl>();102  // classify everything, cap at the end (capDiscovered); the hard ceiling only bounds memory on pathological sitemaps103  const hardCap = Math.max(cfg.discovery.maxUrlsPerRun * 10, 50_000);104  const add = (u: DiscoveredUrl | null) => { if (u && !found.has(u.url) && found.size < hardCap) found.set(u.url, u); };105  const origin = `https://${cfg.domain.replace(/^https?:\/\//, "").replace(/\/$/, "")}`;106107  // seeds108  for (const s of cfg.discovery.seeds) {109    const seed = typeof s === "string" ? { url: s, group: "seed" as string, pageType: undefined as PageType | undefined, minLevel: undefined as number | undefined } : s;110    const url = seed.url.startsWith("http") ? seed.url : origin + seed.url;111    const c = classifyUrl(cfg, url, null);112    add({ url, group: seed.group ?? c?.group ?? "seed", pageType: seed.pageType ?? c?.pageType, priority: 80, minLevel: (seed.minLevel ?? c?.minLevel) as DiscoveredUrl["minLevel"], discoveredFrom: null });113  }114115  // sitemaps116  const sitemapUrls: string[] = [];117  if (cfg.discovery.sitemap === true) sitemapUrls.push(`${origin}/sitemap.xml`, `${origin}/sitemap_index.xml`, `${origin}/sitemap-index.xml`);118  else if (Array.isArray(cfg.discovery.sitemap)) for (const s of cfg.discovery.sitemap) sitemapUrls.push(s.startsWith("http") ? s : origin + s);119  const seenMaps = new Set<string>();120  let mapBudget = 40;121  while (sitemapUrls.length && mapBudget-- > 0) {122    const sm = sitemapUrls.shift()!;123    if (seenMaps.has(sm)) continue;124    seenMaps.add(sm);125    const doc = await ctx.fetch(sm, { group: "sitemap", accept: "application/xml,text/xml,*/*", maxBytes: 60 * 1024 * 1024 });126    if (doc.error || doc.status !== 200) { ctx.log("debug", `sitemap ${sm} → ${doc.error?.code ?? doc.status}`); continue; }127    const parsed = parseSitemap(doc.text);128    for (const child of parsed.sitemaps) if (!seenMaps.has(child)) sitemapUrls.push(child);129    for (const u of parsed.urls) { const c = classifyUrl(cfg, u.loc, sm); if (c) add({ ...c, lastmod: u.lastmod }); }130    ctx.log("info", `sitemap ${sm}: ${parsed.urls.length} urls, ${parsed.sitemaps.length} children`);131  }132133  // rss134  for (const f of cfg.discovery.rss) {135    const url = f.startsWith("http") ? f : origin + f;136    const doc = await ctx.fetch(url, { group: "rss", accept: "application/rss+xml,application/atom+xml,application/xml,text/xml,*/*" });137    if (doc.error || doc.status !== 200) { ctx.log("warn", `rss ${url} → ${doc.error?.code ?? doc.status}`); continue; }138    const items = parseFeed(doc.text);139    for (const it of items) { const c = classifyUrl(cfg, it.link, url); if (c) add({ ...c, group: c.group === "default" ? "newsroom" : c.group, pageType: c.pageType ?? "press_release", priority: 70, lastmod: it.published, meta: { title: it.title, summary: it.summary, published: it.published, categories: it.categories } }); }140    ctx.log("info", `rss ${url}: ${items.length} items`);141  }142143  // link following from index pages (seeds + pagination)144  if (cfg.discovery.follow.length) {145    const follow = compile(cfg.discovery.follow);146    const indexPages = [...found.values()].filter((u) => u.group === "seed" || u.pageType === "facility_index" || u.pageType === "news_index");147    const pages: string[] = indexPages.map((u) => u.url);148    if (cfg.discovery.pagination) for (const ip of indexPages) for (let n = cfg.discovery.pagination.start; n <= cfg.discovery.pagination.max; n++) pages.push(ip.url.replace(/\/$/, "") + cfg.discovery.pagination.template.replace("{n}", String(n)));149    let stale = 0;150    for (const p of pages) {151      if (found.size >= hardCap) break;152      const doc = await ctx.fetch(p, { group: "index" });153      if (doc.error || doc.status !== 200) { if (++stale >= 3) break; continue; }154      const before = found.size;155      for (const l of extractLinks(doc.text, doc.finalUrl)) if (follow.some((re) => re.test(l))) add(classifyUrl(cfg, l, p));156      if (found.size === before) { if (++stale >= 3) break; } else stale = 0;157    }158  }159  const kept = capDiscovered(found.values(), cfg.discovery.maxUrlsPerRun);160  if (kept.length < found.size) ctx.log("warn", `discovery capped at ${kept.length}/${found.size} urls (discovery.maxUrlsPerRun) — highest-priority groups kept`);161  return kept;162}163164/** Robots-declared sitemaps are useful even when discovery.sitemap is false. */165export function pickSitemapsFromRobots(sitemaps: string[], domain: string): string[] {166  const d = domain.replace(/^www\./, "");167  return sitemaps.filter((s) => { try { return new URL(s).hostname.replace(/^www\./, "").endsWith(d); } catch { return false; } });168}169170export function docIsXml(doc: RawDocument): boolean { return /xml/i.test(doc.contentType ?? "") || /^\s*<\?xml/.test(doc.text); }171