import { XMLParser } from "fast-xml-parser"; import { load } from "cheerio"; import type { PageType } from "@dci/core"; import type { ConnectorConfig } from "./config.js"; import type { ConnectorContext, DiscoveredUrl, RawDocument } from "./types.js"; const xml = new XMLParser({ ignoreAttributes: false, attributeNamePrefix: "@_", textNodeName: "#text", trimValues: true, parseTagValue: false }); /** Parse a sitemap or sitemap index. Returns child sitemaps and URL entries. */ export function parseSitemap(text: string): { sitemaps: string[]; urls: Array<{ loc: string; lastmod: string | null }> } { const out = { sitemaps: [] as string[], urls: [] as Array<{ loc: string; lastmod: string | null }> }; let doc: Record; try { doc = xml.parse(text) as Record; } catch { return out; } const idx = (doc.sitemapindex as { sitemap?: unknown })?.sitemap; if (idx) for (const s of Array.isArray(idx) ? idx : [idx]) { const loc = (s as { loc?: string }).loc; if (loc) out.sitemaps.push(String(loc).trim()); } const set = (doc.urlset as { url?: unknown })?.url; if (set) for (const u of Array.isArray(set) ? set : [set]) { const e = u as { loc?: string; lastmod?: string }; if (e.loc) out.urls.push({ loc: String(e.loc).trim(), lastmod: e.lastmod ? String(e.lastmod) : null }); } if (!idx && !set) { // plain-text sitemap for (const line of text.split(/\r?\n/)) { const l = line.trim(); if (/^https?:\/\//.test(l)) out.urls.push({ loc: l, lastmod: null }); } } return out; } export interface FeedItem { title: string; link: string; published: string | null; summary: string | null; id: string | null; categories: string[] } /** RSS 2.0 / Atom / RDF feed parser. */ export function parseFeed(text: string): FeedItem[] { let doc: Record; try { doc = xml.parse(text) as Record; } catch { return []; } const items: FeedItem[] = []; const txt = (v: unknown): string | null => { if (v == null) return null; if (typeof v === "string") return v.trim() || null; if (typeof v === "object") { const o = v as Record; return txt(o["#text"] ?? o["@_href"] ?? null); } return String(v); }; const chan = (doc.rss as { channel?: { item?: unknown } })?.channel?.item ?? (doc["rdf:RDF"] as { item?: unknown })?.item; if (chan) for (const it of Array.isArray(chan) ? chan : [chan]) { const o = it as Record; const link = txt(o.link) ?? txt(o.guid); if (!link) continue; const cats = o.category ? (Array.isArray(o.category) ? o.category : [o.category]).map((c) => txt(c)).filter((c): c is string => Boolean(c)) : []; items.push({ title: txt(o.title) ?? link, link, published: txt(o.pubDate) ?? txt(o["dc:date"]) ?? null, summary: txt(o.description) ?? txt(o["content:encoded"]) ?? null, id: txt(o.guid) ?? null, categories: cats }); } const entries = (doc.feed as { entry?: unknown })?.entry; if (entries) for (const en of Array.isArray(entries) ? entries : [entries]) { const o = en as Record; const links = Array.isArray(o.link) ? o.link : o.link ? [o.link] : []; const alt = (links as Array>).find((l) => !l["@_rel"] || l["@_rel"] === "alternate") ?? (links as Array>)[0]; const link = alt ? String(alt["@_href"] ?? txt(alt) ?? "") : ""; if (!link) continue; const cats = o.category ? (Array.isArray(o.category) ? o.category : [o.category]).map((c) => String((c as Record)["@_term"] ?? txt(c) ?? "")).filter(Boolean) : []; items.push({ title: txt(o.title) ?? link, link, published: txt(o.published) ?? txt(o.updated) ?? null, summary: txt(o.summary) ?? txt(o.content) ?? null, id: txt(o.id) ?? null, categories: cats }); } return items; } /** Absolute, same-site links from an HTML page (anchors only). */ export function extractLinks(html: string, baseUrl: string, sameHostOnly = true): string[] { const $ = load(html); const base = new URL(baseUrl); const out = new Set(); $("a[href]").each((_, a) => { const href = $(a).attr("href"); if (!href || /^(mailto:|tel:|javascript:|#)/i.test(href)) return; try { const u = new URL(href, base); if (!/^https?:$/.test(u.protocol)) return; if (sameHostOnly && u.hostname.replace(/^www\./, "") !== base.hostname.replace(/^www\./, "")) return; u.hash = ""; out.add(u.toString()); } catch { /* ignore */ } }); return [...out]; } function compile(patterns: string[]): RegExp[] { return patterns.map((p) => new RegExp(p, "i")); } /** Apply include/exclude/classify rules from the config to a URL. Returns null when excluded. */ export function classifyUrl(cfg: ConnectorConfig, url: string, discoveredFrom: string | null = null): DiscoveredUrl | null { const d = cfg.discovery; if (d.exclude.length && compile(d.exclude).some((re) => re.test(url))) return null; if (d.include.length && !compile(d.include).some((re) => re.test(url))) return null; for (const rule of d.classify) { if (new RegExp(rule.pattern, "i").test(url)) return { url, group: rule.group, pageType: rule.pageType as PageType | undefined, priority: rule.priority ?? 50, discoveredFrom, minLevel: rule.minLevel as DiscoveredUrl["minLevel"] }; } return { url, group: "default", priority: 30, discoveredFrom }; } /** * Cap a classified URL set to `max`: every candidate is classified first, then the most valuable ones are kept — * higher priority first (seeds 80, facility groups typically 60, newsroom 50–70, `default` 30), ties by discovery * order. Truncating a sitemap by document order instead would silently drop facility pages that happen to be listed * after thousands of blog posts. */ export function capDiscovered(urls: Iterable, max: number): DiscoveredUrl[] { const all = [...urls]; if (!Number.isFinite(max) || max <= 0 || all.length <= max) return all; const rank = (u: DiscoveredUrl) => (u.priority ?? 50) + (u.group && u.group !== "default" ? 0.5 : 0); return all.map((u, i) => ({ u, i })).sort((a, b) => rank(b.u) - rank(a.u) || a.i - b.i).slice(0, max).map((x) => x.u); } /** Generic discovery: sitemaps (auto or listed), RSS feeds, seeds, and link-following from index pages. */ export async function discoverGeneric(cfg: ConnectorConfig, ctx: ConnectorContext): Promise { const found = new Map(); // classify everything, cap at the end (capDiscovered); the hard ceiling only bounds memory on pathological sitemaps const hardCap = Math.max(cfg.discovery.maxUrlsPerRun * 10, 50_000); const add = (u: DiscoveredUrl | null) => { if (u && !found.has(u.url) && found.size < hardCap) found.set(u.url, u); }; const origin = `https://${cfg.domain.replace(/^https?:\/\//, "").replace(/\/$/, "")}`; // seeds for (const s of cfg.discovery.seeds) { const seed = typeof s === "string" ? { url: s, group: "seed" as string, pageType: undefined as PageType | undefined, minLevel: undefined as number | undefined } : s; const url = seed.url.startsWith("http") ? seed.url : origin + seed.url; const c = classifyUrl(cfg, url, null); add({ url, group: seed.group ?? c?.group ?? "seed", pageType: seed.pageType ?? c?.pageType, priority: 80, minLevel: (seed.minLevel ?? c?.minLevel) as DiscoveredUrl["minLevel"], discoveredFrom: null }); } // sitemaps const sitemapUrls: string[] = []; if (cfg.discovery.sitemap === true) sitemapUrls.push(`${origin}/sitemap.xml`, `${origin}/sitemap_index.xml`, `${origin}/sitemap-index.xml`); else if (Array.isArray(cfg.discovery.sitemap)) for (const s of cfg.discovery.sitemap) sitemapUrls.push(s.startsWith("http") ? s : origin + s); const seenMaps = new Set(); let mapBudget = 40; while (sitemapUrls.length && mapBudget-- > 0) { const sm = sitemapUrls.shift()!; if (seenMaps.has(sm)) continue; seenMaps.add(sm); const doc = await ctx.fetch(sm, { group: "sitemap", accept: "application/xml,text/xml,*/*", maxBytes: 60 * 1024 * 1024 }); if (doc.error || doc.status !== 200) { ctx.log("debug", `sitemap ${sm} → ${doc.error?.code ?? doc.status}`); continue; } const parsed = parseSitemap(doc.text); for (const child of parsed.sitemaps) if (!seenMaps.has(child)) sitemapUrls.push(child); for (const u of parsed.urls) { const c = classifyUrl(cfg, u.loc, sm); if (c) add({ ...c, lastmod: u.lastmod }); } ctx.log("info", `sitemap ${sm}: ${parsed.urls.length} urls, ${parsed.sitemaps.length} children`); } // rss for (const f of cfg.discovery.rss) { const url = f.startsWith("http") ? f : origin + f; const doc = await ctx.fetch(url, { group: "rss", accept: "application/rss+xml,application/atom+xml,application/xml,text/xml,*/*" }); if (doc.error || doc.status !== 200) { ctx.log("warn", `rss ${url} → ${doc.error?.code ?? doc.status}`); continue; } const items = parseFeed(doc.text); for (const it of items) { const c = classifyUrl(cfg, it.link, url); if (c) add({ ...c, group: c.group === "default" ? "newsroom" : c.group, pageType: c.pageType ?? "press_release", priority: 70, lastmod: it.published, meta: { title: it.title, summary: it.summary, published: it.published, categories: it.categories } }); } ctx.log("info", `rss ${url}: ${items.length} items`); } // link following from index pages (seeds + pagination) if (cfg.discovery.follow.length) { const follow = compile(cfg.discovery.follow); const indexPages = [...found.values()].filter((u) => u.group === "seed" || u.pageType === "facility_index" || u.pageType === "news_index"); const pages: string[] = indexPages.map((u) => u.url); if (cfg.discovery.pagination) for (const ip of indexPages) for (let n = cfg.discovery.pagination.start; n <= cfg.discovery.pagination.max; n++) pages.push(ip.url.replace(/\/$/, "") + cfg.discovery.pagination.template.replace("{n}", String(n))); let stale = 0; for (const p of pages) { if (found.size >= hardCap) break; const doc = await ctx.fetch(p, { group: "index" }); if (doc.error || doc.status !== 200) { if (++stale >= 3) break; continue; } const before = found.size; for (const l of extractLinks(doc.text, doc.finalUrl)) if (follow.some((re) => re.test(l))) add(classifyUrl(cfg, l, p)); if (found.size === before) { if (++stale >= 3) break; } else stale = 0; } } const kept = capDiscovered(found.values(), cfg.discovery.maxUrlsPerRun); if (kept.length < found.size) ctx.log("warn", `discovery capped at ${kept.length}/${found.size} urls (discovery.maxUrlsPerRun) — highest-priority groups kept`); return kept; } /** Robots-declared sitemaps are useful even when discovery.sitemap is false. */ export function pickSitemapsFromRobots(sitemaps: string[], domain: string): string[] { const d = domain.replace(/^www\./, ""); return sitemaps.filter((s) => { try { return new URL(s).hostname.replace(/^www\./, "").endsWith(d); } catch { return false; } }); } export function docIsXml(doc: RawDocument): boolean { return /xml/i.test(doc.contentType ?? "") || /^\s*<\?xml/.test(doc.text); }