SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
3 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
15.4 KB · 244 lines typescript
Raw Blame History
1import { load, type CheerioAPI } from "cheerio";2import { getDocumentProxy } from "unpdf";3import { parseMw, parseAreaSqm, parseAreaHa, parseMoney, parsePartialDate, normalizeStatus, inferFacilityType, countryFromText, cleanText, htmlToText } from "@dci/core";4import type { SelectorRule } from "./config.js";5import type { RawDocument } from "./types.js";67/** All JSON-LD objects on a page (flattened @graph), optionally filtered by @type. */8export function jsonLd(html: string, type?: string): Record<string, unknown>[] {9  const $ = load(html);10  const out: Record<string, unknown>[] = [];11  $('script[type="application/ld+json"]').each((_, el) => {12    const raw = $(el).contents().text();13    try {14      const parsed = JSON.parse(raw.replace(/^\s*<!--|-->\s*$/g, ""));15      const arr = Array.isArray(parsed) ? parsed : [parsed];16      for (const p of arr) {17        if (p && typeof p === "object" && Array.isArray((p as { "@graph"?: unknown[] })["@graph"])) out.push(...((p as { "@graph": Record<string, unknown>[] })["@graph"]));18        else if (p && typeof p === "object") out.push(p as Record<string, unknown>);19      }20    } catch { /* malformed */ }21  });22  if (!type) return out;23  const t = type.toLowerCase();24  return out.filter((o) => { const ty = o["@type"]; const list = Array.isArray(ty) ? ty : [ty]; return list.some((x) => String(x).toLowerCase() === t); });25}2627/** Embedded JSON blobs: __NEXT_DATA__, window.__INITIAL_STATE__ = {...}, application/json scripts, Nuxt/Drupal settings. */28export function embeddedJson(html: string): Array<{ id: string; data: unknown }> {29  const $ = load(html);30  const out: Array<{ id: string; data: unknown }> = [];31  $('script[type="application/json"], script#__NEXT_DATA__, script[data-drupal-selector="drupal-settings-json"]').each((_, el) => {32    const id = $(el).attr("id") ?? $(el).attr("data-drupal-selector") ?? "json";33    try { out.push({ id, data: JSON.parse($(el).contents().text()) }); } catch { /* skip */ }34  });35  const re = /(?:window\.|var\s+|const\s+|let\s+)?(__INITIAL_STATE__|__NUXT__|__PRELOADED_STATE__|__APOLLO_STATE__|__DATA__|__INITIAL_DATA__|initialData|pageData|__NEXT_DATA__)\s*=\s*(\{[\s\S]*?\})\s*;?\s*(?:<\/script>|\n)/g;36  for (const m of html.matchAll(re)) { try { out.push({ id: m[1]!, data: JSON.parse(m[2]!) }); } catch { /* not plain JSON */ } }37  return out;38}3940/** Tiny JSONPath subset: "$.a.b[0].c", "a.b", "[*]" not supported (use walk). */41export function jsonPath(obj: unknown, path: string): unknown {42  const parts = path.replace(/^\$\.?/, "").split(/\.|\[(\d+)\]/).filter((p) => p !== undefined && p !== "");43  let cur: unknown = obj;44  for (const p of parts) {45    if (cur == null) return undefined;46    cur = (cur as Record<string, unknown>)[p];47  }48  return cur;49}5051/** Depth-first walk collecting objects satisfying a predicate (used to find facility-like nodes in embedded JSON). */52export function walkJson(obj: unknown, pred: (o: Record<string, unknown>) => boolean, out: Record<string, unknown>[] = [], depth = 0): Record<string, unknown>[] {53  if (depth > 12 || obj == null) return out;54  if (Array.isArray(obj)) { for (const x of obj) walkJson(x, pred, out, depth + 1); return out; }55  if (typeof obj === "object") {56    const o = obj as Record<string, unknown>;57    if (pred(o)) out.push(o);58    for (const v of Object.values(o)) walkJson(v, pred, out, depth + 1);59  }60  return out;61}6263export function applyTransform(v: unknown, t: SelectorRule extends string ? never : Exclude<SelectorRule, string>["transform"]): unknown {64  if (v == null) return null;65  const s = typeof v === "string" ? v : String(v);66  switch (t) {67    case "mw": return parseMw(s);68    case "sqm": return parseAreaSqm(s);69    case "ha": return parseAreaHa(s);70    case "money": return parseMoney(s);71    case "date": return parsePartialDate(s);72    case "status": return normalizeStatus(s);73    case "type": return inferFacilityType(s);74    case "country": return /^[A-Z]{2}$/.test(s.trim()) ? s.trim() : countryFromText(s);75    case "int": { const n = parseInt(s.replace(/[^\d-]/g, ""), 10); return Number.isFinite(n) ? n : null; }76    case "float": { const n = parseFloat(s.replace(/[^\d.-]/g, "")); return Number.isFinite(n) ? n : null; }77    case "trim": return cleanText(s);78    case "lower": return s.trim().toLowerCase();79    case "url": return s.trim();80    case "bool": return /^(true|yes|1|y)$/i.test(s.trim());81    default: return typeof v === "string" ? cleanText(v) : v;82  }83}8485/** Evaluate one declarative selector rule against a page. */86export function evalRule($: CheerioAPI, html: string, rule: SelectorRule, scope?: ReturnType<CheerioAPI>): { value: unknown; method: string } {87  const r: Exclude<SelectorRule, string> = typeof rule === "string" ? { selector: rule } : rule;88  let value: unknown = null;89  let method = "";90  if (r.jsonld) {91    const objs = jsonLd(html, r.jsonld);92    value = objs.length && r.jsonPath ? jsonPath(objs[0], r.jsonPath) : objs[0] ?? null;93    method = `json-ld:${r.jsonld}${r.jsonPath ? "." + r.jsonPath : ""}`;94  } else if (r.meta) {95    value = $(`meta[property="${r.meta}"], meta[name="${r.meta}"]`).first().attr("content") ?? null;96    method = `meta:${r.meta}`;97  } else if (r.jsonPath && !r.selector) {98    for (const b of embeddedJson(html)) { const v = jsonPath(b.data, r.jsonPath); if (v != null) { value = v; method = `embedded-json:${b.id}.${r.jsonPath}`; break; } }99  } else if (r.selector) {100    const sel = scope ? scope.find(r.selector) : $(r.selector);101    if (r.all) value = sel.map((_, el) => (r.attr ? $(el).attr(r.attr) : $(el).text())).get().map((x) => cleanText(x)).filter(Boolean);102    else value = r.attr ? sel.first().attr(r.attr) ?? null : cleanText(sel.first().text());103    method = `selector:${r.selector}${r.attr ? "@" + r.attr : ""}`;104  } else if (r.regex) {105    value = htmlToText(html);106    method = "regex";107  }108  if (r.regex && value != null) {109    const re = new RegExp(r.regex, r.regexFlags ?? "i");110    const apply = (s: string) => { const m = s.match(re); return m ? (m[r.group ?? 1] ?? m[0]) : null; };111    value = Array.isArray(value) ? value.map((x) => apply(String(x))).filter(Boolean) : apply(String(value));112    method += `+regex:${r.regex.slice(0, 40)}`;113  }114  if (r.transform && value != null) { value = Array.isArray(value) ? value.map((x) => applyTransform(x, r.transform)) : applyTransform(value, r.transform); method += `>${r.transform}`; }115  if ((value == null || value === "") && r.default !== undefined) value = r.default;116  return { value, method };117}118119export function pageTitle(html: string): string | null {120  const $ = load(html);121  return cleanText($('meta[property="og:title"]').attr("content") ?? $("title").first().text() ?? $("h1").first().text());122}123124export function metaDescription(html: string): string | null {125  const $ = load(html);126  return cleanText($('meta[name="description"]').attr("content") ?? $('meta[property="og:description"]').attr("content") ?? null);127}128129/** Blocks that are never the story: chrome, related / recommended / trending teasers, sidebars, comments. */130export const NON_CONTENT_SELECTOR = "script,style,noscript,nav,header,footer,aside,form,iframe,svg,[role=navigation],[role=complementary],[class*=related],[class*=recommend],[class*=trending],[class*=popular],[class*=sidebar],[class*=read-more],[class*=readmore],[class*=more-stories],[class*=also-like],[class*=comments],[id*=related],[id*=comments]";131/** Heading that opens a trailing teaser section inside the story element itself ("More in Construction & Site Selection", "Related articles", "Tags"). */132export const TEASER_HEADING_RE = /^\s*(?:More (?:in|from|on|like this)\b.*|Related(?: (?:articles?|stories|news|content|posts?|coverage|reading|links?))?|Recommended(?: for you| reading| stories| articles)?|You (?:may|might) also like|Most (?:read|popular|viewed)|Trending(?: now| stories)?|Popular (?:now|posts|stories|articles)|Latest (?:news|stories|posts|articles)|Read (?:more|next|also)|See also|Further reading|Editor'?s picks|Sponsored(?: content)?|Tags|Comments|Leave a (?:comment|reply))\s*:?\s*$/i;133/** Minimum story length before a teaser heading is allowed to cut the text (a short lead titled "Related" is not a teaser section). */134const TEASER_CUT_MIN_CHARS = 300;135136/** Drop everything from the first teaser heading onwards (teasers are the last thing in a story element). */137export function cutTeasers(text: string): string {138  const lines = text.split("\n");139  let consumed = 0;140  for (let i = 0; i < lines.length; i++) {141    const line = lines[i]!;142    if (consumed >= TEASER_CUT_MIN_CHARS && line.trim().length <= 60 && TEASER_HEADING_RE.test(line)) return lines.slice(0, i).join("\n").trimEnd();143    consumed += line.length;144  }145  return text;146}147148/**149 * Main text of an article-like page: the largest <article> / <main> / content block, after removing navigation, footer,150 * aside, related / recommended / trending / popular / sidebar blocks, and cut at the first trailing teaser heading — so a151 * "More in …" list of other headlines never leaks into the story (and never names its operator).152 */153export function mainText(html: string): string {154  const $ = load(html);155  $(NON_CONTENT_SELECTOR).remove();156  const candidates = ["article", "main", '[role="main"]', ".article-body", ".press-release", ".entry-content", ".post-content", ".content", "#content", "body"];157  for (const c of candidates) {158    // several matches (the story plus teaser <article class="card"> cards): the largest one is the story159    const texts = $(c).toArray().map((el) => htmlToText($(el).html() ?? ""));160    if (!texts.length) continue;161    const t = cutTeasers(texts.sort((a, b) => b.length - a.length)[0]!);162    if (t.length > 300) return t;163  }164  return cutTeasers(htmlToText($("body").html() ?? html));165}166167/** Publication date from meta tags / JSON-LD / <time>. */168export function publishedDate(html: string): string | null {169  const $ = load(html);170  const cands = [$('meta[property="article:published_time"]').attr("content"), $('meta[name="pubdate"]').attr("content"), $('meta[name="publish-date"]').attr("content"), $('meta[name="date"]').attr("content"), $('meta[itemprop="datePublished"]').attr("content"), $("time[datetime]").first().attr("datetime"), (jsonLd(html, "NewsArticle")[0]?.datePublished as string | undefined), (jsonLd(html, "Article")[0]?.datePublished as string | undefined)];171  for (const c of cands) if (c) { const d = parsePartialDate(String(c)); if (d) return d; }172  return null;173}174175/** PDFs above this size are not parsed at all (planning documents are a few MB; bigger files are scans or bombs). */176export const PDF_MAX_BYTES = 15 * 1024 * 1024;177/** Hard ceiling on PDF text extraction; a pathological file must never stall a run. */178export const PDF_TIMEOUT_MS = 20_000;179180/**181 * Text of the first `maxPages` pages of a PDF (page by page, so a 3 000-page file costs 60 pages), bounded by182 * `maxBytes` on input and `timeoutMs` on wall time. Throws on oversize / timeout — callers log and skip the document.183 */184export async function pdfText(buf: Buffer, maxPages = 60, opts: { timeoutMs?: number; maxBytes?: number } = {}): Promise<{ text: string; pages: number; title: string | null; truncated: boolean }> {185  const maxBytes = opts.maxBytes ?? PDF_MAX_BYTES;186  const timeoutMs = opts.timeoutMs ?? PDF_TIMEOUT_MS;187  if (buf.length > maxBytes) throw new Error(`pdf too large: ${buf.length} bytes > ${maxBytes}`);188  let pdf: Awaited<ReturnType<typeof getDocumentProxy>> | null = null;189  const work = async () => {190    pdf = await getDocumentProxy(new Uint8Array(buf));191    const pages = pdf.numPages;192    const n = Math.min(pages, Math.max(1, maxPages));193    const parts: string[] = [];194    for (let i = 1; i <= n; i++) {195      const page = await pdf.getPage(i);196      const content = await page.getTextContent();197      parts.push(content.items.map((it) => ("str" in it ? it.str + (it.hasEOL ? "\n" : " ") : "")).join(""));198      page.cleanup();199    }200    const meta = await pdf.getMetadata().catch(() => null);201    const title = (meta?.info as { Title?: string } | undefined)?.Title ?? null;202    return { text: parts.join("\n").replace(/[ \t]+/g, " ").replace(/\n\s*\n+/g, "\n").trim(), pages, title: cleanText(title), truncated: pages > n };203  };204  let timer: ReturnType<typeof setTimeout> | undefined;205  const timeout = new Promise<never>((_, reject) => { timer = setTimeout(() => reject(new Error(`pdf extraction timed out after ${timeoutMs} ms`)), timeoutMs); });206  try {207    return await Promise.race([work(), timeout]);208  } finally {209    clearTimeout(timer);210    // release pdf.js worker memory (also aborts the in-flight parse after a timeout)211    try { void (pdf as unknown as { destroy?: () => Promise<void> } | null)?.destroy?.()?.catch(() => undefined); } catch { /* ignore */ }212  }213}214215export function isPdf(doc: RawDocument): boolean { return /pdf/i.test(doc.contentType ?? "") || doc.body.subarray(0, 5).toString("latin1") === "%PDF-"; }216export function isJson(doc: RawDocument): boolean { return /json/i.test(doc.contentType ?? "") || /^\s*[\[{]/.test(doc.text.slice(0, 20)); }217218/** Loose geo extraction from any HTML: JSON-LD geo, meta geo.position / ICBM, data-lat attributes, Google Maps links. */219export function extractGeo(html: string): { lat: number; lng: number; method: string } | null {220  for (const o of jsonLd(html)) {221    const g = (o.geo ?? (o.location as Record<string, unknown> | undefined)?.geo) as { latitude?: unknown; longitude?: unknown } | undefined;222    if (g && g.latitude != null && g.longitude != null) { const lat = Number(g.latitude), lng = Number(g.longitude); if (Number.isFinite(lat) && Number.isFinite(lng)) return { lat, lng, method: "json-ld:geo" }; }223  }224  const $ = load(html);225  const pos = $('meta[name="geo.position"]').attr("content") ?? $('meta[name="ICBM"]').attr("content");226  if (pos) { const m = pos.match(/(-?\d+(?:\.\d+)?)[;,\s]+(-?\d+(?:\.\d+)?)/); if (m) return { lat: Number(m[1]), lng: Number(m[2]), method: "meta:geo.position" }; }227  const el = $("[data-lat][data-lng], [data-latitude][data-longitude]").first();228  if (el.length) { const lat = Number(el.attr("data-lat") ?? el.attr("data-latitude")), lng = Number(el.attr("data-lng") ?? el.attr("data-longitude")); if (Number.isFinite(lat) && Number.isFinite(lng)) return { lat, lng, method: "attr:data-lat" }; }229  const gm = html.match(/maps\.google\.[a-z.]+\/(?:maps)?\?q=(-?\d+\.\d+),(-?\d+\.\d+)|google\.com\/maps[^"']*?@(-?\d+\.\d+),(-?\d+\.\d+)|!3d(-?\d+\.\d+)!4d(-?\d+\.\d+)/);230  if (gm) { const lat = Number(gm[1] ?? gm[3] ?? gm[5]), lng = Number(gm[2] ?? gm[4] ?? gm[6]); if (Number.isFinite(lat) && Number.isFinite(lng)) return { lat, lng, method: "link:google-maps" }; }231  return null;232}233234/** Postal address from JSON-LD (PostalAddress) on any page. */235export function extractAddress(html: string): { street: string | null; city: string | null; region: string | null; postal: string | null; country: string | null } | null {236  for (const o of jsonLd(html)) {237    const a = (o.address ?? (o.location as Record<string, unknown> | undefined)?.address) as Record<string, unknown> | string | undefined;238    if (!a) continue;239    if (typeof a === "string") return { street: cleanText(a), city: null, region: null, postal: null, country: countryFromText(a) };240    return { street: cleanText(String(a.streetAddress ?? "")), city: cleanText(String(a.addressLocality ?? "")), region: cleanText(String(a.addressRegion ?? "")), postal: cleanText(String(a.postalCode ?? "")), country: a.addressCountry ? (typeof a.addressCountry === "string" ? (/^[A-Z]{2}$/.test(a.addressCountry) ? a.addressCountry : countryFromText(a.addressCountry)) : countryFromText(String((a.addressCountry as Record<string, unknown>).name ?? ""))) : null };241  }242  return null;243}244