spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import { load, type CheerioAPI } from "cheerio";2import { getDocumentProxy } from "unpdf";3import { parseMw, parseAreaSqm, parseAreaHa, parseMoney, parsePartialDate, normalizeStatus, inferFacilityType, countryFromText, cleanText, htmlToText } from "@dci/core";4import type { SelectorRule } from "./config.js";5import type { RawDocument } from "./types.js";67/** All JSON-LD objects on a page (flattened @graph), optionally filtered by @type. */8export function jsonLd(html: string, type?: string): Record<string, unknown>[] {9 const $ = load(html);10 const out: Record<string, unknown>[] = [];11 $('script[type="application/ld+json"]').each((_, el) => {12 const raw = $(el).contents().text();13 try {14 const parsed = JSON.parse(raw.replace(/^\s*<!--|-->\s*$/g, ""));15 const arr = Array.isArray(parsed) ? parsed : [parsed];16 for (const p of arr) {17 if (p && typeof p === "object" && Array.isArray((p as { "@graph"?: unknown[] })["@graph"])) out.push(...((p as { "@graph": Record<string, unknown>[] })["@graph"]));18 else if (p && typeof p === "object") out.push(p as Record<string, unknown>);19 }20 } catch { /* malformed */ }21 });22 if (!type) return out;23 const t = type.toLowerCase();24 return out.filter((o) => { const ty = o["@type"]; const list = Array.isArray(ty) ? ty : [ty]; return list.some((x) => String(x).toLowerCase() === t); });25}2627/** Embedded JSON blobs: __NEXT_DATA__, window.__INITIAL_STATE__ = {...}, application/json scripts, Nuxt/Drupal settings. */28export function embeddedJson(html: string): Array<{ id: string; data: unknown }> {29 const $ = load(html);30 const out: Array<{ id: string; data: unknown }> = [];31 $('script[type="application/json"], script#__NEXT_DATA__, script[data-drupal-selector="drupal-settings-json"]').each((_, el) => {32 const id = $(el).attr("id") ?? $(el).attr("data-drupal-selector") ?? "json";33 try { out.push({ id, data: JSON.parse($(el).contents().text()) }); } catch { /* skip */ }34 });35 const re = /(?:window\.|var\s+|const\s+|let\s+)?(__INITIAL_STATE__|__NUXT__|__PRELOADED_STATE__|__APOLLO_STATE__|__DATA__|__INITIAL_DATA__|initialData|pageData|__NEXT_DATA__)\s*=\s*(\{[\s\S]*?\})\s*;?\s*(?:<\/script>|\n)/g;36 for (const m of html.matchAll(re)) { try { out.push({ id: m[1]!, data: JSON.parse(m[2]!) }); } catch { /* not plain JSON */ } }37 return out;38}3940/** Tiny JSONPath subset: "$.a.b[0].c", "a.b", "[*]" not supported (use walk). */41export function jsonPath(obj: unknown, path: string): unknown {42 const parts = path.replace(/^\$\.?/, "").split(/\.|\[(\d+)\]/).filter((p) => p !== undefined && p !== "");43 let cur: unknown = obj;44 for (const p of parts) {45 if (cur == null) return undefined;46 cur = (cur as Record<string, unknown>)[p];47 }48 return cur;49}5051/** Depth-first walk collecting objects satisfying a predicate (used to find facility-like nodes in embedded JSON). */52export function walkJson(obj: unknown, pred: (o: Record<string, unknown>) => boolean, out: Record<string, unknown>[] = [], depth = 0): Record<string, unknown>[] {53 if (depth > 12 || obj == null) return out;54 if (Array.isArray(obj)) { for (const x of obj) walkJson(x, pred, out, depth + 1); return out; }55 if (typeof obj === "object") {56 const o = obj as Record<string, unknown>;57 if (pred(o)) out.push(o);58 for (const v of Object.values(o)) walkJson(v, pred, out, depth + 1);59 }60 return out;61}6263export function applyTransform(v: unknown, t: SelectorRule extends string ? never : Exclude<SelectorRule, string>["transform"]): unknown {64 if (v == null) return null;65 const s = typeof v === "string" ? v : String(v);66 switch (t) {67 case "mw": return parseMw(s);68 case "sqm": return parseAreaSqm(s);69 case "ha": return parseAreaHa(s);70 case "money": return parseMoney(s);71 case "date": return parsePartialDate(s);72 case "status": return normalizeStatus(s);73 case "type": return inferFacilityType(s);74 case "country": return /^[A-Z]{2}$/.test(s.trim()) ? s.trim() : countryFromText(s);75 case "int": { const n = parseInt(s.replace(/[^\d-]/g, ""), 10); return Number.isFinite(n) ? n : null; }76 case "float": { const n = parseFloat(s.replace(/[^\d.-]/g, "")); return Number.isFinite(n) ? n : null; }77 case "trim": return cleanText(s);78 case "lower": return s.trim().toLowerCase();79 case "url": return s.trim();80 case "bool": return /^(true|yes|1|y)$/i.test(s.trim());81 default: return typeof v === "string" ? cleanText(v) : v;82 }83}8485/** Evaluate one declarative selector rule against a page. */86export function evalRule($: CheerioAPI, html: string, rule: SelectorRule, scope?: ReturnType<CheerioAPI>): { value: unknown; method: string } {87 const r: Exclude<SelectorRule, string> = typeof rule === "string" ? { selector: rule } : rule;88 let value: unknown = null;89 let method = "";90 if (r.jsonld) {91 const objs = jsonLd(html, r.jsonld);92 value = objs.length && r.jsonPath ? jsonPath(objs[0], r.jsonPath) : objs[0] ?? null;93 method = `json-ld:${r.jsonld}${r.jsonPath ? "." + r.jsonPath : ""}`;94 } else if (r.meta) {95 value = $(`meta[property="${r.meta}"], meta[name="${r.meta}"]`).first().attr("content") ?? null;96 method = `meta:${r.meta}`;97 } else if (r.jsonPath && !r.selector) {98 for (const b of embeddedJson(html)) { const v = jsonPath(b.data, r.jsonPath); if (v != null) { value = v; method = `embedded-json:${b.id}.${r.jsonPath}`; break; } }99 } else if (r.selector) {100 const sel = scope ? scope.find(r.selector) : $(r.selector);101 if (r.all) value = sel.map((_, el) => (r.attr ? $(el).attr(r.attr) : $(el).text())).get().map((x) => cleanText(x)).filter(Boolean);102 else value = r.attr ? sel.first().attr(r.attr) ?? null : cleanText(sel.first().text());103 method = `selector:${r.selector}${r.attr ? "@" + r.attr : ""}`;104 } else if (r.regex) {105 value = htmlToText(html);106 method = "regex";107 }108 if (r.regex && value != null) {109 const re = new RegExp(r.regex, r.regexFlags ?? "i");110 const apply = (s: string) => { const m = s.match(re); return m ? (m[r.group ?? 1] ?? m[0]) : null; };111 value = Array.isArray(value) ? value.map((x) => apply(String(x))).filter(Boolean) : apply(String(value));112 method += `+regex:${r.regex.slice(0, 40)}`;113 }114 if (r.transform && value != null) { value = Array.isArray(value) ? value.map((x) => applyTransform(x, r.transform)) : applyTransform(value, r.transform); method += `>${r.transform}`; }115 if ((value == null || value === "") && r.default !== undefined) value = r.default;116 return { value, method };117}118119export function pageTitle(html: string): string | null {120 const $ = load(html);121 return cleanText($('meta[property="og:title"]').attr("content") ?? $("title").first().text() ?? $("h1").first().text());122}123124export function metaDescription(html: string): string | null {125 const $ = load(html);126 return cleanText($('meta[name="description"]').attr("content") ?? $('meta[property="og:description"]').attr("content") ?? null);127}128129/** Blocks that are never the story: chrome, related / recommended / trending teasers, sidebars, comments. */130export const NON_CONTENT_SELECTOR = "script,style,noscript,nav,header,footer,aside,form,iframe,svg,[role=navigation],[role=complementary],[class*=related],[class*=recommend],[class*=trending],[class*=popular],[class*=sidebar],[class*=read-more],[class*=readmore],[class*=more-stories],[class*=also-like],[class*=comments],[id*=related],[id*=comments]";131/** Heading that opens a trailing teaser section inside the story element itself ("More in Construction & Site Selection", "Related articles", "Tags"). */132export const TEASER_HEADING_RE = /^\s*(?:More (?:in|from|on|like this)\b.*|Related(?: (?:articles?|stories|news|content|posts?|coverage|reading|links?))?|Recommended(?: for you| reading| stories| articles)?|You (?:may|might) also like|Most (?:read|popular|viewed)|Trending(?: now| stories)?|Popular (?:now|posts|stories|articles)|Latest (?:news|stories|posts|articles)|Read (?:more|next|also)|See also|Further reading|Editor'?s picks|Sponsored(?: content)?|Tags|Comments|Leave a (?:comment|reply))\s*:?\s*$/i;133/** Minimum story length before a teaser heading is allowed to cut the text (a short lead titled "Related" is not a teaser section). */134const TEASER_CUT_MIN_CHARS = 300;135136/** Drop everything from the first teaser heading onwards (teasers are the last thing in a story element). */137export function cutTeasers(text: string): string {138 const lines = text.split("\n");139 let consumed = 0;140 for (let i = 0; i < lines.length; i++) {141 const line = lines[i]!;142 if (consumed >= TEASER_CUT_MIN_CHARS && line.trim().length <= 60 && TEASER_HEADING_RE.test(line)) return lines.slice(0, i).join("\n").trimEnd();143 consumed += line.length;144 }145 return text;146}147148/**149 * Main text of an article-like page: the largest <article> / <main> / content block, after removing navigation, footer,150 * aside, related / recommended / trending / popular / sidebar blocks, and cut at the first trailing teaser heading — so a151 * "More in …" list of other headlines never leaks into the story (and never names its operator).152 */153export function mainText(html: string): string {154 const $ = load(html);155 $(NON_CONTENT_SELECTOR).remove();156 const candidates = ["article", "main", '[role="main"]', ".article-body", ".press-release", ".entry-content", ".post-content", ".content", "#content", "body"];157 for (const c of candidates) {158 // several matches (the story plus teaser <article class="card"> cards): the largest one is the story159 const texts = $(c).toArray().map((el) => htmlToText($(el).html() ?? ""));160 if (!texts.length) continue;161 const t = cutTeasers(texts.sort((a, b) => b.length - a.length)[0]!);162 if (t.length > 300) return t;163 }164 return cutTeasers(htmlToText($("body").html() ?? html));165}166167/** Publication date from meta tags / JSON-LD / <time>. */168export function publishedDate(html: string): string | null {169 const $ = load(html);170 const cands = [$('meta[property="article:published_time"]').attr("content"), $('meta[name="pubdate"]').attr("content"), $('meta[name="publish-date"]').attr("content"), $('meta[name="date"]').attr("content"), $('meta[itemprop="datePublished"]').attr("content"), $("time[datetime]").first().attr("datetime"), (jsonLd(html, "NewsArticle")[0]?.datePublished as string | undefined), (jsonLd(html, "Article")[0]?.datePublished as string | undefined)];171 for (const c of cands) if (c) { const d = parsePartialDate(String(c)); if (d) return d; }172 return null;173}174175/** PDFs above this size are not parsed at all (planning documents are a few MB; bigger files are scans or bombs). */176export const PDF_MAX_BYTES = 15 * 1024 * 1024;177/** Hard ceiling on PDF text extraction; a pathological file must never stall a run. */178export const PDF_TIMEOUT_MS = 20_000;179180/**181 * Text of the first `maxPages` pages of a PDF (page by page, so a 3 000-page file costs 60 pages), bounded by182 * `maxBytes` on input and `timeoutMs` on wall time. Throws on oversize / timeout — callers log and skip the document.183 */184export async function pdfText(buf: Buffer, maxPages = 60, opts: { timeoutMs?: number; maxBytes?: number } = {}): Promise<{ text: string; pages: number; title: string | null; truncated: boolean }> {185 const maxBytes = opts.maxBytes ?? PDF_MAX_BYTES;186 const timeoutMs = opts.timeoutMs ?? PDF_TIMEOUT_MS;187 if (buf.length > maxBytes) throw new Error(`pdf too large: ${buf.length} bytes > ${maxBytes}`);188 let pdf: Awaited<ReturnType<typeof getDocumentProxy>> | null = null;189 const work = async () => {190 pdf = await getDocumentProxy(new Uint8Array(buf));191 const pages = pdf.numPages;192 const n = Math.min(pages, Math.max(1, maxPages));193 const parts: string[] = [];194 for (let i = 1; i <= n; i++) {195 const page = await pdf.getPage(i);196 const content = await page.getTextContent();197 parts.push(content.items.map((it) => ("str" in it ? it.str + (it.hasEOL ? "\n" : " ") : "")).join(""));198 page.cleanup();199 }200 const meta = await pdf.getMetadata().catch(() => null);201 const title = (meta?.info as { Title?: string } | undefined)?.Title ?? null;202 return { text: parts.join("\n").replace(/[ \t]+/g, " ").replace(/\n\s*\n+/g, "\n").trim(), pages, title: cleanText(title), truncated: pages > n };203 };204 let timer: ReturnType<typeof setTimeout> | undefined;205 const timeout = new Promise<never>((_, reject) => { timer = setTimeout(() => reject(new Error(`pdf extraction timed out after ${timeoutMs} ms`)), timeoutMs); });206 try {207 return await Promise.race([work(), timeout]);208 } finally {209 clearTimeout(timer);210 // release pdf.js worker memory (also aborts the in-flight parse after a timeout)211 try { void (pdf as unknown as { destroy?: () => Promise<void> } | null)?.destroy?.()?.catch(() => undefined); } catch { /* ignore */ }212 }213}214215export function isPdf(doc: RawDocument): boolean { return /pdf/i.test(doc.contentType ?? "") || doc.body.subarray(0, 5).toString("latin1") === "%PDF-"; }216export function isJson(doc: RawDocument): boolean { return /json/i.test(doc.contentType ?? "") || /^\s*[\[{]/.test(doc.text.slice(0, 20)); }217218/** Loose geo extraction from any HTML: JSON-LD geo, meta geo.position / ICBM, data-lat attributes, Google Maps links. */219export function extractGeo(html: string): { lat: number; lng: number; method: string } | null {220 for (const o of jsonLd(html)) {221 const g = (o.geo ?? (o.location as Record<string, unknown> | undefined)?.geo) as { latitude?: unknown; longitude?: unknown } | undefined;222 if (g && g.latitude != null && g.longitude != null) { const lat = Number(g.latitude), lng = Number(g.longitude); if (Number.isFinite(lat) && Number.isFinite(lng)) return { lat, lng, method: "json-ld:geo" }; }223 }224 const $ = load(html);225 const pos = $('meta[name="geo.position"]').attr("content") ?? $('meta[name="ICBM"]').attr("content");226 if (pos) { const m = pos.match(/(-?\d+(?:\.\d+)?)[;,\s]+(-?\d+(?:\.\d+)?)/); if (m) return { lat: Number(m[1]), lng: Number(m[2]), method: "meta:geo.position" }; }227 const el = $("[data-lat][data-lng], [data-latitude][data-longitude]").first();228 if (el.length) { const lat = Number(el.attr("data-lat") ?? el.attr("data-latitude")), lng = Number(el.attr("data-lng") ?? el.attr("data-longitude")); if (Number.isFinite(lat) && Number.isFinite(lng)) return { lat, lng, method: "attr:data-lat" }; }229 const gm = html.match(/maps\.google\.[a-z.]+\/(?:maps)?\?q=(-?\d+\.\d+),(-?\d+\.\d+)|google\.com\/maps[^"']*?@(-?\d+\.\d+),(-?\d+\.\d+)|!3d(-?\d+\.\d+)!4d(-?\d+\.\d+)/);230 if (gm) { const lat = Number(gm[1] ?? gm[3] ?? gm[5]), lng = Number(gm[2] ?? gm[4] ?? gm[6]); if (Number.isFinite(lat) && Number.isFinite(lng)) return { lat, lng, method: "link:google-maps" }; }231 return null;232}233234/** Postal address from JSON-LD (PostalAddress) on any page. */235export function extractAddress(html: string): { street: string | null; city: string | null; region: string | null; postal: string | null; country: string | null } | null {236 for (const o of jsonLd(html)) {237 const a = (o.address ?? (o.location as Record<string, unknown> | undefined)?.address) as Record<string, unknown> | string | undefined;238 if (!a) continue;239 if (typeof a === "string") return { street: cleanText(a), city: null, region: null, postal: null, country: countryFromText(a) };240 return { street: cleanText(String(a.streetAddress ?? "")), city: cleanText(String(a.addressLocality ?? "")), region: cleanText(String(a.addressRegion ?? "")), postal: cleanText(String(a.postalCode ?? "")), country: a.addressCountry ? (typeof a.addressCountry === "string" ? (/^[A-Z]{2}$/.test(a.addressCountry) ? a.addressCountry : countryFromText(a.addressCountry)) : countryFromText(String((a.addressCountry as Record<string, unknown>).name ?? ""))) : null };241 }242 return null;243}244