import { load, type CheerioAPI } from "cheerio"; import { getDocumentProxy } from "unpdf"; import { parseMw, parseAreaSqm, parseAreaHa, parseMoney, parsePartialDate, normalizeStatus, inferFacilityType, countryFromText, cleanText, htmlToText } from "@dci/core"; import type { SelectorRule } from "./config.js"; import type { RawDocument } from "./types.js"; /** All JSON-LD objects on a page (flattened @graph), optionally filtered by @type. */ export function jsonLd(html: string, type?: string): Record[] { const $ = load(html); const out: Record[] = []; $('script[type="application/ld+json"]').each((_, el) => { const raw = $(el).contents().text(); try { const parsed = JSON.parse(raw.replace(/^\s*\s*$/g, "")); const arr = Array.isArray(parsed) ? parsed : [parsed]; for (const p of arr) { if (p && typeof p === "object" && Array.isArray((p as { "@graph"?: unknown[] })["@graph"])) out.push(...((p as { "@graph": Record[] })["@graph"])); else if (p && typeof p === "object") out.push(p as Record); } } catch { /* malformed */ } }); if (!type) return out; const t = type.toLowerCase(); return out.filter((o) => { const ty = o["@type"]; const list = Array.isArray(ty) ? ty : [ty]; return list.some((x) => String(x).toLowerCase() === t); }); } /** Embedded JSON blobs: __NEXT_DATA__, window.__INITIAL_STATE__ = {...}, application/json scripts, Nuxt/Drupal settings. */ export function embeddedJson(html: string): Array<{ id: string; data: unknown }> { const $ = load(html); const out: Array<{ id: string; data: unknown }> = []; $('script[type="application/json"], script#__NEXT_DATA__, script[data-drupal-selector="drupal-settings-json"]').each((_, el) => { const id = $(el).attr("id") ?? $(el).attr("data-drupal-selector") ?? "json"; try { out.push({ id, data: JSON.parse($(el).contents().text()) }); } catch { /* skip */ } }); const re = /(?:window\.|var\s+|const\s+|let\s+)?(__INITIAL_STATE__|__NUXT__|__PRELOADED_STATE__|__APOLLO_STATE__|__DATA__|__INITIAL_DATA__|initialData|pageData|__NEXT_DATA__)\s*=\s*(\{[\s\S]*?\})\s*;?\s*(?:<\/script>|\n)/g; for (const m of html.matchAll(re)) { try { out.push({ id: m[1]!, data: JSON.parse(m[2]!) }); } catch { /* not plain JSON */ } } return out; } /** Tiny JSONPath subset: "$.a.b[0].c", "a.b", "[*]" not supported (use walk). */ export function jsonPath(obj: unknown, path: string): unknown { const parts = path.replace(/^\$\.?/, "").split(/\.|\[(\d+)\]/).filter((p) => p !== undefined && p !== ""); let cur: unknown = obj; for (const p of parts) { if (cur == null) return undefined; cur = (cur as Record)[p]; } return cur; } /** Depth-first walk collecting objects satisfying a predicate (used to find facility-like nodes in embedded JSON). */ export function walkJson(obj: unknown, pred: (o: Record) => boolean, out: Record[] = [], depth = 0): Record[] { if (depth > 12 || obj == null) return out; if (Array.isArray(obj)) { for (const x of obj) walkJson(x, pred, out, depth + 1); return out; } if (typeof obj === "object") { const o = obj as Record; if (pred(o)) out.push(o); for (const v of Object.values(o)) walkJson(v, pred, out, depth + 1); } return out; } export function applyTransform(v: unknown, t: SelectorRule extends string ? never : Exclude["transform"]): unknown { if (v == null) return null; const s = typeof v === "string" ? v : String(v); switch (t) { case "mw": return parseMw(s); case "sqm": return parseAreaSqm(s); case "ha": return parseAreaHa(s); case "money": return parseMoney(s); case "date": return parsePartialDate(s); case "status": return normalizeStatus(s); case "type": return inferFacilityType(s); case "country": return /^[A-Z]{2}$/.test(s.trim()) ? s.trim() : countryFromText(s); case "int": { const n = parseInt(s.replace(/[^\d-]/g, ""), 10); return Number.isFinite(n) ? n : null; } case "float": { const n = parseFloat(s.replace(/[^\d.-]/g, "")); return Number.isFinite(n) ? n : null; } case "trim": return cleanText(s); case "lower": return s.trim().toLowerCase(); case "url": return s.trim(); case "bool": return /^(true|yes|1|y)$/i.test(s.trim()); default: return typeof v === "string" ? cleanText(v) : v; } } /** Evaluate one declarative selector rule against a page. */ export function evalRule($: CheerioAPI, html: string, rule: SelectorRule, scope?: ReturnType): { value: unknown; method: string } { const r: Exclude = typeof rule === "string" ? { selector: rule } : rule; let value: unknown = null; let method = ""; if (r.jsonld) { const objs = jsonLd(html, r.jsonld); value = objs.length && r.jsonPath ? jsonPath(objs[0], r.jsonPath) : objs[0] ?? null; method = `json-ld:${r.jsonld}${r.jsonPath ? "." + r.jsonPath : ""}`; } else if (r.meta) { value = $(`meta[property="${r.meta}"], meta[name="${r.meta}"]`).first().attr("content") ?? null; method = `meta:${r.meta}`; } else if (r.jsonPath && !r.selector) { for (const b of embeddedJson(html)) { const v = jsonPath(b.data, r.jsonPath); if (v != null) { value = v; method = `embedded-json:${b.id}.${r.jsonPath}`; break; } } } else if (r.selector) { const sel = scope ? scope.find(r.selector) : $(r.selector); if (r.all) value = sel.map((_, el) => (r.attr ? $(el).attr(r.attr) : $(el).text())).get().map((x) => cleanText(x)).filter(Boolean); else value = r.attr ? sel.first().attr(r.attr) ?? null : cleanText(sel.first().text()); method = `selector:${r.selector}${r.attr ? "@" + r.attr : ""}`; } else if (r.regex) { value = htmlToText(html); method = "regex"; } if (r.regex && value != null) { const re = new RegExp(r.regex, r.regexFlags ?? "i"); const apply = (s: string) => { const m = s.match(re); return m ? (m[r.group ?? 1] ?? m[0]) : null; }; value = Array.isArray(value) ? value.map((x) => apply(String(x))).filter(Boolean) : apply(String(value)); method += `+regex:${r.regex.slice(0, 40)}`; } if (r.transform && value != null) { value = Array.isArray(value) ? value.map((x) => applyTransform(x, r.transform)) : applyTransform(value, r.transform); method += `>${r.transform}`; } if ((value == null || value === "") && r.default !== undefined) value = r.default; return { value, method }; } export function pageTitle(html: string): string | null { const $ = load(html); return cleanText($('meta[property="og:title"]').attr("content") ?? $("title").first().text() ?? $("h1").first().text()); } export function metaDescription(html: string): string | null { const $ = load(html); return cleanText($('meta[name="description"]').attr("content") ?? $('meta[property="og:description"]').attr("content") ?? null); } /** Blocks that are never the story: chrome, related / recommended / trending teasers, sidebars, comments. */ export const NON_CONTENT_SELECTOR = "script,style,noscript,nav,header,footer,aside,form,iframe,svg,[role=navigation],[role=complementary],[class*=related],[class*=recommend],[class*=trending],[class*=popular],[class*=sidebar],[class*=read-more],[class*=readmore],[class*=more-stories],[class*=also-like],[class*=comments],[id*=related],[id*=comments]"; /** Heading that opens a trailing teaser section inside the story element itself ("More in Construction & Site Selection", "Related articles", "Tags"). */ export const TEASER_HEADING_RE = /^\s*(?:More (?:in|from|on|like this)\b.*|Related(?: (?:articles?|stories|news|content|posts?|coverage|reading|links?))?|Recommended(?: for you| reading| stories| articles)?|You (?:may|might) also like|Most (?:read|popular|viewed)|Trending(?: now| stories)?|Popular (?:now|posts|stories|articles)|Latest (?:news|stories|posts|articles)|Read (?:more|next|also)|See also|Further reading|Editor'?s picks|Sponsored(?: content)?|Tags|Comments|Leave a (?:comment|reply))\s*:?\s*$/i; /** Minimum story length before a teaser heading is allowed to cut the text (a short lead titled "Related" is not a teaser section). */ const TEASER_CUT_MIN_CHARS = 300; /** Drop everything from the first teaser heading onwards (teasers are the last thing in a story element). */ export function cutTeasers(text: string): string { const lines = text.split("\n"); let consumed = 0; for (let i = 0; i < lines.length; i++) { const line = lines[i]!; if (consumed >= TEASER_CUT_MIN_CHARS && line.trim().length <= 60 && TEASER_HEADING_RE.test(line)) return lines.slice(0, i).join("\n").trimEnd(); consumed += line.length; } return text; } /** * Main text of an article-like page: the largest
/
/ content block, after removing navigation, footer, * aside, related / recommended / trending / popular / sidebar blocks, and cut at the first trailing teaser heading — so a * "More in …" list of other headlines never leaks into the story (and never names its operator). */ export function mainText(html: string): string { const $ = load(html); $(NON_CONTENT_SELECTOR).remove(); const candidates = ["article", "main", '[role="main"]', ".article-body", ".press-release", ".entry-content", ".post-content", ".content", "#content", "body"]; for (const c of candidates) { // several matches (the story plus teaser
cards): the largest one is the story const texts = $(c).toArray().map((el) => htmlToText($(el).html() ?? "")); if (!texts.length) continue; const t = cutTeasers(texts.sort((a, b) => b.length - a.length)[0]!); if (t.length > 300) return t; } return cutTeasers(htmlToText($("body").html() ?? html)); } /** Publication date from meta tags / JSON-LD /