import { load } from "cheerio"; import type { ConnectorContext, ExtractedRecord, Parser, RawDocument } from "@dci/connectors"; import { isPdf, mainText, metaDescription, pdfText, publishedDate } from "@dci/connectors"; import { cleanText, htmlToText, urlFingerprint } from "@dci/core"; import { EXTRACTOR_VERSION, extractAnnouncement, isPipelineStatus, parseDateLoose, projectCertainty, qualifiesAsProject, shortSummary, type Announcement, type Money } from "./extract-project.js"; /** * `news_article_v1` — press releases, news articles, government / utility announcements (HTML, Firecrawl * markdown or PDF). Emits one `news_event` per relevant page and, for material announcements, one `project`. * * Params (YAML `extractors..params`): * keepText keep the article body in `data.text` (default false — third-party publishers: title, date, * ≤ 400-char summary and extracted facts only; set true for public-sector sources) * minProjectMw MW gate for the project record (default 5) * minInvestmentUsd investment gate (default 50 000 000) * minAcres site-size gate (default 100) */ export interface ArticleParams { keepText?: boolean; minProjectMw?: number; minInvestmentUsd?: number; minAcres?: number; /** skip the title/lead-signal guards (planning documents) */ lenient?: boolean } const SKIP_GROUPS = new Set(["seed", "index", "rss", "sitemap"]); const TEXT_DATE_RE = /\b(\d{1,2}(?:st|nd|rd|th)?\s+(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)\.?,?\s+20\d\d|(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)\.?\s+\d{1,2}(?:st|nd|rd|th)?,?\s+20\d\d|20\d\d-\d{2}-\d{2})\b/; export interface ArticleContent { title: string | null; titleMethod: string; text: string; html: string; summary: string | null; summaryMethod: string; published: string | null; publishedMethod: string } /** Title / body / date / summary from any document shape (HTML, markdown, PDF). */ export async function articleContent(doc: RawDocument, now = new Date()): Promise { let title: string | null = null, titleMethod = "none"; let text = "", html = ""; let published: string | null = null, publishedMethod = "none"; let summary: string | null = null, summaryMethod = "none"; const meta = doc.meta ?? {}; if (isPdf(doc)) { // pdfText is bounded (15 MB input, 60 pages, 20 s): an oversize or stalled PDF yields empty text, never a stuck run try { const p = await pdfText(doc.body); text = p.text; title = p.title ?? firstLine(text) ?? doc.finalUrl.split("/").pop() ?? null; titleMethod = p.title ? "pdf:meta-title" : "pdf:first-line"; } catch (e) { text = ""; title = doc.finalUrl.split("/").pop() ?? null; titleMethod = `pdf:failed(${(e as Error).message.slice(0, 60)})`; } } else if (/html|xml/.test(doc.contentType ?? "") || / 220 || /^(news( flash)?|press releases?|newsroom|media|announcements?|home|blog|articles?|releases?)$/i.test(h1) || (og != null && h1.length < 25 && og.length > h1.length + 10); if (h1 && !genericH1) { title = h1; titleMethod = "html:h1"; } else if (og) { title = og; titleMethod = "meta:og:title"; } else if (h1 && h1.length >= 10 && h1.length <= 220) { title = h1; titleMethod = "html:h1"; } else if (tt) { title = tt; titleMethod = "html:title"; } text = mainText(html); const pd = publishedDate(html); if (pd) { published = pd; publishedMethod = "meta:published"; } const md = metaDescription(html); if (md) { summary = md; summaryMethod = "meta:description"; } } else if (doc.markdown) { text = doc.markdown; const h = text.match(/^#\s+(.+)$/m); if (h) { title = cleanText(h[1]); titleMethod = "markdown:h1"; } } else { text = doc.text; } if (!title && typeof meta.title === "string") { title = cleanText(meta.title); titleMethod = "rss:title"; } if (!published && typeof meta.published === "string") { const d = parseDateLoose(meta.published); if (d) { published = d; publishedMethod = "rss:published"; } } if (!published) { // dateline in the first 1 500 chars ("SINGAPORE, 21 Aug 2026 —", "September 10, 2026") — never a future date const m = text.slice(0, 1500).match(TEXT_DATE_RE); if (m) { const d = parseDateLoose(m[1]!); if (d && d.length >= 7) { const dt = new Date(d.length === 7 ? `${d}-01` : d); if (Number.isFinite(dt.getTime()) && dt.getTime() <= now.getTime() + 86_400_000 && dt.getFullYear() >= 2000) { published = d; publishedMethod = "text:dateline"; } } } } if (!summary && typeof meta.summary === "string") { const s = cleanText(htmlToText(meta.summary)); if (s) { summary = s; summaryMethod = "rss:description"; } } if (!summary && text) { summary = text.slice(0, 600); summaryMethod = "text:lead"; } return { title, titleMethod, text, html, summary, summaryMethod, published, publishedMethod }; } function firstLine(text: string): string | null { for (const line of text.split(/\n+/)) { const l = cleanText(line); if (l && l.length >= 12 && l.length <= 140 && /[A-Za-z]{3}/.test(l)) return l; } return null; } export function formatMoney(m: Money): string { const abs = m.amount; const n = abs >= 1e9 ? `${trim(abs / 1e9)} billion` : abs >= 1e6 ? `${trim(abs / 1e6)} million` : String(abs); return `${n} ${m.currency}`; } const trim = (v: number) => String(Math.round(v * 100) / 100); export function shouldSkipDocument(doc: RawDocument): boolean { if (doc.group && SKIP_GROUPS.has(doc.group)) return true; if (doc.pageType === "news_index" || doc.pageType === "facility_index" || doc.pageType === "sitemap") return true; return false; } /** * Sentinel for index / seed pages. GenericConnector archives any relevant-looking page (title + full text) as a * news_event when a parser returns nothing; a nameless `project` record is dropped silently by its normalizer, * yields no entity and no validation issue, and keeps that fallback from firing on listing pages. */ export function indexSentinel(connectorId: string, url: string): ExtractedRecord { return { kind: "project", key: `${connectorId}:skip:${urlFingerprint(url)}`, url, data: { _skip: "index_page" }, certainty: 0, pageType: "news_index" }; } /** Build the ExtractedRecords for an announcement — shared with the planning parser. */ export function buildRecords(args: { connectorId: string; url: string; a: Announcement; content: ArticleContent; params: ArticleParams; pageTypeOverride?: ExtractedRecord["pageType"]; extraEvent?: Record; extraProject?: Record; certaintyEvent?: number; forceProject?: boolean }): ExtractedRecord[] { const { connectorId, url, a, content, params } = args; const fp = urlFingerprint(url); const pageType = args.pageTypeOverride ?? a.pageType; const emitProject = (args.forceProject && isPipelineStatus(a.status)) || qualifiesAsProject(a, { minMw: params.minProjectMw, minInvestmentUsd: params.minInvestmentUsd, minAcres: params.minAcres, lenient: params.lenient }); const summary = shortSummary(content.summary ?? a.title); const title = a.title || content.title || url; const country = a.location?.country ?? null; const city = a.location?.city ?? null; // The generic normalizer turns news_event.status + mw into a weaker duplicate project; we only expose the status // on the event when no project record is emitted and none could be derived from it. const statusForEvent = emitProject || (a.headlineMw != null && isPipelineStatus(a.status)) ? null : a.status; const event: ExtractedRecord = { kind: "news_event", key: `${connectorId}:news:${fp}`, url, pageType, certainty: args.certaintyEvent ?? (a.relevance === "strong" ? 0.7 : 0.35), data: { title, url, publishedAt: content.published, summary, pageType, eventType: a.eventType ?? null, mw: a.mwAll, status: statusForEvent, statusInferred: a.status, money: a.money, text: params.keepText ? content.text.slice(0, 20_000) : null, operators: a.operators.map((o) => o.name), countries: country ? [country] : [], cities: city ? [city] : [], relevance: a.relevance, extractor: EXTRACTOR_VERSION, projectClass: a.classification.class, isAi: a.aiEvidence === "confirmed" || a.aiEvidence === "likely", ...(args.extraEvent ?? {}), }, methods: { title: content.titleMethod, publishedAt: content.publishedMethod, summary: content.summaryMethod, mw: "regex:mw_v1", eventType: a.methods.pageType ?? "classify", operators: "lexicon:operator", location: a.methods.location ?? "none" }, }; if (!emitProject) return [event]; const description = [summary, a.money && a.money.currency !== "USD" ? `Investment: ${formatMoney(a.money)}.` : null].filter(Boolean).join(" "); const project: ExtractedRecord = { kind: "project", key: `${connectorId}:project:${fp}`, url, pageType, certainty: projectCertainty(a), data: { name: a.projectName, operatorName: a.operator?.name ?? null, city, regionName: a.location?.region ?? null, countryIso2: country, status: a.status, announcedOn: content.published, expectedOpening: a.expectedOpening, plannedMw: a.headlineMw, investmentUsd: a.investmentUsd, acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url, projectClass: a.classification.class, evidenceLevel: a.classification.evidence.strength, capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics, investmentCurrency: a.money?.currency ?? null, investmentOriginal: a.money?.amount ?? null, claimContext: { ...(a.evidence.plannedMw ? { plannedMw: a.evidence.plannedMw.text } : {}), ...(a.evidence.investment ? { investmentUsd: a.evidence.investment.text } : {}) }, aiEvidence: a.aiEvidence, ...(args.extraProject ?? {}), }, methods: { name: a.methods.name ?? "title", operatorName: a.methods.operatorName ?? "none", city: a.methods.location ?? "none", regionName: a.methods.location ?? "none", countryIso2: a.methods.location ?? "none", status: a.methods.status ?? "none", announcedOn: content.publishedMethod, expectedOpening: a.methods.expectedOpening ?? "none", plannedMw: a.methods.plannedMw ?? "none", investmentUsd: a.methods.investment ?? "none", acreage: a.methods.acreage ?? "none", phaseCount: a.methods.phaseCount ?? "none", }, }; return [event, project]; } export const newsArticleParser: Parser = { name: "news_article_v1", version: EXTRACTOR_VERSION, async parse(doc: RawDocument, ctx: ConnectorContext, params: Record = {}): Promise { if (shouldSkipDocument(doc)) return [indexSentinel(ctx.connectorId, doc.finalUrl)]; const p: ArticleParams = { keepText: params.keepText === true, minProjectMw: numOr(params.minProjectMw), minInvestmentUsd: numOr(params.minInvestmentUsd), minAcres: numOr(params.minAcres) }; let content: ArticleContent; try { content = await articleContent(doc); } catch (e) { ctx.log("warn", `news_article_v1: cannot read ${doc.finalUrl}: ${(e as Error).message}`); return []; } if (!content.text || content.text.length < 80) return []; const primary = params._sourceKind != null && ["operator", "cloud_provider", "government", "utility"].includes(String(params._sourceKind)); const a = extractAnnouncement(content.title, content.text, { publishedAt: content.published, trustLead: primary || p.keepText }); if (a.relevance === "none") { ctx.log("debug", `news_article_v1: not about data centers, skipped ${doc.finalUrl}`); return []; } // Third-party publishers (keepText false): an article only weakly about data centers, or whose data-center vocabulary // sits deep in the body, is not indexed at all. The sentinel keeps the GenericConnector fallback from archiving it. if (!p.keepText && (a.relevance === "weak" || !a.leadRelevance)) { ctx.log("debug", `news_article_v1: off-topic for a third-party publisher, skipped ${doc.finalUrl}`); return [indexSentinel(ctx.connectorId, doc.finalUrl)]; } return buildRecords({ connectorId: ctx.connectorId, url: doc.finalUrl, a, content, params: p }); }, }; function numOr(v: unknown): number | undefined { const n = typeof v === "number" ? v : Number(v); return Number.isFinite(n) && v != null ? n : undefined; }