spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import { load } from "cheerio";2import type { ConnectorContext, ExtractedRecord, Parser, RawDocument } from "@dci/connectors";3import { isPdf, mainText, metaDescription, pdfText, publishedDate } from "@dci/connectors";4import { cleanText, htmlToText, urlFingerprint } from "@dci/core";5import { EXTRACTOR_VERSION, extractAnnouncement, isPipelineStatus, parseDateLoose, projectCertainty, qualifiesAsProject, shortSummary, type Announcement, type Money } from "./extract-project.js";67/**8 * `news_article_v1` — press releases, news articles, government / utility announcements (HTML, Firecrawl9 * markdown or PDF). Emits one `news_event` per relevant page and, for material announcements, one `project`.10 *11 * Params (YAML `extractors.<x>.params`):12 * keepText keep the article body in `data.text` (default false — third-party publishers: title, date,13 * ≤ 400-char summary and extracted facts only; set true for public-sector sources)14 * minProjectMw MW gate for the project record (default 5)15 * minInvestmentUsd investment gate (default 50 000 000)16 * minAcres site-size gate (default 100)17 */18export interface ArticleParams { keepText?: boolean; minProjectMw?: number; minInvestmentUsd?: number; minAcres?: number; /** skip the title/lead-signal guards (planning documents) */ lenient?: boolean }1920const SKIP_GROUPS = new Set(["seed", "index", "rss", "sitemap"]);21const TEXT_DATE_RE = /\b(\d{1,2}(?:st|nd|rd|th)?\s+(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)\.?,?\s+20\d\d|(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)\.?\s+\d{1,2}(?:st|nd|rd|th)?,?\s+20\d\d|20\d\d-\d{2}-\d{2})\b/;2223export interface ArticleContent { title: string | null; titleMethod: string; text: string; html: string; summary: string | null; summaryMethod: string; published: string | null; publishedMethod: string }2425/** Title / body / date / summary from any document shape (HTML, markdown, PDF). */26export async function articleContent(doc: RawDocument, now = new Date()): Promise<ArticleContent> {27 let title: string | null = null, titleMethod = "none";28 let text = "", html = "";29 let published: string | null = null, publishedMethod = "none";30 let summary: string | null = null, summaryMethod = "none";31 const meta = doc.meta ?? {};3233 if (isPdf(doc)) {34 // pdfText is bounded (15 MB input, 60 pages, 20 s): an oversize or stalled PDF yields empty text, never a stuck run35 try {36 const p = await pdfText(doc.body);37 text = p.text;38 title = p.title ?? firstLine(text) ?? doc.finalUrl.split("/").pop() ?? null;39 titleMethod = p.title ? "pdf:meta-title" : "pdf:first-line";40 } catch (e) {41 text = "";42 title = doc.finalUrl.split("/").pop() ?? null;43 titleMethod = `pdf:failed(${(e as Error).message.slice(0, 60)})`;44 }45 } else if (/html|xml/.test(doc.contentType ?? "") || /<html|<!doctype html/i.test(doc.text.slice(0, 2000))) {46 html = doc.text;47 const $ = load(html);48 const og = cleanText($('meta[property="og:title"]').attr("content"));49 const h1 = cleanText($("h1").first().text());50 const tt = cleanText($("title").first().text());51 // generic section headings ("News Flash", "Press Releases") lose to og:title52 const genericH1 = !h1 || h1.length < 10 || h1.length > 220 || /^(news( flash)?|press releases?|newsroom|media|announcements?|home|blog|articles?|releases?)$/i.test(h1) || (og != null && h1.length < 25 && og.length > h1.length + 10);53 if (h1 && !genericH1) { title = h1; titleMethod = "html:h1"; }54 else if (og) { title = og; titleMethod = "meta:og:title"; }55 else if (h1 && h1.length >= 10 && h1.length <= 220) { title = h1; titleMethod = "html:h1"; }56 else if (tt) { title = tt; titleMethod = "html:title"; }57 text = mainText(html);58 const pd = publishedDate(html);59 if (pd) { published = pd; publishedMethod = "meta:published"; }60 const md = metaDescription(html);61 if (md) { summary = md; summaryMethod = "meta:description"; }62 } else if (doc.markdown) {63 text = doc.markdown;64 const h = text.match(/^#\s+(.+)$/m);65 if (h) { title = cleanText(h[1]); titleMethod = "markdown:h1"; }66 } else {67 text = doc.text;68 }6970 if (!title && typeof meta.title === "string") { title = cleanText(meta.title); titleMethod = "rss:title"; }71 if (!published && typeof meta.published === "string") { const d = parseDateLoose(meta.published); if (d) { published = d; publishedMethod = "rss:published"; } }72 if (!published) {73 // dateline in the first 1 500 chars ("SINGAPORE, 21 Aug 2026 —", "September 10, 2026") — never a future date74 const m = text.slice(0, 1500).match(TEXT_DATE_RE);75 if (m) {76 const d = parseDateLoose(m[1]!);77 if (d && d.length >= 7) {78 const dt = new Date(d.length === 7 ? `${d}-01` : d);79 if (Number.isFinite(dt.getTime()) && dt.getTime() <= now.getTime() + 86_400_000 && dt.getFullYear() >= 2000) { published = d; publishedMethod = "text:dateline"; }80 }81 }82 }83 if (!summary && typeof meta.summary === "string") { const s = cleanText(htmlToText(meta.summary)); if (s) { summary = s; summaryMethod = "rss:description"; } }84 if (!summary && text) { summary = text.slice(0, 600); summaryMethod = "text:lead"; }85 return { title, titleMethod, text, html, summary, summaryMethod, published, publishedMethod };86}8788function firstLine(text: string): string | null {89 for (const line of text.split(/\n+/)) { const l = cleanText(line); if (l && l.length >= 12 && l.length <= 140 && /[A-Za-z]{3}/.test(l)) return l; }90 return null;91}9293export function formatMoney(m: Money): string {94 const abs = m.amount;95 const n = abs >= 1e9 ? `${trim(abs / 1e9)} billion` : abs >= 1e6 ? `${trim(abs / 1e6)} million` : String(abs);96 return `${n} ${m.currency}`;97}98const trim = (v: number) => String(Math.round(v * 100) / 100);99100export function shouldSkipDocument(doc: RawDocument): boolean {101 if (doc.group && SKIP_GROUPS.has(doc.group)) return true;102 if (doc.pageType === "news_index" || doc.pageType === "facility_index" || doc.pageType === "sitemap") return true;103 return false;104}105106/**107 * Sentinel for index / seed pages. GenericConnector archives any relevant-looking page (title + full text) as a108 * news_event when a parser returns nothing; a nameless `project` record is dropped silently by its normalizer,109 * yields no entity and no validation issue, and keeps that fallback from firing on listing pages.110 */111export function indexSentinel(connectorId: string, url: string): ExtractedRecord {112 return { kind: "project", key: `${connectorId}:skip:${urlFingerprint(url)}`, url, data: { _skip: "index_page" }, certainty: 0, pageType: "news_index" };113}114115/** Build the ExtractedRecords for an announcement — shared with the planning parser. */116export function buildRecords(args: { connectorId: string; url: string; a: Announcement; content: ArticleContent; params: ArticleParams; pageTypeOverride?: ExtractedRecord["pageType"]; extraEvent?: Record<string, unknown>; extraProject?: Record<string, unknown>; certaintyEvent?: number; forceProject?: boolean }): ExtractedRecord[] {117 const { connectorId, url, a, content, params } = args;118 const fp = urlFingerprint(url);119 const pageType = args.pageTypeOverride ?? a.pageType;120 const emitProject = (args.forceProject && isPipelineStatus(a.status)) || qualifiesAsProject(a, { minMw: params.minProjectMw, minInvestmentUsd: params.minInvestmentUsd, minAcres: params.minAcres, lenient: params.lenient });121 const summary = shortSummary(content.summary ?? a.title);122 const title = a.title || content.title || url;123 const country = a.location?.country ?? null;124 const city = a.location?.city ?? null;125126 // The generic normalizer turns news_event.status + mw into a weaker duplicate project; we only expose the status127 // on the event when no project record is emitted and none could be derived from it.128 const statusForEvent = emitProject || (a.headlineMw != null && isPipelineStatus(a.status)) ? null : a.status;129130 const event: ExtractedRecord = {131 kind: "news_event",132 key: `${connectorId}:news:${fp}`,133 url,134 pageType,135 certainty: args.certaintyEvent ?? (a.relevance === "strong" ? 0.7 : 0.35),136 data: {137 title, url, publishedAt: content.published, summary, pageType, eventType: a.eventType ?? null,138 mw: a.mwAll, status: statusForEvent, statusInferred: a.status, money: a.money,139 text: params.keepText ? content.text.slice(0, 20_000) : null,140 operators: a.operators.map((o) => o.name), countries: country ? [country] : [], cities: city ? [city] : [],141 relevance: a.relevance, extractor: EXTRACTOR_VERSION, projectClass: a.classification.class, isAi: a.aiEvidence === "confirmed" || a.aiEvidence === "likely", ...(args.extraEvent ?? {}),142 },143 methods: { title: content.titleMethod, publishedAt: content.publishedMethod, summary: content.summaryMethod, mw: "regex:mw_v1", eventType: a.methods.pageType ?? "classify", operators: "lexicon:operator", location: a.methods.location ?? "none" },144 };145 if (!emitProject) return [event];146147 const description = [summary, a.money && a.money.currency !== "USD" ? `Investment: ${formatMoney(a.money)}.` : null].filter(Boolean).join(" ");148 const project: ExtractedRecord = {149 kind: "project",150 key: `${connectorId}:project:${fp}`,151 url,152 pageType,153 certainty: projectCertainty(a),154 data: {155 name: a.projectName, operatorName: a.operator?.name ?? null, city, regionName: a.location?.region ?? null, countryIso2: country,156 status: a.status, announcedOn: content.published, expectedOpening: a.expectedOpening, plannedMw: a.headlineMw, investmentUsd: a.investmentUsd,157 acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url,158 projectClass: a.classification.class, evidenceLevel: a.classification.evidence.strength,159 capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics,160 investmentCurrency: a.money?.currency ?? null, investmentOriginal: a.money?.amount ?? null,161 claimContext: { ...(a.evidence.plannedMw ? { plannedMw: a.evidence.plannedMw.text } : {}), ...(a.evidence.investment ? { investmentUsd: a.evidence.investment.text } : {}) },162 aiEvidence: a.aiEvidence, ...(args.extraProject ?? {}),163 },164 methods: {165 name: a.methods.name ?? "title", operatorName: a.methods.operatorName ?? "none", city: a.methods.location ?? "none", regionName: a.methods.location ?? "none", countryIso2: a.methods.location ?? "none",166 status: a.methods.status ?? "none", announcedOn: content.publishedMethod, expectedOpening: a.methods.expectedOpening ?? "none", plannedMw: a.methods.plannedMw ?? "none", investmentUsd: a.methods.investment ?? "none",167 acreage: a.methods.acreage ?? "none", phaseCount: a.methods.phaseCount ?? "none",168 },169 };170 return [event, project];171}172173export const newsArticleParser: Parser = {174 name: "news_article_v1",175 version: EXTRACTOR_VERSION,176 async parse(doc: RawDocument, ctx: ConnectorContext, params: Record<string, unknown> = {}): Promise<ExtractedRecord[]> {177 if (shouldSkipDocument(doc)) return [indexSentinel(ctx.connectorId, doc.finalUrl)];178 const p: ArticleParams = { keepText: params.keepText === true, minProjectMw: numOr(params.minProjectMw), minInvestmentUsd: numOr(params.minInvestmentUsd), minAcres: numOr(params.minAcres) };179 let content: ArticleContent;180 try { content = await articleContent(doc); } catch (e) { ctx.log("warn", `news_article_v1: cannot read ${doc.finalUrl}: ${(e as Error).message}`); return []; }181 if (!content.text || content.text.length < 80) return [];182 const primary = params._sourceKind != null && ["operator", "cloud_provider", "government", "utility"].includes(String(params._sourceKind));183 const a = extractAnnouncement(content.title, content.text, { publishedAt: content.published, trustLead: primary || p.keepText });184 if (a.relevance === "none") { ctx.log("debug", `news_article_v1: not about data centers, skipped ${doc.finalUrl}`); return []; }185 // Third-party publishers (keepText false): an article only weakly about data centers, or whose data-center vocabulary186 // sits deep in the body, is not indexed at all. The sentinel keeps the GenericConnector fallback from archiving it.187 if (!p.keepText && (a.relevance === "weak" || !a.leadRelevance)) { ctx.log("debug", `news_article_v1: off-topic for a third-party publisher, skipped ${doc.finalUrl}`); return [indexSentinel(ctx.connectorId, doc.finalUrl)]; }188 return buildRecords({ connectorId: ctx.connectorId, url: doc.finalUrl, a, content, params: p });189 },190};191192function numOr(v: unknown): number | undefined { const n = typeof v === "number" ? v : Number(v); return Number.isFinite(n) && v != null ? n : undefined; }193