SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
6 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
12.7 KB · 193 lines typescript
Raw Blame History
1import { load } from "cheerio";2import type { ConnectorContext, ExtractedRecord, Parser, RawDocument } from "@dci/connectors";3import { isPdf, mainText, metaDescription, pdfText, publishedDate } from "@dci/connectors";4import { cleanText, htmlToText, urlFingerprint } from "@dci/core";5import { EXTRACTOR_VERSION, extractAnnouncement, isPipelineStatus, parseDateLoose, projectCertainty, qualifiesAsProject, shortSummary, type Announcement, type Money } from "./extract-project.js";67/**8 * `news_article_v1` — press releases, news articles, government / utility announcements (HTML, Firecrawl9 * markdown or PDF). Emits one `news_event` per relevant page and, for material announcements, one `project`.10 *11 * Params (YAML `extractors.<x>.params`):12 *   keepText            keep the article body in `data.text` (default false — third-party publishers: title, date,13 *                       ≤ 400-char summary and extracted facts only; set true for public-sector sources)14 *   minProjectMw        MW gate for the project record (default 5)15 *   minInvestmentUsd    investment gate (default 50 000 000)16 *   minAcres            site-size gate (default 100)17 */18export interface ArticleParams { keepText?: boolean; minProjectMw?: number; minInvestmentUsd?: number; minAcres?: number; /** skip the title/lead-signal guards (planning documents) */ lenient?: boolean }1920const SKIP_GROUPS = new Set(["seed", "index", "rss", "sitemap"]);21const TEXT_DATE_RE = /\b(\d{1,2}(?:st|nd|rd|th)?\s+(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)\.?,?\s+20\d\d|(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)\.?\s+\d{1,2}(?:st|nd|rd|th)?,?\s+20\d\d|20\d\d-\d{2}-\d{2})\b/;2223export interface ArticleContent { title: string | null; titleMethod: string; text: string; html: string; summary: string | null; summaryMethod: string; published: string | null; publishedMethod: string }2425/** Title / body / date / summary from any document shape (HTML, markdown, PDF). */26export async function articleContent(doc: RawDocument, now = new Date()): Promise<ArticleContent> {27  let title: string | null = null, titleMethod = "none";28  let text = "", html = "";29  let published: string | null = null, publishedMethod = "none";30  let summary: string | null = null, summaryMethod = "none";31  const meta = doc.meta ?? {};3233  if (isPdf(doc)) {34    // pdfText is bounded (15 MB input, 60 pages, 20 s): an oversize or stalled PDF yields empty text, never a stuck run35    try {36      const p = await pdfText(doc.body);37      text = p.text;38      title = p.title ?? firstLine(text) ?? doc.finalUrl.split("/").pop() ?? null;39      titleMethod = p.title ? "pdf:meta-title" : "pdf:first-line";40    } catch (e) {41      text = "";42      title = doc.finalUrl.split("/").pop() ?? null;43      titleMethod = `pdf:failed(${(e as Error).message.slice(0, 60)})`;44    }45  } else if (/html|xml/.test(doc.contentType ?? "") || /<html|<!doctype html/i.test(doc.text.slice(0, 2000))) {46    html = doc.text;47    const $ = load(html);48    const og = cleanText($('meta[property="og:title"]').attr("content"));49    const h1 = cleanText($("h1").first().text());50    const tt = cleanText($("title").first().text());51    // generic section headings ("News Flash", "Press Releases") lose to og:title52    const genericH1 = !h1 || h1.length < 10 || h1.length > 220 || /^(news( flash)?|press releases?|newsroom|media|announcements?|home|blog|articles?|releases?)$/i.test(h1) || (og != null && h1.length < 25 && og.length > h1.length + 10);53    if (h1 && !genericH1) { title = h1; titleMethod = "html:h1"; }54    else if (og) { title = og; titleMethod = "meta:og:title"; }55    else if (h1 && h1.length >= 10 && h1.length <= 220) { title = h1; titleMethod = "html:h1"; }56    else if (tt) { title = tt; titleMethod = "html:title"; }57    text = mainText(html);58    const pd = publishedDate(html);59    if (pd) { published = pd; publishedMethod = "meta:published"; }60    const md = metaDescription(html);61    if (md) { summary = md; summaryMethod = "meta:description"; }62  } else if (doc.markdown) {63    text = doc.markdown;64    const h = text.match(/^#\s+(.+)$/m);65    if (h) { title = cleanText(h[1]); titleMethod = "markdown:h1"; }66  } else {67    text = doc.text;68  }6970  if (!title && typeof meta.title === "string") { title = cleanText(meta.title); titleMethod = "rss:title"; }71  if (!published && typeof meta.published === "string") { const d = parseDateLoose(meta.published); if (d) { published = d; publishedMethod = "rss:published"; } }72  if (!published) {73    // dateline in the first 1 500 chars ("SINGAPORE, 21 Aug 2026 —", "September 10, 2026") — never a future date74    const m = text.slice(0, 1500).match(TEXT_DATE_RE);75    if (m) {76      const d = parseDateLoose(m[1]!);77      if (d && d.length >= 7) {78        const dt = new Date(d.length === 7 ? `${d}-01` : d);79        if (Number.isFinite(dt.getTime()) && dt.getTime() <= now.getTime() + 86_400_000 && dt.getFullYear() >= 2000) { published = d; publishedMethod = "text:dateline"; }80      }81    }82  }83  if (!summary && typeof meta.summary === "string") { const s = cleanText(htmlToText(meta.summary)); if (s) { summary = s; summaryMethod = "rss:description"; } }84  if (!summary && text) { summary = text.slice(0, 600); summaryMethod = "text:lead"; }85  return { title, titleMethod, text, html, summary, summaryMethod, published, publishedMethod };86}8788function firstLine(text: string): string | null {89  for (const line of text.split(/\n+/)) { const l = cleanText(line); if (l && l.length >= 12 && l.length <= 140 && /[A-Za-z]{3}/.test(l)) return l; }90  return null;91}9293export function formatMoney(m: Money): string {94  const abs = m.amount;95  const n = abs >= 1e9 ? `${trim(abs / 1e9)} billion` : abs >= 1e6 ? `${trim(abs / 1e6)} million` : String(abs);96  return `${n} ${m.currency}`;97}98const trim = (v: number) => String(Math.round(v * 100) / 100);99100export function shouldSkipDocument(doc: RawDocument): boolean {101  if (doc.group && SKIP_GROUPS.has(doc.group)) return true;102  if (doc.pageType === "news_index" || doc.pageType === "facility_index" || doc.pageType === "sitemap") return true;103  return false;104}105106/**107 * Sentinel for index / seed pages. GenericConnector archives any relevant-looking page (title + full text) as a108 * news_event when a parser returns nothing; a nameless `project` record is dropped silently by its normalizer,109 * yields no entity and no validation issue, and keeps that fallback from firing on listing pages.110 */111export function indexSentinel(connectorId: string, url: string): ExtractedRecord {112  return { kind: "project", key: `${connectorId}:skip:${urlFingerprint(url)}`, url, data: { _skip: "index_page" }, certainty: 0, pageType: "news_index" };113}114115/** Build the ExtractedRecords for an announcement — shared with the planning parser. */116export function buildRecords(args: { connectorId: string; url: string; a: Announcement; content: ArticleContent; params: ArticleParams; pageTypeOverride?: ExtractedRecord["pageType"]; extraEvent?: Record<string, unknown>; extraProject?: Record<string, unknown>; certaintyEvent?: number; forceProject?: boolean }): ExtractedRecord[] {117  const { connectorId, url, a, content, params } = args;118  const fp = urlFingerprint(url);119  const pageType = args.pageTypeOverride ?? a.pageType;120  const emitProject = (args.forceProject && isPipelineStatus(a.status)) || qualifiesAsProject(a, { minMw: params.minProjectMw, minInvestmentUsd: params.minInvestmentUsd, minAcres: params.minAcres, lenient: params.lenient });121  const summary = shortSummary(content.summary ?? a.title);122  const title = a.title || content.title || url;123  const country = a.location?.country ?? null;124  const city = a.location?.city ?? null;125126  // The generic normalizer turns news_event.status + mw into a weaker duplicate project; we only expose the status127  // on the event when no project record is emitted and none could be derived from it.128  const statusForEvent = emitProject || (a.headlineMw != null && isPipelineStatus(a.status)) ? null : a.status;129130  const event: ExtractedRecord = {131    kind: "news_event",132    key: `${connectorId}:news:${fp}`,133    url,134    pageType,135    certainty: args.certaintyEvent ?? (a.relevance === "strong" ? 0.7 : 0.35),136    data: {137      title, url, publishedAt: content.published, summary, pageType, eventType: a.eventType ?? null,138      mw: a.mwAll, status: statusForEvent, statusInferred: a.status, money: a.money,139      text: params.keepText ? content.text.slice(0, 20_000) : null,140      operators: a.operators.map((o) => o.name), countries: country ? [country] : [], cities: city ? [city] : [],141      relevance: a.relevance, extractor: EXTRACTOR_VERSION, projectClass: a.classification.class, isAi: a.aiEvidence === "confirmed" || a.aiEvidence === "likely", ...(args.extraEvent ?? {}),142    },143    methods: { title: content.titleMethod, publishedAt: content.publishedMethod, summary: content.summaryMethod, mw: "regex:mw_v1", eventType: a.methods.pageType ?? "classify", operators: "lexicon:operator", location: a.methods.location ?? "none" },144  };145  if (!emitProject) return [event];146147  const description = [summary, a.money && a.money.currency !== "USD" ? `Investment: ${formatMoney(a.money)}.` : null].filter(Boolean).join(" ");148  const project: ExtractedRecord = {149    kind: "project",150    key: `${connectorId}:project:${fp}`,151    url,152    pageType,153    certainty: projectCertainty(a),154    data: {155      name: a.projectName, operatorName: a.operator?.name ?? null, city, regionName: a.location?.region ?? null, countryIso2: country,156      status: a.status, announcedOn: content.published, expectedOpening: a.expectedOpening, plannedMw: a.headlineMw, investmentUsd: a.investmentUsd,157      acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url,158      projectClass: a.classification.class, evidenceLevel: a.classification.evidence.strength,159      capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics,160      investmentCurrency: a.money?.currency ?? null, investmentOriginal: a.money?.amount ?? null,161      claimContext: { ...(a.evidence.plannedMw ? { plannedMw: a.evidence.plannedMw.text } : {}), ...(a.evidence.investment ? { investmentUsd: a.evidence.investment.text } : {}) },162      aiEvidence: a.aiEvidence, ...(args.extraProject ?? {}),163    },164    methods: {165      name: a.methods.name ?? "title", operatorName: a.methods.operatorName ?? "none", city: a.methods.location ?? "none", regionName: a.methods.location ?? "none", countryIso2: a.methods.location ?? "none",166      status: a.methods.status ?? "none", announcedOn: content.publishedMethod, expectedOpening: a.methods.expectedOpening ?? "none", plannedMw: a.methods.plannedMw ?? "none", investmentUsd: a.methods.investment ?? "none",167      acreage: a.methods.acreage ?? "none", phaseCount: a.methods.phaseCount ?? "none",168    },169  };170  return [event, project];171}172173export const newsArticleParser: Parser = {174  name: "news_article_v1",175  version: EXTRACTOR_VERSION,176  async parse(doc: RawDocument, ctx: ConnectorContext, params: Record<string, unknown> = {}): Promise<ExtractedRecord[]> {177    if (shouldSkipDocument(doc)) return [indexSentinel(ctx.connectorId, doc.finalUrl)];178    const p: ArticleParams = { keepText: params.keepText === true, minProjectMw: numOr(params.minProjectMw), minInvestmentUsd: numOr(params.minInvestmentUsd), minAcres: numOr(params.minAcres) };179    let content: ArticleContent;180    try { content = await articleContent(doc); } catch (e) { ctx.log("warn", `news_article_v1: cannot read ${doc.finalUrl}: ${(e as Error).message}`); return []; }181    if (!content.text || content.text.length < 80) return [];182    const primary = params._sourceKind != null && ["operator", "cloud_provider", "government", "utility"].includes(String(params._sourceKind));183    const a = extractAnnouncement(content.title, content.text, { publishedAt: content.published, trustLead: primary || p.keepText });184    if (a.relevance === "none") { ctx.log("debug", `news_article_v1: not about data centers, skipped ${doc.finalUrl}`); return []; }185    // Third-party publishers (keepText false): an article only weakly about data centers, or whose data-center vocabulary186    // sits deep in the body, is not indexed at all. The sentinel keeps the GenericConnector fallback from archiving it.187    if (!p.keepText && (a.relevance === "weak" || !a.leadRelevance)) { ctx.log("debug", `news_article_v1: off-topic for a third-party publisher, skipped ${doc.finalUrl}`); return [indexSentinel(ctx.connectorId, doc.finalUrl)]; }188    return buildRecords({ connectorId: ctx.connectorId, url: doc.finalUrl, a, content, params: p });189  },190};191192function numOr(v: unknown): number | undefined { const n = typeof v === "number" ? v : Number(v); return Number.isFinite(n) && v != null ? n : undefined; }193