import type { ConnectorContext, ExtractedRecord, Parser, RawDocument } from "@dci/connectors"; import { isPdf } from "@dci/connectors"; import { cleanText } from "@dci/core"; import { articleContent, buildRecords, indexSentinel, shouldSkipDocument, type ArticleContent, type ArticleParams } from "./article-parser.js"; import { EXTRACTOR_VERSION, extractAnnouncement } from "./extract-project.js"; /** * `news_press_release_v1` — first-party press releases from grid operators, utilities and agencies that publish * them as PDF letterhead files (PJM) or as HTML pages without any date markup (ERCOT). Emits exactly the records of * `news_article_v1` (one `news_event`, plus one `project` for material announcements) with two additions: * * - PDF headline: when the PDF carries no metadata title, the first line of the text is letterhead boilerplate * ("Contact: …", "NEWS RELEASE", "FOR IMMEDIATE RELEASE", e-mails, phone numbers). The headline is the first * line that is not boilerplate; a wrapped headline is re-joined when the next line starts in lower case. * - Body starts at the headline: site navigation (HTML mega-menus, breadcrumbs) or letterhead (PDF) that precedes * the headline is dropped from the kept text and from the lead summary. * - `urlDate`: a regex with named groups `y`, `m`, `d` applied to the URL when the document itself gives no date * (ERCOT slugs start with MMDDYYYY). Only a plausible past date (2000 … tomorrow) is accepted. * * Params (YAML `extractors..params`): * keepText default TRUE (first-party public sources) — set false for third-party publishers * urlDate regex source, e.g. "/news/release/(?\\d{2})(?\\d{2})(?\\d{4})-" * minProjectMw / minInvestmentUsd / minAcres — same gates as `news_article_v1` */ export interface PressReleaseParams extends ArticleParams { urlDate?: string } const MONTHS = "January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec"; /** Letterhead / masthead lines that are never the headline of a release. */ const BOILERPLATE_LINE_RE = new RegExp( `^(?:(?:media |press |news )?contacts?\\b|for (?:immediate )?release\\b|embargoed\\b|(?:news|press|media) release\\b|news\\b|release\\b|page \\d|\\d+ of \\d+|www\\.|https?://|[\\w.+-]+@[\\w-]+\\.|\\(?\\d{3}\\)?[ .-]?\\d{3}[ .-]\\d{4}|\\+?\\d[\\d ()-]{8,}\\d$|(?:\\d{1,2}(?:st|nd|rd|th)?\\s+(?:${MONTHS})\\.?,?\\s+20\\d\\d|(?:${MONTHS})\\.?\\s+\\d{1,2}(?:st|nd|rd|th)?,?\\s+20\\d\\d|20\\d\\d-\\d{2}-\\d{2})\\s*$|\\(?[A-Z][A-Za-z .]+,\\s*[A-Z]{2}\\.?\\s*[–—-])`, "i", ); /** First headline-like line of a PDF press release (null when nothing qualifies in the first 40 lines). */ export function pdfHeadline(text: string): string | null { const lines = text.split(/\n+/).map((l) => cleanText(l) ?? "").filter(Boolean).slice(0, 40); for (let i = 0; i < lines.length; i++) { const l = lines[i]!; if (l.length < 15 || l.length > 220) continue; if (!/[a-z]/.test(l) || l.split(/\s+/).length < 3) continue; // all-caps mastheads, single words if (BOILERPLATE_LINE_RE.test(l)) continue; if (/[:|]\s*$/.test(l)) continue; // "Contact:" style labels const next = lines[i + 1]; if (next && !/[.!?:;)]$/.test(l) && /^[a-z]/.test(next) && l.length + next.length < 220) return `${l} ${next}`; return l; } return null; } /** * Drop site navigation / letterhead that precedes the headline in the extracted text: the body starts at the last * occurrence of the headline inside the first 40 % of the text (breadcrumb, then the real heading). Returns the text * unchanged when the headline is not found there, or when what precedes it is too short to be navigation. */ export function stripBeforeHeadline(text: string, title: string | null): string { if (!title || title.length < 12 || text.length < 400) return text; const norm = (s: string) => s.toLowerCase().replace(/[‘’‚′`´]/g, "'").replace(/[“”„]/g, '"').replace(/ /g, " "); // `` values carry a site suffix ("… - Southwest Power Pool"); match on the headline part only, ≤ 60 chars const head = title.split(/\s+[|–—]\s+|\s+-\s+(?=[A-Z])/)[0]!.trim(); if (head.length < 12) return text; const needle = norm(head.slice(0, 60)); const hay = norm(text); // the headline must sit in the first 70 % of the text and leave a body of at least 200 characters const limit = Math.min(Math.floor(text.length * 0.7), text.length - needle.length - 200); let idx = -1, from = 0; for (;;) { const i = hay.indexOf(needle, from); if (i < 0 || i > limit) break; idx = i; from = i + 1; } return idx >= 20 ? text.slice(idx) : text; } /** "YYYY-MM-DD" from a URL when `pattern` (named groups y, m, d) matches and the date is plausible; null otherwise. */ export function dateFromUrl(url: string, pattern: string, now = new Date()): string | null { let re: RegExp; try { re = new RegExp(pattern, "i"); } catch { return null; } const g = url.match(re)?.groups; if (!g?.y || !g.m || !g.d) return null; const y = Number(g.y.length === 2 ? `20${g.y}` : g.y), m = Number(g.m), d = Number(g.d); if (!Number.isInteger(y) || !Number.isInteger(m) || !Number.isInteger(d) || y < 2000 || m < 1 || m > 12 || d < 1 || d > 31) return null; const iso = `${y}-${String(m).padStart(2, "0")}-${String(d).padStart(2, "0")}`; const t = new Date(`${iso}T00:00:00Z`).getTime(); if (!Number.isFinite(t) || t > now.getTime() + 86_400_000) return null; return iso; } export const pressReleaseParser: Parser = { name: "news_press_release_v1", version: EXTRACTOR_VERSION, async parse(doc: RawDocument, ctx: ConnectorContext, params: Record<string, unknown> = {}): Promise<ExtractedRecord[]> { if (shouldSkipDocument(doc)) return [indexSentinel(ctx.connectorId, doc.finalUrl)]; const p: PressReleaseParams = { keepText: params.keepText !== false, minProjectMw: numOr(params.minProjectMw), minInvestmentUsd: numOr(params.minInvestmentUsd), minAcres: numOr(params.minAcres), urlDate: typeof params.urlDate === "string" ? params.urlDate : undefined }; let content: ArticleContent; try { content = await articleContent(doc); } catch (e) { ctx.log("warn", `news_press_release_v1: cannot read ${doc.finalUrl}: ${(e as Error).message}`); return []; } if (!content.text || content.text.length < 80) return []; if (isPdf(doc) && content.titleMethod !== "pdf:meta-title") { const h = pdfHeadline(content.text); if (h) content = { ...content, title: h, titleMethod: "pdf:headline" }; } // navigation (HTML) or letterhead (PDF) before the headline is not part of the release; a summary that merely // repeats that prefix (text lead, or a CMS-generated meta description) is replaced by the lead after the headline const body = stripBeforeHeadline(content.text, content.title); if (body !== content.text) { const lead = cleanText(body.slice(0, 600)); const repeatsPrefix = content.summary != null && (content.summaryMethod === "text:lead" || squash(content.text).startsWith(squash(content.summary).slice(0, 60))); content = { ...content, text: body, ...(repeatsPrefix && lead ? { summary: lead, summaryMethod: "text:lead_after_headline" } : {}) }; } if (!content.published && p.urlDate) { const d = dateFromUrl(doc.finalUrl, p.urlDate); if (d) content = { ...content, published: d, publishedMethod: "url:date" }; } const a = extractAnnouncement(content.title, content.text, { publishedAt: content.published }); if (a.relevance === "none") { ctx.log("debug", `news_press_release_v1: not about data centers, skipped ${doc.finalUrl}`); return []; } if (!p.keepText && (a.relevance === "weak" || !a.leadRelevance)) { ctx.log("debug", `news_press_release_v1: off-topic for a third-party publisher, skipped ${doc.finalUrl}`); return [indexSentinel(ctx.connectorId, doc.finalUrl)]; } return buildRecords({ connectorId: ctx.connectorId, url: doc.finalUrl, a, content, params: p }); }, }; /** Lower-case, whitespace-free form for prefix comparisons across differently wrapped extractions. */ function squash(s: string): string { return s.toLowerCase().replace(/\s+/g, ""); } function numOr(v: unknown): number | undefined { const n = typeof v === "number" ? v : Number(v); return Number.isFinite(n) && v != null ? n : undefined; }