SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
6 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
8.2 KB · 122 lines typescript
Raw Blame History
1import type { ConnectorContext, ExtractedRecord, Parser, RawDocument } from "@dci/connectors";2import { isPdf } from "@dci/connectors";3import { cleanText } from "@dci/core";4import { articleContent, buildRecords, indexSentinel, shouldSkipDocument, type ArticleContent, type ArticleParams } from "./article-parser.js";5import { EXTRACTOR_VERSION, extractAnnouncement } from "./extract-project.js";67/**8 * `news_press_release_v1` — first-party press releases from grid operators, utilities and agencies that publish9 * them as PDF letterhead files (PJM) or as HTML pages without any date markup (ERCOT). Emits exactly the records of10 * `news_article_v1` (one `news_event`, plus one `project` for material announcements) with two additions:11 *12 *   - PDF headline: when the PDF carries no metadata title, the first line of the text is letterhead boilerplate13 *     ("Contact: …", "NEWS RELEASE", "FOR IMMEDIATE RELEASE", e-mails, phone numbers). The headline is the first14 *     line that is not boilerplate; a wrapped headline is re-joined when the next line starts in lower case.15 *   - Body starts at the headline: site navigation (HTML mega-menus, breadcrumbs) or letterhead (PDF) that precedes16 *     the headline is dropped from the kept text and from the lead summary.17 *   - `urlDate`: a regex with named groups `y`, `m`, `d` applied to the URL when the document itself gives no date18 *     (ERCOT slugs start with MMDDYYYY). Only a plausible past date (2000 … tomorrow) is accepted.19 *20 * Params (YAML `extractors.<x>.params`):21 *   keepText          default TRUE (first-party public sources) — set false for third-party publishers22 *   urlDate           regex source, e.g. "/news/release/(?<m>\\d{2})(?<d>\\d{2})(?<y>\\d{4})-"23 *   minProjectMw / minInvestmentUsd / minAcres — same gates as `news_article_v1`24 */25export interface PressReleaseParams extends ArticleParams { urlDate?: string }2627const MONTHS = "January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec";28/** Letterhead / masthead lines that are never the headline of a release. */29const BOILERPLATE_LINE_RE = new RegExp(30  `^(?:(?:media |press |news )?contacts?\\b|for (?:immediate )?release\\b|embargoed\\b|(?:news|press|media) release\\b|news\\b|release\\b|page \\d|\\d+ of \\d+|www\\.|https?://|[\\w.+-]+@[\\w-]+\\.|\\(?\\d{3}\\)?[ .-]?\\d{3}[ .-]\\d{4}|\\+?\\d[\\d ()-]{8,}\\d$|(?:\\d{1,2}(?:st|nd|rd|th)?\\s+(?:${MONTHS})\\.?,?\\s+20\\d\\d|(?:${MONTHS})\\.?\\s+\\d{1,2}(?:st|nd|rd|th)?,?\\s+20\\d\\d|20\\d\\d-\\d{2}-\\d{2})\\s*$|\\(?[A-Z][A-Za-z .]+,\\s*[A-Z]{2}\\.?\\s*[–—-])`,31  "i",32);3334/** First headline-like line of a PDF press release (null when nothing qualifies in the first 40 lines). */35export function pdfHeadline(text: string): string | null {36  const lines = text.split(/\n+/).map((l) => cleanText(l) ?? "").filter(Boolean).slice(0, 40);37  for (let i = 0; i < lines.length; i++) {38    const l = lines[i]!;39    if (l.length < 15 || l.length > 220) continue;40    if (!/[a-z]/.test(l) || l.split(/\s+/).length < 3) continue; // all-caps mastheads, single words41    if (BOILERPLATE_LINE_RE.test(l)) continue;42    if (/[:|]\s*$/.test(l)) continue; // "Contact:" style labels43    const next = lines[i + 1];44    if (next && !/[.!?:;)]$/.test(l) && /^[a-z]/.test(next) && l.length + next.length < 220) return `${l} ${next}`;45    return l;46  }47  return null;48}4950/**51 * Drop site navigation / letterhead that precedes the headline in the extracted text: the body starts at the last52 * occurrence of the headline inside the first 40 % of the text (breadcrumb, then the real heading). Returns the text53 * unchanged when the headline is not found there, or when what precedes it is too short to be navigation.54 */55export function stripBeforeHeadline(text: string, title: string | null): string {56  if (!title || title.length < 12 || text.length < 400) return text;57  const norm = (s: string) => s.toLowerCase().replace(/[‘’‚′`´]/g, "'").replace(/[“”„]/g, '"').replace(/ /g, " ");58  // `<title>` values carry a site suffix ("… - Southwest Power Pool"); match on the headline part only, ≤ 60 chars59  const head = title.split(/\s+[|–—]\s+|\s+-\s+(?=[A-Z])/)[0]!.trim();60  if (head.length < 12) return text;61  const needle = norm(head.slice(0, 60));62  const hay = norm(text);63  // the headline must sit in the first 70 % of the text and leave a body of at least 200 characters64  const limit = Math.min(Math.floor(text.length * 0.7), text.length - needle.length - 200);65  let idx = -1, from = 0;66  for (;;) { const i = hay.indexOf(needle, from); if (i < 0 || i > limit) break; idx = i; from = i + 1; }67  return idx >= 20 ? text.slice(idx) : text;68}6970/** "YYYY-MM-DD" from a URL when `pattern` (named groups y, m, d) matches and the date is plausible; null otherwise. */71export function dateFromUrl(url: string, pattern: string, now = new Date()): string | null {72  let re: RegExp;73  try { re = new RegExp(pattern, "i"); } catch { return null; }74  const g = url.match(re)?.groups;75  if (!g?.y || !g.m || !g.d) return null;76  const y = Number(g.y.length === 2 ? `20${g.y}` : g.y), m = Number(g.m), d = Number(g.d);77  if (!Number.isInteger(y) || !Number.isInteger(m) || !Number.isInteger(d) || y < 2000 || m < 1 || m > 12 || d < 1 || d > 31) return null;78  const iso = `${y}-${String(m).padStart(2, "0")}-${String(d).padStart(2, "0")}`;79  const t = new Date(`${iso}T00:00:00Z`).getTime();80  if (!Number.isFinite(t) || t > now.getTime() + 86_400_000) return null;81  return iso;82}8384export const pressReleaseParser: Parser = {85  name: "news_press_release_v1",86  version: EXTRACTOR_VERSION,87  async parse(doc: RawDocument, ctx: ConnectorContext, params: Record<string, unknown> = {}): Promise<ExtractedRecord[]> {88    if (shouldSkipDocument(doc)) return [indexSentinel(ctx.connectorId, doc.finalUrl)];89    const p: PressReleaseParams = { keepText: params.keepText !== false, minProjectMw: numOr(params.minProjectMw), minInvestmentUsd: numOr(params.minInvestmentUsd), minAcres: numOr(params.minAcres), urlDate: typeof params.urlDate === "string" ? params.urlDate : undefined };90    let content: ArticleContent;91    try { content = await articleContent(doc); } catch (e) { ctx.log("warn", `news_press_release_v1: cannot read ${doc.finalUrl}: ${(e as Error).message}`); return []; }92    if (!content.text || content.text.length < 80) return [];9394    if (isPdf(doc) && content.titleMethod !== "pdf:meta-title") {95      const h = pdfHeadline(content.text);96      if (h) content = { ...content, title: h, titleMethod: "pdf:headline" };97    }98    // navigation (HTML) or letterhead (PDF) before the headline is not part of the release; a summary that merely99    // repeats that prefix (text lead, or a CMS-generated meta description) is replaced by the lead after the headline100    const body = stripBeforeHeadline(content.text, content.title);101    if (body !== content.text) {102      const lead = cleanText(body.slice(0, 600));103      const repeatsPrefix = content.summary != null && (content.summaryMethod === "text:lead" || squash(content.text).startsWith(squash(content.summary).slice(0, 60)));104      content = { ...content, text: body, ...(repeatsPrefix && lead ? { summary: lead, summaryMethod: "text:lead_after_headline" } : {}) };105    }106    if (!content.published && p.urlDate) {107      const d = dateFromUrl(doc.finalUrl, p.urlDate);108      if (d) content = { ...content, published: d, publishedMethod: "url:date" };109    }110111    const a = extractAnnouncement(content.title, content.text, { publishedAt: content.published });112    if (a.relevance === "none") { ctx.log("debug", `news_press_release_v1: not about data centers, skipped ${doc.finalUrl}`); return []; }113    if (!p.keepText && (a.relevance === "weak" || !a.leadRelevance)) { ctx.log("debug", `news_press_release_v1: off-topic for a third-party publisher, skipped ${doc.finalUrl}`); return [indexSentinel(ctx.connectorId, doc.finalUrl)]; }114    return buildRecords({ connectorId: ctx.connectorId, url: doc.finalUrl, a, content, params: p });115  },116};117118/** Lower-case, whitespace-free form for prefix comparisons across differently wrapped extractions. */119function squash(s: string): string { return s.toLowerCase().replace(/\s+/g, ""); }120121function numOr(v: unknown): number | undefined { const n = typeof v === "number" ? v : Number(v); return Number.isFinite(n) && v != null ? n : undefined; }122