spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import { load } from "cheerio";2import type { NormalizedEntity, NormalizedFacility, NormalizedNewsEvent, NormalizedProject, ValidationReport, ValidationIssue, PageType, FacilityStatus, FacilityType } from "@dci/core";3import { parseAllMw, parseMoney, parsePartialDate, validLatLng, inferStatus, inferFacilityType, countryFromText, cleanText, sha256, urlFingerprint } from "@dci/core";4import type { ConnectorConfig, ExtractorConfig } from "./config.js";5import type { Connector, ConnectorContext, DiscoveredUrl, ExtractedRecord, RawDocument } from "./types.js";6import { discoverGeneric } from "./discovery.js";7import { evalRule, pageTitle, mainText, publishedDate, extractGeo, extractAddress, pdfText, isPdf, metaDescription } from "./extract.js";8import { classifyPage, isDataCenterRelevant } from "./classify.js";9import { getParser } from "./registry.js";1011/**12 * Generic, config-driven connector. Covers: sitemap/RSS/seed discovery, escalating fetch, declarative13 * field extraction (selectors / JSON-LD / embedded JSON / regex), code-backed parsers, article14 * classification for newsrooms, PDF text, normalization and validation. Source-specific connectors15 * either extend this via YAML or register an `implementation`.16 */17export class GenericConnector implements Connector {18 id: string; sourceName: string; sourceDomain: string; sourceKind: Connector["sourceKind"]; type: Connector["type"]; parserVersion: string; schedule: Connector["schedule"]; license: string | null; attribution: string | null; priority: number;19 constructor(readonly cfg: ConnectorConfig) {20 this.id = cfg.id; this.sourceName = cfg.name; this.sourceDomain = cfg.domain; this.sourceKind = cfg.kind; this.type = cfg.mode; this.parserVersion = cfg.parserVersion; this.schedule = cfg.schedule; this.license = cfg.license ?? null; this.attribution = cfg.attribution ?? null; this.priority = cfg.priority;21 }2223 async discover(ctx: ConnectorContext): Promise<DiscoveredUrl[]> { return discoverGeneric(this.cfg, ctx); }2425 async fetch(ctx: ConnectorContext, u: DiscoveredUrl): Promise<RawDocument> {26 const doc = await ctx.fetch(u.url, { group: u.group, level: u.minLevel, accept: u.pageType === "planning_document" ? "application/pdf,text/html,*/*" : undefined });27 doc.group = u.group; doc.pageType = u.pageType; doc.meta = { ...(doc.meta ?? {}), ...(u.meta ?? {}) };28 return doc;29 }3031 /** Pick extractors whose `match` regex hits the URL (or whose pageTypes include the doc's page type). */32 selectExtractors(doc: RawDocument, pageType: PageType): Array<[string, ExtractorConfig]> {33 const out: Array<[string, ExtractorConfig]> = [];34 for (const [name, ex] of Object.entries(this.cfg.extractors)) {35 if (ex.match?.length && !ex.match.some((m) => new RegExp(m, "i").test(doc.finalUrl) || new RegExp(m, "i").test(doc.url))) continue;36 if (ex.pageTypes?.length && !ex.pageTypes.includes(pageType)) continue;37 out.push([name, ex]);38 }39 return out;40 }4142 async extract(ctx: ConnectorContext, doc: RawDocument): Promise<ExtractedRecord[]> {43 if (doc.error || doc.notModified || doc.status >= 400 || !doc.body.length) return [];44 let html = doc.text;45 let text = "";46 let title: string | null = null;47 if (isPdf(doc)) {48 try { const p = await pdfText(doc.body); text = p.text; title = p.title ?? doc.url.split("/").pop() ?? null; html = ""; } catch (e) { ctx.log("warn", `pdf parse failed ${doc.url}: ${(e as Error).message}`); return []; }49 } else if (/html|xml/.test(doc.contentType ?? "") || /<html/i.test(html.slice(0, 2000))) {50 title = pageTitle(html);51 text = mainText(html);52 } else if (doc.markdown) { text = doc.markdown; }53 else text = doc.text;54 const cls = classifyPage(doc.finalUrl, title, text);55 const pageType: PageType = doc.pageType && doc.pageType !== "unknown" && !["press_release", "news_index"].includes(doc.pageType) ? doc.pageType : cls.pageType;56 doc.meta = { ...(doc.meta ?? {}), title, pageType, classifier: cls.rule, textLength: text.length };57 const records: ExtractedRecord[] = [];58 const extractors = this.selectExtractors(doc, pageType);59 for (const [name, ex] of extractors) {60 if (ex.parser) {61 const parser = getParser(ex.parser);62 const recs = await parser.parse(doc, ctx, { ...(ex.params ?? {}), _sourceKind: this.sourceKind });63 for (const r of recs) records.push({ ...r, pageType: r.pageType ?? pageType });64 continue;65 }66 if (ex.fields && html) {67 const $ = load(html);68 const scopes = ex.each ? $(ex.each).toArray().map((el) => $(el)) : [undefined];69 for (const scope of scopes) {70 const data: Record<string, unknown> = {};71 const methods: Record<string, string> = {};72 for (const [field, rule] of Object.entries(ex.fields)) {73 const { value, method } = evalRule($, html, rule, scope);74 if (value != null && value !== "") { data[field] = value; methods[field] = method; }75 }76 if (!Object.keys(data).length) continue;77 const key = ex.key ? ex.key.replace(/\{(\w+)\}/g, (_, k: string) => (k === "url" ? doc.finalUrl : String(data[k] ?? ""))) : `${this.id}:${urlFingerprint(doc.finalUrl)}`;78 const certainty = Math.min(1, Object.keys(data).length / Math.max(3, Object.keys(ex.fields).length * 0.6));79 if (ex.minCertainty && certainty < ex.minCertainty) continue;80 records.push({ kind: ex.kind, key, data: { ...data, _title: title, _url: doc.finalUrl, _extractor: name }, methods, certainty, url: doc.finalUrl, pageType });81 }82 }83 }84 // Newsroom / announcement pages with no dedicated extractor: produce a news_event record when relevant.85 // Third-party publishers (kind = news) keep only title, date, link and a ≤ 400-char summary — never the text.86 if (!records.length && ["press_release", "project_announcement", "acquisition", "expansion", "closure", "construction_update", "planning_document", "power_infrastructure", "cloud_region", "financial_disclosure"].includes(pageType) && isDataCenterRelevant(`${title ?? ""} ${text.slice(0, 5000)}`)) {87 const thirdParty = this.sourceKind === "news";88 const summary = trimSummary(cleanText((doc.meta?.summary as string | undefined) ?? metaDescription(html) ?? text.slice(0, 400)), THIRD_PARTY_SUMMARY_MAX);89 const published = (doc.meta?.published as string | undefined) ? parsePartialDate(String(doc.meta?.published)) : html ? publishedDate(html) : null;90 const money = parseMoney(`${title ?? ""} ${text.slice(0, 3000)}`);91 records.push({ kind: "news_event", key: `${this.id}:news:${urlFingerprint(doc.finalUrl)}`, data: { title: title ?? doc.finalUrl, url: doc.finalUrl, publishedAt: published, summary, pageType, eventType: cls.eventType, mw: cls.mw, status: cls.status, money, text: thirdParty ? null : text.slice(0, 20_000) }, methods: { title: "html:title", publishedAt: "meta:published", mw: "regex:mw_v1" }, certainty: 0.6, url: doc.finalUrl, pageType });92 }93 return records;94 }9596 async normalize(ctx: ConnectorContext, records: ExtractedRecord[]): Promise<NormalizedEntity[]> {97 const out: NormalizedEntity[] = [];98 const d = this.cfg.defaults ?? {};99 for (const r of records) {100 const prov = ctx.provenance(r.url, { method: r.methods ? Object.values(r.methods).slice(0, 3).join(",") : undefined, extractorVersion: this.parserVersion });101 const x = r.data;102 const str = (k: string): string | null => { const v = x[k]; return v == null ? null : cleanText(String(v)); };103 const num = (k: string): number | null => { const v = x[k]; if (v == null || v === "") return null; const n = typeof v === "number" ? v : Number(String(v).replace(/[^\d.-]/g, "")); return Number.isFinite(n) ? n : null; };104 if (r.kind === "facility") {105 const lat = num("lat"), lng = num("lng");106 const geo = validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: (str("geoPrecision") as NormalizedFacility["geo"] extends infer G ? (G extends { precision: infer P } ? P : never) : never) ?? "exact", source: `operator:${this.id}` } : null;107 const name = str("name") ?? str("_title");108 if (!name) continue;109 const f: NormalizedFacility = {110 entityType: "facility", key: r.key, name, aliases: Array.isArray(x.aliases) ? (x.aliases as string[]) : [],111 operatorName: str("operatorName") ?? d.operatorName ?? null, campusName: str("campusName"), address: str("address"), city: str("city"), regionName: str("regionName"),112 countryIso2: (str("countryIso2") ?? (str("country") ? (/^[A-Z]{2}$/.test(str("country")!) ? str("country") : countryFromText(str("country"))) : null) ?? d.countryIso2 ?? null) as string | null,113 postalCode: str("postalCode"), geo,114 status: ((str("status") as FacilityStatus | null) ?? (d.status as FacilityStatus | undefined) ?? inferStatus(str("statusText")) ?? "operational"),115 facilityType: ((str("facilityType") as FacilityType | null) ?? (d.facilityType as FacilityType | undefined) ?? inferFacilityType(`${name} ${str("description") ?? ""}`) ?? "colocation"),116 tier: str("tier"), buildingSqm: num("buildingSqm"), siteAreaHa: num("siteAreaHa"), itCapacityMw: num("itCapacityMw"), totalPowerMw: num("totalPowerMw"), plannedPowerMw: num("plannedPowerMw"), rackCount: num("rackCount"), pue: num("pue"),117 coolingType: str("coolingType"), renewableClaim: str("renewableClaim"), openedOn: str("openedOn") ? parsePartialDate(str("openedOn")) : null, constructionStartedOn: str("constructionStartedOn") ? parsePartialDate(str("constructionStartedOn")) : null, announcedOn: str("announcedOn") ? parsePartialDate(str("announcedOn")) : null,118 website: str("website") ?? r.url, isAi: typeof x.isAi === "boolean" ? x.isAi : null, isHyperscale: typeof x.isHyperscale === "boolean" ? x.isHyperscale : d.isHyperscale ?? null,119 certifications: Array.isArray(x.certifications) ? (x.certifications as string[]) : [], carriers: Array.isArray(x.carriers) ? (x.carriers as string[]) : [], cloudProviders: Array.isArray(x.cloudProviders) ? (x.cloudProviders as string[]) : [], ixps: Array.isArray(x.ixps) ? (x.ixps as string[]) : [],120 externalIds: (x.externalIds as Record<string, string | number> | undefined) ?? {}, description: str("description"), provenance: prov,121 };122 // when no explicit geo but page has JSON-LD/meta geo, the extractor may already have set lat/lng; otherwise leave null (never fake coordinates)123 out.push(f);124 } else if (r.kind === "news_event") {125 const n: NormalizedNewsEvent = { entityType: "news_event", key: r.key, title: str("title") ?? r.url, url: (str("url") ?? r.url)!, publishedAt: str("publishedAt"), summary: str("summary"), pageType: (str("pageType") as PageType | null) ?? r.pageType, eventType: (str("eventType") as NormalizedNewsEvent["eventType"]) ?? undefined, mentions: { mw: Array.isArray(x.mw) ? (x.mw as number[]) : [], operators: [...(Array.isArray(x.operators) ? (x.operators as string[]) : []), ...(d.operatorName ? [d.operatorName] : [])], countriesIso2: [...(Array.isArray(x.countries) ? (x.countries as string[]) : []), countryFromText(`${str("title")} ${str("summary")}`) ?? ""].filter((c, i, a) => c && a.indexOf(c) === i), cities: Array.isArray(x.cities) ? (x.cities as string[]) : [] }, projectClass: str("projectClass"), isAi: typeof x.isAi === "boolean" ? x.isAi : null, provenance: prov };126 out.push(n);127 // Project candidate from a fallback news_event (no dedicated parser). Third-party publishers (kind = news) go128 // through `news_article_v1`, which has the real gate — the fallback never creates projects for them. For129 // operator / government newsrooms the announcement must be visible in the HEADLINE: a MW figure in the title130 // and a pipeline status keyword in the title (appointments, awards and financing stories quote MW in the body).131 const gate = this.sourceKind === "news" ? null : fallbackProjectGate(n.title, n.summary);132 const mw = gate?.mw;133 const st = gate?.status as FacilityStatus | undefined;134 if (gate && mw && st && ["project_announcement", "expansion", "construction_update", "planning_document", "press_release"].includes(n.pageType ?? "")) {135 const money = x.money as { amount: number; currency: string } | null;136 const p: NormalizedProject = { entityType: "project", key: `${r.key}:project`, name: n.title.slice(0, 160), operatorName: d.operatorName ?? null, countryIso2: n.mentions?.countriesIso2?.[0] ?? d.countryIso2 ?? null, status: st, announcedOn: n.publishedAt, plannedMw: mw, investmentUsd: money && money.currency === "USD" ? money.amount : null, description: n.summary, sourceUrl: n.url, timeline: n.publishedAt ? [{ date: n.publishedAt, type: st === "under_construction" ? "construction_started" : "project_announced", description: n.title, url: n.url }] : [], provenance: { ...prov, confidence: "moderate", method: "regex:announcement_v1" } };137 out.push(p);138 }139 } else if (r.kind === "project") {140 const name = str("name") ?? str("_title"); if (!name) continue;141 const lat = num("lat"), lng = num("lng");142 const claimContext = x.claimContext && typeof x.claimContext === "object" ? (x.claimContext as Record<string, string>) : undefined;143 out.push({ entityType: "project", key: r.key, name, operatorName: str("operatorName") ?? d.operatorName ?? null, city: str("city"), regionName: str("regionName"), countryIso2: str("countryIso2") ?? d.countryIso2 ?? null, geo: validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: "approximate", source: this.id } : null, status: (str("status") as FacilityStatus | null) ?? "announced", announcedOn: str("announcedOn"), expectedOpening: str("expectedOpening"), plannedMw: num("plannedMw"), investmentUsd: num("investmentUsd"), acreage: num("acreage"), phaseCount: num("phaseCount"), description: str("description"), sourceUrl: r.url,144 projectClass: str("projectClass"), evidenceLevel: (str("evidenceLevel") as NormalizedProject["evidenceLevel"]) ?? null, capacityScope: str("capacityScope"), capacitySemantics: str("capacitySemantics"), investmentScope: str("investmentScope"), investmentSemantics: str("investmentSemantics"), investmentCurrency: str("investmentCurrency"), investmentOriginal: num("investmentOriginal"), claimContext, aiEvidence: (str("aiEvidence") as NormalizedProject["aiEvidence"]) ?? null, developerName: str("developerName"), tenantName: str("tenantName"), campusName: str("campusName"), constructionStartedOn: str("constructionStartedOn"), approvedOn: str("approvedOn"), permitFiledOn: str("permitFiledOn"), provenance: prov });145 } else if (r.kind === "operator") {146 const name = str("name"); if (!name) continue;147 out.push({ entityType: "operator", key: r.key, name, website: str("website"), kind: (str("kind") as NormalizedEntity extends { kind?: infer K } ? K : never) ?? null, hqCountryIso2: str("hqCountryIso2"), description: str("description"), provenance: prov });148 } else if (r.kind === "cloud_region") {149 const code = str("code"), name = str("name"); if (!code || !name) continue;150 const lat = num("lat"), lng = num("lng");151 out.push({ entityType: "cloud_region", key: r.key, providerName: str("providerName") ?? d.providerName ?? this.sourceName, code, name, city: str("city"), regionName: str("regionName"), countryIso2: str("countryIso2") ?? (str("country") ? countryFromText(str("country")) : null), geo: validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: (str("geoPrecision") as "city" | "exact" | null) ?? "city", source: this.id } : null, availabilityZones: num("availabilityZones"), launchedOn: str("launchedOn") ? parsePartialDate(str("launchedOn")) : null, status: (str("status") as "announced" | "operational" | null) ?? "operational", isSovereign: Boolean(x.isSovereign), sourceUrl: r.url, provenance: prov });152 }153 }154 return out;155 }156157 async validate(_ctx: ConnectorContext, entities: NormalizedEntity[]): Promise<ValidationReport> { return validateEntities(entities); }158}159160/** Third-party publishers: the stored summary never exceeds this many characters (copyright). */161export const THIRD_PARTY_SUMMARY_MAX = 400;162export function trimSummary(s: string | null, max = THIRD_PARTY_SUMMARY_MAX): string | null {163 if (!s || s.length <= max) return s;164 const cut = s.slice(0, max - 1);165 return `${cut.slice(0, Math.max(cut.lastIndexOf(" "), max - 40)).trim()}…`;166}167168/** Headlines about people, awards, deals, money, markets or policy — a MW figure in them is never a new site (fallback path). */169export const NON_PROJECT_HEADLINE_RE = /\b(appoint|names? .+ as|joins|award|recogni[sz]ed|honou?red|named .+ of the year|certified|interview|webinar|podcast|featured in|partnership|partners with|join forces|work together|agreement|contract|ppa\b|peering|pop at|financ|refinanc|raises \$|series [a-e]\b|loan|bond|esg\b|results|earnings|revenue|report|survey|market (to|will|set to) |lawsuit|sues|bill would|tax|moratorium|opens|opened|now (open|online|live)|we are online|delivered|celebrat|headquarters|head office|offices?\b|why |how |what )/i;170/** Headlines that describe planning or building a facility (fallback path). */171export const BUILD_HEADLINE_RE = /\b(breaks? ground|broke ground|groundbreaking|to build|will build|builds?|build of|plans? (for|to|new|\d)|planned|proposes?|proposed|approv(es|ed|al)|files? (plans|for)|rezon\w+|to develop|develops?|to construct|construction|unveils?|announces? (plans|new|a new|two|three|its|\d|\$|groundbreaking)|to invest|invests?|expansion|expands?|expanding|adds? \d+ ?\+?(mw|gw)|new (\d+ ?\+?(mw|gw)|hyperscale|data ?cent(er|re)s?|campus|facilit(y|ies))|(\d+ ?\+?(mw|gw))[- ](campus|data ?cent(er|re)|facility|site|project|hyperscale|phase)|topped out|topping out|enters? [A-Z]\w+ with|first data ?cent(er|re) in|further facilit(y|ies)|additional \d+ ?mw|second|third|fourth|fifth|phase \d)\b/i;172173/**174 * Gate for the fallback project candidate (pages with no dedicated parser). The announcement must be visible up175 * front: a build / plan headline that is not a people / deal / money story, a pipeline status keyword in the176 * headline, and a MW figure in the headline or the summary (the first ≤ 400 characters). Returns the figure and177 * status to use, or null.178 */179export function fallbackProjectGate(title: string, summary: string | null | undefined): { mw: number; status: FacilityStatus } | null {180 if (!title || NON_PROJECT_HEADLINE_RE.test(title) || !BUILD_HEADLINE_RE.test(title)) return null;181 const st = inferStatus(title) as FacilityStatus | null;182 if (!st || !["announced", "proposed", "permitting", "approved", "under_construction", "expansion", "rumored", "delayed"].includes(st)) return null;183 const plausible = (v: number) => v > 0 && v <= 20_000;184 const mw = parseAllMw(title).filter(plausible)[0] ?? parseAllMw(`${title} ${summary ?? ""}`.slice(0, 600)).filter(plausible)[0];185 return mw ? { mw, status: st } : null;186}187188/** Facility names that are page / article titles or listings rather than a site ("Data Centers in Virginia | Acme", "Our Locations"). */189export const TITLE_LIKE_NAME_RE = /\s[|]\s|\?$|^(?:how|why|what|when|where|inside|top \d+|the \d+ (?:best|largest))\b|\b(?:data cent(?:er|re)s? (?:in|near|for|across|around) [A-Z]|colocation (?:in|near) [A-Z]|(?:our|all|view|browse|explore) (?:data cent(?:er|re)s?|sites|facilities|locations)|locations\b|list of|news(?:room)?|blog|contact us|privacy|cookie|login|sign in|search results?|page not found|404|about us|careers?|press releases?|sitemap|home ?page|learn more|read more|coming soon)\b/i;190/** Names that describe a business other than a data center (crowd-sourced tags). */191export const NOT_A_FACILITY_NAME_RE = /\b(process server|training cent(?:er|re)|kursu|web (?:design|developer|development)|software developer|repair|shop|store|cafe|restaurant|hotel|school|clinic|pharmacy|translation|law firm|accounting|real estate agency)\b/i;192193/** Shared validation: hard errors reject the record; warnings are recorded. Never accept impossible figures. */194export function validateEntities(entities: NormalizedEntity[]): ValidationReport {195 const issues: ValidationIssue[] = [];196 let rejected = 0;197 for (const e of entities) {198 const errs: ValidationIssue[] = [];199 const warn = (field: string, message: string) => issues.push({ key: e.key, field, level: "warn", message });200 const err = (field: string, message: string) => errs.push({ key: e.key, field, level: "error", message });201 if (e.entityType === "facility") {202 if (!e.name || e.name.length < 2 || e.name.length > 200) err("name", "missing or implausible name");203 else {204 if (TITLE_LIKE_NAME_RE.test(e.name)) err("name", "name looks like a page or article title, not a facility");205 if (NOT_A_FACILITY_NAME_RE.test(e.name)) err("name", "name describes another kind of business");206 if (e.name.length > 120) warn("name", "unusually long facility name");207 }208 if (e.geo && !validLatLng(e.geo.lat, e.geo.lng)) err("geo", "invalid coordinates");209 if (e.geo && e.geo.precision === "exact" && Number.isInteger(e.geo.lat) && Number.isInteger(e.geo.lng)) err("geo", "integer coordinates cannot be exact");210 for (const f of ["itCapacityMw", "totalPowerMw", "plannedPowerMw"] as const) { const v = e[f]; if (v != null && (v <= 0 || v > 10_000)) err(f, `implausible MW ${v}`); }211 // a single building above 1 000 MW is a campus-level or portfolio figure unless the record says "campus"212 const isCampus = /\b(campus|park|cluster|hub|gigafactory|complex)\b/i.test(`${e.name} ${e.campusName ?? ""}`);213 for (const f of ["itCapacityMw", "totalPowerMw"] as const) { const v = e[f]; if (v != null && v > 1_000 && !isCampus) err(f, `single-facility ${f} ${v} MW above 1 000 MW without a campus designation`); }214 if (e.plannedPowerMw != null && e.plannedPowerMw > 5_000 && !isCampus) warn("plannedPowerMw", `planned ${e.plannedPowerMw} MW for a single facility`);215 if (e.itCapacityMw != null && e.totalPowerMw != null && e.itCapacityMw > e.totalPowerMw * 1.05) warn("itCapacityMw", "IT capacity exceeds total power");216 if (e.pue != null && (e.pue < 1 || e.pue > 3)) err("pue", `implausible PUE ${e.pue}`);217 if (e.buildingSqm != null && (e.buildingSqm < 20 || e.buildingSqm > 2_000_000)) err("buildingSqm", "implausible building size");218 if (e.buildingSqm != null && e.buildingSqm > 500_000 && !isCampus) err("buildingSqm", `building ${e.buildingSqm} m² above 500 000 m² without a campus designation`);219 if (e.siteAreaHa != null && (e.siteAreaHa <= 0 || e.siteAreaHa > 20_000)) err("siteAreaHa", `implausible site area ${e.siteAreaHa} ha`);220 if (e.rackCount != null && (e.rackCount <= 0 || e.rackCount > 100_000)) err("rackCount", `implausible rack count ${e.rackCount}`);221 if (e.countryIso2 && !/^[A-Z]{2}$/.test(e.countryIso2)) err("countryIso2", "bad country code");222 if (!e.countryIso2 && !e.geo) warn("countryIso2", "no country and no coordinates");223 if (e.openedOn && Number(e.openedOn.slice(0, 4)) < 1950) err("openedOn", "opening year before 1950");224 if (e.openedOn && Number(e.openedOn.slice(0, 4)) < 1980 && e.provenance.connectorId !== "wikidata") warn("openedOn", "opening year before 1980 from a non-Wikidata source (building date?)");225 for (const f of ["openedOn", "constructionStartedOn", "announcedOn"] as const) { const v = e[f]; if (v && Number(v.slice(0, 4)) > new Date().getFullYear() + 12) err(f, `${f} ${v} too far in the future`); }226 if (e.openedOn && e.constructionStartedOn && e.openedOn.slice(0, 4) < e.constructionStartedOn.slice(0, 4)) warn("openedOn", "opened before construction started");227 } else if (e.entityType === "project") {228 if (!e.name) err("name", "missing name");229 else if (TITLE_LIKE_NAME_RE.test(e.name)) warn("name", "project name looks like an article title");230 if (e.plannedMw != null && (e.plannedMw <= 0 || e.plannedMw > 20_000)) err("plannedMw", `implausible MW ${e.plannedMw}`);231 if (e.investmentUsd != null && (e.investmentUsd < 1e5 || e.investmentUsd > 1e12)) warn("investmentUsd", "investment outside plausible range");232 if (e.investmentUsd != null && e.investmentUsd > 200e9) err("investmentUsd", `investment ${e.investmentUsd} above $200B is an industry statistic`);233 if (e.expectedOpening && Number(e.expectedOpening.slice(0, 4)) > new Date().getFullYear() + 15) err("expectedOpening", `expected opening ${e.expectedOpening} too far in the future`);234 if (!e.countryIso2 && !e.geo) warn("countryIso2", "no location");235 } else if (e.entityType === "cloud_region") {236 if (!e.code || !e.name) err("code", "missing code/name");237 if (e.geo && !validLatLng(e.geo.lat, e.geo.lng)) err("geo", "invalid coordinates");238 } else if (e.entityType === "news_event") {239 if (!/^https?:\/\//.test(e.url)) err("url", "bad url");240 if (!e.title) err("title", "missing title");241 }242 if (errs.length) { rejected++; issues.push(...errs); }243 }244 return { total: entities.length, valid: entities.length - rejected, rejected, issues };245}246247export function entityIsValid(report: ValidationReport, key: string): boolean { return !report.issues.some((i) => i.key === key && i.level === "error"); }248249export function newsKey(connectorId: string, url: string): string { return `${connectorId}:news:${sha256(url).slice(0, 24)}`; }250