SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
2 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%

Phase 1 — claim-first data layer: claims + quality flags + snapshots (migration 0003), scope/semantics/authority classifiers, announcement class veto + evidence threshold, capacity/investment sanity engines, HQ guard, containment-aware aggregation, effective parser versions, connector quarantine + health states, extraction debugger (dci trace), quality sweep + regression checks, parseMw decimal fix, API contract 2.0 types

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Simon-Pierre Boucher committed 20 days ago (Sep 12, 2026) parent d55af7a

42 changed files +3,188 −163

modified apps/worker/src/cli.ts +25 −2
@@ -8,7 +8,9 @@
8 8 * discover <id> [--dry-run]
9 9 * docs <id> [--due] [--limit n] [--group g]
10 10 * inspect <url-or-docId> document row + latest version + provenance rows referencing it
11 − * reprocess <id> [--limit n] [--group g] re-extract from archived bodies (no network)
11 + * reprocess <id> [--limit n] [--group g] [--stale] re-extract from archived bodies (no network); --stale = older parser version only
12 + * trace <doc-id|url> [--live] [--json] extraction debugger: every pipeline stage for one document, nothing persisted
13 + * quarantine <id> [on|off] preview-only mode for a connector (ingest rolled back) · quality · snapshot · gaps
12 14 * stats global counts (entities, documents, runs, events, queues)
13 15 * doctor Postgres / Redis / ClickHouse / MinIO, budgets, stale connectors, error rates → system_alerts
14 16 * rank | metrics | refresh-stats maintenance jobs, run inline
@@ -31,6 +33,8 @@ import { computeRankings } from "./rankings.js";
31 33 import { closeScheduler, enqueueRun, pauseConnector, queueSnapshot, schedulerTick, startSchedulerLoop } from "./scheduler.js";
32 34 import { parseDiscoveredFrom } from "./scheduling.js";
33 35 import { getRaw } from "./storage.js";
36 +import { traceDocument } from "./trace.js";
37 +import { dataGaps, qualitySweep, snapshotAndCheck } from "./quality.js";
34 38 import { ensureClickHouse } from "@dci/db/clickhouse";
35 39
36 40 /* ---------- arg parsing ---------- */
@@ -137,7 +141,7 @@ async function cmdRun(a: Args): Promise<number> {
137 141 await ensureClickHouseQuiet();
138 142 if (!dryRun) await syncConnectorsToDb([requireConnector(id)]);
139 143 const urls = list(a, "url");
140 − const r = await runConnector(id, { task: urls.length ? "crawl" : taskOf(a, "full"), group: str(a, "group"), limit: num(a, "limit"), dryRun, urls: urls.length ? urls : undefined, force: bool(a, "force"), logLevel: bool(a, "verbose") ? "debug" : undefined, sampleEntities: dryRun ? num(a, "samples") ?? 20 : num(a, "samples") ?? 0 });
144 + const r = await runConnector(id, { task: urls.length ? "crawl" : taskOf(a, "full"), group: str(a, "group"), limit: num(a, "limit"), dryRun, urls: urls.length ? urls : undefined, force: bool(a, "force"), staleOnly: bool(a, "stale"), quarantine: bool(a, "quarantine"), logLevel: bool(a, "verbose") ? "debug" : undefined, sampleEntities: dryRun ? num(a, "samples") ?? 20 : num(a, "samples") ?? 0 });
141 145 printRun(r, bool(a, "json"));
142 146 return r.status === "failed" ? 1 : 0;
143 147 }
@@ -254,6 +258,20 @@ async function cmdValidateYaml(a: Args): Promise<number> {
254 258 return 0;
255 259 }
256 260
261 +function printTrace(t: Awaited<ReturnType<typeof traceDocument>>): void {
262 + const d = t.document ?? {};
263 + console.log(paint("bold", `SOURCE DOCUMENT ${String(d.id ?? "?")}`)); console.log(` ${String(d.url ?? "")}\n connector ${String(d.connectorId ?? "")} · pageType ${String(d.pageType ?? "")} · extractor ${String(d.extractorVersion ?? "")} · refs ${JSON.stringify(d.entityRefs ?? [])}`);
264 + if (t.fetch) console.log(`\n${paint("bold", "RAW FETCH")} ${t.fetch.source} · HTTP ${t.fetch.status} · ${t.fetch.contentType ?? "?"} · ${t.fetch.bytes} B · L${t.fetch.level} ${t.fetch.fetcher}`);
265 + console.log(`\n${paint("bold", "PARSED TEXT")} title: ${t.text.title ?? "—"} · ${t.text.length} chars · classified ${t.text.classification?.pageType ?? "?"} (${t.text.classification?.rule ?? ""}) · MW in text: ${t.text.classification?.mw.join(", ") || "none"}`);
266 + if (t.announcement) { const a = t.announcement as Record<string, unknown>; const c = a.classification as { class: string; mayCreateProject: boolean; evidence: Record<string, unknown> }; console.log(`\n${paint("bold", "ANNOUNCEMENT")} class ${paint(c.mayCreateProject ? "green" : "yellow", c.class)} · mayCreateProject ${c.mayCreateProject} · evidence ${JSON.stringify(c.evidence)}\n status ${String(a.status)} · headline MW ${String(a.headlineMw)} (scope ${String(a.capacityScope)}, ${String(a.capacitySemantics)}) · money ${JSON.stringify(a.money)} (${String(a.investmentScope)}/${String(a.investmentSemantics)}) · location ${JSON.stringify(a.location)} · operator ${String(a.operator)} · AI ${String(a.aiEvidence)}${a.hqGuarded ? " · HQ mention stripped" : ""}`); const ev = a.evidence as { plannedMw?: { text: string } | null; investment?: { text: string } | null }; if (ev?.plannedMw) console.log(paint("dim", ` MW evidence: "${ev.plannedMw.text.slice(0, 220)}"`)); if (ev?.investment) console.log(paint("dim", ` $ evidence: "${ev.investment.text.slice(0, 220)}"`)); }
267 + console.log(`\n${paint("bold", `STRUCTURED DATA ${t.records.length} record(s)`)}`); for (const r of t.records.slice(0, 20)) console.log(` ${r.kind.padEnd(11)} ${r.key} · certainty ${r.certainty ?? "—"} · ${Object.keys(r.data).filter((k) => r.data[k] != null && !k.startsWith("_")).slice(0, 12).join(", ")}`);
268 + console.log(`\n${paint("bold", `EXTRACTED ENTITIES ${t.entities.length} · valid ${t.validation.valid} · rejected ${t.validation.rejected}`)}`); for (const i of t.validation.issues.slice(0, 20)) console.log(paint(i.level === "error" ? "red" : "yellow", ` ${i.level} ${i.key}${i.field ? "." + i.field : ""}: ${i.message}`));
269 + console.log(`\n${paint("bold", `EXTRACTED CLAIMS ${t.claims.length}`)}`); for (const c of t.claims) console.log(` ${c.field.padEnd(14)} ${String(c.value).padStart(10)} ${c.unit} · scope ${paint(["building", "facility", "campus"].includes(c.scope) ? "green" : "yellow", c.scope)} (${c.scopeReason}) · ${c.semantics ?? "—"}${c.evidence ? paint("dim", `\n "${c.evidence.text.slice(0, 200)}"`) : paint("red", "\n no supporting sentence")}`);
270 + console.log(`\n${paint("bold", `MATCH CANDIDATES ${t.matches.length} facility record(s)`)}`); for (const m of t.matches) { console.log(` ${m.entityKey} → ${m.how}${m.matchedId ? ` ${m.matchedId}` : ""}`); for (const c of m.candidates.slice(0, 5)) console.log(paint("dim", ` ${c.score.toFixed(3)} ${c.name} (${c.operatorName ?? "—"}, ${c.city ?? "—"}) ${c.reasons.join(" ")}`)); }
271 + if (t.reconciliation) { const r = t.reconciliation; console.log(`\n${paint("bold", "RECONCILIATION (dry run)")} created ${r.created} · updated ${r.updated} · unchanged ${r.unchanged} · merged ${r.merged} · pending ${r.pendingMatches} · rejected ${r.rejected} · events ${r.events} · provenance ${r.provenanceRows} · claims ${r.claims} (${r.unscopedClaims} unscoped) · flags ${r.qualityFlags} · projects vetoed ${r.projectsVetoed}`); console.log(`\n${paint("bold", "RESULTING DATABASE CHANGES")}`); for (const c of r.changes.slice(0, 20)) console.log(` ${JSON.stringify(c).slice(0, 220)}`); for (const ref of r.refs.slice(0, 20)) console.log(paint("dim", ` ref ${ref.type} ${ref.id}`)); }
272 + if (t.error) console.log(paint("red", `\nERROR ${t.error}`));
273 +}
274 +
257 275 function help(): number {
258 276 console.log(readFileSync(new URL(import.meta.url), "utf8").split("\n").slice(1, 22).map((l) => l.replace(/^ \*\s?/, "")).join("\n"));
259 277 return 0;
@@ -284,6 +302,11 @@ async function main(): Promise<number> {
284 302 case "scheduler": return cmdScheduler(a);
285 303 case "clickhouse-init": await ensureClickHouse(); console.log("clickhouse tables ensured"); return 0;
286 304 case "validate": return cmdValidateYaml(a);
305 + case "trace": { const key = a.positional[0]; if (!key) throw new Error("usage: dci trace <doc-id-or-url> [--live] [--json]"); await registerAllConnectors(); const t = await traceDocument(key, { live: bool(a, "live") }); if (bool(a, "json")) { console.log(JSON.stringify(t, null, 2)); return t.error ? 1 : 0; } printTrace(t); return t.error ? 1 : 0; }
306 + case "quarantine": { const id = a.positional[0]; const on = (a.positional[1] ?? "on") !== "off"; if (!id) throw new Error("usage: dci quarantine <connector-id> [on|off]"); await getDb().execute(sql`update connectors set quarantine = ${on}, updated_at = now() where id = ${id}`); console.log(`${id}: quarantine ${on ? "ON (ingest rolled back, preview only)" : "OFF"}`); return 0; }
307 + case "quality": { const r = await qualitySweep(); console.log(JSON.stringify(r, null, 2)); return 0; }
308 + case "snapshot": { const r = await snapshotAndCheck(); console.log(JSON.stringify(r, null, 2)); return 0; }
309 + case "gaps": { const r = await dataGaps(); console.log(table(Object.entries(r).map(([k, v]) => ({ gap: k, count: v })), ["gap", "count"])); return 0; }
287 310 case "worker": { const { startWorker } = await import("./main.js"); const w = await startWorker(); await new Promise<void>((resolve) => { const stop = () => void w.stop().then(resolve); process.once("SIGINT", stop); process.once("SIGTERM", stop); }); return 0; }
288 311 case "help": case "--help": case "-h": return help();
289 312 default: console.error(paint("red", `unknown command "${a.cmd}"`)); help(); return 2;
modified apps/worker/src/configs.ts +38 −8
@@ -3,8 +3,8 @@
3 3 * GenericConnector) and mirror them into the `connectors` + `sources` tables for the admin UI / scheduler.
4 4 */
5 5 import { existsSync } from "node:fs";
6 −import { GenericConnector, getImplementation, loadConnectorConfigs, type Connector, type ConnectorConfig } from "@dci/connectors";
7 −import { stableId } from "@dci/core";
6 +import { GenericConnector, getImplementation, getParser, loadConnectorConfigs, type Connector, type ConnectorConfig } from "@dci/connectors";
7 +import { sha256, stableId } from "@dci/core";
8 8 import { getDb, connectors as connectorsTable, sources as sourcesTable, sql, eq, type Db } from "@dci/db";
9 9 import { getEnv } from "./env.js";
10 10
@@ -13,6 +13,23 @@ export interface LoadedConnector {
13 13 connector: Connector;
14 14 /** stableId("source", connectorId) — one source row per connector */
15 15 sourceId: string;
16 + /**
17 + * Effective extractor version = the YAML `parserVersion` + the names and versions of every parser the extractors
18 + * reference (+ the implementation's own version). Bumping a parser's `version` therefore invalidates every cached
19 + * extraction of every connector that uses it — `shouldSkipExtraction` compares this, not the YAML string alone.
20 + */
21 + extractorVersion: string;
22 +}
23 +
24 +export function effectiveExtractorVersion(cfg: ConnectorConfig, connector: Connector): string {
25 + const parts: string[] = [];
26 + for (const ex of Object.values(cfg.extractors ?? {})) {
27 + if (!ex.parser) continue;
28 + try { const p = getParser(ex.parser); parts.push(`${p.name}@${p.version}`); } catch { parts.push(`${ex.parser}@?`); }
29 + }
30 + if (cfg.implementation) parts.push(`impl:${cfg.implementation}@${connector.parserVersion}`);
31 + parts.sort();
32 + return parts.length ? `${cfg.parserVersion}+${sha256(parts.join("|")).slice(0, 10)}` : cfg.parserVersion;
16 33 }
17 34
18 35 export function sourceIdFor(connectorId: string): string { return stableId("source", connectorId); }
@@ -34,7 +51,7 @@ export function loadAllConnectors(opts: { reload?: boolean; dir?: string } = {})
34 51 const dir = opts.dir ?? getEnv().configDir;
35 52 const out = new Map<string, LoadedConnector>();
36 53 if (existsSync(dir)) {
37 − for (const cfg of loadConnectorConfigs(dir)) out.set(cfg.id, { cfg, connector: buildConnector(cfg), sourceId: sourceIdFor(cfg.id) });
54 + for (const cfg of loadConnectorConfigs(dir)) { const connector = buildConnector(cfg); out.set(cfg.id, { cfg, connector, sourceId: sourceIdFor(cfg.id), extractorVersion: effectiveExtractorVersion(cfg, connector) }); }
38 55 }
39 56 cache = out;
40 57 return [...out.values()];
@@ -60,22 +77,22 @@ export interface SyncResult { connectors: number; sources: number; disabledInDb:
60 77 export async function syncConnectorsToDb(loaded: LoadedConnector[] = loadAllConnectors(), db: Db = getDb()): Promise<SyncResult> {
61 78 let n = 0, s = 0;
62 79 const now = new Date().toISOString();
63 − for (const { cfg, connector, sourceId } of loaded) {
80 + for (const { cfg, sourceId, extractorVersion } of loaded) {
64 81 const config = JSON.parse(JSON.stringify(cfg)) as Record<string, unknown>;
65 82 await db
66 83 .insert(connectorsTable)
67 − .values({ id: cfg.id, sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: connector.parserVersion, schedule: cfg.schedule })
84 + .values({ id: cfg.id, sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: extractorVersion, schedule: cfg.schedule })
68 85 .onConflictDoUpdate({
69 86 target: connectorsTable.id,
70 − set: { sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: connector.parserVersion, schedule: cfg.schedule, updatedAt: now },
87 + set: { sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: extractorVersion, schedule: cfg.schedule, updatedAt: now },
71 88 });
72 89 n++;
73 90 await db
74 91 .insert(sourcesTable)
75 − .values({ id: sourceId, connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null })
92 + .values({ id: sourceId, connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null, redistribution: redistributionOf(cfg.license), attributionRequired: attributionRequiredOf(cfg.license, cfg.attribution) })
76 93 .onConflictDoUpdate({
77 94 target: sourcesTable.id,
78 − set: { connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null, updatedAt: now },
95 + set: { connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null, redistribution: redistributionOf(cfg.license), attributionRequired: attributionRequiredOf(cfg.license, cfg.attribution), updatedAt: now },
79 96 });
80 97 s++;
81 98 }
@@ -94,3 +111,16 @@ export async function syncConnectorsToDb(loaded: LoadedConnector[] = loadAllConn
94 111 }
95 112 return { connectors: n, sources: s, disabledInDb };
96 113 }
114 +
115 +/** Redistribution policy derived from the declared licence (drives /download and the API `sources` block). */
116 +export function redistributionOf(license: string | null | undefined): "allowed" | "attribution" | "restricted" | "unknown" {
117 + const l = (license ?? "").toLowerCase();
118 + if (!l) return "unknown";
119 + if (/cc0|public domain|pddl|open government|ogl|us government|unrestricted/.test(l)) return "allowed";
120 + if (/odbl|cc[- ]by|odc-by|attribution|mit|apache|peeringdb|open data/.test(l)) return "attribution";
121 + if (/all rights reserved|proprietary|copyright|terms of (use|service)|facts only|no redistribution|non-?commercial|nc\b/.test(l)) return "restricted";
122 + return "unknown";
123 +}
124 +export function attributionRequiredOf(license: string | null | undefined, attribution: string | null | undefined): boolean {
125 + return redistributionOf(license) === "attribution" || !!attribution;
126 +}
modified apps/worker/src/connectors/news/article-parser.ts +7 −2
@@ -131,7 +131,7 @@ export function buildRecords(args: { connectorId: string; url: string; a: Announ
131 131 mw: a.mwAll, status: statusForEvent, statusInferred: a.status, money: a.money,
132 132 text: params.keepText ? content.text.slice(0, 20_000) : null,
133 133 operators: a.operators.map((o) => o.name), countries: country ? [country] : [], cities: city ? [city] : [],
134 − relevance: a.relevance, extractor: EXTRACTOR_VERSION, ...(args.extraEvent ?? {}),
134 + relevance: a.relevance, extractor: EXTRACTOR_VERSION, projectClass: a.classification.class, isAi: a.aiEvidence === "confirmed" || a.aiEvidence === "likely", ...(args.extraEvent ?? {}),
135 135 },
136 136 methods: { title: content.titleMethod, publishedAt: content.publishedMethod, summary: content.summaryMethod, mw: "regex:mw_v1", eventType: a.methods.pageType ?? "classify", operators: "lexicon:operator", location: a.methods.location ?? "none" },
137 137 };
@@ -147,7 +147,12 @@ export function buildRecords(args: { connectorId: string; url: string; a: Announ
147 147 data: {
148 148 name: a.projectName, operatorName: a.operator?.name ?? null, city, regionName: a.location?.region ?? null, countryIso2: country,
149 149 status: a.status, announcedOn: content.published, expectedOpening: a.expectedOpening, plannedMw: a.headlineMw, investmentUsd: a.investmentUsd,
150 − acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url, ...(args.extraProject ?? {}),
150 + acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url,
151 + projectClass: a.classification.class, evidenceLevel: a.classification.evidence.strength,
152 + capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics,
153 + investmentCurrency: a.money?.currency ?? null, investmentOriginal: a.money?.amount ?? null,
154 + claimContext: { ...(a.evidence.plannedMw ? { plannedMw: a.evidence.plannedMw.text } : {}), ...(a.evidence.investment ? { investmentUsd: a.evidence.investment.text } : {}) },
155 + aiEvidence: a.aiEvidence, ...(args.extraProject ?? {}),
151 156 },
152 157 methods: {
153 158 name: a.methods.name ?? "title", operatorName: a.methods.operatorName ?? "none", city: a.methods.location ?? "none", regionName: a.methods.location ?? "none", countryIso2: a.methods.location ?? "none",
modified apps/worker/src/connectors/news/extract-project.test.ts +39 −0
@@ -406,3 +406,42 @@ describe("news_edgar_fts_v1", () => {
406 406 expect(r.data).toMatchObject({ title: "Riot Platforms, Inc.: Form 8-K (EX-99.2)", publishedAt: "2026-08-10", operators: ["Riot Platforms"], countries: ["US"], cities: ["Castle Rock"], text: null });
407 407 });
408 408 });
409 +
410 +describe("announcement classifier veto (2026-09 upgrade)", () => {
411 + const body = (extra: string) => `${extra} The company operates more than 13 GW of capacity across its global portfolio, with campuses in Northern Virginia, Dallas and Phoenix. Its newest Lancaster campus will deliver 500 MW at full build-out.`;
412 + it("never turns an executive appointment into a project, whatever MW the body quotes", () => {
413 + const a = extractAnnouncement("STACK Infrastructure Appoints Matt VanderZanden as Chief Executive Officer, STACK Americas", body("STACK Infrastructure today announced the appointment of Matt VanderZanden as CEO of STACK Americas."), { publishedAt: "2025-12-12" });
414 + expect(a.classification.class).toBe("EXECUTIVE_APPOINTMENT");
415 + expect(qualifiesAsProject(a)).toBe(false);
416 + });
417 + it("keeps portfolio totals out of the headline figure and scopes the site figure", () => {
418 + const a = extractAnnouncement("STACK breaks ground on new Lancaster data center campus", body("STACK Infrastructure broke ground today on its Lancaster, Texas campus."), { publishedAt: "2026-05-01" });
419 + expect(a.classification.class).toBe("CONSTRUCTION_START");
420 + expect(a.headlineMw).toBe(500);
421 + expect(a.capacityScope).toBe("campus");
422 + expect(a.evidence.plannedMw?.text).toMatch(/500 MW/);
423 + expect(qualifiesAsProject(a)).toBe(true);
424 + });
425 + it("does not locate a project at the company's headquarters", () => {
426 + const a = extractAnnouncement("Denver-based Vantage plans 192 MW data center campus in Abilene, Texas", "Denver-based Vantage Data Centers said it will build a 192 MW campus in Abilene, Texas, with the first building expected in 2027.", { publishedAt: "2026-03-01" });
427 + expect(a.location?.city).toBe("Abilene");
428 + expect(a.hqGuarded).toBe(true);
429 + });
430 + it("classifies PPA, financing and market-research headlines as non-physical", () => {
431 + for (const [title, cls] of [
432 + ["Ormat Technologies Signs 20-Year PPA with Switch for ~13 MW of Carbon-Free Geothermal Capacity to Power Data Centers", "POWER_AGREEMENT"],
433 + ["STACK Secures $1.3B Financing for Development", "FINANCING"],
434 + ["Hyperscale Data Center Market to Surpass USD 624.2 Billion by 2035", "GENERAL_COMPANY_NEWS"],
435 + ] as Array<[string, string]>) {
436 + const a = extractAnnouncement(title, `${title}. Data center capacity of 300 MW is mentioned in passing.`, { publishedAt: "2026-01-02" });
437 + expect(a.classification.class).toBe(cls);
438 + expect(qualifiesAsProject(a)).toBe(false);
439 + }
440 + });
441 + it("grades AI evidence from explicit wording only", () => {
442 + const a = extractAnnouncement("Crusoe to build 1.2 GW AI factory campus in Abilene, Texas", "Crusoe will develop a purpose-built AI campus in Abilene, Texas with NVIDIA GB200 systems.", { publishedAt: "2026-01-02" });
443 + expect(a.aiEvidence).toBe("confirmed");
444 + const b = extractAnnouncement("Equinix opens DA11 colocation data center in Dallas", "Equinix opened DA11, a colocation facility in Dallas, Texas.", { publishedAt: "2026-01-02" });
445 + expect(b.aiEvidence).toBe("unknown");
446 + });
447 +});
modified apps/worker/src/connectors/news/extract-project.ts +56 −8
@@ -1,5 +1,5 @@
1 −import type { EventType, FacilityStatus, PageType } from "@dci/core";
2 −import { PIPELINE_STATUSES, cleanText, parseAllMw, parseAreaHa, parseMoney, parsePartialDate, partialDateSortKey, round } from "@dci/core";
1 +import type { EventType, FacilityStatus, PageType, ClaimScope, Evidence, ProjectClassification, AiEvidence } from "@dci/core";
2 +import { PIPELINE_STATUSES, cleanText, parseAllMw, parseAreaHa, parseMoney, parsePartialDate, partialDateSortKey, round, classifyProjectEvent, classifyScope, classifyCapacitySemantics, classifyInvestmentSemantics, classifyAiEvidence, findEvidence } from "@dci/core";
3 3 import { classifyPage, eventTypeFor, isDataCenterRelevant } from "@dci/connectors";
4 4 import { detectOperators, primaryOperator, type OperatorHit } from "./operators-lexicon.js";
5 5 import { detectLocation, type LocationHit } from "./locations-lexicon.js";
@@ -39,9 +39,33 @@ export interface Announcement {
39 39 explicitName: string | null;
40 40 projectName: string;
41 41 methods: Record<string, string>;
42 + /** what the announcement is about (people / money / deal / build…) — decides whether a project may exist at all */
43 + classification: ProjectClassification;
44 + /** supporting sentences for the headline figures (no sentence → no claim) */
45 + evidence: { plannedMw: Evidence | null; investment: Evidence | null };
46 + capacityScope: ClaimScope;
47 + capacitySemantics: string | null;
48 + investmentScope: ClaimScope;
49 + investmentSemantics: string | null;
50 + aiEvidence: AiEvidence;
51 + /** a "City, State-based" / "headquartered in" mention was removed before locating the project */
52 + hqGuarded: boolean;
42 53 }
43 54
44 −export const EXTRACTOR_VERSION = "news_v1";
55 +export const EXTRACTOR_VERSION = "news_v2";
56 +
57 +/**
58 + * Company-domicile mentions must not locate a project: "Denver-based Vantage plans a campus in Abilene, Texas" is in
59 + * Abilene. Strips "<Place>-based", "headquartered in <Place>" and "<company>, based in <Place>" before location detection.
60 + */
61 +export function stripHqMentions(text: string): { text: string; stripped: boolean } {
62 + let stripped = false;
63 + const out = text
64 + .replace(/\b([A-Z][a-zA-Z.'-]+(?:,? [A-Z][a-zA-Z.'-]+){0,3})[- ]based\b/g, () => { stripped = true; return " "; })
65 + .replace(/\b(?:headquartered|HQ'?d|domiciled) in ([A-Z][a-zA-Z.'-]+(?:,? [A-Z][a-zA-Z.'-]+){0,3})/g, () => { stripped = true; return " "; })
66 + .replace(/\b(Inc\.?|LLC|Ltd\.?|Limited|Group|plc|Corp\.?|Corporation|company|firm|developer|operator|provider|startup|REIT|fund)[,]? (?:which is |that is )?based in ([A-Z][a-zA-Z.'-]+(?:,? [A-Z][a-zA-Z.'-]+){0,3})/g, (_m, w: string) => { stripped = true; return `${w} `; });
67 + return { text: out, stripped };
68 +}
45 69
46 70 const FULL_MONTHS: Record<string, string> = { january: "Jan", february: "Feb", march: "Mar", april: "Apr", june: "Jun", july: "Jul", august: "Aug", september: "Sep", october: "Oct", november: "Nov", december: "Dec" };
47 71 /**
@@ -405,12 +429,17 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op
405 429 const exp = parseExpectedOpening(`${title}. ${body}`, { minYear, maxYear: now.getFullYear() + 15, after: opts.publishedAt ?? null });
406 430 if (exp.value) methods.expectedOpening = exp.method;
407 431
408 − // location: title first, then lead, then body; a title hit without a city is completed from the lead when consistent
409 − let location = detectLocation(title);
410 − const leadLoc = detectLocation(text.slice(0, 1500));
411 − if (!location) location = leadLoc ?? detectLocation(body);
432 + // location: title first, then lead, then body; a title hit without a city is completed from the lead when consistent.
433 + // Company domiciles ("Denver-based", "headquartered in Ashburn") are removed first — they are not the project's location.
434 + const hqTitle = stripHqMentions(title);
435 + const hqLead = stripHqMentions(text.slice(0, 1500));
436 + const hqBody = stripHqMentions(body);
437 + let location = detectLocation(hqTitle.text);
438 + const leadLoc = detectLocation(hqLead.text);
439 + if (!location) location = leadLoc ?? detectLocation(hqBody.text);
412 440 else if (!location.city && leadLoc?.city && (!location.country || leadLoc.country === location.country) && (!location.region || !leadLoc.region || leadLoc.region === location.region)) location = { ...leadLoc, method: `${location.method}+${leadLoc.method}` };
413 441 if (location) { methods.location = location.method; }
442 + const hqGuarded = hqTitle.stripped || hqLead.stripped || hqBody.stripped;
414 443
415 444 const operators = detectOperators(`${title}\n${body}`);
416 445 let operator = primaryOperator(title, body);
@@ -429,7 +458,22 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op
429 458 // people / deals / money / market headlines are plain news whatever the body's vocabulary says
430 459 if (nonProject && !["acquisition", "closure", "cloud_region", "financial_disclosure"].includes(pageType)) { pageType = "press_release"; eventType = "news"; methods.pageType = `${methods.pageType}+veto:title`; }
431 460
432 − return { title, pageType, eventType, relevance, leadRelevance: leadRelevanceOf(title, text), titleSignal: titleSignalOf(title), nonProjectTitle: nonProject, mwAll: mwBody, headlineMw, money, investmentUsd, investmentUsdApprox, acreage, phaseCount, expectedOpening: exp.value, status: st.status, location, operators, operator, explicitName, projectName: gen.name, methods };
461 + // classification BEFORE anything becomes a project: people / money / deal / market headlines never do
462 + const classification = classifyProjectEvent({ title, lead: head, hasExplicitName: !!explicitName, hasOperator: !!operator, hasLocation: !!(location?.city || location?.region || location?.country), status: st.status });
463 + // evidence + scope + semantics of the headline figures (the sentence decides; no sentence → the claim is not site-scoped)
464 + const campusHint: ClaimScope = /\b(campus|park|complex|hub)\b/i.test(`${explicitName ?? ""} ${title}`) ? "campus" : "facility";
465 + const mwEvidence = headlineMw != null ? findEvidence(`${title}. ${body}`, headlineMw, "mw") : null;
466 + const mwScope = headlineMw != null ? (mwEvidence ? classifyScope(mwEvidence.text, campusHint) : { scope: "unknown" as ClaimScope, reason: "no-evidence" }) : { scope: "unknown" as ClaimScope, reason: "none" };
467 + const mwSem = mwEvidence ? classifyCapacitySemantics(mwEvidence.text) : { predicate: null, reason: "none" };
468 + const invEvidence = money ? findEvidence(`${title}. ${body}`, money.amount, "usd") : null;
469 + const invSem = money ? classifyInvestmentSemantics(invEvidence?.text ?? title, campusHint) : { predicate: "project_investment_usd" as const, scope: "unknown" as ClaimScope, reason: "none" };
470 + const ai = classifyAiEvidence(`${title}. ${head}`);
471 + methods.classification = classification.reasons.join("+") || "none";
472 + if (mwEvidence) methods.plannedMwEvidence = "sentence"; if (invEvidence) methods.investmentEvidence = "sentence";
473 + return {
474 + title, pageType, eventType, relevance, leadRelevance: leadRelevanceOf(title, text), titleSignal: titleSignalOf(title), nonProjectTitle: nonProject, mwAll: mwBodySite.length ? mwBodySite : mwBody.slice(0, 3), headlineMw, money, investmentUsd, investmentUsdApprox, acreage, phaseCount, expectedOpening: exp.value, status: st.status, location, operators, operator, explicitName, projectName: gen.name, methods,
475 + classification, evidence: { plannedMw: mwEvidence, investment: invEvidence }, capacityScope: mwScope.scope, capacitySemantics: mwSem.predicate ?? (headlineMw != null ? "planned_power_mw" : null), investmentScope: invSem.scope, investmentSemantics: invSem.predicate, aiEvidence: ai.level, hqGuarded,
476 + };
433 477 }
434 478
435 479 /** Portfolio / company-wide context around a MW figure — not the size of the announced site. */
@@ -467,6 +511,10 @@ const PROJECT_STATUSES: readonly FacilityStatus[] = [...PIPELINE_STATUSES, "expa
467 511 export function qualifiesAsProject(a: Announcement, opts: { minMw?: number; minInvestmentUsd?: number; minAcres?: number; lenient?: boolean } = {}): boolean {
468 512 if (a.relevance !== "strong") return false;
469 513 if (a.nonProjectTitle) return false;
514 + // the announcement classifier: only physical development classes with (name or operator) + location + development verb
515 + if (!a.classification.physical) return false;
516 + if (!opts.lenient && !a.classification.mayCreateProject) return false;
517 + if (opts.lenient && a.classification.evidence.strength === "none") return false;
470 518 if (!a.status || !PROJECT_STATUSES.includes(a.status)) return false;
471 519 if (!PROJECT_PAGE_TYPES.includes(a.pageType)) return false;
472 520 // An announcement names its subject up front: data-center vocabulary in the lead, and a title that mentions the
modified apps/worker/src/context.ts +1 −1
@@ -148,7 +148,7 @@ export function createRunContext(loaded: LoadedConnector, opts: CreateContextOpt
148 148
149 149 provenance(url: string, extra: Partial<Provenance> = {}): Provenance {
150 150 const now = new Date().toISOString();
151 − return { sourceId: loaded.sourceId, connectorId: cfg.id, url, firstObserved: now, lastObserved: now, retrievedAt: now, confidence: SOURCE_KIND_BASE[cfg.kind] ?? "moderate", extractorVersion: loaded.connector.parserVersion, ...extra };
151 + return { sourceId: loaded.sourceId, connectorId: cfg.id, url, firstObserved: now, lastObserved: now, retrievedAt: now, confidence: SOURCE_KIND_BASE[cfg.kind] ?? "moderate", extractorVersion: loaded.extractorVersion, ...extra };
152 152 },
153 153
154 154 async fetch(url: string, o: FetchOptions = {}): Promise<RawDocument> {
modified apps/worker/src/documents.ts +3 −1
@@ -225,7 +225,7 @@ export interface VersionResult { versionId: string; diff: LineDiff | null; signi
225 225 * Record a new content version. `prev` is the document row as it was BEFORE recordFetch (old hash / storage key).
226 226 * The old text is loaded from object storage when available; `detectedChanges` come from the ingest layer.
227 227 */
228 −export async function recordVersion(prev: DocumentRow, raw: RawDocument, newHash: string, storageKey: string | null, detectedChanges: DetectedChange[], opts: { dryRun?: boolean; now?: Date; runId?: string } = {}): Promise<VersionResult> {
228 +export async function recordVersion(prev: DocumentRow, raw: RawDocument, newHash: string, storageKey: string | null, detectedChanges: DetectedChange[], opts: { dryRun?: boolean; now?: Date; runId?: string; extractorVersion?: string } = {}): Promise<VersionResult> {
229 229 const now = opts.now ?? new Date();
230 230 let diff: LineDiff | null = null;
231 231 if (prev.storageKey && prev.contentHash && prev.contentHash !== newHash) {
@@ -251,6 +251,8 @@ export async function recordVersion(prev: DocumentRow, raw: RawDocument, newHash
251 251 diffSummary: diff ? { addedCount: diff.addedCount, removedCount: diff.removedCount, ratio: Math.round(diff.ratio * 10_000) / 10_000, added: diff.added.slice(0, 60), removed: diff.removed.slice(0, 60) } : null,
252 252 detectedChanges: detectedChanges as unknown as Array<Record<string, unknown>>,
253 253 significance,
254 + runId: opts.runId ?? null,
255 + extractorVersion: opts.extractorVersion ?? null,
254 256 })
255 257 .onConflictDoNothing();
256 258 if (prev.contentHash) {
modified apps/worker/src/health-http.ts +10 −0
@@ -15,6 +15,8 @@ export interface HealthHttpOptions {
15 15 /** refresh snapshot gauges (queue depth, budgets…) right before rendering */
16 16 beforeScrape?: () => Promise<void>;
17 17 log?: (msg: string) => void;
18 + /** extra JSON routes keyed by path prefix ("/trace" matches "/trace/<id>") */
19 + routes?: Record<string, (url: URL) => Promise<{ status: number; body: unknown }>>;
18 20 }
19 21
20 22 export function startHealthHttp(o: HealthHttpOptions): Server {
@@ -35,6 +37,14 @@ export function startHealthHttp(o: HealthHttpOptions): Server {
35 37 res.end(renderMetrics());
36 38 return;
37 39 }
40 + for (const [prefix, handler] of Object.entries(o.routes ?? {})) {
41 + if (url.pathname === prefix || url.pathname.startsWith(`${prefix}/`)) {
42 + const r = await handler(url);
43 + res.writeHead(r.status, { "content-type": "application/json", "cache-control": "no-store" });
44 + res.end(JSON.stringify(r.body));
45 + return;
46 + }
47 + }
38 48 res.writeHead(404, { "content-type": "application/json" });
39 49 res.end(JSON.stringify({ error: "not found" }));
40 50 } catch (e) {
modified apps/worker/src/ingest/campuses.ts +2 −0
@@ -30,6 +30,8 @@ export async function resolveCampus(tx: Tx, ctx: IngestContext, c: CampusInput):
30 30 }
31 31 if (!id && c.externalIds) {
32 32 for (const [k, v] of Object.entries(c.externalIds)) {
33 + // allowlist: only namespaces that name one campus may link records
34 + if (v == null || v === "" || !/^(osm|wikidata|peeringdb_campus|[a-z0-9]+_campus(_slug|_code)?)$/.test(k) || /^(operator|owner|state|region|country|postal|market|source)_/.test(k)) continue;
33 35 const r = await tx.execute(sql`select id from campuses where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`);
34 36 if (r[0]) {
35 37 id = String(r[0].id);
added apps/worker/src/ingest/claims.ts +218 −0
@@ -0,0 +1,218 @@
1 +/**
2 + * Claim store + quality flags + winner bookkeeping (docs/CLAIMS.md).
3 + *
4 + * Every numerical figure observed about a subject becomes a `claims` row carrying its scope, semantics, supporting
5 + * sentence, parser and field-level authority tier. The ingest layer then asks `resolveCapacityColumn()` which claim
6 + * may populate a column: only site-scoped claims (building / facility / campus) that pass the deterministic sanity
7 + * engine; everything else is stored with status `unscoped` / `review` and surfaces in the admin quality dashboard.
8 + */
9 +import { sql } from "@dci/db";
10 +import {
11 + authorityFieldFor,
12 + authorityTier,
13 + capacitySanity,
14 + investmentSanity,
15 + isSiteScope,
16 + newId,
17 + stableId,
18 + TIER_RANK,
19 + type AuthorityTier,
20 + type CapacityPredicate,
21 + type ClaimScope,
22 + type ClaimStatus,
23 + type Evidence,
24 + type InvestmentPredicate,
25 + type Provenance,
26 + type SanityFlag,
27 +} from "@dci/core";
28 +import type { IngestContext, Tx } from "./common.js";
29 +
30 +export interface ClaimInput {
31 + predicate: string;
32 + value?: number | null;
33 + valueText?: string | null;
34 + unit?: string | null;
35 + scope: ClaimScope;
36 + scopeReason?: string | null;
37 + evidence?: Evidence | null;
38 + publishedAt?: string | null;
39 + provenance: Provenance;
40 + parserName?: string | null;
41 + /** override the computed status (e.g. "review" after a sanity flag) */
42 + status?: ClaimStatus;
43 + rejectionReason?: string | null;
44 +}
45 +
46 +export interface WrittenClaim { id: string; status: ClaimStatus; tier: AuthorityTier }
47 +
48 +export function claimId(subjectType: string, subjectId: string, c: { predicate: string; value?: number | null; valueText?: string | null; provenance: Provenance }, url: string): string {
49 + return stableId("claim", `${subjectType}|${subjectId}|${c.predicate}|${c.provenance.sourceId}|${url}|${c.value ?? ""}|${c.valueText ?? ""}`);
50 +}
51 +
52 +/** Upsert one claim. Same source + url + predicate with a different value supersedes the earlier claim. */
53 +export async function writeClaim(tx: Tx, ctx: IngestContext, subjectType: string, subjectId: string, c: ClaimInput): Promise<WrittenClaim> {
54 + const p = c.provenance;
55 + const url = p.url || ctx.doc?.url || "";
56 + const sourceId = p.sourceId || ctx.run.sourceId;
57 + const tier = authorityTier({ field: authorityFieldFor(c.predicate), sourceKind: ctx.run.sourceKind, isEstimate: p.isEstimate, method: p.method });
58 + const status: ClaimStatus = c.status ?? (isSiteScope(c.scope) ? "current" : "unscoped");
59 + const id = claimId(subjectType, subjectId, c, url);
60 + const observed = p.lastObserved || ctx.now;
61 + if (!ctx.run.dryRun) {
62 + await tx.execute(sql`
63 + insert into claims (id, subject_type, subject_id, predicate, value, value_text, unit, scope, scope_reason, source_id, connector_id, document_id, url, published_at, retrieved_at,
64 + confidence, is_estimate, authority_tier, evidence_text, evidence_start, evidence_end, parser_name, parser_version, run_id, status, rejection_reason, first_observed, last_observed)
65 + values (${id}, ${subjectType}, ${subjectId}, ${c.predicate}, ${c.value ?? null}, ${c.valueText ?? null}, ${c.unit ?? null}, ${c.scope}, ${c.scopeReason ?? null}, ${sourceId}, ${p.connectorId || ctx.run.connectorId},
66 + ${p.documentId ?? ctx.doc?.documentId ?? null}, ${url}, ${c.publishedAt ?? null}, ${p.retrievedAt || ctx.now}, ${p.confidence ?? "moderate"}, ${!!p.isEstimate}, ${tier},
67 + ${c.evidence?.text ?? null}, ${c.evidence?.start ?? null}, ${c.evidence?.end ?? null}, ${c.parserName ?? p.method ?? null}, ${p.extractorVersion ?? null}, ${ctx.run.runId}, ${status}, ${c.rejectionReason ?? null}, ${p.firstObserved || observed}, ${observed})
68 + on conflict (id) do update set last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, confidence = excluded.confidence, authority_tier = excluded.authority_tier,
69 + evidence_text = coalesce(excluded.evidence_text, claims.evidence_text), evidence_start = coalesce(excluded.evidence_start, claims.evidence_start), evidence_end = coalesce(excluded.evidence_end, claims.evidence_end),
70 + scope = excluded.scope, scope_reason = excluded.scope_reason, run_id = excluded.run_id, parser_version = excluded.parser_version,
71 + status = case when claims.status in ('rejected') then claims.status else excluded.status end, rejection_reason = coalesce(excluded.rejection_reason, claims.rejection_reason)`);
72 + // the same source, same page, same predicate now says something else → the older claim is superseded
73 + await tx.execute(sql`update claims set status = 'superseded' where subject_type = ${subjectType} and subject_id = ${subjectId} and predicate = ${c.predicate} and source_id = ${sourceId} and url = ${url} and id <> ${id} and status in ('current', 'review')`);
74 + }
75 + ctx.stats.claims = (ctx.stats.claims ?? 0) + 1;
76 + return { id, status, tier };
77 +}
78 +
79 +/** Review priority 0–100: impact-weighted (MW / money size, low confidence, new country, AI, ambiguity, unexpected change). */
80 +export function reviewPriority(i: { mw?: number | null; investmentUsd?: number | null; confidence?: string | null; newCountry?: boolean; ai?: boolean; ambiguous?: boolean; unexpectedChange?: boolean; severity?: "info" | "warn" | "critical"; homepageVisible?: boolean }): number {
81 + let p = 0;
82 + const mw = i.mw ?? 0;
83 + p += mw >= 1000 ? 40 : mw >= 300 ? 30 : mw >= 100 ? 20 : mw >= 20 ? 10 : mw > 0 ? 4 : 0;
84 + const inv = i.investmentUsd ?? 0;
85 + p += inv >= 10e9 ? 25 : inv >= 1e9 ? 15 : inv >= 100e6 ? 8 : 0;
86 + if (i.confidence === "unverified" || i.confidence === "estimated") p += 10;
87 + if (i.newCountry) p += 10;
88 + if (i.ai) p += 5;
89 + if (i.ambiguous) p += 8;
90 + if (i.unexpectedChange) p += 12;
91 + if (i.severity === "critical") p += 15; else if (i.severity === "warn") p += 5;
92 + if (i.homepageVisible) p += 10;
93 + return Math.max(0, Math.min(100, Math.round(p)));
94 +}
95 +
96 +/** Upsert deterministic quality flags for an entity (deduped by entity + code + field). Resolved flags are not reopened for the same key. */
97 +export async function writeQualityFlags(tx: Tx, ctx: IngestContext, entityType: string, entityId: string, flags: SanityFlag[], opts: { claimId?: string | null; priority?: number; details?: Record<string, unknown> } = {}): Promise<number> {
98 + let n = 0;
99 + for (const f of flags) {
100 + if (f.severity === "info") continue;
101 + const dedupeKey = `${entityType}|${entityId}|${f.code}|${f.field ?? ""}`;
102 + const id = stableId("flag", dedupeKey);
103 + const priority = opts.priority ?? reviewPriority({ mw: /mw/.test(f.code) ? f.value : null, investmentUsd: /inv/.test(f.code) ? f.value : null, severity: f.severity });
104 + if (!ctx.run.dryRun) {
105 + await tx.execute(sql`
106 + insert into quality_flags (id, entity_type, entity_id, claim_id, code, severity, field, message, details, priority, status, run_id, dedupe_key)
107 + values (${id}, ${entityType}, ${entityId}, ${opts.claimId ?? null}, ${f.code}, ${f.severity}, ${f.field ?? null}, ${f.message.slice(0, 1000)}, ${JSON.stringify({ ...(opts.details ?? {}), value: f.value ?? null })}::jsonb, ${priority}, 'open', ${ctx.run.runId}, ${dedupeKey})
108 + on conflict (dedupe_key) do update set message = excluded.message, details = excluded.details, priority = greatest(quality_flags.priority, excluded.priority), claim_id = coalesce(excluded.claim_id, quality_flags.claim_id),
109 + run_id = excluded.run_id, updated_at = now(), status = case when quality_flags.status = 'dismissed' then 'dismissed' else 'open' end`);
110 + }
111 + n++;
112 + }
113 + ctx.stats.qualityFlags = (ctx.stats.qualityFlags ?? 0) + n;
114 + return n;
115 +}
116 +
117 +/** Close open flags of the given codes for an entity (the condition no longer holds). */
118 +export async function resolveQualityFlags(tx: Tx, ctx: IngestContext, entityType: string, entityId: string, codes: string[], resolution = "auto: condition cleared"): Promise<void> {
119 + if (!codes.length || ctx.run.dryRun) return;
120 + await tx.execute(sql`update quality_flags set status = 'resolved', resolution = ${resolution}, resolved_by = 'system', resolved_at = now(), updated_at = now()
121 + where entity_type = ${entityType} and entity_id = ${entityId} and status = 'open' and code in ${codes}`);
122 +}
123 +
124 +/**
125 + * Mark, per field, the current provenance row whose value equals the stored column value as the winner (and unmark
126 + * the others). This is what makes "why does the page show 300 MW?" answerable without a diff.
127 + */
128 +export async function markWinners(tx: Tx, ctx: IngestContext, entityType: string, entityId: string, fields: Array<{ field: string; value: unknown }>): Promise<void> {
129 + if (ctx.run.dryRun) return;
130 + for (const f of fields) {
131 + if (f.value == null) continue;
132 + const valueJson = JSON.stringify(f.value);
133 + await tx.execute(sql`update provenance set is_winner = (value = ${valueJson}::jsonb) where entity_type = ${entityType} and entity_id = ${entityId} and field = ${f.field} and is_current and is_winner <> (value = ${valueJson}::jsonb)`);
134 + }
135 +}
136 +
137 +export interface CapacityDecision {
138 + /** write the figure into the record's column */
139 + assign: boolean;
140 + claim: WrittenClaim;
141 + flags: SanityFlag[];
142 + blockedBy: string[];
143 +}
144 +
145 +export interface CapacityClaimInput {
146 + field: "itCapacityMw" | "totalPowerMw" | "plannedPowerMw" | "plannedMw" | "utilityCapacityMw" | "gridConnectionMw" | "ultimateCampusMw";
147 + predicate: CapacityPredicate;
148 + value: number;
149 + scope: ClaimScope;
150 + scopeReason?: string | null;
151 + evidence?: Evidence | null;
152 + context?: string | null;
153 + previous?: number | null;
154 + recordScope: "building" | "facility" | "campus" | "project";
155 + campusDesignation?: boolean;
156 + publishedAt?: string | null;
157 + provenance: Provenance;
158 + parserName?: string | null;
159 + /** the sentence did not say what kind of MW this is; semantics were defaulted from the record status */
160 + semanticsDefaulted?: boolean;
161 +}
162 +
163 +/** Store a capacity claim, run the sanity engine, write flags and decide whether the column may take the value. */
164 +export async function recordCapacityClaim(tx: Tx, ctx: IngestContext, subjectType: "facility" | "project" | "campus", subjectId: string, i: CapacityClaimInput): Promise<CapacityDecision> {
165 + const flags = capacitySanity({ value: i.value, predicate: i.semanticsDefaulted ? null : i.predicate, scope: i.scope, recordScope: i.recordScope, previous: i.previous ?? null, context: i.context ?? i.evidence?.text ?? null, campusDesignation: i.campusDesignation });
166 + const blockedBy = flags.filter((f) => f.blocks).map((f) => f.code);
167 + const critical = flags.some((f) => f.severity === "critical");
168 + const status: ClaimStatus = blockedBy.length ? (isSiteScope(i.scope) ? "review" : "unscoped") : critical ? "review" : "current";
169 + const claim = await writeClaim(tx, ctx, subjectType, subjectId, { predicate: i.predicate, value: i.value, unit: "MW", scope: i.scope, scopeReason: i.scopeReason ?? null, evidence: i.evidence ?? null, publishedAt: i.publishedAt ?? null, provenance: i.provenance, parserName: i.parserName ?? null, status, rejectionReason: blockedBy.length ? blockedBy.join(",") : null });
170 + await writeQualityFlags(tx, ctx, subjectType, subjectId, flags.map((f) => ({ ...f, field: f.field ?? i.field })), { claimId: claim.id, priority: reviewPriority({ mw: i.value, severity: flags.some((f) => f.severity === "critical") ? "critical" : flags.some((f) => f.severity === "warn") ? "warn" : "info" }) });
171 + if (!blockedBy.length) {
172 + const cleared = ["scope_unknown", "scope_company", "scope_portfolio", "scope_country", "scope_metro", "mw_market_statistic", "mw_invalid"].filter((c) => !flags.some((f) => f.code === c));
173 + await resolveQualityFlags(tx, ctx, subjectType, subjectId, cleared.map((c) => c));
174 + }
175 + // a critical (non-blocking) flag such as "> 1 000 MW single site" still assigns — the figure is what the source says — but
176 + // stays visible for review; a blocking flag never assigns
177 + return { assign: blockedBy.length === 0, claim, flags, blockedBy };
178 +}
179 +
180 +export interface InvestmentClaimInput {
181 + value: number;
182 + currency?: string | null;
183 + predicate: InvestmentPredicate;
184 + scope: ClaimScope;
185 + scopeReason?: string | null;
186 + evidence?: Evidence | null;
187 + context?: string | null;
188 + previous?: number | null;
189 + recordScope: "facility" | "campus" | "project";
190 + publishedAt?: string | null;
191 + provenance: Provenance;
192 + parserName?: string | null;
193 +}
194 +
195 +export async function recordInvestmentClaim(tx: Tx, ctx: IngestContext, subjectType: "facility" | "project" | "campus" | "operator", subjectId: string, i: InvestmentClaimInput): Promise<CapacityDecision> {
196 + const flags = investmentSanity({ value: i.value, scope: i.scope, predicate: i.predicate, recordScope: i.recordScope, previous: i.previous ?? null, context: i.context ?? i.evidence?.text ?? null });
197 + const blockedBy = flags.filter((f) => f.blocks).map((f) => f.code);
198 + const status: ClaimStatus = blockedBy.length ? (isSiteScope(i.scope) && i.predicate === "project_investment_usd" ? "review" : "unscoped") : flags.some((f) => f.severity === "critical") ? "review" : "current";
199 + const claim = await writeClaim(tx, ctx, subjectType, subjectId, { predicate: i.predicate, value: i.value, unit: i.currency ?? "USD", scope: i.scope, scopeReason: i.scopeReason ?? null, evidence: i.evidence ?? null, publishedAt: i.publishedAt ?? null, provenance: i.provenance, parserName: i.parserName ?? null, status, rejectionReason: blockedBy.length ? blockedBy.join(",") : null });
200 + await writeQualityFlags(tx, ctx, subjectType, subjectId, flags.map((f) => ({ ...f, field: f.field ?? "investmentUsd" })), { claimId: claim.id, priority: reviewPriority({ investmentUsd: i.value, severity: flags.some((f) => f.severity === "critical") ? "critical" : "warn" }) });
201 + return { assign: blockedBy.length === 0 && (i.currency ?? "USD") === "USD", claim, flags, blockedBy };
202 +}
203 +
204 +/** Best current claim for a predicate (highest tier, then most recent), for the evidence drawer and reconciliation. */
205 +export async function bestCurrentClaim(tx: Tx, subjectType: string, subjectId: string, predicate: string): Promise<{ id: string; value: number | null; tier: AuthorityTier; url: string; evidence: string | null } | null> {
206 + const rows = await tx.execute(sql`select id, value, authority_tier, url, evidence_text from claims where subject_type = ${subjectType} and subject_id = ${subjectId} and predicate = ${predicate} and status = 'current' order by last_observed desc limit 50`);
207 + if (!rows.length) return null;
208 + const sorted = [...rows].sort((a, b) => (TIER_RANK[String(b.authority_tier) as AuthorityTier] ?? 0) - (TIER_RANK[String(a.authority_tier) as AuthorityTier] ?? 0));
209 + const r = sorted[0]!;
210 + return { id: String(r.id), value: r.value == null ? null : Number(r.value), tier: String(r.authority_tier) as AuthorityTier, url: String(r.url), evidence: r.evidence_text == null ? null : String(r.evidence_text) };
211 +}
212 +
213 +/** A generic non-numeric claim (status, operator, opening date…) — same store, text value. */
214 +export async function writeTextClaim(tx: Tx, ctx: IngestContext, subjectType: string, subjectId: string, predicate: string, value: string, provenance: Provenance, opts: { scope?: ClaimScope; evidence?: Evidence | null; publishedAt?: string | null } = {}): Promise<WrittenClaim> {
215 + return writeClaim(tx, ctx, subjectType, subjectId, { predicate, valueText: value, scope: opts.scope ?? "facility", evidence: opts.evidence ?? null, publishedAt: opts.publishedAt ?? null, provenance, status: "current" });
216 +}
217 +
218 +export { newId as _newClaimId };
modified apps/worker/src/ingest/contract.ts +10 −0
@@ -42,6 +42,16 @@ export interface IngestStats {
42 42 implausibleDropped?: number;
43 43 /** projects folded into an already-known announcement (same operator, city, MW ±10 %, within 30 days) */
44 44 projectDedup?: number;
45 + /** claims rows written */
46 + claims?: number;
47 + /** quality flags written */
48 + qualityFlags?: number;
49 + /** figures kept as claims but not assigned to a column (portfolio / company / unknown scope, market statistics) */
50 + unscopedClaims?: number;
51 + /** project candidates vetoed by the announcement classifier (appointments, financing, PPAs, market research…) */
52 + projectsVetoed?: number;
53 + /** building records linked to their campus record instead of being flagged as duplicates */
54 + campusLinks?: number;
45 55 }
46 56
47 57 export type IngestFn = (run: IngestRun, entities: NormalizedEntity[], doc: IngestDocRef | null) => Promise<IngestStats>;
modified apps/worker/src/ingest/events.ts +5 −2
@@ -110,6 +110,9 @@ export interface EventInput {
110 110 reviewStatus?: "auto" | "pending";
111 111 /** override the fingerprint day/value (default: newValue + ctx.day) */
112 112 fingerprint?: string;
113 + isAi?: boolean;
114 + /** documents describing the same announcement share a cluster id */
115 + clusterId?: string | null;
113 116 }
114 117
115 118 /** Insert an event unless the same fingerprint already exists today. Returns the id when inserted. */
@@ -118,12 +121,12 @@ export async function recordEvent(tx: Tx, ctx: IngestContext, e: EventInput): Pr
118 121 const id = newId("event");
119 122 const rows = await tx.execute(sql`
120 123 insert into events (id, entity_type, entity_id, event_type, detected_at, effective_date, old_value, new_value, source_id, document_id, url, title, summary,
121 − significance, confidence, review_status, country_iso2, operator_id, metro_id, project_id, fingerprint)
124 + significance, confidence, review_status, country_iso2, operator_id, metro_id, project_id, fingerprint, run_id, source_kind, is_ai, cluster_id)
122 125 values (${id}, ${e.entityType}, ${e.entityId}, ${e.eventType}, ${ctx.now}, ${e.effectiveDate ?? null},
123 126 ${e.oldValue === undefined ? null : JSON.stringify(e.oldValue)}::jsonb, ${e.newValue === undefined ? null : JSON.stringify(e.newValue)}::jsonb,
124 127 ${ctx.run.sourceId}, ${ctx.doc?.documentId ?? null}, ${e.url}, ${e.title.slice(0, 300)}, ${e.summary ?? null},
125 128 ${Math.max(0, Math.min(100, Math.round(e.significance)))}, ${e.confidence ?? "moderate"}, ${e.reviewStatus ?? "auto"},
126 − ${e.countryIso2 ?? null}, ${e.operatorId ?? null}, ${e.metroId ?? null}, ${e.projectId ?? null}, ${fp})
129 + ${e.countryIso2 ?? null}, ${e.operatorId ?? null}, ${e.metroId ?? null}, ${e.projectId ?? null}, ${fp}, ${ctx.run.runId}, ${ctx.run.sourceKind}, ${!!e.isAi}, ${e.clusterId ?? null})
127 130 on conflict (fingerprint) do nothing
128 131 returning id`);
129 132 if (!rows.length) return null;
modified apps/worker/src/ingest/facilities.ts +120 −18
@@ -21,7 +21,9 @@ import { bboxAround, validGeo } from "./geo.js";
21 21 import { countryFromPoint } from "./country-lookup.js";
22 22 import { facilityIdForKey, upsertKey } from "./keys.js";
23 23 import { attachIxpsByName } from "./ixps.js";
24 −import { bestMatch, bestMw, completenessScore, decide, facilityConfidence, isPipelineStatus, shouldReplace, shouldReplaceGeo, shouldReplaceMw, type FacilityCandidate, type FieldObservation, type MatchDecision, type MatchScore } from "./match.js";
24 +import { bestMatch, bestMw, completenessScore, decide, facilityConfidence, isPipelineStatus, scoreFacilityMatch, shouldReplace, shouldReplaceGeo, shouldReplaceMw, type FacilityCandidate, type FieldObservation, type MatchDecision, type MatchScore } from "./match.js";
25 +import { CAPACITY_COLUMN, classifyAiEvidence, classifyCapacitySemantics, classifyScope, findEvidence, type CapacityPredicate, type ClaimScope } from "@dci/core";
26 +import { markWinners, recordCapacityClaim, writeQualityFlags } from "./claims.js";
25 27 import { assignMetro } from "./metros.js";
26 28 import { operatorNames, resolveOperator } from "./operators.js";
27 29 import { backingObservation, loadCurrentProvenance, summarizeSources, writeProvenance, type CurrentProvenance, type ObservedField } from "./provenance.js";
@@ -70,6 +72,16 @@ const COLUMNS: Record<string, string> = {
70 72 externalIds: "external_ids",
71 73 sourceCount: "source_count",
72 74 lastVerified: "last_verified",
75 + parentFacilityId: "parent_facility_id",
76 + recordScope: "record_scope",
77 + aiEvidence: "ai_evidence",
78 + utilityCapacityMw: "utility_capacity_mw",
79 + gridConnectionMw: "grid_connection_mw",
80 + ultimateCampusMw: "ultimate_campus_mw",
81 + capacityScope: "capacity_scope",
82 + capacitySemantics: "capacity_semantics",
83 + developerId: "developer_id",
84 + landownerId: "landowner_id",
73 85 };
74 86
75 87 const SCALAR_FIELDS = ["name", "city", "regionName", "address", "postalCode", "status", "facilityType", "tier", "buildingSqm", "siteAreaHa", "rackCount", "pue", "coolingType", "renewableClaim", "openedOn", "constructionStartedOn", "announcedOn", "website", "description"] as const;
@@ -108,14 +120,21 @@ async function followMerged(tx: Tx, id: string): Promise<string> {
108 120 * id, a state name, an investment figure, a shared `ref` tag…) is informative but must never link records — a
109 121 * shared `operator_wikidata` once folded 121 AWS OpenStreetMap features into a single facility.
110 122 */
111 −export const IDENTIFYING_EXTERNAL_ID_KEYS: ReadonlySet<string> = new Set(["osm", "wikidata", "wikipedia_en", "peeringdb_fac", "peeringdb", "pdb_fac", "dcmap", "geonames", "facebook_page", "meta_info_sheet", "google_detail_page", "cyrusone_slug", "qts_slug", "coresite_code"]);
112 −const IDENTIFYING_KEY_RE = /(^|_)(id|slug|code|page|sheet|url|qid|uid)$/;
123 +export const IDENTIFYING_EXTERNAL_ID_KEYS: ReadonlySet<string> = new Set([
124 + "osm", "wikidata", "wikipedia_en", "peeringdb_fac", "peeringdb", "pdb_fac", "dcmap", "geonames",
125 + "facebook_page", "meta_info_sheet", "meta_location", "google_detail_page", "google_location",
126 + "equinix_ibx", "digitalrealty_node", "digitalrealty_site_code", "ntt_slug", "cyrusone_slug", "stack_slug", "qts_slug", "vantage_slug", "coresite_code", "switch_slug",
127 + "databank_code", "edgeconnex_slug", "tierpoint_slug", "airtrunk_code", "nextdc_code", "atnorth_code", "cloudhq_slug", "skybox_slug", "odata_code", "telehouse_slug", "coltdcs_slug", "globalswitch_slug", "virtus_slug", "digitaledge_code",
128 +]);
113 129
114 −/** true when `key` names one facility (curated namespaces, or `*_id` / `*_slug` / `*_code` / `*_page` keys that are not operator-level). */
130 +/**
131 + * ALLOWLIST ONLY. `state_code`, `region_code`, `postal_code`, `source_url`, `osm_ref` (a shared `ref` tag), `stack_campus`
132 + * (one campus, many buildings) or `investment_currency` all end in an id-looking suffix and would fold unrelated
133 + * facilities into one row (the `*_code` heuristic once folded 121 AWS OpenStreetMap features). A new connector that
134 + * needs its key to link records adds it here explicitly.
135 + */
115 136 export function isIdentifyingExternalId(key: string): boolean {
116 − if (IDENTIFYING_EXTERNAL_ID_KEYS.has(key)) return true;
117 − if (/^(operator|owner|brand|company|parent|network)_/.test(key)) return false;
118 − return IDENTIFYING_KEY_RE.test(key);
137 + return IDENTIFYING_EXTERNAL_ID_KEYS.has(key);
119 138 }
120 139
121 140 async function byExternalIds(tx: Tx, ext: Record<string, string | number> | undefined): Promise<string | null> {
@@ -141,10 +160,15 @@ async function loadCandidates(tx: Tx, nf: NormalizedFacility, operatorId: string
141 160 }
142 161 if (operatorId && countryIso2) conds.push(sql`(f.operator_id = ${operatorId} and f.country_iso2 = ${countryIso2})`);
143 162 if (!conds.length) return [];
163 + // deterministic and relevance-ordered: same operator first, then same normalized name, then nearest — a dense metro
164 + // (Ashburn, Dallas, Singapore) has far more than 400 rows in a 40 km box and the true duplicate must be on the first page
165 + const orderGeo = geo ? sql`, ((f.lat - ${geo.lat}) * (f.lat - ${geo.lat}) + (f.lng - ${geo.lng}) * (f.lng - ${geo.lng})) asc nulls last` : sql``;
144 166 const rows = await tx.execute(sql`
145 167 select f.id, f.name, f.normalized_name, f.operator_id, o.name as operator_name, f.country_iso2, f.city, f.address, f.lat, f.lng, f.geo_precision, f.external_ids,
146 168 (select coalesce(array_agg(alias), '{}'::text[]) from facility_aliases a where a.facility_id = f.id) as aliases
147 − from facilities f left join operators o on o.id = f.operator_id where f.merged_into is null and (${sql.join(conds, sql` or `)}) limit 400`);
169 + from facilities f left join operators o on o.id = f.operator_id where f.merged_into is null and (${sql.join(conds, sql` or `)})
170 + order by (${operatorId ?? null}::text is not null and f.operator_id = ${operatorId ?? null}) desc, (f.normalized_name in ${aliasNorms.length ? aliasNorms : ["__none__"]}) desc${orderGeo}, f.created_at asc
171 + limit 400`);
148 172 return rows.map((r) => ({
149 173 id: String(r.id),
150 174 name: String(r.name),
@@ -164,10 +188,23 @@ async function loadCandidates(tx: Tx, nf: NormalizedFacility, operatorId: string
164 188
165 189 interface Resolution {
166 190 id: string | null;
167 − how: "key" | "external_id" | "merge" | "pending" | "create";
191 + how: "key" | "external_id" | "merge" | "pending" | "create" | "campus_link";
168 192 match?: { candidate: FacilityCandidate; match: MatchScore } | null;
169 193 }
170 194
195 +/** true when a facility name designates a campus / park / multi-building site rather than one building. */
196 +export function isCampusName(name: string | null | undefined, campusName?: string | null): boolean {
197 + return /\b(campus|park|cluster|hub|gigafactory|complex|estate|mega ?site)\b/i.test(`${name ?? ""} ${campusName ?? ""}`);
198 +}
199 +
200 +/** Record scope from the source's statement or the name: campus designation → campus, building code (DC12, Hall 3, Building B) → building, else facility. */
201 +export function inferRecordScope(nf: { name: string; recordScope?: string | null; campusName?: string | null }): "building" | "facility" | "campus" {
202 + if (nf.recordScope === "building" || nf.recordScope === "facility" || nf.recordScope === "campus") return nf.recordScope;
203 + if (isCampusName(nf.name)) return "campus";
204 + if (/\b(building|bldg|hall|data hall|phase)\s*[A-Z0-9]{1,3}\b/i.test(nf.name) || (nf.campusName && nf.campusName !== nf.name)) return "building";
205 + return "facility";
206 +}
207 +
171 208 async function resolveFacility(tx: Tx, ctx: IngestContext, nf: NormalizedFacility, operatorId: string | null, operatorName: string | null, countryIso2: string | null): Promise<Resolution> {
172 209 const byKey = await facilityIdForKey(tx, ctx, nf.key);
173 210 if (byKey) return { id: byKey, how: "key" };
@@ -183,6 +220,8 @@ async function resolveFacility(tx: Tx, ctx: IngestContext, nf: NormalizedFacilit
183 220 if (!best) return { id: null, how: "create" };
184 221 const d: MatchDecision = decide(best.match.score);
185 222 if (d === "merge") return { id: best.candidate.id, how: "merge", match: best };
223 + // campus vs one of its buildings: not a duplicate but a containment — create the record and link it to its parent
224 + if (best.match.reasons.includes("rule:campus-vs-building")) return { id: null, how: "campus_link", match: best };
186 225 if (d === "pending") return { id: null, how: "pending", match: best };
187 226 return { id: null, how: "create", match: best.match.score >= 0.3 ? best : null };
188 227 }
@@ -241,8 +280,21 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF
241 280
242 281 // --- field merge -----------------------------------------------------------------------------------------------
243 282 const before: Row = existing ? { ...existing } : {};
244 − const next: Row = existing ? { ...existing } : { id, geoPrecision: "unknown", status: "unknown", facilityType: "unknown", confidence: "unverified", completeness: 0, sourceCount: 0, isAi: false, isHyperscale: false, mwIsEstimate: false, externalIds: {}, certifications: [], aliases: [] };
283 + const next: Row = existing ? { ...existing } : { id, geoPrecision: "unknown", status: "unknown", facilityType: "unknown", confidence: "unverified", completeness: 0, sourceCount: 0, isAi: false, isHyperscale: false, mwIsEstimate: false, externalIds: {}, certifications: [], aliases: [], recordScope: "facility", aiEvidence: "unknown" };
245 284 const observed: ObservedField[] = [];
285 + // containment: a building that matched its campus record (or names its campus record by key) points at the parent
286 + if (res.how === "campus_link" && res.match) {
287 + const candIsCampus = isCampusName(res.match.candidate.name) && !isCampusName(name);
288 + if (candIsCampus) { next.parentFacilityId = res.match.candidate.id; next.recordScope = "building"; }
289 + else if (isCampusName(name) && !isCampusName(res.match.candidate.name) && !ctx.run.dryRun) {
290 + // the incoming record is the campus: the stored building becomes its child
291 + await tx.execute(sql`update facilities set parent_facility_id = ${id}, record_scope = 'building', updated_at = now() where id = ${res.match.candidate.id} and parent_facility_id is null`);
292 + }
293 + ctx.stats.campusLinks = (ctx.stats.campusLinks ?? 0) + 1;
294 + }
295 + if (nf.parentFacilityKey && !next.parentFacilityId) { const pid = await facilityIdForKey(tx, ctx, nf.parentFacilityKey); if (pid && pid !== id) { next.parentFacilityId = pid; next.recordScope = "building"; } }
296 + if (nf.developerName) { const dev = await resolveOperator(tx, ctx, { name: nf.developerName }); if (dev) { next.developerId = dev.id; observed.push({ field: "developerName", value: dev.name, provenance: pf("developerName") }); } }
297 + if (nf.landownerName) { const lo = await resolveOperator(tx, ctx, { name: nf.landownerName }); if (lo) { next.landownerId = lo.id; observed.push({ field: "landownerName", value: lo.name, provenance: pf("landownerName") }); } }
246 298 const incoming: Record<string, unknown> = {
247 299 name,
248 300 city: cleanText(nf.city),
@@ -278,13 +330,38 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF
278 330 const dec = shouldReplace(obs(v, p, ctx), storedObs(prov, field, existing?.[field]));
279 331 if (dec.replace) next[field] = v;
280 332 }
333 + // capacity figures go through the claim store: scope + semantics from the supporting sentence (structured operator
334 + // specs have none → the record's own scope), sanity engine, then the authority policy for the column
335 + const recordScope = inferRecordScope(nf);
336 + const campusDesignation = isCampusName(name, nf.campusName);
337 + const MW_PREDICATE: Record<(typeof MW_FIELDS)[number], CapacityPredicate> = { itCapacityMw: "it_capacity_mw", totalPowerMw: "current_power_mw", plannedPowerMw: "planned_power_mw" };
281 338 for (const field of MW_FIELDS) {
282 339 const v = nf[field];
283 340 if (v == null || !Number.isFinite(v) || v <= 0) continue;
284 341 const p = pf(field);
285 − observed.push({ field, value: v, provenance: p });
286 − const dec = shouldReplaceMw(obs(v, p, ctx), storedObs(prov, field, existing?.[field]));
287 − if (dec.replace) next[field] = v;
342 + const context = nf.claimContext?.[field] ?? null;
343 + const structured = !context; // a parser read a spec table / JSON field: the figure describes the record itself
344 + const sc = structured ? { scope: recordScope as ClaimScope, reason: "structured:record" } : classifyScope(context, recordScope as ClaimScope);
345 + const sem = structured ? { predicate: MW_PREDICATE[field], reason: "field" } : classifyCapacitySemantics(context);
346 + const predicate: CapacityPredicate = sem.predicate ?? MW_PREDICATE[field];
347 + const evidence = context ? findEvidence(context, v, "mw") ?? { text: context.slice(0, 600), start: 0, end: Math.min(600, context.length) } : null;
348 + const decision = await recordCapacityClaim(tx, ctx, "facility", id, { field, predicate, value: v, scope: sc.scope, scopeReason: sc.reason, evidence, context, previous: (existing?.[field] as number | null) ?? null, recordScope, campusDesignation, provenance: p, semanticsDefaulted: !structured && !sem.predicate });
349 + if (!decision.assign) { ctx.stats.unscopedClaims = (ctx.stats.unscopedClaims ?? 0) + 1; continue; }
350 + // utility / grid / ultimate figures live in their own columns, never in IT / total / planned
351 + const column = CAPACITY_COLUMN[predicate];
352 + const target = column && column !== "itCapacityMw" && column !== "totalPowerMw" && column !== "plannedPowerMw" ? column : field;
353 + observed.push({ field: target, value: v, provenance: { ...p, note: p.note ?? `${predicate} · ${sc.scope}` } });
354 + const dec = shouldReplaceMw(obs(v, p, ctx), storedObs(prov, target, existing?.[target]));
355 + if (dec.replace) { next[target] = v; if (target === field) { next.capacityScope = sc.scope; next.capacitySemantics = predicate; } }
356 + }
357 + next.recordScope = recordScope;
358 + // AI evidence: graded from the source text, never from one keyword
359 + {
360 + const ai = classifyAiEvidence(`${nf.aiEvidence === "confirmed" ? "AI campus" : ""} ${name} ${nf.description ?? ""} ${nf.facilityType === "ai" || nf.facilityType === "hpc" ? "AI data center" : ""}`);
361 + const level = nf.aiEvidence && nf.aiEvidence !== "unknown" ? nf.aiEvidence : ai.level;
362 + const rank: Record<string, number> = { unknown: 0, associated: 1, likely: 2, confirmed: 3 };
363 + if ((rank[level] ?? 0) > (rank[String(next.aiEvidence ?? "unknown")] ?? 0)) next.aiEvidence = level;
364 + if (level === "confirmed" || level === "likely") { next.isAi = true; if (nf.isAi !== true) observed.push({ field: "isAi", value: true, provenance: { ...P, method: `ai-evidence:${level}${ai.evidence ? `:${ai.evidence}` : ""}` } }); }
288 365 }
289 366 // operator / owner: authority policy like other fields (a news source never re-assigns an operator set by the operator itself);
290 367 // an operator inferred from the name never replaces one stated by a source
@@ -352,12 +429,14 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF
352 429 next.confidence = res.how === "pending" ? "unverified" : "moderate";
353 430 await tx.execute(sql`insert into facilities (id, slug, name, normalized_name, operator_id, owner_id, campus_id, metro_id, country_iso2, city, region_name, address, postal_code, lat, lng, geo_precision, geo_source, geohash,
354 431 status, facility_type, tier, building_sqm, site_area_ha, it_capacity_mw, total_power_mw, planned_power_mw, mw_is_estimate, rack_count, pue, cooling_type, renewable_claim, opened_on, construction_started_on, announced_on,
355 − website, description, is_ai, is_hyperscale, certifications, confidence, completeness, external_ids, source_count, first_seen)
432 + website, description, is_ai, is_hyperscale, certifications, confidence, completeness, external_ids, source_count, first_seen,
433 + parent_facility_id, record_scope, ai_evidence, utility_capacity_mw, grid_connection_mw, ultimate_campus_mw, capacity_scope, capacity_semantics, developer_id, landowner_id)
356 434 values (${id}, ${slug}, ${next.name}, ${normalizedName}, ${next.operatorId ?? null}, ${next.ownerId ?? null}, ${next.campusId ?? null}, ${next.metroId ?? null}, ${next.countryIso2 ?? null}, ${next.city ?? null}, ${next.regionName ?? null},
357 435 ${next.address ?? null}, ${next.postalCode ?? null}, ${next.lat ?? null}, ${next.lng ?? null}, ${next.geoPrecision ?? "unknown"}, ${next.geoSource ?? null}, ${next.geohash ?? null},
358 436 ${next.status ?? "unknown"}, ${next.facilityType ?? "unknown"}, ${next.tier ?? null}, ${next.buildingSqm ?? null}, ${next.siteAreaHa ?? null}, ${next.itCapacityMw ?? null}, ${next.totalPowerMw ?? null}, ${next.plannedPowerMw ?? null}, false,
359 437 ${next.rackCount ?? null}, ${next.pue ?? null}, ${next.coolingType ?? null}, ${next.renewableClaim ?? null}, ${next.openedOn ?? null}, ${next.constructionStartedOn ?? null}, ${next.announcedOn ?? null},
360 − ${next.website ?? null}, ${next.description ?? null}, ${!!next.isAi}, ${!!next.isHyperscale}, ${textArray((next.certifications as string[]) ?? [])}::text[], ${next.confidence}, 0, ${JSON.stringify(next.externalIds ?? {})}::jsonb, 0, ${ctx.now})`);
438 + ${next.website ?? null}, ${next.description ?? null}, ${!!next.isAi}, ${!!next.isHyperscale}, ${textArray((next.certifications as string[]) ?? [])}::text[], ${next.confidence}, 0, ${JSON.stringify(next.externalIds ?? {})}::jsonb, 0, ${ctx.now},
439 + ${next.parentFacilityId ?? null}, ${next.recordScope ?? "facility"}, ${next.aiEvidence ?? "unknown"}, ${next.utilityCapacityMw ?? null}, ${next.gridConnectionMw ?? null}, ${next.ultimateCampusMw ?? null}, ${next.capacityScope ?? null}, ${next.capacitySemantics ?? null}, ${next.developerId ?? null}, ${next.landownerId ?? null})`);
361 440 } else {
362 441 const sets = [];
363 442 for (const [camel, col] of Object.entries(COLUMNS)) {
@@ -385,8 +464,9 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF
385 464 const aliasValues = uniqStrings([...(nf.aliases ?? []), name !== next.name ? name : null, existing && String(existing.name) !== String(next.name) ? String(existing.name) : null]).filter((a) => normalizeName(a) && a !== next.name);
386 465 for (const a of aliasValues) await tx.execute(sql`insert into facility_aliases (facility_id, alias, normalized, source_id) values (${id}, ${a}, ${normalizeName(a)}, ${ctx.run.sourceId}) on conflict do nothing`);
387 466
388 − // provenance for every observed field
467 + // provenance for every observed field, then mark which observation backs each displayed value
389 468 await writeProvenance(tx, ctx, "facility", id, observed, nf.key);
469 + await markWinners(tx, ctx, "facility", id, [...SCALAR_FIELDS, ...MW_FIELDS, "utilityCapacityMw", "gridConnectionMw", "ultimateCampusMw", "countryIso2"].map((f) => ({ field: f, value: next[f] })).concat([{ field: "operatorId", value: next.operatorId }, { field: "ownerId", value: next.ownerId }]));
390 470
391 471 // tenants / IXPs
392 472 if (nf.carriers?.length) await attachTenants(tx, ctx, id, { names: nf.carriers, role: "carrier" });
@@ -408,6 +488,13 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF
408 488 } else if (res.how === "create" && res.match) {
409 489 await tx.execute(sql`insert into entity_matches (id, connector_id, candidate_key, candidate, matched_facility_id, score, reasons, status, decided_by, decided_at)
410 490 values (${newId("match")}, ${ctx.run.connectorId}, ${nf.key}, ${JSON.stringify(candidateJson(nf, id))}::jsonb, ${res.match.candidate.id}, ${res.match.match.score}, ${textArray(res.match.match.reasons)}::text[], 'auto_created', 'system', ${ctx.now})`);
491 + } else if (res.how === "campus_link" && res.match) {
492 + await tx.execute(sql`insert into entity_matches (id, connector_id, candidate_key, candidate, matched_facility_id, score, reasons, status, decided_by, decided_at)
493 + values (${newId("match")}, ${ctx.run.connectorId}, ${nf.key}, ${JSON.stringify(candidateJson(nf, id))}::jsonb, ${res.match.candidate.id}, ${res.match.match.score}, ${textArray(res.match.match.reasons)}::text[], 'related_campus', 'system', ${ctx.now})`);
494 + }
495 + // duplicate candidates awaiting review are a quality flag too (they inflate every aggregate until decided)
496 + if (res.how === "pending" && res.match) {
497 + await writeQualityFlags(tx, ctx, "facility", id, [{ code: "possible_duplicate", severity: "warn", message: `possible duplicate of ${res.match.candidate.name} (score ${res.match.match.score})`, field: "name" }], { details: { matchedFacilityId: res.match.candidate.id, reasons: res.match.match.reasons } });
411 498 }
412 499
413 500 // --- events ----------------------------------------------------------------------------------------------------
@@ -485,7 +572,7 @@ export async function refreshDerived(tx: Tx, id: string, forceConfidence: Confid
485 572 // a facility created from an ambiguous match stays "unverified" until the admin decides
486 573 const pending = forceConfidence ? [] : await tx.execute(sql`select 1 from entity_matches where status = 'pending' and candidate->>'createdFacilityId' = ${id} limit 1`);
487 574 const confidence = forceConfidence ?? (pending.length ? "unverified" : facilityConfidence({ sourceKinds: kinds, sourceCount: sourceIds.length, lastVerifiedIso: lastPrimaryObserved, onlyEstimates }));
488 − const counts = (await tx.execute(sql`select carriers_count, ixp_count from facilities where id = ${id}`))[0];
575 + const counts = (await tx.execute(sql`select carriers_count, ixp_count, (select coalesce(max(priority), 0) from quality_flags q where q.entity_type = 'facility' and q.entity_id = ${id} and q.status = 'open') as review_priority from facilities where id = ${id}`))[0];
489 576 const completeness = completenessScore({
490 577 geoPrecision: row.geoPrecision as string,
491 578 hasCoords: validLatLng(row.lat, row.lng),
@@ -503,7 +590,7 @@ export async function refreshDerived(tx: Tx, id: string, forceConfidence: Confid
503 590 carriersCount: (counts?.carriers_count as number | null) ?? null,
504 591 ixpCount: (counts?.ixp_count as number | null) ?? null,
505 592 });
506 − await tx.execute(sql`update facilities set mw_is_estimate = ${mwIsEstimate}, source_count = ${sourceIds.length}, last_verified = ${lastPrimaryObserved}, confidence = ${confidence}, completeness = ${completeness} where id = ${id}`);
593 + await tx.execute(sql`update facilities set mw_is_estimate = ${mwIsEstimate}, source_count = ${sourceIds.length}, last_verified = ${lastPrimaryObserved}, confidence = ${confidence}, completeness = ${completeness}, review_priority = ${Number(counts?.review_priority ?? 0)} where id = ${id}`);
507 594 }
508 595
509 596 /** Admin approval of a pending match: fold `fromId` into `intoId` (keys, aliases, provenance, links, projects). */
@@ -528,3 +615,18 @@ export async function mergeFacilities(tx: Tx, fromId: string, intoId: string, de
528 615 await refreshDerived(tx, intoId);
529 616 }
530 617
618 +
619 +/** Extraction debugger: how would this normalized facility resolve right now (candidates, scores, decision)? Read-only. */
620 +export async function previewFacilityResolution(tx: Tx, ctx: IngestContext, nf: NormalizedFacility): Promise<{ how: string; matchedId: string | null; candidates: Array<{ id: string; name: string; operatorName: string | null; city: string | null; score: number; reasons: string[]; distanceKm: number | null }> }> {
621 + const operator = nf.operatorName ? await resolveOperator(tx, ctx, { name: nf.operatorName, key: nf.operatorKey ?? null }) : null;
622 + const country = await safeCountry(tx, ctx, nf.countryIso2);
623 + const byKey = await facilityIdForKey(tx, ctx, nf.key);
624 + const byExt = byKey ? null : await byExternalIds(tx, nf.externalIds);
625 + const candidates = await loadCandidates(tx, nf, operator?.id ?? null, country);
626 + const geo = validGeo(nf.geo) ? nf.geo : null;
627 + const probe = { name: nf.name, aliases: nf.aliases, operatorId: operator?.id ?? null, operatorName: operator?.name ?? nf.operatorName ?? null, countryIso2: country, city: nf.city ?? null, address: nf.address ?? null, lat: geo?.lat ?? null, lng: geo?.lng ?? null, geoPrecision: geo?.precision ?? null };
628 + const scored = candidates.map((c) => { const m = scoreFacilityMatch(probe, c); return { id: c.id, name: c.name, operatorName: c.operatorName ?? null, city: c.city ?? null, score: m.score, reasons: m.reasons, distanceKm: m.distanceKm ?? null }; }).sort((a, b) => b.score - a.score).slice(0, 10);
629 + const best = scored[0];
630 + const how = byKey ? "key" : byExt ? "external_id" : !best ? "create" : best.reasons.includes("rule:campus-vs-building") ? "campus_link" : decide(best.score);
631 + return { how, matchedId: byKey ?? byExt ?? (best && (how === "merge") ? best.id : null), candidates: scored };
632 +}
added apps/worker/src/ingest/geocode.ts +43 −0
@@ -0,0 +1,43 @@
1 +/**
2 + * Offline city-level geocoding for projects (and facilities without coordinates): a metro seed (name / alias) or the
3 + * curated city table gives CITY-level coordinates, never more precise. Nothing is guessed: unknown city → null.
4 + */
5 +import { normalizeName, type GeoPoint } from "@dci/core";
6 +import { CITIES } from "../connectors/cloud/city-coords.js";
7 +import type { Tx } from "./common.js";
8 +import { loadMetros } from "./metros.js";
9 +
10 +const cityIndex = new Map<string, { lat: number; lng: number; iso: string; precision: GeoPoint["precision"]; name: string }>();
11 +for (const c of CITIES) {
12 + for (const a of [c.city, ...(c.aliases ?? [])]) {
13 + const k = `${c.iso}|${normalizeName(a)}`;
14 + if (!cityIndex.has(k)) cityIndex.set(k, { lat: c.lat, lng: c.lng, iso: c.iso, precision: c.precision ?? "city", name: c.city });
15 + }
16 +}
17 +
18 +export interface GeocodeInput { city?: string | null; regionName?: string | null; countryIso2?: string | null }
19 +
20 +/** City-level point for a place name, or null. `source` records how it was derived so provenance stays honest. */
21 +export async function geocodeCity(tx: Tx, i: GeocodeInput): Promise<(GeoPoint & { metroId?: string | null; label: string }) | null> {
22 + const city = i.city?.trim();
23 + if (!city) return null;
24 + const norm = normalizeName(city);
25 + if (!norm) return null;
26 + const metros = await loadMetros(tx);
27 + const metroHit = metros.find((m) => (!i.countryIso2 || m.countryIso2 === i.countryIso2) && m.normalizedAliases.includes(norm));
28 + if (metroHit) return { lat: metroHit.lat, lng: metroHit.lng, precision: "metro", source: `geocoder:metro:${metroHit.slug}`, metroId: metroHit.id, label: metroHit.name };
29 + if (i.countryIso2) {
30 + const hit = cityIndex.get(`${i.countryIso2}|${norm}`);
31 + if (hit) return { lat: hit.lat, lng: hit.lng, precision: hit.precision === "exact" ? "city" : hit.precision, source: "geocoder:city-table", label: hit.name };
32 + } else {
33 + const hits = [...cityIndex.entries()].filter(([k]) => k.endsWith(`|${norm}`));
34 + if (hits.length === 1) { const h = hits[0]![1]; return { lat: h.lat, lng: h.lng, precision: h.precision === "exact" ? "city" : h.precision, source: "geocoder:city-table", label: h.name }; }
35 + }
36 + // a county / parish in the lexicon may be a metro alias without the suffix
37 + const stripped = normalizeName(city.replace(/\b(county|parish|township|borough)\b/gi, ""));
38 + if (stripped && stripped !== norm) {
39 + const m2 = metros.find((m) => (!i.countryIso2 || m.countryIso2 === i.countryIso2) && m.normalizedAliases.includes(stripped));
40 + if (m2) return { lat: m2.lat, lng: m2.lng, precision: "approximate", source: `geocoder:metro:${m2.slug}`, metroId: m2.id, label: m2.name };
41 + }
42 + return null;
43 +}
modified apps/worker/src/ingest/index.ts +4 −1
@@ -98,7 +98,10 @@ export const ingestEntities: IngestFn = async (run: IngestRun, entities: Normali
98 98 };
99 99
100 100 export type { IngestDocRef, IngestFn, IngestRun, IngestStats } from "./contract.js";
101 −export { mergeFacilities, refreshDerived } from "./facilities.js";
101 +export { mergeFacilities, refreshDerived, previewFacilityResolution, isCampusName, inferRecordScope } from "./facilities.js";
102 +export { hideProject, mergeProjects } from "./projects.js";
103 +export { writeClaim, writeQualityFlags, recordCapacityClaim, recordInvestmentClaim, reviewPriority, bestCurrentClaim } from "./claims.js";
104 +export { geocodeCity } from "./geocode.js";
102 105 export { resolveOperator, lookupOperatorId, inferOperatorKind, mergeOperators } from "./operators.js";
103 106 export { findCanonicalOperator, CANONICAL_OPERATORS, HYPERSCALERS } from "./canonical-operators.js";
104 107 export { assignMetro, loadMetros, nearestMetro, invalidateMetroCache } from "./metros.js";
modified apps/worker/src/ingest/ixps.ts +1 −0
@@ -29,6 +29,7 @@ export async function resolveIxp(tx: Tx, ctx: IngestContext, i: IxpInput): Promi
29 29 }
30 30 if (!id && i.externalIds) {
31 31 for (const [k, v] of Object.entries(i.externalIds)) {
32 + if (v == null || v === "" || !/^(peeringdb_ix|peeringdb|pdb_ix|wikidata|ixpdb|pch_id|euro_ix)$/.test(k)) continue;
32 33 const r = await tx.execute(sql`select id from ixps where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`);
33 34 if (r[0]) {
34 35 id = String(r[0].id);
modified apps/worker/src/ingest/keys.ts +1 −1
@@ -25,6 +25,6 @@ export async function facilityIdForKey(tx: Tx, ctx: IngestContext, key: string):
25 25
26 26 export async function upsertKey(tx: Tx, ctx: IngestContext, key: string, entityType: string, entityId: string): Promise<void> {
27 27 await tx.execute(sql`insert into entity_keys (key, entity_type, entity_id, connector_id) values (${key}, ${entityType}, ${entityId}, ${ctx.run.connectorId})
28 − on conflict (key) do update set entity_id = excluded.entity_id, connector_id = excluded.connector_id`);
28 + on conflict (key, entity_type) do update set entity_id = excluded.entity_id, connector_id = excluded.connector_id`);
29 29 if (entityType === "facility") ctx.caches.facilityIdByKey.set(key, entityId);
30 30 }
modified apps/worker/src/ingest/news.ts +22 −7
@@ -3,7 +3,8 @@
3 3 * Items carrying an eventType with significance ≥ 40 also produce an `events` row (entityType news_event).
4 4 */
5 5 import { sql, textArray } from "@dci/db";
6 −import { cleanText, countryFromText, normalizeName, parseAllMw, stableId, type EventType, type NormalizedNewsEvent } from "@dci/core";
6 +import { cleanText, classifyAiEvidence, normalizeName, parseAllMw, stableId, type EventType, type NormalizedNewsEvent } from "@dci/core";
7 +import { countryFromTextSafe } from "../connectors/news/locations-lexicon.js";
7 8 import { addRef, bump, isoOrNull, knownCountries, type IngestContext, type Tx } from "./common.js";
8 9 import { CANONICAL_OPERATORS, findCanonicalOperator } from "./canonical-operators.js";
9 10 import { recordEvent } from "./events.js";
@@ -24,6 +25,16 @@ const EVENT_SIGNIFICANCE: Partial<Record<EventType, number>> = {
24 25 cloud_region_launched: 55,
25 26 incident: 70,
26 27 phase_announced: 50,
28 + land_acquired: 60,
29 + grid_connection: 65,
30 + grid_constraint: 70,
31 + utility_event: 50,
32 + operator_expansion: 65,
33 + customer_agreement: 45,
34 + partnership: 35,
35 + executive_change: 20,
36 + project_delayed: 70,
37 + project_cancelled: 80,
27 38 news: 30,
28 39 page_changed: 10,
29 40 };
@@ -98,7 +109,8 @@ export async function linkNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEven
98 109 }
99 110 const known = await knownCountries(tx, ctx);
100 111 const countries = new Set<string>();
101 − const fromText = countryFromText(text);
112 + // the hardened matcher: "North America" is not the US, "Georgia" is a US state, "… Jordan" is a person
113 + const fromText = countryFromTextSafe(text);
102 114 if (fromText && known.has(fromText)) countries.add(fromText);
103 115 for (const c of n.mentions?.countriesIso2 ?? []) if (c && known.has(c.toUpperCase())) countries.add(c.toUpperCase());
104 116 const facilityIds: string[] = [];
@@ -110,8 +122,9 @@ export async function linkNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEven
110 122 : await tx.execute(sql`select id from facilities where merged_into is null and normalized_name = ${norm} limit 2`);
111 123 if (rows.length === 1) facilityIds.push(String(rows[0]!.id));
112 124 }
113 − const mwList = n.mentions?.mw?.filter((v) => typeof v === "number" && v > 0) ?? [];
114 − const mw = mwList.length ? Math.max(...mwList) : (parseAllMw(text)[0] ?? null);
125 + // `mentions.mw` is already portfolio-filtered by the news parser; the fallback only reads the headline + summary
126 + const mwList = n.mentions?.mw?.filter((v) => typeof v === "number" && v > 0 && v <= 20_000) ?? [];
127 + const mw = mwList.length ? Math.max(...mwList) : (parseAllMw(text).filter((v) => v <= 20_000)[0] ?? null);
115 128 return { operatorIds, operatorNames: [...operatorNames], countryIso2s: [...countries], facilityIds, mw };
116 129 }
117 130
@@ -127,12 +140,13 @@ export async function ingestNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEv
127 140 const mentions = { ...(n.mentions ?? {}), linkedOperators: links.operatorNames };
128 141 // third-party publishers (kind = news): title, date, link and a ≤ 400-character summary only — never more text
129 142 const summary = ctx.run.sourceKind === "news" ? shortSummary(n.summary, THIRD_PARTY_SUMMARY_MAX) : cleanText(n.summary);
130 − const inserted = await tx.execute(sql`insert into news_items (id, source_id, connector_id, url, title, published_at, summary, page_type, event_type, mentions, operator_ids, country_iso2s, facility_ids, mw, significance)
143 + const ai = n.isAi ?? ["confirmed", "likely"].includes(classifyAiEvidence(`${title} ${n.summary ?? ""}`).level);
144 + const inserted = await tx.execute(sql`insert into news_items (id, source_id, connector_id, url, title, published_at, summary, page_type, event_type, mentions, operator_ids, country_iso2s, facility_ids, mw, significance, project_class)
131 145 values (${id}, ${ctx.run.sourceId}, ${ctx.run.connectorId}, ${url}, ${title.slice(0, 500)}, ${publishedAt}, ${summary}, ${n.pageType ?? "unknown"}, ${eventType}, ${JSON.stringify(mentions)}::jsonb,
132 − ${textArray(links.operatorIds)}::text[], ${textArray(links.countryIso2s)}::text[], ${textArray(links.facilityIds)}::text[], ${links.mw}, ${significance})
146 + ${textArray(links.operatorIds)}::text[], ${textArray(links.countryIso2s)}::text[], ${textArray(links.facilityIds)}::text[], ${links.mw}, ${significance}, ${n.projectClass ?? null})
133 147 on conflict (url) do update set title = excluded.title, summary = coalesce(excluded.summary, news_items.summary), published_at = coalesce(excluded.published_at, news_items.published_at),
134 148 event_type = coalesce(excluded.event_type, news_items.event_type), mentions = excluded.mentions, operator_ids = excluded.operator_ids, country_iso2s = excluded.country_iso2s,
135 − facility_ids = excluded.facility_ids, mw = coalesce(excluded.mw, news_items.mw), significance = greatest(excluded.significance, news_items.significance)
149 + facility_ids = excluded.facility_ids, mw = coalesce(excluded.mw, news_items.mw), significance = greatest(excluded.significance, news_items.significance), project_class = coalesce(excluded.project_class, news_items.project_class)
136 150 returning (xmax = 0) as inserted`);
137 151 const isNew = Boolean(inserted[0]?.inserted);
138 152 if (isNew) ctx.stats.created++;
@@ -156,6 +170,7 @@ export async function ingestNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEv
156 170 countryIso2: links.countryIso2s[0] ?? null,
157 171 operatorId: links.operatorIds[0] ?? null,
158 172 fingerprint: `news:${id}:${eventType}`,
173 + isAi: ai,
159 174 });
160 175 }
161 176 }
modified apps/worker/src/ingest/operators.ts +3 −1
@@ -76,10 +76,12 @@ async function byKey(tx: Tx, key: string): Promise<OperatorRow | null> {
76 76 return rows[0] ? rowFrom(rows[0]) : null;
77 77 }
78 78
79 +/** External-id namespaces that identify ONE operator (allowlist — a shared `hq_country` or `stock_exchange` must never fold two companies). */
80 +export const OPERATOR_IDENTIFYING_KEYS: ReadonlySet<string> = new Set(["wikidata", "wikipedia_en", "peeringdb_org", "peeringdb_net", "lei", "cik", "asn", "crunchbase", "linkedin", "gleif"]);
79 81 async function byExternalIds(tx: Tx, ext: Record<string, string | number> | undefined): Promise<OperatorRow | null> {
80 82 if (!ext) return null;
81 83 for (const [k, v] of Object.entries(ext)) {
82 − if (v == null || v === "") continue;
84 + if (v == null || v === "" || !OPERATOR_IDENTIFYING_KEYS.has(k)) continue;
83 85 const rows = await tx.execute(sql`select ${COLS} from operators where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`);
84 86 if (rows[0]) return rowFrom(rows[0]);
85 87 }
modified apps/worker/src/ingest/projects.ts +257 −58
@@ -1,17 +1,47 @@
1 1 /**
2 − * Project reconciliation: key → external ids → same sourceUrl → (operator, trigram(normalized name) ≥ 0.85, same country).
3 − * Timeline rows are deduplicated by (date, type, sha(description)); status changes become project_status_changed events.
2 + * Project reconciliation + persistence (docs/PROJECT-EXTRACTION.md, docs/CLAIMS.md).
3 + *
4 + * - The announcement CLASS decides first: appointments, financing, PPAs, partnerships, customer deals, market research
5 + * never create a project. Non-physical classes attach a timeline row + event to an existing, identifiable project.
6 + * - Evidence threshold: a new project needs (explicit name or operator) + location + a development verb.
7 + * - Capacity and investment go through the claim store (scope + sanity engine); a company-wide or portfolio figure is
8 + * kept as a claim and never written to the project's columns.
9 + * - Lifecycle transitions follow the state machine; a backward move needs a source that outranks the stored one.
10 + * - Projects without coordinates are geocoded at CITY / METRO level (never more precise) from the metro seeds.
4 11 */
5 12 import { sql } from "@dci/db";
6 −import { cleanText, newId, normalizeName, sha256, validLatLng, type ConfidenceLevel, type NormalizedProject } from "@dci/core";
7 −import { addRef, bump, provenanceFor, safeCountry, uniqueSlug, type IngestContext, type Tx } from "./common.js";
13 +import {
14 + ASSOCIATED_CLASSES,
15 + PHYSICAL_CLASSES,
16 + cleanText,
17 + classifyAiEvidence,
18 + classifyInvestmentSemantics,
19 + classifyScope,
20 + findEvidence,
21 + newId,
22 + normalizeName,
23 + projectTransition,
24 + sha256,
25 + validLatLng,
26 + type CapacityPredicate,
27 + type ClaimScope,
28 + type ConfidenceLevel,
29 + type EventType,
30 + type InvestmentPredicate,
31 + type NormalizedProject,
32 + type ProjectClass,
33 +} from "@dci/core";
34 +import { addRef, authority, bump, provenanceFor, safeCountry, uniqueSlug, type IngestContext, type Tx } from "./common.js";
8 35 import { emitDiffEvents, recordEvent, TRACKED_PROJECT_FIELDS } from "./events.js";
9 36 import { validGeo } from "./geo.js";
37 +import { geocodeCity } from "./geocode.js";
10 38 import { entityIdForKey, facilityIdForKey, upsertKey } from "./keys.js";
11 −import { isPipelineStatus, shouldReplace, shouldReplaceGeo, shouldReplaceMw, type FieldObservation } from "./match.js";
39 +import { isPipelineStatus, shouldReplace, shouldReplaceGeo, type FieldObservation } from "./match.js";
12 40 import { assignMetro } from "./metros.js";
13 41 import { operatorNames, resolveOperator } from "./operators.js";
14 42 import { backingObservation, loadCurrentProvenance, writeProvenance, type CurrentProvenance, type ObservedField } from "./provenance.js";
43 +import { markWinners, recordCapacityClaim, recordInvestmentClaim, reviewPriority, writeQualityFlags, writeTextClaim } from "./claims.js";
44 +import { resolveCampus } from "./campuses.js";
15 45
16 46 type Row = Record<string, unknown>;
17 47
@@ -31,6 +61,8 @@ const COLS: Record<string, string> = {
31 61 expectedOpening: "expected_opening",
32 62 plannedMw: "planned_mw",
33 63 investmentUsd: "investment_usd",
64 + investmentCurrency: "investment_currency",
65 + investmentOriginal: "investment_original",
34 66 acreage: "acreage",
35 67 phaseCount: "phase_count",
36 68 isAi: "is_ai",
@@ -38,8 +70,30 @@ const COLS: Record<string, string> = {
38 70 sourceUrl: "source_url",
39 71 confidence: "confidence",
40 72 externalIds: "external_ids",
73 + projectClass: "project_class",
74 + evidenceLevel: "evidence_level",
75 + aiEvidence: "ai_evidence",
76 + capacityScope: "capacity_scope",
77 + capacitySemantics: "capacity_semantics",
78 + investmentScope: "investment_scope",
79 + investmentSemantics: "investment_semantics",
80 + developerId: "developer_id",
81 + tenantId: "tenant_id",
82 + campusId: "campus_id",
83 + constructionStartedOn: "construction_started_on",
84 + approvedOn: "approved_on",
85 + permitFiledOn: "permit_filed_on",
86 + openedOn: "opened_on",
87 + reviewPriority: "review_priority",
88 + hidden: "hidden",
41 89 };
42 90
91 +/** External-id namespaces that identify ONE project (allowlist). */
92 +export const PROJECT_IDENTIFYING_KEYS: ReadonlySet<string> = new Set(["wikidata", "planning_ref", "permit_id", "case_number", "application_id", "docket", "planning_application", "rezoning_case", "eia_ref"]);
93 +
94 +/** Timeline / event type for a non-physical (associated) class. */
95 +const ASSOCIATED_EVENT: Record<string, EventType> = { POWER_AGREEMENT: "power_agreement", FINANCING: "investment_announced", ACQUISITION: "acquisition", PARTNERSHIP: "partnership", CUSTOMER_AGREEMENT: "customer_agreement", GRID_CONNECTION: "grid_connection", LAND_ACQUISITION: "land_acquired", PERMIT: "planning_filed", CONSTRUCTION_START: "construction_started", EXPANSION: "expansion_announced", NEW_BUILD: "project_announced" };
96 +
43 97 async function loadProject(tx: Tx, id: string): Promise<Row | null> {
44 98 const r = (await tx.execute(sql`select * from projects where id = ${id}`))[0];
45 99 if (!r) return null;
@@ -48,25 +102,38 @@ async function loadProject(tx: Tx, id: string): Promise<Row | null> {
48 102 return out;
49 103 }
50 104
105 +async function followMerged(tx: Tx, id: string): Promise<string> {
106 + let cur = id;
107 + for (let i = 0; i < 5; i++) {
108 + const r = await tx.execute(sql`select merged_into from projects where id = ${cur}`);
109 + const m = r[0]?.merged_into;
110 + if (!m) break;
111 + cur = String(m);
112 + }
113 + return cur;
114 +}
115 +
51 116 async function resolveProject(tx: Tx, ctx: IngestContext, p: NormalizedProject, operatorId: string | null, country: string | null): Promise<string | null> {
52 117 const byKey = await entityIdForKey(tx, p.key, "project");
53 − if (byKey) return byKey;
118 + if (byKey) return followMerged(tx, byKey);
54 119 if (p.externalIds) {
55 120 for (const [k, v] of Object.entries(p.externalIds)) {
56 − const r = await tx.execute(sql`select id from projects where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`);
121 + if (v == null || v === "" || !PROJECT_IDENTIFYING_KEYS.has(k)) continue;
122 + const r = await tx.execute(sql`select id from projects where merged_into is null and external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`);
57 123 if (r[0]) return String(r[0].id);
58 124 }
59 125 }
60 126 const src = p.sourceUrl ?? null;
61 127 if (src) {
62 − const r = await tx.execute(sql`select id from projects where source_url = ${src} limit 1`);
63 − if (r[0]) return String(r[0].id);
128 + const r = await tx.execute(sql`select id from projects where merged_into is null and source_url = ${src} limit 1`);
129 + if (r[0]) return followMerged(tx, String(r[0].id));
64 130 }
65 131 const norm = normalizeName(p.name);
66 132 if (!norm) return null;
67 133 const rows = await tx.execute(sql`
68 134 select id, similarity(normalized_name, ${norm}) as sim from projects
69 − where (${country}::text is null or country_iso2 is null or country_iso2 = ${country})
135 + where merged_into is null
136 + and (${country}::text is null or country_iso2 is null or country_iso2 = ${country})
70 137 and (${operatorId}::text is null or operator_id is null or operator_id = ${operatorId})
71 138 and (normalized_name = ${norm} or similarity(normalized_name, ${norm}) >= 0.85)
72 139 order by (operator_id = ${operatorId}) desc nulls last, sim desc limit 1`);
@@ -78,24 +145,25 @@ async function resolveProject(tx: Tx, ctx: IngestContext, p: NormalizedProject,
78 145 export const ANNOUNCEMENT_WINDOW_DAYS = 30;
79 146
80 147 /**
81 − * The same announcement covered by several outlets: same operator, same city (or same country when neither side
82 − * names a city), planned MW within ±10 % and announced within 30 days → one project; the other articles become
83 − * timeline rows / provenance. Requires an operator AND a size: without both, two "Ohio data center project"
84 − * rows may well be different sites.
148 + * The same announcement covered by several outlets: same operator (or, without an operator, the same city), planned MW
149 + * within ±10 % and announced within 30 days → one project. Without an operator the city is mandatory.
85 150 */
86 151 async function sameAnnouncement(tx: Tx, ctx: IngestContext, p: NormalizedProject, operatorId: string | null, country: string | null): Promise<string | null> {
87 − if (!operatorId || p.plannedMw == null || p.plannedMw <= 0 || !p.announcedOn) return null;
152 + if (p.plannedMw == null || p.plannedMw <= 0 || !p.announcedOn) return null;
88 153 const day = p.announcedOn.length >= 10 ? p.announcedOn.slice(0, 10) : null;
89 154 if (!day) return null;
90 155 const city = p.city ? normalizeName(p.city) : null;
156 + if (!operatorId && !city) return null;
91 157 const rows = await tx.execute(sql`
92 158 select id from projects
93 − where operator_id = ${operatorId} and planned_mw is not null
159 + where merged_into is null and planned_mw is not null
94 160 and planned_mw between ${p.plannedMw * 0.9} and ${p.plannedMw * 1.1}
95 161 and announced_on is not null and length(announced_on) >= 10
96 162 and abs(announced_on::date - ${day}::date) <= ${ANNOUNCEMENT_WINDOW_DAYS}
97 163 and (${country}::text is null or country_iso2 is null or country_iso2 = ${country})
98 − and ((${city}::text is not null and city is not null and lower(regexp_replace(city, '[^A-Za-z0-9]+', ' ', 'g')) = ${city}) or (${city}::text is null and city is null))
164 + and (${operatorId}::text is null or operator_id is null or operator_id = ${operatorId})
165 + and (${city}::text is null or city is null or lower(regexp_replace(city, '[^A-Za-z0-9]+', ' ', 'g')) = ${city})
166 + and (${operatorId}::text is not null or (${city}::text is not null and city is not null))
99 167 order by created_at asc limit 1`);
100 168 if (!rows[0]) return null;
101 169 ctx.stats.projectDedup = (ctx.stats.projectDedup ?? 0) + 1;
@@ -112,85 +180,189 @@ function stored(prov: CurrentProvenance[], field: string, current: unknown): Fie
112 180 return { value: current, sourceKind: b?.sourceKind ?? null, confidence: b?.confidence ?? null, isEstimate: b?.isEstimate ?? false, observedAt: b?.lastObserved ?? null, sourceId: b?.sourceId ?? null, url: b?.url ?? null };
113 181 }
114 182
183 +const isCampusName = (s: string | null | undefined) => /\b(campus|park|complex|hub|cluster|gigafactory|estate)\b/i.test(s ?? "");
184 +
115 185 export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedProject): Promise<void> {
116 186 const name = cleanText(p.name);
117 187 if (!name) throw new Error(`project ${p.key}: name is required`);
118 188 const url = p.provenance.url || p.sourceUrl || ctx.doc?.url || "";
119 189 if (!url) throw new Error(`project ${p.key}: provenance url is required`);
190 +
191 + const cls = (p.projectClass ?? null) as ProjectClass | null;
192 + const physical = cls == null || PHYSICAL_CLASSES.has(cls);
193 + const associated = cls != null && ASSOCIATED_CLASSES.has(cls);
194 +
120 195 const operator = p.operatorName ? await resolveOperator(tx, ctx, { name: p.operatorName }) : null;
121 196 const country0 = await safeCountry(tx, ctx, p.countryIso2);
122 − const geo = validGeo(p.geo) ? p.geo : null;
197 + let geo = validGeo(p.geo) ? p.geo : null;
198 + let geoMethod = geo ? "source" : null;
199 + if (!geo && p.city) {
200 + const g = await geocodeCity(tx, { city: p.city, regionName: p.regionName, countryIso2: country0 });
201 + if (g) { geo = { lat: g.lat, lng: g.lng, precision: g.precision, source: g.source }; geoMethod = g.source; }
202 + }
123 203 const metro = await assignMetro(tx, { lat: geo?.lat, lng: geo?.lng, city: p.city, countryIso2: country0 });
124 204 const country = country0 ?? metro.countryIso2;
125 205 const facilityId = p.facilityKey ? await facilityIdForKey(tx, ctx, p.facilityKey) : null;
126 206
127 207 const existingId = await resolveProject(tx, ctx, p, operator?.id ?? null, country);
128 208 const existing = existingId ? await loadProject(tx, existingId) : null;
209 +
210 + // ─── veto: non-physical announcements never create a project; weak evidence never creates a project ─────────────
211 + if (!existing) {
212 + if (!physical) { ctx.stats.projectsVetoed = (ctx.stats.projectsVetoed ?? 0) + 1; return; }
213 + if (p.evidenceLevel === "none") { ctx.stats.projectsVetoed = (ctx.stats.projectsVetoed ?? 0) + 1; return; }
214 + }
215 + // ─── associated class on an existing project: timeline row + event only, no field changes ─────────────────────────
216 + if (existing && !physical) {
217 + const evType = (cls && ASSOCIATED_EVENT[cls]) || "news";
218 + const date = p.announcedOn ?? ctx.day;
219 + const desc = cleanText(p.description) ?? name;
220 + const tid = `ptl_${sha256(`${existing.id}|${date}|${evType}|${desc.toLowerCase()}`).slice(0, 20)}`;
221 + const ins = await tx.execute(sql`insert into project_timeline (id, project_id, event_date, event_type, description, source_id, document_id, url)
222 + values (${tid}, ${existing.id}, ${date}, ${evType}, ${desc.slice(0, 400)}, ${ctx.run.sourceId}, ${ctx.doc?.documentId ?? null}, ${url}) on conflict (id) do nothing returning id`);
223 + if (ins.length && associated) {
224 + await recordEvent(tx, ctx, { entityType: "project", entityId: String(existing.id), eventType: evType, title: `${String(existing.name)}: ${cls!.toLowerCase().replace(/_/g, " ")} — ${name}`.slice(0, 300), summary: desc.slice(0, 400), newValue: { class: cls, title: name, mw: p.plannedMw ?? null, investmentUsd: p.investmentUsd ?? null }, significance: cls === "POWER_AGREEMENT" || cls === "GRID_CONNECTION" ? 60 : cls === "FINANCING" ? 55 : 45, confidence: p.provenance.confidence, effectiveDate: p.announcedOn ?? null, url, countryIso2: existing.countryIso2 as string | null, operatorId: existing.operatorId as string | null, metroId: existing.metroId as string | null, projectId: String(existing.id) });
225 + await tx.execute(sql`update projects set last_update = ${ctx.now} where id = ${existing.id}`);
226 + }
227 + // money attached to a deal / financing headline is a claim about the project, never its investment column
228 + if (p.investmentUsd != null && p.investmentUsd > 0) {
229 + const sem = classifyInvestmentSemantics(p.claimContext?.investmentUsd ?? name);
230 + await recordInvestmentClaim(tx, ctx, "project", String(existing.id), { value: p.investmentUsd, currency: p.investmentCurrency ?? "USD", predicate: cls === "FINANCING" || cls === "ACQUISITION" ? "deal_value_usd" : (sem.predicate as InvestmentPredicate), scope: cls === "FINANCING" || cls === "ACQUISITION" ? "company" : sem.scope, evidence: p.claimContext?.investmentUsd ? findEvidence(p.claimContext.investmentUsd, p.investmentUsd, "usd") : null, recordScope: "project", publishedAt: p.announcedOn ?? null, provenance: p.provenance });
231 + }
232 + addRef(ctx, "project", String(existing.id));
233 + bump(ctx, "project");
234 + return;
235 + }
236 +
129 237 const prov = existing ? await loadCurrentProvenance(tx, "project", existingId!) : [];
130 238 const id = existing ? String(existing.id) : newId("project");
131 239 const before: Row = existing ? { ...existing } : {};
132 − const next: Row = existing ? { ...existing } : { id, status: "announced", geoPrecision: "unknown", isAi: false, confidence: "moderate", externalIds: {} };
240 + const next: Row = existing ? { ...existing } : { id, status: "announced", geoPrecision: "unknown", isAi: false, confidence: "moderate", externalIds: {}, aiEvidence: "unknown", hidden: false, reviewPriority: 0 };
133 241 const observed: ObservedField[] = [];
242 + const campusDesignation = isCampusName(name) || isCampusName(p.campusName);
134 243
135 244 const scalars: Record<string, unknown> = {
136 245 name,
137 246 city: cleanText(p.city),
138 247 regionName: cleanText(p.regionName),
139 − status: p.status && p.status !== "unknown" ? p.status : null,
140 248 announcedOn: p.announcedOn ?? null,
141 249 expectedOpening: p.expectedOpening ?? null,
142 − investmentUsd: p.investmentUsd ?? null,
143 250 acreage: p.acreage ?? null,
144 251 phaseCount: p.phaseCount ?? null,
145 252 description: cleanText(p.description),
146 253 sourceUrl: p.sourceUrl ?? null,
254 + constructionStartedOn: p.constructionStartedOn ?? null,
255 + approvedOn: p.approvedOn ?? null,
256 + permitFiledOn: p.permitFiledOn ?? null,
257 + countryIso2: country,
147 258 };
148 259 for (const [field, v] of Object.entries(scalars)) {
149 260 if (v == null) continue;
150 261 observed.push({ field, value: v, provenance: p.provenance });
151 262 if (shouldReplace(obs(v, field, p, ctx), stored(prov, field, existing?.[field])).replace) next[field] = v;
152 263 }
264 +
265 + // ─── status: lifecycle state machine ────────────────────────────────────────────────────────────────────────────────
266 + const incomingStatus = p.status && p.status !== "unknown" ? p.status : null;
267 + if (incomingStatus) {
268 + observed.push({ field: "status", value: incomingStatus, provenance: p.provenance });
269 + const verdict = projectTransition(existing?.status as string | null, incomingStatus);
270 + if (verdict === "forward" || verdict === "side" || verdict === "resume" || (verdict === "same" && !existing)) next.status = incomingStatus;
271 + else if (verdict === "backward") {
272 + // a backward move is accepted only from a source that outranks the stored one (a government filing correcting a news story)
273 + const st = stored(prov, "status", existing?.status);
274 + if (authority(ctx.run.sourceKind, p.provenance.confidence, false) > authority(st?.sourceKind, st?.confidence, false)) next.status = incomingStatus;
275 + else await writeQualityFlags(tx, ctx, "project", id, [{ code: "status_backward", severity: "warn", message: `source says ${incomingStatus} but project is ${String(existing?.status)} — kept, lower-authority source`, field: "status" }]);
276 + } else if (verdict === "invalid") {
277 + await writeQualityFlags(tx, ctx, "project", id, [{ code: "status_invalid_transition", severity: "warn", message: `invalid lifecycle transition ${String(existing?.status)} → ${incomingStatus}`, field: "status" }]);
278 + }
279 + // stage dates derived from the announcement that moved the stage
280 + const day = p.announcedOn ?? null;
281 + if (day) {
282 + if (incomingStatus === "under_construction" && !next.constructionStartedOn) next.constructionStartedOn = day;
283 + if (incomingStatus === "approved" && !next.approvedOn) next.approvedOn = day;
284 + if (incomingStatus === "permitting" && !next.permitFiledOn) next.permitFiledOn = day;
285 + if ((incomingStatus === "operational" || incomingStatus === "partially_operational") && !next.openedOn) next.openedOn = day;
286 + }
287 + await writeTextClaim(tx, ctx, "project", id, "status", incomingStatus, p.provenance, { scope: campusDesignation ? "campus" : "facility", publishedAt: p.announcedOn ?? null });
288 + }
289 +
290 + // ─── capacity through the claim store ───────────────────────────────────────────────────────────────────────────────
153 291 if (p.plannedMw != null && p.plannedMw > 0) {
154 − observed.push({ field: "plannedMw", value: p.plannedMw, provenance: p.provenance });
155 − if (shouldReplaceMw(obs(p.plannedMw, "plannedMw", p, ctx), stored(prov, "plannedMw", existing?.plannedMw)).replace) next.plannedMw = p.plannedMw;
292 + const context = p.claimContext?.plannedMw ?? null;
293 + const scopeHint: ClaimScope = campusDesignation ? "campus" : "facility";
294 + const sc = p.capacityScope ? { scope: p.capacityScope as ClaimScope, reason: "extractor" } : classifyScope(context, context ? null : scopeHint);
295 + const predicate = (p.capacitySemantics ?? "planned_power_mw") as CapacityPredicate;
296 + const evidence = context ? findEvidence(context, p.plannedMw, "mw") ?? { text: context.slice(0, 600), start: 0, end: Math.min(600, context.length) } : null;
297 + const decision = await recordCapacityClaim(tx, ctx, "project", id, { field: "plannedMw", predicate, value: p.plannedMw, scope: sc.scope, scopeReason: sc.reason, evidence, context, previous: (existing?.plannedMw as number | null) ?? null, recordScope: "project", campusDesignation, publishedAt: p.announcedOn ?? null, provenance: p.provenance, semanticsDefaulted: !p.capacitySemantics });
298 + if (decision.assign) {
299 + observed.push({ field: "plannedMw", value: p.plannedMw, provenance: p.provenance, scope: sc.scope });
300 + if (shouldReplace(obs(p.plannedMw, "plannedMw", p, ctx), stored(prov, "plannedMw", existing?.plannedMw)).replace) { next.plannedMw = p.plannedMw; next.capacityScope = sc.scope; next.capacitySemantics = predicate; }
301 + } else ctx.stats.unscopedClaims = (ctx.stats.unscopedClaims ?? 0) + 1;
156 302 }
303 + // ─── investment through the claim store ─────────────────────────────────────────────────────────────────────────────
304 + const money = p.investmentOriginal != null && p.investmentOriginal > 0 ? { amount: p.investmentOriginal, currency: p.investmentCurrency ?? "USD" } : p.investmentUsd != null && p.investmentUsd > 0 ? { amount: p.investmentUsd, currency: "USD" } : null;
305 + if (money) {
306 + const context = p.claimContext?.investmentUsd ?? null;
307 + const sem = p.investmentSemantics && p.investmentScope ? { predicate: p.investmentSemantics as InvestmentPredicate, scope: p.investmentScope as ClaimScope, reason: "extractor" } : classifyInvestmentSemantics(context ?? name, campusDesignation ? "campus" : "facility");
308 + const evidence = context ? findEvidence(context, money.amount, "usd") ?? { text: context.slice(0, 600), start: 0, end: Math.min(600, context.length) } : null;
309 + const decision = await recordInvestmentClaim(tx, ctx, "project", id, { value: money.amount, currency: money.currency, predicate: sem.predicate, scope: sem.scope, scopeReason: sem.reason, evidence, context, previous: (existing?.investmentUsd as number | null) ?? null, recordScope: "project", publishedAt: p.announcedOn ?? null, provenance: p.provenance });
310 + if (decision.assign && p.investmentUsd != null) {
311 + observed.push({ field: "investmentUsd", value: p.investmentUsd, provenance: p.provenance, scope: sem.scope });
312 + if (shouldReplace(obs(p.investmentUsd, "investmentUsd", p, ctx), stored(prov, "investmentUsd", existing?.investmentUsd)).replace) { next.investmentUsd = p.investmentUsd; next.investmentScope = sem.scope; next.investmentSemantics = sem.predicate; }
313 + } else ctx.stats.unscopedClaims = (ctx.stats.unscopedClaims ?? 0) + 1;
314 + if (money.currency !== "USD") { next.investmentCurrency = money.currency; next.investmentOriginal = money.amount; }
315 + }
316 +
317 + // ─── related entities ───────────────────────────────────────────────────────────────────────────────────────────────
157 318 if (operator) {
158 319 observed.push({ field: "operatorName", value: operator.name, provenance: p.provenance });
159 320 if (shouldReplace(obs(operator.id, "operatorName", p, ctx), stored(prov, "operatorId", existing?.operatorId)).replace) next.operatorId = operator.id;
160 321 }
322 + if (p.developerName) { const d = await resolveOperator(tx, ctx, { name: p.developerName }); if (d) { next.developerId = d.id; observed.push({ field: "developerName", value: d.name, provenance: p.provenance }); } }
323 + if (p.tenantName) { const t = await resolveOperator(tx, ctx, { name: p.tenantName }); if (t) { next.tenantId = t.id; observed.push({ field: "tenantName", value: t.name, provenance: p.provenance }); } }
161 324 if (facilityId) next.facilityId = facilityId;
325 + if (p.campusName) { const c = await resolveCampus(tx, ctx, { name: p.campusName, operatorId: operator?.id ?? null, countryIso2: country, city: p.city ?? null, lat: geo?.lat ?? null, lng: geo?.lng ?? null }); if (c) next.campusId = c.id; }
162 326 if (geo) {
163 − observed.push({ field: "geo", value: { lat: geo.lat, lng: geo.lng, precision: geo.precision, source: geo.source }, provenance: p.provenance });
164 − if (shouldReplaceGeo(geo.precision, existing?.geoPrecision as string | null, validLatLng(existing?.lat, existing?.lng))) {
165 − next.lat = geo.lat;
166 − next.lng = geo.lng;
167 − next.geoPrecision = geo.precision;
168 − }
169 − }
170 − if (country) {
171 − observed.push({ field: "countryIso2", value: country, provenance: p.provenance });
172 − next.countryIso2 = next.countryIso2 ?? country;
327 + observed.push({ field: "geo", value: { lat: geo.lat, lng: geo.lng, precision: geo.precision, source: geo.source }, provenance: { ...p.provenance, method: geoMethod ?? p.provenance.method } });
328 + if (shouldReplaceGeo(geo.precision, existing?.geoPrecision as string | null, validLatLng(existing?.lat, existing?.lng))) { next.lat = geo.lat; next.lng = geo.lng; next.geoPrecision = geo.precision; }
173 329 }
174 330 {
175 331 const m = validLatLng(next.lat, next.lng) ? await assignMetro(tx, { lat: next.lat as number, lng: next.lng as number, city: next.city as string | null, countryIso2: next.countryIso2 as string | null }) : metro;
176 332 if (m.metroId) next.metroId = m.metroId;
333 + if (!next.countryIso2 && m.countryIso2) next.countryIso2 = m.countryIso2;
334 + }
335 + // ─── AI evidence: graded, never from one keyword ──────────────────────────────────────────────────────────────────
336 + {
337 + const rank: Record<string, number> = { unknown: 0, associated: 1, likely: 2, confirmed: 3 };
338 + const graded = classifyAiEvidence(`${name} ${p.description ?? ""}`);
339 + const level = p.aiEvidence && p.aiEvidence !== "unknown" ? p.aiEvidence : graded.level;
340 + if ((rank[level] ?? 0) > (rank[String(next.aiEvidence ?? "unknown")] ?? 0)) next.aiEvidence = level;
341 + next.isAi = next.aiEvidence === "confirmed" || next.aiEvidence === "likely";
177 342 }
178 − if (/\b(ai|gpu|hpc|accelerated|inference|training)\b/i.test(`${name} ${p.description ?? ""}`)) next.isAi = true;
343 + if (cls) { next.projectClass = cls; if (p.evidenceLevel) next.evidenceLevel = p.evidenceLevel; }
179 344 if (p.externalIds && Object.keys(p.externalIds).length) next.externalIds = { ...((next.externalIds as Record<string, unknown>) ?? {}), ...p.externalIds };
180 345
346 + // ─── persist ────────────────────────────────────────────────────────────────────────────────────────────────────────
181 347 const normalizedName = normalizeName(String(next.name));
182 348 if (!existing) {
183 349 const slug = await uniqueSlug(tx, "projects", `${operator && !normalizeName(name).includes(normalizeName(operator.name)) ? `${operator.name} ` : ""}${name}`, (next.city as string | null) ?? (next.countryIso2 as string | null));
350 + next.confidence = p.evidenceLevel === "weak" ? "unverified" : (p.provenance.confidence ?? "moderate");
184 351 await tx.execute(sql`insert into projects (id, slug, name, normalized_name, operator_id, facility_id, metro_id, country_iso2, city, region_name, lat, lng, geo_precision, status, announced_on, expected_opening, planned_mw,
185 − investment_usd, acreage, phase_count, is_ai, description, source_url, confidence, external_ids, last_update)
352 + investment_usd, investment_currency, investment_original, acreage, phase_count, is_ai, description, source_url, confidence, external_ids, last_update,
353 + project_class, evidence_level, ai_evidence, capacity_scope, capacity_semantics, investment_scope, investment_semantics, developer_id, tenant_id, campus_id, construction_started_on, approved_on, permit_filed_on, opened_on, hidden)
186 354 values (${id}, ${slug}, ${next.name}, ${normalizedName}, ${next.operatorId ?? null}, ${next.facilityId ?? null}, ${next.metroId ?? null}, ${next.countryIso2 ?? null}, ${next.city ?? null}, ${next.regionName ?? null},
187 355 ${next.lat ?? null}, ${next.lng ?? null}, ${next.geoPrecision ?? "unknown"}, ${next.status ?? "announced"}, ${next.announcedOn ?? null}, ${next.expectedOpening ?? null}, ${next.plannedMw ?? null},
188 − ${next.investmentUsd ?? null}, ${next.acreage ?? null}, ${next.phaseCount ?? null}, ${!!next.isAi}, ${next.description ?? null}, ${next.sourceUrl ?? null}, ${p.provenance.confidence ?? "moderate"}, ${JSON.stringify(next.externalIds ?? {})}::jsonb, ${ctx.now})`);
356 + ${next.investmentUsd ?? null}, ${next.investmentCurrency ?? null}, ${next.investmentOriginal ?? null}, ${next.acreage ?? null}, ${next.phaseCount ?? null}, ${!!next.isAi}, ${next.description ?? null}, ${next.sourceUrl ?? null}, ${next.confidence}, ${JSON.stringify(next.externalIds ?? {})}::jsonb, ${ctx.now},
357 + ${next.projectClass ?? null}, ${next.evidenceLevel ?? null}, ${next.aiEvidence ?? "unknown"}, ${next.capacityScope ?? null}, ${next.capacitySemantics ?? null}, ${next.investmentScope ?? null}, ${next.investmentSemantics ?? null}, ${next.developerId ?? null}, ${next.tenantId ?? null}, ${next.campusId ?? null},
358 + ${next.constructionStartedOn ?? null}, ${next.approvedOn ?? null}, ${next.permitFiledOn ?? null}, ${next.openedOn ?? null}, false)`);
189 359 ctx.stats.created++;
360 + if (p.evidenceLevel === "weak") await writeQualityFlags(tx, ctx, "project", id, [{ code: "project_weak_evidence", severity: "warn", message: "created from weak evidence (missing operator/name or location) — review", field: "name" }]);
361 + if (/\s[|]\s|\s[–—]\s| - /.test(name) || name.length > 90) await writeQualityFlags(tx, ctx, "project", id, [{ code: "project_title_like_name", severity: "warn", message: "project name looks like an article headline", field: "name" }]);
190 362 } else {
191 363 const sets = [];
192 364 for (const [camel, col] of Object.entries(COLS)) {
193 − if (camel === "confidence") continue;
365 + if (camel === "confidence" || camel === "reviewPriority" || camel === "hidden") continue;
194 366 if (JSON.stringify(before[camel] ?? null) === JSON.stringify(next[camel] ?? null)) continue;
195 367 if (camel === "externalIds") sets.push(sql`${sql.identifier(col)} = ${JSON.stringify(next[camel] ?? {})}::jsonb`);
196 368 else sets.push(sql`${sql.identifier(col)} = ${next[camel] as string | number | boolean | null}`);
@@ -206,10 +378,21 @@ export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedPro
206 378 addRef(ctx, "project", id);
207 379 bump(ctx, "project");
208 380 await writeProvenance(tx, ctx, "project", id, observed, p.key);
381 + await markWinners(tx, ctx, "project", id, ["status", "plannedMw", "investmentUsd", "expectedOpening", "announcedOn", "countryIso2", "city", "name"].map((f) => ({ field: f, value: next[f] })).concat([{ field: "operatorId", value: next.operatorId }]));
382 + // review priority from open flags + size
383 + {
384 + const r = (await tx.execute(sql`select coalesce(max(priority), 0) as p from quality_flags where entity_type = 'project' and entity_id = ${id} and status = 'open'`))[0];
385 + const rp = Math.max(Number(r?.p ?? 0), reviewPriority({ mw: next.plannedMw as number | null, investmentUsd: next.investmentUsd as number | null, confidence: String(next.confidence ?? ""), ai: !!next.isAi }) - 30);
386 + if (!ctx.run.dryRun) await tx.execute(sql`update projects set review_priority = ${Math.max(0, rp)} where id = ${id}`);
387 + }
209 388
210 − // timeline (deduplicated); another outlet's coverage of an existing project is kept as a "reported" row
389 + // ─── timeline (deduplicated) ────────────────────────────────────────────────────────────────────────────────────────
211 390 let timelineAdded = 0;
212 391 const timeline = [...(p.timeline ?? [])];
392 + if (!timeline.length && p.announcedOn && cls) {
393 + const evType = ASSOCIATED_EVENT[cls] ?? "project_announced";
394 + timeline.push({ date: p.announcedOn, type: incomingStatus === "under_construction" ? "construction_started" : incomingStatus === "approved" ? "planning_approved" : incomingStatus === "permitting" ? "planning_filed" : evType, description: cleanText(p.description)?.slice(0, 300) ?? name, url });
395 + }
213 396 if (existing && p.sourceUrl && existing.sourceUrl && p.sourceUrl !== existing.sourceUrl && p.announcedOn && p.description) {
214 397 timeline.push({ date: p.announcedOn, type: "reported", description: `Also reported: ${String(p.description).slice(0, 300)}`, url: p.sourceUrl });
215 398 }
@@ -222,18 +405,19 @@ export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedPro
222 405 }
223 406 if (timelineAdded && existing) await tx.execute(sql`update projects set last_update = ${ctx.now} where id = ${id}`);
224 407
225 − // events
226 − const confidence = (existing ? String(existing.confidence) : (p.provenance.confidence ?? "moderate")) as ConfidenceLevel;
408 + // ─── events ─────────────────────────────────────────────────────────────────────────────────────────────────────────
409 + const confidence = (existing ? String(existing.confidence) : String(next.confidence ?? "moderate")) as ConfidenceLevel;
227 410 if (!existing) {
228 411 const mw = next.plannedMw as number | null;
229 412 const where = [next.city, next.countryIso2].filter(Boolean).join(", ");
413 + const evType: EventType = incomingStatus === "under_construction" ? "construction_started" : incomingStatus === "approved" ? "planning_approved" : incomingStatus === "permitting" ? "planning_filed" : cls === "EXPANSION" ? "expansion_announced" : cls === "LAND_ACQUISITION" ? "land_acquired" : cls === "GRID_CONNECTION" ? "grid_connection" : "project_announced";
230 414 await recordEvent(tx, ctx, {
231 415 entityType: "project",
232 416 entityId: id,
233 − eventType: "project_announced",
417 + eventType: evType,
234 418 title: `New project: ${next.name}${operator && !String(next.name).toLowerCase().includes(operator.name.toLowerCase()) ? ` (${operator.name})` : ""}${where ? ` — ${where}` : ""}`,
235 − summary: [mw != null ? `${mw} MW planned.` : null, next.status ? `Status: ${String(next.status).replace(/_/g, " ")}.` : null, next.expectedOpening ? `Expected ${String(next.expectedOpening)}.` : null].filter(Boolean).join(" ") || null,
236 − newValue: { name: next.name, status: next.status, plannedMw: mw },
419 + summary: [mw != null ? `${mw} MW planned (${String(next.capacityScope ?? "site")} scope).` : null, next.status ? `Status: ${String(next.status).replace(/_/g, " ")}.` : null, next.expectedOpening ? `Expected ${String(next.expectedOpening)}.` : null].filter(Boolean).join(" ") || null,
420 + newValue: { name: next.name, status: next.status, plannedMw: mw, class: cls },
237 421 significance: mw != null && mw >= 100 ? 75 : isPipelineStatus(next.status as string) ? 60 : 50,
238 422 confidence,
239 423 effectiveDate: (next.announcedOn as string | null) ?? null,
@@ -242,23 +426,38 @@ export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedPro
242 426 operatorId: next.operatorId as string | null,
243 427 metroId: next.metroId as string | null,
244 428 projectId: id,
429 + isAi: !!next.isAi,
430 + reviewStatus: p.evidenceLevel === "weak" ? "pending" : "auto",
245 431 });
246 432 } else {
247 433 const names = await operatorNames(tx, ctx, [before.operatorId as string | null, next.operatorId as string | null]);
248 − await emitDiffEvents(tx, ctx, {
249 − entityType: "project",
250 − entityId: id,
251 − entityName: String(next.name),
252 − before,
253 − after: next,
254 − specs: TRACKED_PROJECT_FIELDS,
255 − url,
256 − confidence,
257 − countryIso2: next.countryIso2 as string | null,
258 − operatorId: next.operatorId as string | null,
259 − metroId: next.metroId as string | null,
260 − projectId: id,
261 − operatorNames: names,
262 − });
434 + await emitDiffEvents(tx, ctx, { entityType: "project", entityId: id, entityName: String(next.name), before, after: next, specs: TRACKED_PROJECT_FIELDS, url, confidence, countryIso2: next.countryIso2 as string | null, operatorId: next.operatorId as string | null, metroId: next.metroId as string | null, projectId: id, operatorNames: names });
435 + // delayed / cancelled get their own, louder event types on top of the generic status change
436 + if (before.status !== next.status && (next.status === "delayed" || next.status === "cancelled")) {
437 + await recordEvent(tx, ctx, { entityType: "project", entityId: id, eventType: next.status === "delayed" ? "project_delayed" : "project_cancelled", title: `${String(next.name)}: ${next.status === "delayed" ? "delayed" : "cancelled"}`, summary: cleanText(p.description)?.slice(0, 400) ?? null, oldValue: before.status, newValue: next.status, significance: next.status === "cancelled" ? 85 : 70, confidence, effectiveDate: p.announcedOn ?? null, url, countryIso2: next.countryIso2 as string | null, operatorId: next.operatorId as string | null, metroId: next.metroId as string | null, projectId: id, isAi: !!next.isAi });
438 + }
263 439 }
264 440 }
441 +
442 +/** Admin: hide a false-positive project (kept for audit, removed from every listing and aggregate). */
443 +export async function hideProject(tx: Tx, id: string, reason: string, decidedBy = "admin"): Promise<void> {
444 + await tx.execute(sql`update projects set hidden = true, updated_at = now() where id = ${id}`);
445 + await tx.execute(sql`update events set review_status = 'rejected' where project_id = ${id}`);
446 + await tx.execute(sql`insert into quality_flags (id, entity_type, entity_id, code, severity, field, message, priority, status, resolution, resolved_by, resolved_at, dedupe_key)
447 + values (${`flg_${sha256(`project|${id}|hidden`).slice(0, 16)}`}, 'project', ${id}, 'project_false_positive', 'critical', 'name', ${reason.slice(0, 1000)}, 0, 'resolved', 'hidden', ${decidedBy}, now(), ${`project|${id}|project_false_positive|`})
448 + on conflict (dedupe_key) do update set status = 'resolved', resolution = 'hidden', resolved_by = ${decidedBy}, resolved_at = now(), message = excluded.message`);
449 +}
450 +
451 +/** Admin: fold `fromId` into `intoId` (keys, provenance, claims, timeline, events, news). */
452 +export async function mergeProjects(tx: Tx, fromId: string, intoId: string, decidedBy = "admin"): Promise<void> {
453 + if (fromId === intoId) return;
454 + await tx.execute(sql`update entity_keys set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${fromId}`);
455 + await tx.execute(sql`update provenance set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${fromId} and not exists (select 1 from provenance q where q.entity_type = 'project' and q.entity_id = ${intoId} and q.field = provenance.field and q.source_id = provenance.source_id and q.url = provenance.url)`);
456 + await tx.execute(sql`delete from provenance where entity_type = 'project' and entity_id = ${fromId}`);
457 + await tx.execute(sql`update claims set subject_id = ${intoId} where subject_type = 'project' and subject_id = ${fromId}`);
458 + await tx.execute(sql`update project_timeline set project_id = ${intoId} where project_id = ${fromId}`);
459 + await tx.execute(sql`update events set project_id = ${intoId}, entity_id = case when entity_type = 'project' and entity_id = ${fromId} then ${intoId} else entity_id end where project_id = ${fromId} or (entity_type = 'project' and entity_id = ${fromId})`);
460 + await tx.execute(sql`update news_items set project_id = ${intoId} where project_id = ${fromId}`);
461 + await tx.execute(sql`update projects set merged_into = ${intoId}, hidden = true, updated_at = now() where id = ${fromId}`);
462 + await tx.execute(sql`update quality_flags set status = 'resolved', resolution = ${`merged into ${intoId}`}, resolved_by = ${decidedBy}, resolved_at = now() where entity_type = 'project' and entity_id = ${fromId} and status = 'open'`);
463 +}
modified apps/worker/src/ingest/provenance.ts +5 −3
@@ -51,6 +51,8 @@ export interface ObservedField {
51 51 value: unknown;
52 52 provenance: Provenance;
53 53 entityKey?: string;
54 + /** claim scope of the observation (building / facility / campus / …) when known */
55 + scope?: string | null;
54 56 }
55 57
56 58 export function provenanceId(entityType: string, entityId: string, field: string, sourceId: string, url: string): string {
@@ -71,13 +73,13 @@ export async function writeProvenance(tx: Tx, ctx: IngestContext, entityType: st
71 73 const observed = p.lastObserved || ctx.now;
72 74 await tx.execute(sql`
73 75 insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, document_id, url, first_observed, last_observed, retrieved_at,
74 − confidence, is_estimate, method, extractor_version, is_current, note)
76 + confidence, is_estimate, method, extractor_version, is_current, note, run_id, scope)
75 77 values (${id}, ${entityType}, ${entityId}, ${f.field}, ${valueJson}::jsonb, ${sourceId}, ${p.connectorId || ctx.run.connectorId}, ${p.documentId ?? ctx.doc?.documentId ?? null}, ${url},
76 − ${p.firstObserved || observed}, ${observed}, ${p.retrievedAt || ctx.now}, ${p.confidence ?? "moderate"}, ${!!p.isEstimate}, ${p.method ?? null}, ${p.extractorVersion ?? null}, true, ${p.note ?? null})
78 + ${p.firstObserved || observed}, ${observed}, ${p.retrievedAt || ctx.now}, ${p.confidence ?? "moderate"}, ${!!p.isEstimate}, ${p.method ?? null}, ${p.extractorVersion ?? null}, true, ${p.note ?? null}, ${ctx.run.runId}, ${f.scope ?? null})
77 79 on conflict (entity_type, entity_id, field, source_id, url) do update set
78 80 value = excluded.value, last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, confidence = excluded.confidence,
79 81 is_estimate = excluded.is_estimate, method = excluded.method, extractor_version = excluded.extractor_version, document_id = coalesce(excluded.document_id, provenance.document_id),
80 − is_current = true, note = excluded.note`);
82 + is_current = true, note = excluded.note, run_id = excluded.run_id, scope = coalesce(excluded.scope, provenance.scope)`);
81 83 // the same source reporting the field from another URL earlier → superseded
82 84 await tx.execute(sql`update provenance set is_current = false where entity_type = ${entityType} and entity_id = ${entityId} and field = ${f.field} and source_id = ${sourceId} and id <> ${id} and is_current = true`);
83 85 n++;
modified apps/worker/src/main.ts +17 −1
@@ -28,6 +28,8 @@ import { computeRankings } from "./rankings.js";
28 28 import { CRAWL_QUEUE, MAINTENANCE_QUEUE, QUEUE_PREFIX, RUN_LOCK_TTL_SECONDS, WORKER_STATUS_KEY, closeScheduler, getRedis, publishQueueMetrics, queueSnapshot, runLockKey, startSchedulerLoop, type CrawlJobData, type MaintenanceJobData } from "./scheduler.js";
29 29 import { ensureBucket } from "./storage.js";
30 30 import { cleanupMaintenance, reconcileOrphanRuns } from "./maintenance.js";
31 +import { snapshotAndCheck, qualitySweep, dataGaps } from "./quality.js";
32 +import { traceDocument } from "./trace.js";
31 33
32 34 export interface WorkerStatus {
33 35 startedAt: string;
@@ -156,6 +158,8 @@ export async function startWorker(opts: { withScheduler?: boolean } = {}): Promi
156 158 case "metrics": return await computeDailyMetrics();
157 159 case "refresh-stats": return await refreshStats();
158 160 case "cleanup": return await cleanupMaintenance();
161 + case "snapshot": return await snapshotAndCheck();
162 + case "quality": return await qualitySweep();
159 163 default: throw new Error(`unknown maintenance kind ${String((job.data as { kind?: string }).kind)}`);
160 164 }
161 165 } finally { running.delete(String(job.id)); }
@@ -176,7 +180,19 @@ export async function startWorker(opts: { withScheduler?: boolean } = {}): Promi
176 180 onTick: (r) => { lastScheduler = { at: new Date().toISOString(), enqueued: r.enqueued.length, checked: r.checked }; if (r.enqueued.length) console.error(`[scheduler] enqueued ${r.enqueued.join(", ")}`); },
177 181 onError: (e) => console.error(`[scheduler] tick failed: ${e.message}`),
178 182 });
179 − const http = startHealthHttp({ port: env.workerPort, role: "worker", health: async () => ({ ...(await status()), queueDepth: await queueSnapshot().catch(() => []) }), ready: () => !shuttingDown, beforeScrape: refreshGauges, log: (m) => console.error(`[worker] ${m}`) });
183 + const http = startHealthHttp({
184 + port: env.workerPort,
185 + role: "worker",
186 + health: async () => ({ ...(await status()), queueDepth: await queueSnapshot().catch(() => []) }),
187 + ready: () => !shuttingDown,
188 + beforeScrape: refreshGauges,
189 + log: (m) => console.error(`[worker] ${m}`),
190 + routes: {
191 + // extraction debugger (internal network only; the API proxies it behind the admin token)
192 + "/trace": async (url) => { const id = url.pathname.split("/")[2]; if (!id) return { status: 400, body: { error: "usage: /trace/<document-id>" } }; return { status: 200, body: await traceDocument(decodeURIComponent(id), { live: url.searchParams.get("live") === "1" }) }; },
193 + "/data-gaps": async () => ({ status: 200, body: await dataGaps() }),
194 + },
195 + });
180 196 await heartbeat();
181 197 const hb = setInterval(() => void heartbeat(), 15_000);
182 198 console.error(`[worker] ready · queues ${env.queues.join(",")} · crawl concurrency ${env.crawlConcurrency} · scheduler ${withScheduler ? "embedded" : "external"} · ${loaded.length} connectors · host ${env.hostname}`);
modified apps/worker/src/pipeline.ts +43 −13
@@ -10,14 +10,14 @@ import type { DiscoveredUrl, RawDocument } from "@dci/connectors";
10 10 import { entityIsValid, pageTitle } from "@dci/connectors";
11 11 import type { NormalizedEntity, ValidationIssue } from "@dci/core";
12 12 import { contentFingerprint, sha256, newId } from "@dci/core";
13 −import { getDb, connectorRuns, connectors as connectorsTable, eq, inArray } from "@dci/db";
13 +import { getDb, connectorRuns, connectors as connectorsTable, eq, inArray, sql } from "@dci/db";
14 14 import { loadAllConnectors, requireConnector, type LoadedConnector } from "./configs.js";
15 15 import { createRunContext, type LogLevel, type LogLine, type RunContext } from "./context.js";
16 16 import { connectorDocStats, documentIdFor, dueDocuments, nextCheckByGroup, recordFetch, recordVersion, registerDiscovered, textProjection, updateExtraction, type DocumentRow } from "./documents.js";
17 17 import { ingestEntities } from "./ingest/index.js";
18 18 import type { IngestDocRef, IngestRun, IngestStats } from "./ingest/contract.js";
19 19 import { metrics } from "./prom.js";
20 −import { healthFrom, minIntervalMs, parseDiscoveredFrom, shouldSkipExtraction } from "./scheduling.js";
20 +import { connectorHealthFrom, healthFrom, minIntervalMs, parseDiscoveredFrom, shouldSkipExtraction } from "./scheduling.js";
21 21 import { getRaw, putRaw } from "./storage.js";
22 22
23 23 export type RunTask = "discover" | "crawl" | "full" | "reprocess";
@@ -35,6 +35,10 @@ export interface RunOptions {
35 35 logLevel?: LogLevel;
36 36 /** collect up to N normalized entities for display (dry runs) */
37 37 sampleEntities?: number;
38 + /** quarantine: fetch, archive and extract normally but roll the ingest back (set from connectors.quarantine) */
39 + quarantine?: boolean;
40 + /** reprocess only documents whose stored extractor version differs from the current one */
41 + staleOnly?: boolean;
38 42 }
39 43
40 44 export interface RunStats extends Record<string, number> {
@@ -137,6 +141,7 @@ function docToDiscovered(doc: DocumentRow): DiscoveredUrl {
137 141
138 142 function sumIngest(stats: RunStats, s: IngestStats): void {
139 143 stats.created += s.created; stats.updated += s.updated; stats.merged += s.merged; stats.pendingMatches += s.pendingMatches; stats.events += s.events; stats.provenanceRows += s.provenanceRows;
144 + stats.claims = (stats.claims ?? 0) + (s.claims ?? 0); stats.qualityFlags = (stats.qualityFlags ?? 0) + (s.qualityFlags ?? 0); stats.unscopedClaims = (stats.unscopedClaims ?? 0) + (s.unscopedClaims ?? 0); stats.projectsVetoed = (stats.projectsVetoed ?? 0) + (s.projectsVetoed ?? 0); stats.campusLinks = (stats.campusLinks ?? 0) + (s.campusLinks ?? 0);
140 145 }
141 146
142 147 interface DocOutcome { kind: "fetched" | "notModified" | "failed" | "unchanged" | "changed" | "aborted"; ms: number }
@@ -162,11 +167,11 @@ async function extractAndIngest(ctx: RunContext, loaded: LoadedConnector, doc: D
162 167 const classifier = (raw.meta?.classifier as string | undefined) ?? null;
163 168 metrics.ingest.inc({ connector: loaded.cfg.id, result: "rejected" }, report.rejected);
164 169 if (!valid.length) return { ingest: null, valid: 0, rejected: report.rejected, classifier, pageType };
165 − const run: IngestRun = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId: ctx.runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: Boolean(opts.dryRun) };
170 + const run: IngestRun = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId: ctx.runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: Boolean(opts.dryRun) || Boolean(opts.quarantine) };
166 171 const ref: IngestDocRef = { documentId: doc.id, url: raw.finalUrl || doc.url, pageType, fetchedAt: raw.fetchedAt };
167 172 const ingest = await ingestEntities(run, valid, ref);
168 173 sumIngest(stats, ingest);
169 − if (!opts.dryRun) {
174 + if (!opts.dryRun && !opts.quarantine) {
170 175 metrics.ingest.inc({ connector: loaded.cfg.id, result: "created" }, ingest.created);
171 176 metrics.ingest.inc({ connector: loaded.cfg.id, result: "updated" }, ingest.updated);
172 177 metrics.ingest.inc({ connector: loaded.cfg.id, result: "merged" }, ingest.merged);
@@ -226,7 +231,7 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do
226 231 try { storageKey = (await putRaw(cfg.id, doc.id, hash, raw.body, raw.contentType, raw.finalUrl)).key; } catch (e) { ctx.log("warn", `archive failed for ${doc.url}: ${(e as Error).message}`); }
227 232 }
228 233
229 − const skip = markupOnly || shouldSkipExtraction({ newHash: hash, storedHash: doc.contentHash, force, extractorVersion: connector.parserVersion, storedExtractorVersion: doc.extractorVersion, notModified: false });
234 + const skip = markupOnly || shouldSkipExtraction({ newHash: hash, storedHash: doc.contentHash, force, extractorVersion: loaded.extractorVersion, storedExtractorVersion: doc.extractorVersion, notModified: false });
230 235 // markup-only: adopt the new fingerprint (so a stabilised page re-syncs after one fetch) but keep the archived
231 236 // body — its text projection is identical, which is what versions/diffs are built from.
232 237 const plan = await recordFetch(doc, raw, { contentHash: hash, changed, storageKey, title }, cfg.schedule, { dryRun });
@@ -250,9 +255,9 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do
250 255 extractError = `extract: ${(e as Error).message}`.slice(0, 500);
251 256 ctx.log("error", `${doc.url} ${extractError}`);
252 257 }
253 − await updateExtraction(doc.id, { entityRefs: ingestStats?.refs ?? [], extractOk: extractError === null, extractCount: validCount, extractorVersion: connector.parserVersion, error: extractError, pageType, classifier }, dryRun);
258 + await updateExtraction(doc.id, { entityRefs: ingestStats?.refs ?? [], extractOk: extractError === null, extractCount: validCount, extractorVersion: loaded.extractorVersion, error: extractError, pageType, classifier }, dryRun);
254 259 if (changed || !doc.contentHash || force) {
255 − await recordVersion(doc, raw, hash, storageKey, ingestStats?.changes ?? [], { dryRun, runId: ctx.runId });
260 + await recordVersion(doc, raw, hash, storageKey, ingestStats?.changes ?? [], { dryRun, runId: ctx.runId, extractorVersion: loaded.extractorVersion });
256 261 }
257 262 ctx.log("info", `${doc.url} → ${raw.status} ${raw.fetcher} L${raw.level} ${raw.durationMs}ms ${changed ? "CHANGED" : "same"} · ${validCount} entities${ingestStats ? ` (+${ingestStats.created} new, ${ingestStats.updated} upd, ${ingestStats.events} events)` : ""} next=${plan.nextCheck ?? "never"}`);
258 263 return { kind: changed ? "changed" : "fetched", ms: Date.now() - t0 };
@@ -262,7 +267,7 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do
262 267 stats.failed++;
263 268 ctx.log("error", `${doc.url} unexpected: ${(e as Error).stack ?? (e as Error).message}`);
264 269 try {
265 − if (!dryRun) await updateExtraction(doc.id, { entityRefs: [], extractOk: false, extractCount: 0, extractorVersion: connector.parserVersion, error: `pipeline: ${(e as Error).message}`.slice(0, 500) });
270 + if (!dryRun) await updateExtraction(doc.id, { entityRefs: [], extractOk: false, extractCount: 0, extractorVersion: loaded.extractorVersion, error: `pipeline: ${(e as Error).message}`.slice(0, 500) });
266 271 } catch { /* ignore */ }
267 272 return { kind: "failed", ms: Date.now() - t0 };
268 273 }
@@ -281,7 +286,7 @@ async function reprocessDocument(ctx: RunContext, loaded: LoadedConnector, doc:
281 286 const { group } = parseDiscoveredFrom(doc.discoveredFrom);
282 287 const raw: RawDocument = { url: doc.url, finalUrl: doc.url, fetchedAt: doc.lastFetched ?? new Date().toISOString(), status: doc.statusCode ?? 200, contentType, body: stored.body, text: isText ? stored.body.toString("utf8") : "", headers: {}, etag: doc.etag, lastModified: doc.lastModified, notModified: false, fetcher: "cache", level: doc.fetchLevel as RawDocument["level"], durationMs: 0, credits: 0, group, pageType: doc.pageType !== "unknown" ? (doc.pageType as RawDocument["pageType"]) : undefined };
283 288 const r = await extractAndIngest(ctx, loaded, doc, raw, stats, result, opts);
284 − await updateExtraction(doc.id, { entityRefs: r.ingest?.refs ?? [], extractOk: true, extractCount: r.valid, extractorVersion: loaded.connector.parserVersion, error: null, pageType: r.pageType, classifier: r.classifier }, dryRun);
289 + await updateExtraction(doc.id, { entityRefs: r.ingest?.refs ?? [], extractOk: true, extractCount: r.valid, extractorVersion: loaded.extractorVersion, error: null, pageType: r.pageType, classifier: r.classifier }, dryRun);
285 290 ctx.log("info", `${doc.url} reprocessed → ${r.valid} entities${r.ingest ? ` (+${r.ingest.created} new, ${r.ingest.updated} upd, ${r.ingest.events} events)` : ""}`);
286 291 return { kind: "fetched", ms: Date.now() - t0 };
287 292 } catch (e) {
@@ -306,8 +311,17 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi
306 311 let fatal: string | null = null;
307 312 const durations: number[] = [];
308 313
309 − if (!dryRun) await db.insert(connectorRuns).values({ id: runId, connectorId, task: opts.task, startedAt, status: "running" });
310 − ctx.log("info", `run ${runId} task=${opts.task}${opts.group ? ` group=${opts.group}` : ""}${opts.limit ? ` limit=${opts.limit}` : ""}${dryRun ? " DRY-RUN" : ""}${opts.force ? " force" : ""}`);
314 + // quarantine mode (connectors.quarantine): everything runs, nothing is published
315 + let quarantine = Boolean(opts.quarantine);
316 + let previouslyDiscovered: number | null = null;
317 + if (!dryRun) {
318 + const row = (await db.execute<{ quarantine: boolean; last_discovered: number | null }>(sql`select quarantine, last_discovered from connectors where id = ${connectorId}`))[0];
319 + if (row?.quarantine) quarantine = true;
320 + previouslyDiscovered = row?.last_discovered == null ? null : Number(row.last_discovered);
321 + }
322 + opts = { ...opts, quarantine };
323 + if (!dryRun) await db.insert(connectorRuns).values({ id: runId, connectorId, task: opts.task, startedAt, status: "running", quarantined: quarantine });
324 + ctx.log("info", `run ${runId} task=${opts.task}${opts.group ? ` group=${opts.group}` : ""}${opts.limit ? ` limit=${opts.limit}` : ""}${dryRun ? " DRY-RUN" : ""}${quarantine ? " QUARANTINE (ingest rolled back)" : ""}${opts.force ? " force" : ""}`);
311 325
312 326 try {
313 327 // ---- discover
@@ -335,7 +349,9 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi
335 349
336 350 // ---- crawl / reprocess
337 351 if (opts.task === "crawl" || opts.task === "reprocess" || (opts.task === "full" && !dryRun) || (opts.urls?.length && opts.task !== "discover")) {
338 − const docs = await dueDocuments(connectorId, { group: opts.group, limit: opts.limit, force: Boolean(opts.force) || Boolean(targetIds) || opts.task === "reprocess", ids: targetIds });
352 + let docs = await dueDocuments(connectorId, { group: opts.group, limit: opts.limit, force: Boolean(opts.force) || Boolean(targetIds) || opts.task === "reprocess", ids: targetIds });
353 + // `reprocess --stale`: only documents extracted with an older parser version (parser fix → controlled re-extraction)
354 + if (opts.task === "reprocess" && opts.staleOnly) docs = docs.filter((d) => d.extractorVersion !== loaded.extractorVersion);
339 355 stats.selected = docs.length;
340 356 ctx.log("info", `${docs.length} document(s) selected${opts.task === "reprocess" ? " for reprocessing" : ""}`);
341 357 const limit = pLimit(cfg.fetch.concurrency);
@@ -383,10 +399,24 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi
383 399 const nexts = groups.map((g) => g.nextCheck).filter((x): x is string => Boolean(x)).map((x) => Date.parse(x));
384 400 const nextRunAt = new Date(nexts.length ? Math.min(...nexts) : Date.now() + minIntervalMs(cfg.schedule)).toISOString();
385 401 const docStats = await connectorDocStats(connectorId);
402 + const blocked = Object.values(ctx.costByFetcher).reduce((n, c) => n + (c.errors ?? 0), 0);
403 + const health = connectorHealthFrom({ runHealth: result.health, status: result.status, fetched: stats.fetched, blocked: Math.min(blocked, stats.failed), discovered: stats.discovered, previouslyDiscovered: opts.task === "discover" || opts.task === "full" ? previouslyDiscovered : null, extracted: stats.extracted, entities: stats.entities, docsTotal: docStats.total });
404 + const failedRun = result.status === "failed" || health === "blocked";
405 + const prev = (await db.execute<{ consecutive_failures: number; blocked_since: string | null }>(sql`select consecutive_failures, blocked_since from connectors where id = ${connectorId}`))[0];
406 + const consecutive = failedRun ? Number(prev?.consecutive_failures ?? 0) + 1 : 0;
407 + const blockedSince = health === "blocked" ? (prev?.blocked_since ?? finishedAt) : null;
408 + // guard rails: a connector failing 5 runs in a row, or a non-full run that suddenly creates hundreds of records, is quarantined
409 + const suspiciousSpike = opts.task !== "full" && !quarantine && stats.created >= 500;
410 + const autoQuarantine = consecutive >= 5 || suspiciousSpike;
386 411 await db
387 412 .update(connectorsTable)
388 − .set({ health: result.status === "failed" ? "failing" : result.health, lastRunAt: finishedAt, lastStatus: result.status, lastError: result.error, nextRunAt, stats: { ...stats, docsTotal: docStats.total, docsDue: docStats.due, docsQuarantined: docStats.quarantined, docsErrors: docStats.errors, docsExtracted: docStats.extracted, lastRunMs: result.durationMs }, updatedAt: finishedAt })
413 + .set({ health, lastRunAt: finishedAt, lastStatus: result.status, lastError: result.error, nextRunAt, consecutiveFailures: consecutive, blockedSince, lastDiscovered: opts.task === "discover" || opts.task === "full" ? stats.discovered : undefined, ...(autoQuarantine ? { quarantine: true } : {}), stats: { ...stats, docsTotal: docStats.total, docsDue: docStats.due, docsQuarantined: docStats.quarantined, docsErrors: docStats.errors, docsExtracted: docStats.extracted, lastRunMs: result.durationMs, blockedFetches: blocked, unscopedClaims: stats.unscopedClaims ?? 0, projectsVetoed: stats.projectsVetoed ?? 0 }, updatedAt: finishedAt })
389 414 .where(eq(connectorsTable.id, connectorId));
415 + if (autoQuarantine) {
416 + const why = suspiciousSpike ? `run ${runId} created ${stats.created} records in a ${opts.task} run — quarantined pending review` : `${consecutive} consecutive failed/blocked runs — quarantined`;
417 + ctx.log("error", why);
418 + await db.execute(sql`insert into system_alerts (id, level, component, message, details) values (${`alr_${Date.now().toString(36)}_q_${connectorId.replace(/[^a-z0-9]/gi, "")}`}, 'error', ${`connector:${connectorId}`}, ${why}, ${JSON.stringify({ runId, stats })}::jsonb)`).catch(() => undefined);
419 + }
390 420 } catch (e) {
391 421 console.error(`[${connectorId}] finishing run ${runId} failed: ${(e as Error).message}`);
392 422 }
added apps/worker/src/quality.ts +189 −0
@@ -0,0 +1,189 @@
1 +/**
2 + * Data-quality maintenance:
3 + * - `snapshotAndCheck()` — daily JSON snapshots (global totals, per-operator / per-country totals, project stages,
4 + * facility statuses) into `entity_snapshots` + automated regression checks against the previous snapshot
5 + * (facilities drop > 5 %, known MW jumps > 20 %, one operator gains > 10 GW, one connector creates > 500 projects,
6 + * one source changes hundreds of locations) → `system_alerts`.
7 + * - `qualitySweep()` — deterministic flags over the live database (largest values, scope, duplicates, orphans,
8 + * stale entities, project false-positive candidates…) into `quality_flags`, plus review priorities.
9 + * - `dataGaps()` — counts for the admin data-gaps page (facilities without operator / coordinates / capacity…).
10 + */
11 +import { getDb, sql, type Db } from "@dci/db";
12 +import { sha256 } from "@dci/core";
13 +import { reviewPriority } from "./ingest/claims.js";
14 +
15 +const day = () => new Date().toISOString().slice(0, 10);
16 +
17 +export interface SnapshotResult { day: string; snapshots: number; alerts: string[] }
18 +
19 +async function alert(db: Db, component: string, level: "warn" | "error", message: string, details: Record<string, unknown>): Promise<void> {
20 + await db.execute(sql`insert into system_alerts (id, level, component, message, details) values (${`alr_${Date.now().toString(36)}_${Math.random().toString(36).slice(2, 6)}`}, ${level}, ${component}, ${message.slice(0, 1000)}, ${JSON.stringify(details)}::jsonb)`);
21 +}
22 +
23 +export async function snapshotAndCheck(db: Db = getDb(), today = day()): Promise<SnapshotResult> {
24 + const alerts: string[] = [];
25 + const g = (await db.execute(sql`
26 + select
27 + (select count(*) from facilities where merged_into is null)::int as facilities,
28 + (select count(*) from operators)::int as operators,
29 + (select count(*) from projects where merged_into is null and not hidden)::int as projects,
30 + (select count(distinct country_iso2) from facilities where merged_into is null and country_iso2 is not null)::int as countries,
31 + coalesce((select sum(coalesce(it_capacity_mw, total_power_mw)) from facilities where merged_into is null and status in ('operational','partially_operational','expansion') and not exists (select 1 from facilities c where c.parent_facility_id = facilities.id and c.merged_into is null and coalesce(c.it_capacity_mw, c.total_power_mw) is not null)), 0)::double precision as known_mw,
32 + coalesce((select sum(coalesce(planned_power_mw, it_capacity_mw, total_power_mw)) from facilities where merged_into is null and status = 'under_construction'), 0)::double precision as construction_mw,
33 + coalesce((select sum(planned_mw) from projects where merged_into is null and not hidden and status in ('rumored','proposed','announced','permitting','approved','under_construction','delayed')), 0)::double precision as project_mw,
34 + (select count(*) from events where detected_at > now() - interval '24 hours')::int as events_24h,
35 + (select count(*) from claims)::int as claims,
36 + (select count(*) from quality_flags where status = 'open')::int as open_flags`))[0]!;
37 + const totals = Object.fromEntries(Object.entries(g).map(([k, v]) => [k, Number(v)]));
38 + const statuses = await db.execute(sql`select status, count(*)::int as n from facilities where merged_into is null group by 1`);
39 + const stages = await db.execute(sql`select status, count(*)::int as n, coalesce(sum(planned_mw), 0)::double precision as mw from projects where merged_into is null and not hidden group by 1`);
40 + const ops = await db.execute(sql`select o.id, o.slug, count(f.id)::int as facilities, coalesce(sum(coalesce(f.it_capacity_mw, f.total_power_mw)), 0)::double precision as known_mw,
41 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden)::double precision as project_mw
42 + from operators o left join facilities f on f.operator_id = o.id and f.merged_into is null group by o.id, o.slug having count(f.id) > 0 or exists (select 1 from projects p where p.operator_id = o.id and not p.hidden)`);
43 + const countries = await db.execute(sql`select country_iso2 as iso2, count(*)::int as facilities, coalesce(sum(coalesce(it_capacity_mw, total_power_mw)), 0)::double precision as known_mw from facilities where merged_into is null and country_iso2 is not null group by 1`);
44 + const rankings = await db.execute(sql`select key, rows from rankings where is_current`);
45 + const connectors = await db.execute(sql`select connector_id, coalesce(sum((stats->>'created')::int), 0)::int as created, count(*)::int as runs from connector_runs where started_at > now() - interval '24 hours' and status <> 'running' and task <> 'full' group by 1`);
46 + const locChanges = await db.execute(sql`select p.connector_id, count(*)::int as n from provenance p where p.field = 'geo' and p.last_observed > now() - interval '24 hours' and p.first_observed < p.last_observed - interval '1 hour' group by 1`);
47 +
48 + const previous = (await db.execute(sql`select payload from entity_snapshots where kind = 'global_totals' and key = 'global' and day < ${today}::date order by day desc limit 1`))[0]?.payload as Record<string, number> | undefined;
49 + const prevOps = new Map<string, Record<string, unknown>>();
50 + for (const r of await db.execute(sql`select key, payload from entity_snapshots where kind = 'operator_totals' and day = (select max(day) from entity_snapshots where kind = 'operator_totals' and day < ${today}::date)`)) prevOps.set(String(r.key), r.payload as Record<string, unknown>);
51 +
52 + // ── regression checks
53 + if (previous) {
54 + const pf = Number(previous.facilities ?? 0), nf = totals.facilities!;
55 + if (pf > 100 && nf < pf * 0.95) { const m = `facilities dropped ${pf} → ${nf} (${(((nf - pf) / pf) * 100).toFixed(1)} %)`; alerts.push(m); await alert(db, "regression:facilities", "error", m, { previous: pf, now: nf }); }
56 + const pm = Number(previous.known_mw ?? 0), nm = totals.known_mw!;
57 + if (pm > 500 && nm > pm * 1.2) { const m = `known MW jumped ${Math.round(pm)} → ${Math.round(nm)} (+${(((nm - pm) / pm) * 100).toFixed(1)} %)`; alerts.push(m); await alert(db, "regression:known_mw", "warn", m, { previous: pm, now: nm }); }
58 + const pp = Number(previous.projects ?? 0), np = totals.projects!;
59 + if (pp > 50 && np > pp * 1.5) { const m = `projects jumped ${pp} → ${np}`; alerts.push(m); await alert(db, "regression:projects", "warn", m, { previous: pp, now: np }); }
60 + const pc = Number(previous.countries ?? 0), nc = totals.countries!;
61 + if (pc > 20 && nc < pc - 3) { const m = `countries dropped ${pc} → ${nc}`; alerts.push(m); await alert(db, "regression:countries", "warn", m, { previous: pc, now: nc }); }
62 + }
63 + for (const o of ops) {
64 + const prev = prevOps.get(String(o.id));
65 + const gain = Number(o.known_mw) + Number(o.project_mw) - (prev ? Number(prev.known_mw ?? 0) + Number(prev.project_mw ?? 0) : 0);
66 + if (prev && gain > 10_000) { const m = `operator ${String(o.slug)} gained ${Math.round(gain)} MW in one day`; alerts.push(m); await alert(db, "regression:operator_mw", "error", m, { operator: o.slug, gainMw: gain }); }
67 + }
68 + for (const c of connectors) if (Number(c.created) > 500) { const m = `connector ${String(c.connector_id)} created ${Number(c.created)} records in 24 h`; alerts.push(m); await alert(db, `regression:connector:${String(c.connector_id)}`, "warn", m, { created: Number(c.created), runs: Number(c.runs) }); }
69 + for (const l of locChanges) if (Number(l.n) > 200) { const m = `connector ${String(l.connector_id)} changed ${Number(l.n)} locations in 24 h`; alerts.push(m); await alert(db, `regression:locations:${String(l.connector_id)}`, "warn", m, { changed: Number(l.n) }); }
70 +
71 + // ── snapshots
72 + let n = 0;
73 + await db.transaction(async (tx) => {
74 + const put = async (kind: string, key: string, payload: unknown) => { await tx.execute(sql`insert into entity_snapshots (day, kind, key, payload) values (${today}::date, ${kind}, ${key}, ${JSON.stringify(payload)}::jsonb) on conflict (day, kind, key) do update set payload = excluded.payload, created_at = now()`); n++; };
75 + await put("global_totals", "global", totals);
76 + await put("facility_status", "global", Object.fromEntries(statuses.map((r) => [String(r.status), Number(r.n)])));
77 + await put("project_stage", "global", Object.fromEntries(stages.map((r) => [String(r.status), { count: Number(r.n), mw: Number(r.mw) }])));
78 + for (const o of ops) await put("operator_totals", String(o.id), { slug: o.slug, facilities: Number(o.facilities), known_mw: Number(o.known_mw), project_mw: Number(o.project_mw) });
79 + for (const c of countries) await put("country_totals", String(c.iso2), { facilities: Number(c.facilities), known_mw: Number(c.known_mw) });
80 + for (const r of rankings) await put("ranking", String(r.key), { rows: (r.rows as unknown[]).slice(0, 100) });
81 + });
82 + return { day: today, snapshots: n, alerts };
83 +}
84 +
85 +export interface SweepResult { flagged: number; resolved: number; byCode: Record<string, number> }
86 +
87 +/** Batch quality sweep over the live database. Idempotent (dedupe keys). */
88 +export async function qualitySweep(db: Db = getDb()): Promise<SweepResult> {
89 + const byCode: Record<string, number> = {};
90 + let flagged = 0, resolved = 0;
91 + const flag = async (entityType: string, entityId: string, code: string, severity: "warn" | "critical", message: string, field: string | null, details: Record<string, unknown>, priority: number) => {
92 + const dedupeKey = `${entityType}|${entityId}|${code}|${field ?? ""}`;
93 + await db.execute(sql`insert into quality_flags (id, entity_type, entity_id, code, severity, field, message, details, priority, status, dedupe_key)
94 + values (${`flg_${sha256(dedupeKey).slice(0, 20)}`}, ${entityType}, ${entityId}, ${code}, ${severity}, ${field}, ${message.slice(0, 1000)}, ${JSON.stringify(details)}::jsonb, ${priority}, 'open', ${dedupeKey})
95 + on conflict (dedupe_key) do update set message = excluded.message, details = excluded.details, priority = excluded.priority, updated_at = now(), status = case when quality_flags.status = 'dismissed' then 'dismissed' when quality_flags.status = 'resolved' and quality_flags.resolved_by <> 'system' then 'resolved' else 'open' end`);
96 + byCode[code] = (byCode[code] ?? 0) + 1; flagged++;
97 + };
98 + const clear = async (code: string, keepIds: string[], entityType: string) => {
99 + const r = await db.execute(sql`update quality_flags set status = 'resolved', resolution = 'auto: condition cleared', resolved_by = 'system', resolved_at = now(), updated_at = now() where code = ${code} and entity_type = ${entityType} and status = 'open' ${keepIds.length ? sql`and entity_id not in ${keepIds}` : sql``} returning id`);
100 + resolved += r.length;
101 + };
102 +
103 + // suspicious MW on facilities (single site > 1 GW without campus designation, building > 500 MW, IT > total)
104 + const bigFac = await db.execute(sql`select id, name, slug, coalesce(it_capacity_mw, total_power_mw) as mw, record_scope, confidence from facilities where merged_into is null and coalesce(it_capacity_mw, total_power_mw) > 1000 and record_scope <> 'campus' and name !~* '(campus|park|complex|hub|cluster|mega ?site)'`);
105 + for (const r of bigFac) await flag("facility", String(r.id), "mw_single_site_gt_1000", "critical", `${String(r.name)}: ${Number(r.mw)} MW on a single ${String(r.record_scope)} record without a campus designation`, "itCapacityMw", { slug: r.slug, mw: Number(r.mw) }, reviewPriority({ mw: Number(r.mw), confidence: String(r.confidence), severity: "critical", homepageVisible: true }));
106 + await clear("mw_single_site_gt_1000", bigFac.map((r) => String(r.id)), "facility");
107 + const itGtTotal = await db.execute(sql`select id, name, slug, it_capacity_mw, total_power_mw from facilities where merged_into is null and it_capacity_mw is not null and total_power_mw is not null and it_capacity_mw > total_power_mw * 1.05`);
108 + for (const r of itGtTotal) await flag("facility", String(r.id), "mw_it_gt_total", "warn", `${String(r.name)}: IT capacity ${Number(r.it_capacity_mw)} MW exceeds total power ${Number(r.total_power_mw)} MW`, "itCapacityMw", { slug: r.slug }, reviewPriority({ mw: Number(r.it_capacity_mw), severity: "warn" }));
109 + await clear("mw_it_gt_total", itGtTotal.map((r) => String(r.id)), "facility");
110 + // tiny buildings with huge MW (the 0.779 → 779 class of unit errors)
111 + const dense = await db.execute(sql`select id, name, slug, it_capacity_mw, building_sqm from facilities where merged_into is null and it_capacity_mw is not null and building_sqm is not null and building_sqm > 0 and it_capacity_mw / building_sqm > 0.05`);
112 + for (const r of dense) await flag("facility", String(r.id), "mw_density_implausible", "critical", `${String(r.name)}: ${Number(r.it_capacity_mw)} MW in ${Number(r.building_sqm)} m² (${(Number(r.it_capacity_mw) * 1000 / Number(r.building_sqm)).toFixed(0)} kW/m²) — unit error?`, "itCapacityMw", { slug: r.slug }, reviewPriority({ mw: Number(r.it_capacity_mw), severity: "critical" }));
113 + await clear("mw_density_implausible", dense.map((r) => String(r.id)), "facility");
114 +
115 + // projects: false-positive candidates, headline names, huge figures, missing location, unknown scope
116 + const fp = await db.execute(sql`select id, name, slug, planned_mw, investment_usd, project_class, evidence_level from projects where merged_into is null and not hidden and (
117 + name ~* '\\m(appoint|names? [A-Z]\\w+ [A-Z]\\w+ as|joins|welcomes|promot|retire|award|shortlist|finalist|webinar|podcast|interview|market to (surpass|reach|hit|grow)|market size|cagr|report|survey|headquarters|head office|academy|workforce|partner page|homepage|bubble|comparing|why |how |what |ppa\\M|power purchase|sustainab|net.zero|carbon|financ|refinanc|raises \\$|series [a-e]\\M|bond|loan|credit facility|earnings|results|revenue|acquires [A-Z]\\w+ (group|holdings|inc|ltd|llc)|merger|takeover)'
118 + or (project_class is not null and project_class not in ('NEW_BUILD','EXPANSION','CONSTRUCTION_START','PERMIT','LAND_ACQUISITION','GRID_CONNECTION')))`);
119 + for (const r of fp) await flag("project", String(r.id), "project_false_positive_candidate", "critical", `${String(r.name)}: headline / class (${String(r.project_class ?? "n/a")}) does not describe physical development`, "name", { slug: r.slug, plannedMw: r.planned_mw, investmentUsd: r.investment_usd }, reviewPriority({ mw: Number(r.planned_mw ?? 0), investmentUsd: Number(r.investment_usd ?? 0), severity: "critical", homepageVisible: true }));
120 + await clear("project_false_positive_candidate", fp.map((r) => String(r.id)), "project");
121 + const headline = await db.execute(sql`select id, name, slug, planned_mw from projects where merged_into is null and not hidden and (name ~ '\\s[|]\\s|\\s[–—]\\s| - ' or length(name) > 90)`);
122 + for (const r of headline) await flag("project", String(r.id), "project_title_like_name", "warn", `${String(r.name)}: project name looks like an article headline`, "name", { slug: r.slug }, reviewPriority({ mw: Number(r.planned_mw ?? 0), severity: "warn" }));
123 + await clear("project_title_like_name", headline.map((r) => String(r.id)), "project");
124 + const bigPrj = await db.execute(sql`select id, name, slug, planned_mw, capacity_scope from projects where merged_into is null and not hidden and planned_mw >= 2000 and name !~* '(campus|park|complex|hub|cluster|gigafactory)'`);
125 + for (const r of bigPrj) await flag("project", String(r.id), "mw_project_gt_2000", "critical", `${String(r.name)}: ${Number(r.planned_mw)} MW planned (scope ${String(r.capacity_scope ?? "unknown")}) — verify it is one site`, "plannedMw", { slug: r.slug }, reviewPriority({ mw: Number(r.planned_mw), severity: "critical", homepageVisible: true }));
126 + await clear("mw_project_gt_2000", bigPrj.map((r) => String(r.id)), "project");
127 + const bigInv = await db.execute(sql`select id, name, slug, investment_usd from projects where merged_into is null and not hidden and investment_usd > 50e9`);
128 + for (const r of bigInv) await flag("project", String(r.id), "inv_single_site_gt_50b", "critical", `${String(r.name)}: $${(Number(r.investment_usd) / 1e9).toFixed(1)}B on one project — verify scope`, "investmentUsd", { slug: r.slug }, reviewPriority({ investmentUsd: Number(r.investment_usd), severity: "critical" }));
129 + await clear("inv_single_site_gt_50b", bigInv.map((r) => String(r.id)), "project");
130 + const noLoc = await db.execute(sql`select id, name, slug from projects where merged_into is null and not hidden and country_iso2 is null and lat is null`);
131 + for (const r of noLoc) await flag("project", String(r.id), "project_no_location", "warn", `${String(r.name)}: no country and no coordinates`, "countryIso2", { slug: r.slug }, 15);
132 + await clear("project_no_location", noLoc.map((r) => String(r.id)), "project");
133 + const unknownScope = await db.execute(sql`select id, name, slug, planned_mw from projects where merged_into is null and not hidden and planned_mw is not null and (capacity_scope is null or capacity_scope not in ('building','facility','campus'))`);
134 + for (const r of unknownScope) await flag("project", String(r.id), "project_unknown_scope", "warn", `${String(r.name)}: ${Number(r.planned_mw)} MW with no site-scoped evidence`, "plannedMw", { slug: r.slug }, reviewPriority({ mw: Number(r.planned_mw), severity: "warn" }));
135 + await clear("project_unknown_scope", unknownScope.map((r) => String(r.id)), "project");
136 +
137 + // facilities: location / country mismatch, missing country, orphan operators, stale
138 + const noCountry = await db.execute(sql`select id, name, slug from facilities where merged_into is null and country_iso2 is null`);
139 + for (const r of noCountry) await flag("facility", String(r.id), "facility_no_country", "warn", `${String(r.name)}: no country`, "countryIso2", { slug: r.slug }, 8);
140 + await clear("facility_no_country", noCountry.map((r) => String(r.id)), "facility");
141 + const stale = await db.execute(sql`select id, name, slug, last_verified from facilities where merged_into is null and status in ('announced','under_construction','permitting','approved') and coalesce(last_verified, first_seen) < now() - interval '400 days'`);
142 + for (const r of stale) await flag("facility", String(r.id), "facility_stale_pipeline", "warn", `${String(r.name)}: pipeline status not verified for over 400 days`, "status", { slug: r.slug, lastVerified: r.last_verified }, 12);
143 + await clear("facility_stale_pipeline", stale.map((r) => String(r.id)), "facility");
144 + const orphanOps = await db.execute(sql`select o.id, o.name, o.slug from operators o where not exists (select 1 from facilities f where f.operator_id = o.id or f.owner_id = o.id) and not exists (select 1 from projects p where p.operator_id = o.id) and not exists (select 1 from cloud_regions c where c.provider_id = o.id) and not exists (select 1 from facility_tenants t where t.operator_id = o.id) and o.created_at < now() - interval '7 days'`);
145 + for (const r of orphanOps) await flag("operator", String(r.id), "operator_orphan", "warn", `${String(r.name)}: operator with no facility, project, region or tenancy`, null, { slug: r.slug }, 5);
146 + await clear("operator_orphan", orphanOps.map((r) => String(r.id)), "operator");
147 + const badSlug = await db.execute(sql`select id, name, slug from operators where slug ~ '^(item|op|operator)(-\\d+)?$' or slug = ''`);
148 + for (const r of badSlug) await flag("operator", String(r.id), "operator_bad_slug", "warn", `${String(r.name)}: slug "${String(r.slug)}" is not derived from the name (non-Latin script?)`, "slug", {}, 6);
149 + await clear("operator_bad_slug", badSlug.map((r) => String(r.id)), "operator");
150 +
151 + // review priority roll-up
152 + await db.execute(sql`update facilities f set review_priority = coalesce((select max(priority) from quality_flags q where q.entity_type = 'facility' and q.entity_id = f.id and q.status = 'open'), 0) where merged_into is null`);
153 + await db.execute(sql`update projects p set review_priority = coalesce((select max(priority) from quality_flags q where q.entity_type = 'project' and q.entity_id = p.id and q.status = 'open'), 0) where merged_into is null`);
154 + return { flagged, resolved, byCode };
155 +}
156 +
157 +export interface DataGaps { [key: string]: number }
158 +
159 +/** Counts for /admin/data-gaps — what we do not know, ranked for enrichment. */
160 +export async function dataGaps(db: Db = getDb()): Promise<DataGaps> {
161 + const r = (await db.execute(sql`
162 + select
163 + (select count(*) from facilities where merged_into is null and operator_id is null)::int as facilities_without_operator,
164 + (select count(*) from facilities where merged_into is null and lat is null)::int as facilities_without_coordinates,
165 + (select count(*) from facilities where merged_into is null and geo_precision in ('city','metro','approximate') )::int as facilities_imprecise_coordinates,
166 + (select count(*) from facilities where merged_into is null and coalesce(it_capacity_mw, total_power_mw, planned_power_mw) is null and parent_facility_id is null)::int as facilities_without_capacity,
167 + (select count(*) from facilities where merged_into is null and status = 'unknown')::int as facilities_unknown_status,
168 + (select count(*) from facilities where merged_into is null and facility_type = 'unknown')::int as facilities_unknown_type,
169 + (select count(*) from facilities where merged_into is null and opened_on is null)::int as facilities_without_opening_date,
170 + (select count(*) from facilities where merged_into is null and source_count <= 1)::int as facilities_single_source,
171 + (select count(*) from facilities where merged_into is null and coalesce(last_verified, first_seen) < now() - interval '180 days')::int as facilities_stale_sources,
172 + (select count(*) from facilities where merged_into is null and country_iso2 is null)::int as facilities_without_country,
173 + (select count(*) from projects where merged_into is null and not hidden and lat is null)::int as projects_without_coordinates,
174 + (select count(*) from projects where merged_into is null and not hidden and country_iso2 is null)::int as projects_without_country,
175 + (select count(*) from projects where merged_into is null and not hidden and operator_id is null)::int as projects_without_operator,
176 + (select count(*) from projects where merged_into is null and not hidden and planned_mw is null)::int as projects_without_capacity,
177 + (select count(*) from projects where merged_into is null and not hidden and expected_opening is null)::int as projects_without_expected_opening,
178 + (select count(*) from projects where merged_into is null and not hidden and source_url is null)::int as projects_without_source,
179 + (select count(*) from projects where merged_into is null and not hidden and facility_id is null)::int as projects_without_facility_link,
180 + (select count(*) from operators where hq_country_iso2 is null)::int as operators_without_hq,
181 + (select count(*) from operators where website is null)::int as operators_without_website,
182 + (select count(*) from entity_matches where status = 'pending')::int as pending_matches,
183 + (select count(*) from quality_flags where status = 'open')::int as open_quality_flags,
184 + (select count(*) from claims where status = 'unscoped')::int as unscoped_claims,
185 + (select count(*) from claims where status = 'review')::int as claims_in_review,
186 + (select count(*) from cloud_regions where lat is null)::int as cloud_regions_without_coordinates,
187 + (select count(*) from ixps where metro_id is null)::int as ixps_without_metro`))[0]!;
188 + return Object.fromEntries(Object.entries(r).map(([k, v]) => [k, Number(v)]));
189 +}
modified apps/worker/src/rankings.ts +71 −20
@@ -34,6 +34,11 @@ export interface Aggregate {
34 34 metros: number;
35 35 cloudRegions: number;
36 36 projects: number;
37 + /** planned MW published on project records (pipeline statuses) — kept apart from facility figures */
38 + projectPlannedMw: number;
39 + projectConstructionMw: number;
40 + expansionMw: number;
41 + aiConfirmed: number;
37 42 ixps: number;
38 43 population: number | null;
39 44 gdpUsd: number | null;
@@ -47,18 +52,41 @@ function recentYear(): string {
47 52 return String(new Date().getUTCFullYear() - 3);
48 53 }
49 54
55 +/**
56 + * Containment-aware facility view used by every aggregate:
57 + * - `agg_mw` is a facility's own known MW (IT, else total) unless it is a campus whose buildings publish their own
58 + * figures (then the campus row contributes nothing — the buildings do); a building without a figure under a campus
59 + * with one contributes nothing either (the campus figure already covers it). Never both.
60 + * - `counted` excludes campus rows that have building rows (a campus with 4 buildings is 4 facilities, not 5) and
61 + * hidden / merged rows.
62 + */
63 +export const FACILITY_VIEW = sql`
64 + select f.*,
65 + exists (select 1 from facilities c where c.parent_facility_id = f.id and c.merged_into is null) as has_children,
66 + exists (select 1 from facilities c where c.parent_facility_id = f.id and c.merged_into is null and coalesce(c.it_capacity_mw, c.total_power_mw, c.planned_power_mw) is not null) as children_have_mw,
67 + (f.parent_facility_id is not null and exists (select 1 from facilities pp where pp.id = f.parent_facility_id and pp.merged_into is null and coalesce(pp.it_capacity_mw, pp.total_power_mw, pp.planned_power_mw) is not null)
68 + and coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is null) as covered_by_parent
69 + from facilities f where f.merged_into is null`;
70 +
71 +/** MW expressions honouring containment (campus vs buildings never double count). */
72 +const KNOWN_MW_EXPR = sql`case when f.children_have_mw then null else coalesce(f.it_capacity_mw, f.total_power_mw) end`;
73 +const PIPELINE_MW_EXPR = sql`case when f.children_have_mw then null else coalesce(f.planned_power_mw, f.it_capacity_mw, f.total_power_mw) end`;
74 +const COUNTED = sql`(not f.has_children)`;
75 +
50 76 const FACILITY_AGG = (dim: ReturnType<typeof sql>) => sql`
51 − count(f.id)::int as facilities,
52 − count(f.id) filter (where f.status in ${OPERATIONAL})::int as operational,
53 − count(f.id) filter (where f.status = 'under_construction')::int as construction,
54 − count(f.id) filter (where f.status in ${PLANNED})::int as planned,
55 − coalesce(sum(coalesce(f.it_capacity_mw, f.total_power_mw)) filter (where f.status in ${OPERATIONAL}), 0) as known_mw,
56 − coalesce(sum(coalesce(f.planned_power_mw, f.it_capacity_mw, f.total_power_mw)) filter (where f.status = 'under_construction'), 0) as construction_mw,
57 − coalesce(sum(coalesce(f.planned_power_mw, f.it_capacity_mw, f.total_power_mw)) filter (where f.status in ${PLANNED}), 0) as planned_mw,
58 − count(f.id) filter (where f.is_hyperscale)::int as hyperscale,
59 − count(f.id) filter (where f.is_ai)::int as ai,
60 − count(f.id) filter (where coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is not null)::int as with_mw,
61 − count(f.id) filter (where f.opened_on >= ${recentYear()})::int as opened_recent,
77 + count(f.id) filter (where ${COUNTED})::int as facilities,
78 + count(f.id) filter (where ${COUNTED} and f.status in ${OPERATIONAL})::int as operational,
79 + count(f.id) filter (where ${COUNTED} and f.status = 'under_construction')::int as construction,
80 + count(f.id) filter (where ${COUNTED} and f.status in ${PLANNED})::int as planned,
81 + coalesce(sum(${KNOWN_MW_EXPR}) filter (where f.status in ${OPERATIONAL}), 0) as known_mw,
82 + coalesce(sum(${PIPELINE_MW_EXPR}) filter (where f.status = 'under_construction'), 0) as construction_mw,
83 + coalesce(sum(${PIPELINE_MW_EXPR}) filter (where f.status in ${PLANNED}), 0) as planned_mw,
84 + coalesce(sum(case when f.children_have_mw then null else f.planned_power_mw end) filter (where f.status = 'expansion'), 0) as expansion_mw,
85 + count(f.id) filter (where ${COUNTED} and f.is_hyperscale)::int as hyperscale,
86 + count(f.id) filter (where ${COUNTED} and f.is_ai)::int as ai,
87 + count(f.id) filter (where ${COUNTED} and f.ai_evidence = 'confirmed')::int as ai_confirmed,
88 + count(f.id) filter (where ${COUNTED} and (coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is not null or f.covered_by_parent))::int as with_mw,
89 + count(f.id) filter (where ${COUNTED} and f.opened_on ~ '^\d{4}' and substring(f.opened_on from 1 for 4) >= ${recentYear()} and substring(f.opened_on from 1 for 4) <= ${String(new Date().getUTCFullYear())})::int as opened_recent,
62 90 count(distinct f.operator_id)::int as operators,
63 91 count(distinct f.country_iso2)::int as countries,
64 92 count(distinct f.metro_id)::int as metros,
@@ -87,6 +115,10 @@ function toAgg(r: Record<string, unknown>): Aggregate {
87 115 metros: num(r.metros),
88 116 cloudRegions: num(r.cloud_regions),
89 117 projects: num(r.projects),
118 + projectPlannedMw: num(r.project_planned_mw),
119 + projectConstructionMw: num(r.project_construction_mw),
120 + expansionMw: num(r.expansion_mw),
121 + aiConfirmed: num(r.ai_confirmed),
90 122 ixps: num(r.ixps),
91 123 population: numOrNull(r.population),
92 124 gdpUsd: numOrNull(r.gdp_usd),
@@ -98,10 +130,12 @@ export async function aggregateCountries(db: Db): Promise<Aggregate[]> {
98 130 const rows = await db.execute(sql`
99 131 select c.iso2 as id, c.slug, c.name, c.iso2 as country_iso2, c.population, c.gdp_usd,
100 132 (select count(*) from cloud_regions cr where cr.country_iso2 = c.iso2 and cr.status <> 'retired')::int as cloud_regions,
101 − (select count(*) from projects p where p.country_iso2 = c.iso2)::int as projects,
133 + (select count(*) from projects p where p.country_iso2 = c.iso2 and p.merged_into is null and not p.hidden)::int as projects,
134 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.country_iso2 = c.iso2 and p.merged_into is null and not p.hidden and p.status in ${PLANNED})::double precision as project_planned_mw,
135 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.country_iso2 = c.iso2 and p.merged_into is null and not p.hidden and p.status = 'under_construction')::double precision as project_construction_mw,
102 136 (select count(*) from ixps x where x.country_iso2 = c.iso2)::int as ixps,
103 137 ${FACILITY_AGG(sql`1 as _`)}
104 − from countries c left join facilities f on f.country_iso2 = c.iso2 and f.merged_into is null
138 + from countries c left join (${FACILITY_VIEW}) f on f.country_iso2 = c.iso2
105 139 group by c.iso2, c.slug, c.name, c.population, c.gdp_usd`);
106 140 return rows.map(toAgg);
107 141 }
@@ -110,10 +144,12 @@ export async function aggregateMetros(db: Db): Promise<Aggregate[]> {
110 144 const rows = await db.execute(sql`
111 145 select m.id, m.slug, m.name, m.country_iso2, null::bigint as population, null::double precision as gdp_usd,
112 146 (select count(*) from cloud_regions cr where cr.metro_id = m.id and cr.status <> 'retired')::int as cloud_regions,
113 − (select count(*) from projects p where p.metro_id = m.id)::int as projects,
147 + (select count(*) from projects p where p.metro_id = m.id and p.merged_into is null and not p.hidden)::int as projects,
148 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.metro_id = m.id and p.merged_into is null and not p.hidden and p.status in ${PLANNED})::double precision as project_planned_mw,
149 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.metro_id = m.id and p.merged_into is null and not p.hidden and p.status = 'under_construction')::double precision as project_construction_mw,
114 150 (select count(*) from ixps x where x.metro_id = m.id)::int as ixps,
115 151 ${FACILITY_AGG(sql`1 as _`)}
116 − from metros m left join facilities f on f.metro_id = m.id and f.merged_into is null
152 + from metros m left join (${FACILITY_VIEW}) f on f.metro_id = m.id
117 153 group by m.id, m.slug, m.name, m.country_iso2`);
118 154 return rows.map(toAgg);
119 155 }
@@ -122,10 +158,12 @@ export async function aggregateOperators(db: Db, limit = 5000): Promise<Aggregat
122 158 const rows = await db.execute(sql`
123 159 select o.id, o.slug, o.name, o.hq_country_iso2 as country_iso2, null::bigint as population, null::double precision as gdp_usd,
124 160 (select count(*) from cloud_regions cr where cr.provider_id = o.id and cr.status <> 'retired')::int as cloud_regions,
125 − (select count(*) from projects p where p.operator_id = o.id)::int as projects,
161 + (select count(*) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden)::int as projects,
162 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden and p.status in ${PLANNED})::double precision as project_planned_mw,
163 + (select coalesce(sum(p.planned_mw), 0) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden and p.status = 'under_construction')::double precision as project_construction_mw,
126 164 0 as ixps,
127 165 ${FACILITY_AGG(sql`1 as _`)}
128 − from operators o left join facilities f on f.operator_id = o.id and f.merged_into is null
166 + from operators o left join (${FACILITY_VIEW}) f on f.operator_id = o.id
129 167 group by o.id, o.slug, o.name, o.hq_country_iso2
130 168 order by count(f.id) desc, o.name asc
131 169 limit ${limit}`);
@@ -148,7 +186,7 @@ interface RankingSpec {
148 186
149 187 const mwGate = (a: Aggregate) => a.facilities >= MIN_FACILITIES && a.coverage >= MIN_MW_COVERAGE;
150 188 const OPERATIONAL_TXT = "operational, partially operational or expanding";
151 −const KNOWN_MW_TXT = `Known MW = sum of IT capacity (or total power when IT capacity is not published) over ${OPERATIONAL_TXT} facilities with a published or filed figure. Estimates are included only when no measured figure exists and are flagged on the facility. Coverage = share of the entity's facilities with any MW figure.`;
189 +const KNOWN_MW_TXT = `Known MW = sum of IT capacity (or total power when IT capacity is not published) over ${OPERATIONAL_TXT} facilities with a published or filed figure. A campus and its buildings are never both counted: when buildings publish their own figures the campus figure is ignored, otherwise the campus figure stands for its buildings. Estimates are included only when no measured figure exists and are flagged on the facility. Coverage = share of the entity's facilities with any MW figure (a building covered by its campus figure counts as covered). Figures are published lower bounds, never extrapolated.`;
152 190 const GATE_TXT = `Entities with fewer than ${MIN_FACILITIES} facilities or MW coverage below ${Math.round(MIN_MW_COVERAGE * 100)} % are excluded to avoid ranking on sparse data.`;
153 191
154 192 export const RANKING_SPECS: RankingSpec[] = [
@@ -167,14 +205,23 @@ export const RANKING_SPECS: RankingSpec[] = [
167 205 // metros
168 206 { key: "metros_facilities", scope: "metros", label: "Metros by facility count", unit: "facilities", methodology: "Facilities assigned to the metro: within the metro radius of its reference point, or in a town listed as part of the market when no coordinates are known.", value: (a) => a.facilities, secondary: (a) => a.knownMw },
169 207 { key: "metros_known_mw", scope: "metros", label: "Metros by known operational MW", unit: "MW", methodology: `${KNOWN_MW_TXT} ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.knownMw, secondary: (a) => a.operational, eligible: mwGate },
170 − { key: "metros_pipeline_mw", scope: "metros", label: "Metros by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction (planned power, or IT/total power when no planned figure exists) over facilities not yet operational. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate },
208 + { key: "metros_pipeline_mw", scope: "metros", label: "Metros by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction (planned power, or IT/total power when no planned figure exists) over facilities not yet operational, plus the planned power of expanding facilities. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw + a.expansionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate },
209 + { key: "metros_ai", scope: "metros", label: "Metros by AI/HPC facilities", unit: "facilities", methodology: "Facilities whose AI evidence is confirmed or likely (source explicitly describes AI/HPC/accelerated computing or GPU / high-density infrastructure). Keyword mentions alone never qualify.", value: (a) => a.ai, secondary: (a) => a.facilities },
210 + { key: "metros_cloud_regions", scope: "metros", label: "Metros by cloud regions", unit: "regions", methodology: "Public cloud regions associated with the metro (announced or operational).", value: (a) => a.cloudRegions, secondary: (a) => a.facilities },
211 + { key: "metros_projects", scope: "metros", label: "Metros by project count", unit: "projects", methodology: "Tracked infrastructure projects (announced, permitting, approved, under construction, delayed) located in the metro; false positives hidden by review are excluded.", value: (a) => a.projects, secondary: (a) => a.projectPlannedMw + a.projectConstructionMw },
171 212 { key: "metros_operators", scope: "metros", label: "Metros by operator count", unit: "operators", methodology: "Distinct operators with at least one facility assigned to the metro.", value: (a) => a.operators, secondary: (a) => a.facilities },
172 213 // operators
173 214 { key: "operators_facilities", scope: "operators", label: "Operators by facility count", unit: "facilities", methodology: "Facilities operated (not merely owned or tenanted) by the operator, subsidiaries folded into the parent brand where the curated alias table says so.", value: (a) => a.facilities, secondary: (a) => a.countries },
174 215 { key: "operators_countries", scope: "operators", label: "Operators by country footprint", unit: "countries", methodology: "Distinct countries with at least one facility operated by the operator.", value: (a) => a.countries, secondary: (a) => a.facilities, eligible: (a) => a.facilities > 0 },
175 216 { key: "operators_metros", scope: "operators", label: "Operators by metro footprint", unit: "metros", methodology: "Distinct metros with at least one facility operated by the operator (facilities outside any defined metro are not counted).", value: (a) => a.metros, secondary: (a) => a.facilities, eligible: (a) => a.facilities > 0 },
176 217 { key: "operators_known_mw", scope: "operators", label: "Operators by known operational MW", unit: "MW", methodology: `${KNOWN_MW_TXT} ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.knownMw, secondary: (a) => a.operational, eligible: mwGate },
177 − { key: "operators_pipeline_mw", scope: "operators", label: "Operators by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction over the operator's facilities not yet operational. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate },
218 + { key: "operators_pipeline_mw", scope: "operators", label: "Operators by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction over the operator's facilities not yet operational, plus the planned power of expanding facilities. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw + a.expansionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate },
219 + { key: "operators_ai", scope: "operators", label: "Operators by AI/HPC facilities", unit: "facilities", methodology: "Facilities whose AI evidence is confirmed or likely. Keyword mentions alone never qualify.", value: (a) => a.ai, secondary: (a) => a.facilities },
220 + { key: "operators_projects", scope: "operators", label: "Operators by tracked projects", unit: "projects", methodology: "Infrastructure projects (announced, permitting, approved, under construction, delayed) attributed to the operator; the secondary figure is the published planned MW on those records.", value: (a) => a.projects, secondary: (a) => a.projectPlannedMw + a.projectConstructionMw },
221 + { key: "operators_project_mw", scope: "operators", label: "Operators by published project MW", unit: "MW", methodology: "Sum of planned MW published on the operator's project records (pipeline statuses), site-scoped figures only — company-wide or portfolio totals are stored as claims and never summed here.", value: (a) => a.projectPlannedMw + a.projectConstructionMw, secondary: (a) => a.projects, eligible: (a) => a.projects >= 2 },
222 + { key: "countries_project_mw", scope: "countries", label: "Countries by published project MW", unit: "MW", methodology: "Sum of planned MW published on project records located in the country (pipeline statuses), site-scoped figures only.", value: (a) => a.projectPlannedMw + a.projectConstructionMw, secondary: (a) => a.projects, eligible: (a) => a.projects >= 2 },
223 + { key: "countries_projects", scope: "countries", label: "Countries by tracked projects", unit: "projects", methodology: "Infrastructure projects located in the country (announced, permitting, approved, under construction, delayed).", value: (a) => a.projects, secondary: (a) => a.projectPlannedMw + a.projectConstructionMw },
224 + { key: "countries_ixps", scope: "countries", label: "Countries by internet exchanges", unit: "IXPs", methodology: "Internet exchange points indexed in the country.", value: (a) => a.ixps, secondary: (a) => a.facilities },
178 225 ];
179 226
180 227 const HREF: Record<Ranking["scope"], (slug: string) => string> = { countries: (s) => `/countries/${s}`, metros: (s) => `/metros/${s}`, operators: (s) => `/operators/${s}`, facilities: (s) => `/facilities/${s}` };
@@ -253,6 +300,10 @@ function statsJson(a: Aggregate, breakdown: Record<string, number> | undefined,
253 300 metroCount: a.metros,
254 301 cloudRegionCount: a.cloudRegions,
255 302 projectCount: a.projects,
303 + projectPlannedMw: a.projectPlannedMw || null,
304 + projectConstructionMw: a.projectConstructionMw || null,
305 + expansionMw: a.expansionMw || null,
306 + aiConfirmedCount: a.aiConfirmed,
256 307 ixpCount: a.ixps,
257 308 openedLast3Years: a.openedRecent,
258 309 mwCoverage: a.coverage,
modified apps/worker/src/scheduler.ts +4 −1
@@ -39,7 +39,7 @@ export const RUN_LOCK_PREFIX = "dci:run-lock";
39 39 export const RUN_LOCK_TTL_SECONDS = 4 * 3600;
40 40
41 41 export interface CrawlJobData { connectorId: string; task: RunTask; group?: string; limit?: number; force?: boolean; urls?: string[]; requestedBy?: string }
42 −export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup";
42 +export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup" | "snapshot" | "quality";
43 43 export interface MaintenanceJobData { kind: MaintenanceKind; requestedBy?: string }
44 44
45 45 let _redis: Redis | null = null;
@@ -114,6 +114,9 @@ export async function ensureMaintenanceSchedulers(): Promise<void> {
114 114 await q.upsertJobScheduler("daily-rankings", { pattern: "30 0 * * *", tz }, { name: "rankings", data: { kind: "rankings", requestedBy: "scheduler" } });
115 115 await q.upsertJobScheduler("hourly-refresh-stats", { pattern: "0 * * * *", tz }, { name: "refresh-stats", data: { kind: "refresh-stats", requestedBy: "scheduler" } });
116 116 await q.upsertJobScheduler("daily-cleanup", { pattern: "0 1 * * *", tz }, { name: "cleanup", data: { kind: "cleanup", requestedBy: "scheduler" } });
117 + // daily snapshots + regression checks after rankings; quality sweep every 6 hours
118 + await q.upsertJobScheduler("daily-snapshot", { pattern: "45 0 * * *", tz }, { name: "snapshot", data: { kind: "snapshot", requestedBy: "scheduler" } });
119 + await q.upsertJobScheduler("quality-sweep", { pattern: "20 */6 * * *", tz }, { name: "quality", data: { kind: "quality", requestedBy: "scheduler" } });
117 120 }
118 121
119 122 export interface TickResult { checked: number; enqueued: string[]; skipped: string[]; lockHeldElsewhere: boolean }
modified apps/worker/src/scheduling.test.ts +11 −0
@@ -195,3 +195,14 @@ describe("pLimit", () => {
195 195 expect(await limit(async () => 42)).toBe(42);
196 196 });
197 197 });
198 +
199 +describe("connectorHealthFrom", () => {
200 + it("labels blocked, schema change and no-new-content connectors", async () => {
201 + const { connectorHealthFrom } = await import("./scheduling.js");
202 + expect(connectorHealthFrom({ runHealth: "failing", status: "partial", fetched: 10, blocked: 8, discovered: 0, previouslyDiscovered: null, extracted: 0, entities: 0, docsTotal: 50 })).toBe("blocked");
203 + expect(connectorHealthFrom({ runHealth: "ok", status: "ok", fetched: 12, blocked: 0, discovered: 0, previouslyDiscovered: null, extracted: 12, entities: 0, docsTotal: 200 })).toBe("schema_change");
204 + expect(connectorHealthFrom({ runHealth: "ok", status: "ok", fetched: 0, blocked: 0, discovered: 0, previouslyDiscovered: 40, extracted: 0, entities: 0, docsTotal: 40 })).toBe("no_new_content");
205 + expect(connectorHealthFrom({ runHealth: "ok", status: "ok", fetched: 8, blocked: 0, discovered: 3, previouslyDiscovered: 40, extracted: 8, entities: 8, docsTotal: 40 })).toBe("ok");
206 + expect(connectorHealthFrom({ runHealth: "degraded", status: "failed", fetched: 8, blocked: 0, discovered: 3, previouslyDiscovered: 40, extracted: 8, entities: 8, docsTotal: 40 })).toBe("failing");
207 + });
208 +});
modified apps/worker/src/scheduling.ts +16 −1
@@ -126,7 +126,11 @@ export function shouldSkipExtraction(i: { newHash: string; storedHash: string |
126 126 return (i.storedExtractorVersion ?? i.extractorVersion) === i.extractorVersion;
127 127 }
128 128
129 −/** Health label from a run: failure rate < 20 % ok, < 50 % degraded, else failing. */
129 +/**
130 + * Health label from a run: failure rate < 20 % ok, < 50 % degraded, else failing. A run that attempted nothing is
131 + * `ok` only when it had nothing to do (no due documents); a run whose discovery yielded nothing for a connector that
132 + * used to discover is reported by the caller as `no_new_content` / `blocked` (see pipeline connector health).
133 + */
130 134 export function healthFrom(fetched: number, failed: number): "ok" | "degraded" | "failing" {
131 135 if (fetched <= 0) return failed > 0 ? "failing" : "ok";
132 136 const rate = failed / fetched;
@@ -155,3 +159,14 @@ export function premiumAllowedAfterErrors(errorCount: number): boolean {
155 159 if (!Number.isFinite(errorCount) || errorCount < 2) return true;
156 160 return errorCount % 4 === 0;
157 161 }
162 +
163 +/** Connector health label (admin health center): HEALTHY | DEGRADED | BLOCKED | SCHEMA_CHANGE | NO_NEW_CONTENT | FAILED. */
164 +export type ConnectorHealth = "ok" | "degraded" | "failing" | "blocked" | "schema_change" | "no_new_content" | "never_run";
165 +export function connectorHealthFrom(i: { runHealth: "ok" | "degraded" | "failing"; status: string; fetched: number; blocked: number; discovered: number; previouslyDiscovered: number | null; extracted: number; entities: number; docsTotal: number }): ConnectorHealth {
166 + if (i.status === "failed") return "failing";
167 + if (i.fetched > 0 && i.blocked >= Math.max(1, Math.ceil(i.fetched * 0.5))) return "blocked";
168 + // pages fetched fine but the parsers found nothing on pages that used to yield entities → the site's HTML changed
169 + if (i.fetched >= 5 && i.extracted >= 5 && i.entities === 0 && i.docsTotal > 0) return "schema_change";
170 + if (i.previouslyDiscovered != null && i.previouslyDiscovered > 0 && i.discovered === 0 && i.fetched === 0) return "no_new_content";
171 + return i.runHealth;
172 +}
added apps/worker/src/trace.ts +149 −0
@@ -0,0 +1,149 @@
1 +/**
2 + * Extraction debugger — the full pipeline for ONE document, stage by stage, without publishing anything:
3 + *
4 + * SOURCE DOCUMENT → RAW FETCH (archived body or live) → PARSED TEXT → STRUCTURED DATA (extracted records) →
5 + * EXTRACTED ENTITIES (normalized) → EXTRACTED CLAIMS (figures with scope / semantics / evidence sentence) →
6 + * NORMALIZED VALUES → MATCH CANDIDATES (facility resolution preview) → RECONCILIATION (dry-run ingest: created /
7 + * updated / events / rejected) → RESULTING DATABASE CHANGES (what a real run would write).
8 + *
9 + * Exposed by `dci trace <doc-id|url>` and by the worker HTTP endpoint `GET /trace/<doc-id>` (internal network; the API
10 + * proxies it as `/api/admin/documents/:id/trace`).
11 + */
12 +import type { RawDocument } from "@dci/connectors";
13 +import { entityIsValid, mainText, pageTitle, isPdf, pdfText, classifyPage } from "@dci/connectors";
14 +import type { NormalizedEntity, NormalizedFacility, NormalizedProject, ValidationIssue } from "@dci/core";
15 +import { classifyAiEvidence, classifyCapacitySemantics, classifyInvestmentSemantics, classifyProjectEvent, classifyScope, findEvidence, newId, parseAllMw } from "@dci/core";
16 +import { getDb } from "@dci/db";
17 +import { requireConnector } from "./configs.js";
18 +import { createRunContext } from "./context.js";
19 +import { getDocument } from "./documents.js";
20 +import { parseDiscoveredFrom } from "./scheduling.js";
21 +import { ingestEntities } from "./ingest/index.js";
22 +import { newContext } from "./ingest/common.js";
23 +import { previewFacilityResolution } from "./ingest/facilities.js";
24 +import { extractAnnouncement } from "./connectors/news/extract-project.js";
25 +import { getRaw } from "./storage.js";
26 +
27 +export interface TraceClaim { entityKey: string; field: string; value: number; unit: "MW" | "USD"; scope: string; scopeReason: string; semantics: string | null; evidence: { text: string; start: number; end: number } | null; }
28 +
29 +export interface TraceResult {
30 + document: Record<string, unknown> | null;
31 + fetch: { source: "archive" | "live"; status: number; contentType: string | null; bytes: number; storageKey: string | null; level: number; fetcher: string } | null;
32 + text: { title: string | null; length: number; excerpt: string; classification: { pageType: string; rule: string; eventType: string | null; mw: number[] } | null };
33 + announcement: Record<string, unknown> | null;
34 + records: Array<{ kind: string; key: string; certainty: number | null; pageType: string | null; data: Record<string, unknown>; methods: Record<string, string> }>;
35 + entities: NormalizedEntity[];
36 + validation: { total: number; valid: number; rejected: number; issues: ValidationIssue[] };
37 + claims: TraceClaim[];
38 + matches: Array<{ entityKey: string; how: string; matchedId: string | null; candidates: Array<{ id: string; name: string; operatorName: string | null; city: string | null; score: number; reasons: string[]; distanceKm: number | null }> }>;
39 + reconciliation: { created: number; updated: number; unchanged: number; merged: number; pendingMatches: number; rejected: number; events: number; provenanceRows: number; claims: number; qualityFlags: number; unscopedClaims: number; projectsVetoed: number; changes: unknown[]; refs: Array<{ type: string; id: string }> } | null;
40 + logs: string[];
41 + error: string | null;
42 +}
43 +
44 +/** Highlight offsets of every MW / money figure in the text (for the debugger UI). */
45 +export function highlightFigures(text: string): Array<{ start: number; end: number; kind: "mw" | "usd"; raw: string }> {
46 + const out: Array<{ start: number; end: number; kind: "mw" | "usd"; raw: string }> = [];
47 + for (const m of text.matchAll(/(\d{1,3}(?:[.,]\d{3})+|\d+(?:[.,]\d+)?)\+?\s?(gigawatts?|gw|megawatts?|mw)\b/gi)) out.push({ start: m.index ?? 0, end: (m.index ?? 0) + m[0].length, kind: "mw", raw: m[0] });
48 + for (const m of text.matchAll(/(?:US\$|USD|\$|€|£|A\$|C\$|S\$)\s?\d[\d.,]*\s*(?:trillion|billion|million|bn|m\b|b\b|k\b)?/gi)) out.push({ start: m.index ?? 0, end: (m.index ?? 0) + m[0].length, kind: "usd", raw: m[0] });
49 + return out.sort((a, b) => a.start - b.start);
50 +}
51 +
52 +export async function traceDocument(idOrUrl: string, opts: { live?: boolean } = {}): Promise<TraceResult> {
53 + const logs: string[] = [];
54 + const result: TraceResult = { document: null, fetch: null, text: { title: null, length: 0, excerpt: "", classification: null }, announcement: null, records: [], entities: [], validation: { total: 0, valid: 0, rejected: 0, issues: [] }, claims: [], matches: [], reconciliation: null, logs, error: null };
55 + const doc = await getDocument(idOrUrl);
56 + if (!doc) { result.error = `no document for ${idOrUrl}`; return result; }
57 + result.document = { id: doc.id, connectorId: doc.connectorId, url: doc.url, pageType: doc.pageType, classifier: doc.classifier, contentHash: doc.contentHash, extractorVersion: doc.extractorVersion, extractOk: doc.extractOk, extractCount: doc.extractCount, error: doc.error, lastFetched: doc.lastFetched, lastChanged: doc.lastChanged, fetchLevel: doc.fetchLevel, storageKey: doc.storageKey, title: doc.title, entityRefs: doc.entityRefs, quarantined: doc.quarantined };
58 + const loaded = requireConnector(doc.connectorId);
59 + const runId = newId("run");
60 + const ctx = createRunContext(loaded, { runId, dryRun: true, onLog: (l) => logs.push(`${l.level} ${l.msg}`), logLevel: "debug" });
61 + try {
62 + // ── raw
63 + let raw: RawDocument | null = null;
64 + if (!opts.live && doc.storageKey) {
65 + const stored = await getRaw(doc.storageKey);
66 + if (stored) {
67 + const contentType = stored.contentType ?? doc.contentType;
68 + const isText = !contentType || /text|json|xml|javascript|html|csv|markdown/i.test(contentType);
69 + const { group } = parseDiscoveredFrom(doc.discoveredFrom);
70 + raw = { url: doc.url, finalUrl: doc.url, fetchedAt: doc.lastFetched ?? new Date().toISOString(), status: doc.statusCode ?? 200, contentType, body: stored.body, text: isText ? stored.body.toString("utf8") : "", headers: {}, etag: doc.etag, lastModified: doc.lastModified, notModified: false, fetcher: "cache", level: doc.fetchLevel as RawDocument["level"], durationMs: 0, credits: 0, group, pageType: doc.pageType !== "unknown" ? (doc.pageType as RawDocument["pageType"]) : undefined };
71 + result.fetch = { source: "archive", status: raw.status, contentType, bytes: stored.body.length, storageKey: doc.storageKey, level: doc.fetchLevel, fetcher: "cache" };
72 + }
73 + }
74 + if (!raw) {
75 + const { group } = parseDiscoveredFrom(doc.discoveredFrom);
76 + raw = await loaded.connector.fetch(ctx, { url: doc.url, group, pageType: doc.pageType !== "unknown" ? (doc.pageType as RawDocument["pageType"]) : undefined, priority: 100 });
77 + result.fetch = { source: "live", status: raw.status, contentType: raw.contentType, bytes: raw.body.length, storageKey: null, level: raw.level, fetcher: raw.fetcher };
78 + if (raw.error) { result.error = `fetch: ${raw.error.code} ${raw.error.message}`; return result; }
79 + }
80 + // ── text
81 + let title: string | null = null;
82 + let text = "";
83 + if (isPdf(raw)) { const p = await pdfText(raw.body); text = p.text; title = p.title ?? null; }
84 + else if (/html|xml/.test(raw.contentType ?? "") || /<html/i.test(raw.text.slice(0, 2000))) { title = pageTitle(raw.text); text = mainText(raw.text); }
85 + else text = raw.markdown ?? raw.text;
86 + const cls = classifyPage(raw.finalUrl, title, text);
87 + result.text = { title, length: text.length, excerpt: text.slice(0, 6000), classification: { pageType: cls.pageType, rule: cls.rule, eventType: cls.eventType ?? null, mw: cls.mw ?? [] } };
88 + // ── announcement analysis (news-like pages)
89 + if (text.length > 80 && /news|press|announcement|planning|construction|expansion|acquisition|power|financial|project|unknown/.test(cls.pageType)) {
90 + try {
91 + const a = extractAnnouncement(title, text);
92 + result.announcement = { title: a.title, pageType: a.pageType, eventType: a.eventType ?? null, relevance: a.relevance, leadRelevance: a.leadRelevance, titleSignal: a.titleSignal, nonProjectTitle: a.nonProjectTitle, classification: a.classification, status: a.status, headlineMw: a.headlineMw, mwAll: a.mwAll, money: a.money, investmentUsd: a.investmentUsd, acreage: a.acreage, phaseCount: a.phaseCount, expectedOpening: a.expectedOpening, location: a.location, operator: a.operator?.name ?? null, operators: a.operators.map((o) => o.name), explicitName: a.explicitName, projectName: a.projectName, capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics, aiEvidence: a.aiEvidence, hqGuarded: a.hqGuarded, evidence: a.evidence, methods: a.methods, figures: highlightFigures(`${a.title}\n${text.slice(0, 6000)}`) };
93 + } catch (e) { logs.push(`warn announcement analysis failed: ${(e as Error).message}`); }
94 + }
95 + // ── structured data
96 + const records = await loaded.connector.extract(ctx, raw);
97 + result.records = records.map((r) => ({ kind: r.kind, key: r.key, certainty: r.certainty ?? null, pageType: r.pageType ?? null, data: r.data, methods: r.methods ?? {} }));
98 + // ── entities + validation
99 + const entities = await loaded.connector.normalize(ctx, records);
100 + const report = await loaded.connector.validate(ctx, entities);
101 + result.entities = entities;
102 + result.validation = { total: report.total, valid: report.valid, rejected: report.rejected, issues: report.issues };
103 + // ── claims (what the claim store would receive)
104 + for (const e of entities) {
105 + if (e.entityType === "facility") {
106 + const f = e as NormalizedFacility;
107 + for (const field of ["itCapacityMw", "totalPowerMw", "plannedPowerMw"] as const) {
108 + const v = f[field];
109 + if (v == null || v <= 0) continue;
110 + const context = f.claimContext?.[field] ?? null;
111 + const sc = context ? classifyScope(context, "facility") : { scope: "facility", reason: "structured:record" };
112 + const sem = context ? classifyCapacitySemantics(context) : { predicate: field === "itCapacityMw" ? "it_capacity_mw" : field === "totalPowerMw" ? "current_power_mw" : "planned_power_mw", reason: "field" };
113 + result.claims.push({ entityKey: f.key, field, value: v, unit: "MW", scope: sc.scope, scopeReason: sc.reason, semantics: sem.predicate, evidence: context ? findEvidence(context, v, "mw") : null });
114 + }
115 + } else if (e.entityType === "project") {
116 + const p = e as NormalizedProject;
117 + if (p.plannedMw != null && p.plannedMw > 0) { const c = p.claimContext?.plannedMw ?? null; result.claims.push({ entityKey: p.key, field: "plannedMw", value: p.plannedMw, unit: "MW", scope: p.capacityScope ?? classifyScope(c, "facility").scope, scopeReason: p.capacityScope ? "extractor" : classifyScope(c, "facility").reason, semantics: p.capacitySemantics ?? null, evidence: c ? findEvidence(c, p.plannedMw, "mw") : null }); }
118 + if (p.investmentUsd != null && p.investmentUsd > 0) { const c = p.claimContext?.investmentUsd ?? null; const sem = classifyInvestmentSemantics(c ?? p.name); result.claims.push({ entityKey: p.key, field: "investmentUsd", value: p.investmentUsd, unit: "USD", scope: p.investmentScope ?? sem.scope, scopeReason: p.investmentScope ? "extractor" : sem.reason, semantics: p.investmentSemantics ?? sem.predicate, evidence: c ? findEvidence(c, p.investmentUsd, "usd") : null }); }
119 + } else if (e.entityType === "news_event") {
120 + for (const mw of e.mentions?.mw ?? []) result.claims.push({ entityKey: e.key, field: "mw", value: mw, unit: "MW", scope: "unknown", scopeReason: "news mention (never assigned)", semantics: null, evidence: findEvidence(text, mw, "mw") });
121 + }
122 + }
123 + // ── match candidates (facilities) — read-only, inside a rolled-back transaction
124 + const valid = entities.filter((x) => entityIsValid(report, x.key));
125 + const db = getDb();
126 + const run = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: true };
127 + await db.transaction(async (tx) => {
128 + const ictx = newContext(run, { documentId: doc.id, url: raw!.finalUrl, pageType: cls.pageType, fetchedAt: raw!.fetchedAt });
129 + for (const e of valid) {
130 + if (e.entityType !== "facility") continue;
131 + try { const m = await previewFacilityResolution(tx, ictx, e as NormalizedFacility); result.matches.push({ entityKey: e.key, ...m }); } catch (err) { logs.push(`warn match preview ${e.key}: ${(err as Error).message}`); }
132 + }
133 + throw new Error("__trace_rollback__");
134 + }).catch((e: Error) => { if (e.message !== "__trace_rollback__") throw e; });
135 + // ── reconciliation (dry-run ingest: everything runs, nothing is committed)
136 + if (valid.length) {
137 + const st = await ingestEntities(run, valid, { documentId: doc.id, url: raw.finalUrl, pageType: cls.pageType, fetchedAt: raw.fetchedAt });
138 + result.reconciliation = { created: st.created, updated: st.updated, unchanged: st.unchanged, merged: st.merged, pendingMatches: st.pendingMatches, rejected: st.rejected, events: st.events, provenanceRows: st.provenanceRows, claims: st.claims ?? 0, qualityFlags: st.qualityFlags ?? 0, unscopedClaims: st.unscopedClaims ?? 0, projectsVetoed: st.projectsVetoed ?? 0, changes: st.changes, refs: st.refs };
139 + } else result.reconciliation = { created: 0, updated: 0, unchanged: 0, merged: 0, pendingMatches: 0, rejected: report.rejected, events: 0, provenanceRows: 0, claims: 0, qualityFlags: 0, unscopedClaims: 0, projectsVetoed: 0, changes: [], refs: [] };
140 + // AI grading of the whole text for the debugger
141 + logs.push(`debug ai-evidence: ${JSON.stringify(classifyAiEvidence(`${title ?? ""} ${text.slice(0, 3000)}`))}`);
142 + logs.push(`debug mw figures in text: ${parseAllMw(text.slice(0, 6000)).join(", ") || "none"}`);
143 + // the announcement classifier verdict for non-news pages too
144 + if (!result.announcement && title) logs.push(`debug classifier: ${JSON.stringify(classifyProjectEvent({ title, lead: text.slice(0, 600) }))}`);
145 + } catch (e) {
146 + result.error = (e as Error).stack ?? (e as Error).message;
147 + }
148 + return result;
149 +}
added docs/AUDIT-2026-09-11.md +140 −0
@@ -0,0 +1,140 @@
1 +# DataCenterIndex — full repository & data audit (2026-09-11)
2 +
3 +Internal audit written before the "infrastructure intelligence graph" upgrade. Sections A–K follow the brief. Figures come
4 +from the production database on BHS128 (`deploy/bin/psql.sh`) on 2026-09-11 evening, one day after launch.
5 +
6 +## A. Current architecture
7 +
8 +- **Monorepo** pnpm + TypeScript strict/ESM: `packages/core` (types, ids, normalize, geo, confidence, ssrf, api contract),
9 + `packages/db` (Drizzle schema, plain SQL migrations, ClickHouse DDL), `packages/connectors` (connector SDK), `apps/worker`
10 + (runtime, ingest, scheduler, rankings, CLI), `apps/api` (Fastify 5, `/api/v1` public + `/api/admin` token), `apps/web`
11 + (Next 16 App Router, Tailwind v4, MapLibre + OpenFreeMap, hand-rolled SVG charts).
12 +- **Rendering**: every public page is ISR (`revalidate` 30–300 s) on top of a typed API client that never throws; map page
13 + is a static shell with client fetches to `/api/v1/map` (server-side clustering, ≤ 5 000 points).
14 +- **Deployment**: two OVH servers joined by WireGuard. BHS128 = data/public (Caddy edge :8300 → web :8310 / api :8311,
15 + Postgres 17, ClickHouse, Redis, MinIO, Prometheus, Grafana). BHS64b = crawl worker (concurrency 6). Public route via the
16 + MacLustr Tunnel gateway BHS64. Nightly backup timer (pg_dump, config tar, ClickHouse backup best-effort, MinIO mirror,
17 + rsync off-node to BHS64b). No CI, no Alertmanager.
18 +- **Finding (ops)**: the production `scheduler` container is crash-looping (851 restarts, exit 0, no log) because the deployed
19 + image predates commit `ff67cee` (standalone scheduler entry). Crawling continues because the BHS64b worker embeds the
20 + scheduler loop. Fixed by the redeploy at the end of this upgrade.
21 +
22 +## B. Current database schema (Postgres)
23 +
24 +28 tables. Core: `facilities` (2 912 rows; `it_capacity_mw`, `total_power_mw`, `planned_power_mw`, `mw_is_estimate`,
25 +`geo_precision`, `status`, `confidence`, `completeness`, `campus_id`, `merged_into`), `operators` (470), `projects` (324;
26 +`planned_mw`, `investment_usd`, `status`), `campuses` (114), `metros` (214), `countries` (250), `cloud_regions` (307),
27 +`ixps` (12), `facility_ixps`, `facility_tenants`, `facility_aliases`, `entity_keys` (connector-scoped keys, **global PK on
28 +`key`**), `provenance` (38 533 rows, one row per entity×field×source×url, `is_current`, `method`, `extractor_version`),
29 +`events` (5 549, fingerprint-deduplicated, `significance` 0–100), `news_items` (4 404), `documents` (7 092) +
30 +`document_versions` (8 785, content-hash versions with `diff_summary`), `entity_matches` (1 730: 1 200 auto_created, 81
31 +auto_merged, **449 pending never reviewed**), `rankings` (snapshots with `min_coverage`), `daily_metrics`, `sources`,
32 +`connectors`, `connector_runs`, `connector_state`, `project_timeline`, `system_alerts`, `admin_sessions`.
33 +ClickHouse holds append-only `observations`, `crawl_log`, `page_changes`, `entity_daily`, `api_requests` (best-effort, silently
34 +dropped on failure).
35 +
36 +Indexes: btree on the obvious FKs/status, trigram GIN on names/cities, generated tsvector on facilities, partial `(lat,lng)`.
37 +**No PostGIS** (haversine + bbox pre-filter + geohash). No `claims`/observation table in Postgres; no entity-version table;
38 +no `run_id` on `provenance`/`document_versions`.
39 +
40 +## C. Current crawler architecture
41 +
42 +Declarative YAML per source (`config/connectors/*.yaml`) → `GenericConnector` or registered parser/implementation →
43 +`discover → fetch (L1 direct → L2 browser identity → L3 Firecrawl → L4 Scrapfly) → archive (MinIO, zstd, per content hash)
44 +→ extract → normalize → validate → ingest (reconcile, provenance, events)`. Robots.txt honoured, per-host token bucket,
45 +conditional GET, content-hash change detection, extraction skipped when hash and `extractor_version` unchanged, adaptive
46 +`next_check` with EMA change score, per-document quarantine after 3 × 404/410. BullMQ `crawl` + `maintenance` queues,
47 +Redis run locks, per-run/per-connector/per-provider premium budgets, Prometheus metrics, `dci doctor`.
48 +
49 +## D. Current connectors
50 +
51 +104 YAML configs (87 enabled): 60 operator directories, 20 industry/hyperscaler newsrooms (RSS), 10 cloud-region JSON,
52 +6 government, 3 utilities, 3 datasets (PeeringDB — disabled pending AUP approval —, Wikidata, World Bank), SEC EDGAR
53 +full-text, OSM Overpass. Facility rows by connector: OSM 1 346, Equinix 262, Digital Realty 259, Wikidata 190, STACK 78,
54 +DataBank 75, NTT 72, EdgeConneX 59, CyrusOne 52, Google 47, QTS 47 … 58 named parsers; **zero configs use the declarative
55 +extractor** (dead code path); **zero HTML fixtures**; `packages/core` and `packages/connectors` have no tests.
56 +
57 +## E. Current data-quality issues (measured)
58 +
59 +Facilities: 2 640 operational / 154 unknown / 76 announced / 37 under construction; only **536 (18 %) have any MW**;
60 +797 have no coordinates; 1 377 have `facility_type = unknown`; only 4 flagged AI; 109 duplicate normalized-name pairs and
61 +469 coordinate pairs within ~100 m (OSM building-vs-campus, e.g. atNorth ICE02 vs buildings M01…M16).
62 +
63 +Confirmed extraction errors:
64 +- **Decimal MW parsed as thousands**: DataBank MSP4 shows 779 MW; source says "0.779MW Critical IT Load". Same bug hits
65 + DFW6 (675), AUS1 (405) and any "0.xxx MW" figure (`parseMw` strips the dot when three decimals follow).
66 +- **Headlines stored as project names**: 182/324 projects are named after an article title.
67 +- **Executive appointments as projects**: "STACK Appoints Matt VanderZanden as CEO" = 13 000 MW project;
68 + "AirTrunk appoints Laura Coad" = 1 400 MW; VIRTUS, 365 Data Centers, Element Critical, Cassava appointments likewise.
69 +- **Portfolio/company figures on one project**: "Microsoft Georgia data center project" 12 GW / $175 B (company capex);
70 + "Google Alabama" 17 GW (Southern Company pipeline); Nvidia "Washington" 20 GW (Starcloud funding story).
71 +- **Market-research articles as projects**: "Hyperscale Data Center Market to Surpass USD 624.2 Billion" ($80.9 B).
72 +- **Wrong operator**: "Meta's Canadian AI Data Center" assigned to Google.
73 +- **PPA / sustainability / HQ / workforce stories as projects** (Ormat–Switch PPA 3 400 MW, AirTrunk HQ 1 400 MW,
74 + Meta Workforce Academy 5 000 MW).
75 +- **Projects have no coordinates at all** (0/324), 158 have a country, 0 link to a facility.
76 +- **Broken operator slug** `item` for 中華電信數據通信分公司 (26 facilities).
77 +- Ranking coverage gate leaves 3 countries in `countries_known_mw` — the MW coverage story must be visible, not hidden.
78 +
79 +Structural causes (code): one-row-per-entity with no claim scope; `isIdentifyingExternalId` folds unrelated facilities on
80 +`*_code/*_url` keys; `loadCandidates` truncates at 400 rows without ordering; news country via unsafe `countryFromText`
81 +("North America" → US); no HQ guard on project location; `projects.country_iso2` write-once; campus/building double counting
82 +in every MW aggregate; equal-authority ping-pong after 30 days creates event churn; `entity_keys.key` global PK;
83 +`parserVersion` is `v1` everywhere so parser fixes never invalidate cached extractions; `healthFrom(0,0) = ok`.
84 +
85 +## F. Current frontend strengths
86 +
87 +Unified status/confidence/precision visual language (CSS variables shared by badges, charts and MapLibre paint);
88 +per-field provenance popovers + provenance table + source history + sources footer with licences; dense typography with
89 +mono tabular figures; URL-as-state everywhere; ⌘K/`/` command palette with API-side query interpretation; server-clustered
90 +map with precision rings; full SEO (canonical, per-entity OG images, JSON-LD, sharded sitemaps); a11y depth; an automated
91 +UX sweep harness; a substantial `/admin` console (connectors, documents, matches, events, dev tool).
92 +
93 +## G. Current frontend weaknesses
94 +
95 +No explore/query builder, no compare, no evidence drawer (only popover + bottom table), no export, no watchlist, no
96 +coverage page, no AI index, no pulse, no power/connectivity layers, no map density/time modes, no virtualization,
97 +long single-scroll detail pages without active-section tracking, hand-rolled charts without brushing, keyboard story stops
98 +at the palette, project pages lack a state machine/timeline funnel, MapLibre controls < 40 px on touch.
99 +
100 +## H. Current API strengths
101 +
102 +27 public GET endpoints, envelope `{data, meta, sources}`, zod-validated params, weak ETags + `s-maxage`, rate limits,
103 +consistent 404/400 JSON, OpenAPI + Swagger UI, `/map` zoom-tiered clustering, `/search` with interpretation, admin API with
104 +constant-time token compare and same-origin proxy with CSRF guard. Weaknesses: no provenance/history/nearby/coverage/export
105 +endpoints; list responses carry no `sources`; OpenAPI has no response schemas; placeholder-token check is a no-op.
106 +
107 +## I. Schema changes proposed (migration 0003)
108 +
109 +1. `claims` — claim-first store: subject (type,id), predicate, value/unit, **scope** (building/facility/campus/metro/
110 + country/portfolio/company/unknown), source/document/url, published/retrieved, confidence, is_estimate, evidence text +
111 + offsets, parser name/version, `run_id`, status (current/superseded/rejected/review/unscoped), rejection reason.
112 +2. `quality_flags` — deterministic sanity flags (capacity, investment, scope, location, operator, duplicate, project
113 + false-positive) with severity, priority score, status, resolution, linking to entity + claim.
114 +3. `facilities.parent_facility_id`, `facilities.record_scope` (building/facility/campus) for containment-aware aggregation;
115 + `facilities.ai_evidence` (confirmed/likely/associated/unknown); `facilities.utility_capacity_mw`, `grid_connection_mw`,
116 + `ultimate_campus_mw` (capacity ontology columns; IT/total/planned kept).
117 +4. `projects.project_class` (NEW_BUILD … EXECUTIVE_APPOINTMENT), `projects.evidence_level`, `projects.merged_into`,
118 + `projects.lat/lng` geocoded from metro/city with `geo_precision = city`, `projects.ai_evidence`, `projects.investment_scope`.
119 +5. `document_versions.run_id`, `provenance.run_id`, `provenance.scope`; `connectors.quarantine` (bool) +
120 + `consecutive_failures`, `blocked_since`.
121 +6. `field_authority` (per-field source-kind tiers) as code table in `packages/core` + methodology page (no table needed).
122 +7. `entity_snapshots` (daily JSON snapshot of global totals, rankings, facility status counts, project stage counts) for
123 + "as of" views and regression checks.
124 +8. Indexes: `provenance(is_current)`, `facilities(merged_into)`, `facilities(campus_id)`, `facilities(parent_facility_id)`,
125 + `events(significance, detected_at)`, `daily_metrics(metric, dim, day)`, GIN on `news_items.operator_ids`.
126 +
127 +## J. Components to preserve
128 +
129 +Everything in section F/H, the connector SDK and escalation ladder, the content-hash/archive pipeline, entity keys,
130 +provenance table, event fingerprinting, the matcher (with fixes), rankings coverage gates, the design tokens, the map
131 +architecture, the admin console, deploy scripts, backups, the QA sweep.
132 +
133 +## K. Implementation order
134 +
135 +Phase 1 data quality (parseMw fix, claim/scope layer, project classifier + evidence threshold, capacity/investment sanity
136 +engine, external-id allowlist, HQ/country guards, campus containment aggregation, extraction debugger, connector health,
137 +fixtures, cleanup of current bad records) → Phase 2 core UX (home 2.0, map 2.0, facility 3.0, operator/market/country/project
138 +2.0, brand, mobile) → Phase 3 intelligence (AI index, pulse, live feed 2.0, compare, explore, coverage, quality dashboards)
139 +→ Phase 4 graph (connectivity, power, cloud, IXP pages) → Phase 5 historical (time machine, capacity history, announced vs
140 +delivered, velocity) → API 2.0 + docs + download → regression checks, security fixes, deploy, final data audit.
modified packages/connectors/src/generic.ts +4 −2
@@ -122,7 +122,7 @@ export class GenericConnector implements Connector {
122 122 // when no explicit geo but page has JSON-LD/meta geo, the extractor may already have set lat/lng; otherwise leave null (never fake coordinates)
123 123 out.push(f);
124 124 } else if (r.kind === "news_event") {
125 − const n: NormalizedNewsEvent = { entityType: "news_event", key: r.key, title: str("title") ?? r.url, url: (str("url") ?? r.url)!, publishedAt: str("publishedAt"), summary: str("summary"), pageType: (str("pageType") as PageType | null) ?? r.pageType, eventType: (str("eventType") as NormalizedNewsEvent["eventType"]) ?? undefined, mentions: { mw: Array.isArray(x.mw) ? (x.mw as number[]) : [], operators: [...(Array.isArray(x.operators) ? (x.operators as string[]) : []), ...(d.operatorName ? [d.operatorName] : [])], countriesIso2: [...(Array.isArray(x.countries) ? (x.countries as string[]) : []), countryFromText(`${str("title")} ${str("summary")}`) ?? ""].filter((c, i, a) => c && a.indexOf(c) === i), cities: Array.isArray(x.cities) ? (x.cities as string[]) : [] }, provenance: prov };
125 + const n: NormalizedNewsEvent = { entityType: "news_event", key: r.key, title: str("title") ?? r.url, url: (str("url") ?? r.url)!, publishedAt: str("publishedAt"), summary: str("summary"), pageType: (str("pageType") as PageType | null) ?? r.pageType, eventType: (str("eventType") as NormalizedNewsEvent["eventType"]) ?? undefined, mentions: { mw: Array.isArray(x.mw) ? (x.mw as number[]) : [], operators: [...(Array.isArray(x.operators) ? (x.operators as string[]) : []), ...(d.operatorName ? [d.operatorName] : [])], countriesIso2: [...(Array.isArray(x.countries) ? (x.countries as string[]) : []), countryFromText(`${str("title")} ${str("summary")}`) ?? ""].filter((c, i, a) => c && a.indexOf(c) === i), cities: Array.isArray(x.cities) ? (x.cities as string[]) : [] }, projectClass: str("projectClass"), isAi: typeof x.isAi === "boolean" ? x.isAi : null, provenance: prov };
126 126 out.push(n);
127 127 // Project candidate from a fallback news_event (no dedicated parser). Third-party publishers (kind = news) go
128 128 // through `news_article_v1`, which has the real gate — the fallback never creates projects for them. For
@@ -139,7 +139,9 @@ export class GenericConnector implements Connector {
139 139 } else if (r.kind === "project") {
140 140 const name = str("name") ?? str("_title"); if (!name) continue;
141 141 const lat = num("lat"), lng = num("lng");
142 − out.push({ entityType: "project", key: r.key, name, operatorName: str("operatorName") ?? d.operatorName ?? null, city: str("city"), regionName: str("regionName"), countryIso2: str("countryIso2") ?? d.countryIso2 ?? null, geo: validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: "approximate", source: this.id } : null, status: (str("status") as FacilityStatus | null) ?? "announced", announcedOn: str("announcedOn"), expectedOpening: str("expectedOpening"), plannedMw: num("plannedMw"), investmentUsd: num("investmentUsd"), acreage: num("acreage"), phaseCount: num("phaseCount"), description: str("description"), sourceUrl: r.url, provenance: prov });
142 + const claimContext = x.claimContext && typeof x.claimContext === "object" ? (x.claimContext as Record<string, string>) : undefined;
143 + out.push({ entityType: "project", key: r.key, name, operatorName: str("operatorName") ?? d.operatorName ?? null, city: str("city"), regionName: str("regionName"), countryIso2: str("countryIso2") ?? d.countryIso2 ?? null, geo: validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: "approximate", source: this.id } : null, status: (str("status") as FacilityStatus | null) ?? "announced", announcedOn: str("announcedOn"), expectedOpening: str("expectedOpening"), plannedMw: num("plannedMw"), investmentUsd: num("investmentUsd"), acreage: num("acreage"), phaseCount: num("phaseCount"), description: str("description"), sourceUrl: r.url,
144 + projectClass: str("projectClass"), evidenceLevel: (str("evidenceLevel") as NormalizedProject["evidenceLevel"]) ?? null, capacityScope: str("capacityScope"), capacitySemantics: str("capacitySemantics"), investmentScope: str("investmentScope"), investmentSemantics: str("investmentSemantics"), investmentCurrency: str("investmentCurrency"), investmentOriginal: num("investmentOriginal"), claimContext, aiEvidence: (str("aiEvidence") as NormalizedProject["aiEvidence"]) ?? null, developerName: str("developerName"), tenantName: str("tenantName"), campusName: str("campusName"), constructionStartedOn: str("constructionStartedOn"), approvedOn: str("approvedOn"), permitFiledOn: str("permitFiledOn"), provenance: prov });
143 145 } else if (r.kind === "operator") {
144 146 const name = str("name"); if (!name) continue;
145 147 out.push({ entityType: "operator", key: r.key, name, website: str("website"), kind: (str("kind") as NormalizedEntity extends { kind?: infer K } ? K : never) ?? null, hqCountryIso2: str("hqCountryIso2"), description: str("description"), provenance: prov });
modified packages/core/src/api-types.ts +580 −4
@@ -3,6 +3,7 @@
3 3 * Every response is wrapped: { data, meta?, sources? }. Dates are ISO strings; partial dates use "YYYY", "YYYY-MM", "YYYY-Qn".
4 4 */
5 5 import type { ConfidenceLevel, EntityType, EventType, FacilityStatus, FacilityType, GeoPrecision, OperatorKind, ReviewStatus, SourceKind } from "./types.js";
6 +import type { AiEvidence, AuthorityTier, ClaimScope, ClaimStatus, ProjectClass, Significance } from "./claims.js";
6 7
7 8 export interface ApiEnvelope<T> {
8 9 data: T;
@@ -18,6 +19,9 @@ export interface SourceRef {
18 19 url?: string | null;
19 20 license?: string | null;
20 21 attribution?: string | null;
22 + /** allowed | attribution | restricted | unknown — drives /download and re-use guidance */
23 + redistribution?: "allowed" | "attribution" | "restricted" | "unknown" | null;
24 + attributionRequired?: boolean | null;
21 25 }
22 26
23 27 export interface ProvenanceDTO {
@@ -33,6 +37,82 @@ export interface ProvenanceDTO {
33 37 confidence: ConfidenceLevel;
34 38 isEstimate: boolean;
35 39 method?: string | null;
40 + /** true for the observation backing the displayed column value */
41 + isWinner?: boolean;
42 + scope?: string | null;
43 + runId?: string | null;
44 + documentId?: string | null;
45 +}
46 +
47 +/** One figure asserted by one document about one subject (claim-first architecture). */
48 +export interface ClaimDTO {
49 + id: string;
50 + predicate: string;
51 + label: string;
52 + value: number | null;
53 + valueText: string | null;
54 + unit: string | null;
55 + scope: ClaimScope;
56 + scopeReason: string | null;
57 + sourceId: string;
58 + sourceName: string;
59 + sourceKind: SourceKind;
60 + url: string;
61 + documentId: string | null;
62 + publishedAt: string | null;
63 + retrievedAt: string;
64 + confidence: ConfidenceLevel;
65 + isEstimate: boolean;
66 + authorityTier: AuthorityTier;
67 + evidenceText: string | null;
68 + parserName: string | null;
69 + parserVersion: string | null;
70 + status: ClaimStatus;
71 + rejectionReason: string | null;
72 + /** true when this claim backs the displayed value */
73 + isWinner: boolean;
74 + firstObserved: string;
75 + lastObserved: string;
76 +}
77 +
78 +/** A dated value of one field, for capacity history charts and change diffs. */
79 +export interface HistoryPoint {
80 + date: string;
81 + field: string;
82 + predicate?: string | null;
83 + value: unknown;
84 + oldValue?: unknown;
85 + sourceId: string | null;
86 + sourceName: string | null;
87 + sourceKind: SourceKind | null;
88 + url: string | null;
89 + claimId?: string | null;
90 + eventId?: string | null;
91 + kind: "observed" | "changed" | "claim";
92 +}
93 +
94 +export interface EntityHistory {
95 + entityType: "facility" | "project" | "operator";
96 + entityId: string;
97 + fields: Record<string, HistoryPoint[]>;
98 + /** field value changes with old → new and the evidence behind each */
99 + changes: HistoryPoint[];
100 +}
101 +
102 +export interface Distance { distanceKm: number }
103 +export interface NearbyInfrastructure {
104 + center: { lat: number; lng: number };
105 + radiusKm: number;
106 + facilities: Array<FacilitySummary & Distance>;
107 + projects: Array<ProjectSummary & Distance>;
108 + ixps: Array<IxpSummary & Distance & { lat: number | null; lng: number | null }>;
109 + cloudRegions: Array<CloudRegionSummary & Distance>;
110 + metros: Array<{ id: string; slug: string; name: string; countryIso2: string; distanceKm: number }>;
111 + /** cable landing stations / substations / power plants — empty until their connectors exist (never faked) */
112 + landingStations: Array<{ id: string; name: string; lat: number; lng: number; distanceKm: number; sourceName: string }>;
113 + substations: Array<{ id: string; name: string; lat: number; lng: number; distanceKm: number; sourceName: string }>;
114 + powerPlants: Array<{ id: string; name: string; lat: number; lng: number; distanceKm: number; kind: string | null; capacityMw: number | null; sourceName: string }>;
115 + note: string;
36 116 }
37 117
38 118 export interface OperatorSummary {
@@ -47,10 +127,44 @@ export interface OperatorSummary {
47 127 metroCount?: number;
48 128 knownMw?: number | null;
49 129 plannedMw?: number | null;
130 + constructionMw?: number | null;
50 131 projectCount?: number;
132 + projectPlannedMw?: number | null;
133 + aiCount?: number;
134 + cloudRegionCount?: number;
135 + mwCoverage?: number | null;
136 + isCloudProvider?: boolean;
137 + isCarrier?: boolean;
138 +}
139 +
140 +export interface PipelineBreakdown {
141 + /** facility counts + known MW by lifecycle bucket (containment-aware, published figures only) */
142 + operational: { count: number; mw: number | null };
143 + construction: { count: number; mw: number | null };
144 + approved: { count: number; mw: number | null };
145 + announced: { count: number; mw: number | null };
146 + /** project records (not facilities) by stage */
147 + projects: Array<{ status: FacilityStatus; count: number; mw: number | null }>;
148 + mwCoverage: number;
149 +}
150 +
151 +export interface ExpansionVelocity {
152 + windows: Array<{ label: "12m" | "3y" | "5y"; newFacilities: number; newProjects: number; newCountries: number; newMetros: number; openedMw: number | null; announcedMw: number | null }>;
153 + /** cumulative countries / metros entered by year (from opened_on / announced_on / first_seen — labelled per point) */
154 + countriesOverTime: Array<{ year: number; countries: number; metros: number; facilities: number; basis: "opened" | "first_seen" }>;
51 155 }
52 156
53 157 export interface OperatorDetail extends OperatorSummary {
158 + pipeline: PipelineBreakdown;
159 + velocity: ExpansionVelocity;
160 + topCountries: Array<{ iso2: string; name: string; slug: string; facilityCount: number; knownMw: number | null; share: number }>;
161 + topMetros: Array<{ id: string; slug: string; name: string; countryIso2: string; facilityCount: number; knownMw: number | null; share: number }>;
162 + aiFacilities: FacilitySummary[];
163 + cloudRegions: CloudRegionSummary[];
164 + /** acquisitions, financing, partnerships, executive changes — never mixed with physical projects */
165 + corporateEvents: EventDTO[];
166 + claims: ClaimDTO[];
167 + dataQuality: DataQualitySummary;
54 168 aliases: string[];
55 169 description?: string | null;
56 170 parent?: { id: string; slug: string; name: string } | null;
@@ -93,9 +207,43 @@ export interface FacilitySummary {
93 207 carriersCount?: number | null;
94 208 ixpCount?: number | null;
95 209 networksCount?: number | null;
210 + /** building | facility | campus (containment) */
211 + recordScope?: "building" | "facility" | "campus";
212 + parentFacility?: { id: string; slug: string; name: string } | null;
213 + aiEvidence?: AiEvidence;
214 + /** what the displayed MW figure means and describes */
215 + capacityScope?: ClaimScope | null;
216 + capacitySemantics?: string | null;
217 + utilityCapacityMw?: number | null;
218 + gridConnectionMw?: number | null;
219 + ultimateCampusMw?: number | null;
220 + sourceCount?: number;
221 +}
222 +
223 +export interface DataQualitySummary {
224 + sourceCount: number;
225 + primarySourceCount: number;
226 + lastVerified: string | null;
227 + completeness: number;
228 + fieldsWithProvenance: number;
229 + claimsTotal: number;
230 + claimsCurrent: number;
231 + claimsUnscoped: number;
232 + claimsInReview: number;
233 + openFlags: Array<{ code: string; severity: "info" | "warn" | "critical"; message: string; field: string | null }>;
234 + pendingDuplicate: boolean;
96 235 }
97 236
98 237 export interface FacilityDetail extends FacilitySummary {
238 + buildings: FacilitySummary[];
239 + developer: { id: string; slug: string; name: string } | null;
240 + landowner: { id: string; slug: string; name: string } | null;
241 + tenants: Array<{ id: string; slug: string; name: string; role: string }>;
242 + claims: ClaimDTO[];
243 + capacityHistory: HistoryPoint[];
244 + nearbyInfrastructure: NearbyInfrastructure | null;
245 + powerContext: { utilityCapacityMw: number | null; gridConnectionMw: number | null; powerEvents: EventDTO[]; gridConstraints: GridConstraintDTO[]; countryEnergy: { renewableShare: number | null; electricityTwh: number | null; statsYear: number | null } | null; note: string };
246 + dataQuality: DataQualitySummary;
99 247 aliases: string[];
100 248 owner: { id: string; slug: string; name: string } | null;
101 249 campus: { id: string; slug: string; name: string } | null;
@@ -177,13 +325,50 @@ export interface ProjectSummary {
177 325 phaseCount: number | null;
178 326 confidence: ConfidenceLevel;
179 327 lastUpdate: string;
328 + geoPrecision?: GeoPrecision;
329 + projectClass?: ProjectClass | null;
330 + evidenceLevel?: "strong" | "weak" | "none" | null;
331 + isAi?: boolean;
332 + aiEvidence?: AiEvidence;
333 + capacityScope?: ClaimScope | null;
334 + capacitySemantics?: string | null;
335 + investmentScope?: ClaimScope | null;
336 + investmentSemantics?: string | null;
337 + investmentCurrency?: string | null;
338 + investmentOriginal?: number | null;
339 + developer?: { id: string; slug: string; name: string } | null;
340 + tenant?: { id: string; slug: string; name: string } | null;
341 + constructionStartedOn?: string | null;
342 + approvedOn?: string | null;
343 + permitFiledOn?: string | null;
344 + openedOn?: string | null;
345 +}
346 +
347 +/** Lifecycle stage view: one row per stage, dated when known, with the evidence that moved the project there. */
348 +export interface ProjectStageDTO {
349 + stage: "rumored" | "proposed" | "announced" | "permitting" | "approved" | "under_construction" | "partially_operational" | "operational" | "delayed" | "cancelled";
350 + date: string | null;
351 + reached: boolean;
352 + current: boolean;
353 + url: string | null;
354 + sourceName: string | null;
355 + eventId: string | null;
180 356 }
181 357
182 358 export interface ProjectDetail extends ProjectSummary {
183 359 description: string | null;
184 360 sourceUrl: string | null;
361 + campus: { id: string; slug: string; name: string } | null;
362 + stages: ProjectStageDTO[];
363 + /** days between reached stages (announcement → permitting → approval → construction → opening), null when unknown */
364 + velocityDays: { announcedToPermitting: number | null; permittingToApproval: number | null; approvalToConstruction: number | null; constructionToOpening: number | null; announcedToConstruction: number | null };
185 365 timeline: Array<{ date: string; type: string; description: string; url: string | null; sourceName: string | null }>;
186 366 statusHistory: Array<{ date: string; from: FacilityStatus | null; to: FacilityStatus; url: string | null }>;
367 + claims: ClaimDTO[];
368 + history: HistoryPoint[];
369 + nearbyInfrastructure: NearbyInfrastructure | null;
370 + relatedEvents: EventDTO[];
371 + dataQuality: DataQualitySummary;
187 372 provenance: ProvenanceDTO[];
188 373 events: EventDTO[];
189 374 sourceHistory: SourceHistoryItem[];
@@ -210,6 +395,52 @@ export interface EventDTO {
210 395 reviewStatus: ReviewStatus;
211 396 countryIso2: string | null;
212 397 operator: { id: string; slug: string; name: string } | null;
398 + metro?: { id: string; slug: string; name: string } | null;
399 + project?: { id: string; slug: string; name: string } | null;
400 + isAi?: boolean;
401 + significanceBand?: Significance;
402 + /** documents describing the same announcement (deduplicated feed): count + the other sources */
403 + evidenceCount?: number;
404 + clusterId?: string | null;
405 + otherSources?: Array<{ sourceName: string; sourceKind: SourceKind; url: string }>;
406 + documentId?: string | null;
407 +}
408 +
409 +export interface EventFilters {
410 + type?: string; // comma list of EventType
411 + country?: string;
412 + metro?: string;
413 + operator?: string;
414 + project?: string;
415 + entity_type?: string;
416 + entity_id?: string;
417 + min_significance?: number;
418 + significance?: "major" | "medium" | "minor";
419 + confidence?: string;
420 + source_kind?: string;
421 + ai?: boolean;
422 + since?: string;
423 + until?: string;
424 + q?: string;
425 + /** collapse events sharing a cluster id into one row (default true) */
426 + dedupe?: boolean;
427 + page?: number;
428 + per_page?: number;
429 +}
430 +
431 +export interface GridConstraintDTO {
432 + id: string;
433 + kind: "moratorium" | "grid_delay" | "capacity_restriction" | "load_cap" | "new_transmission" | "new_substation" | "regulation" | "large_load_queue" | string;
434 + title: string;
435 + summary: string | null;
436 + effectiveDate: string | null;
437 + url: string;
438 + sourceName: string | null;
439 + sourceKind: SourceKind | null;
440 + metro: { id: string; slug: string; name: string } | null;
441 + countryIso2: string | null;
442 + confidence: ConfidenceLevel;
443 + eventId: string | null;
213 444 }
214 445
215 446 export interface CountrySummary {
@@ -233,12 +464,34 @@ export interface CountrySummary {
233 464 hyperscaleCount: number;
234 465 aiCount: number;
235 466 projectCount: number;
467 + projectPlannedMw?: number | null;
468 + projectConstructionMw?: number | null;
469 + ixpCount?: number;
236 470 mwCoverage: number; // share of facilities with a known MW figure, 0..1
237 471 lat: number | null;
238 472 lng: number | null;
239 473 }
240 474
475 +export interface EnergyContext {
476 + renewableShare: number | null;
477 + electricityTwh: number | null;
478 + statsYear: number | null;
479 + /** grams CO2/kWh where a reliable public source exists, else null */
480 + gridCarbonIntensity: number | null;
481 + sourceName: string | null;
482 + sourceUrl: string | null;
483 + note: string;
484 +}
485 +
241 486 export interface CountryDetail extends CountrySummary {
487 + ixps: IxpSummary[];
488 + gridConstraints: GridConstraintDTO[];
489 + energy: EnergyContext;
490 + aiFacilities: FacilitySummary[];
491 + aiProjects: ProjectSummary[];
492 + pipeline: PipelineBreakdown;
493 + coverage: CoverageRow;
494 + claims: ClaimDTO[];
242 495 topOperators: OperatorSummary[];
243 496 metros: MetroSummary[];
244 497 cloudRegions: CloudRegionSummary[];
@@ -274,9 +527,42 @@ export interface MetroSummary {
274 527 cloudRegionCount: number;
275 528 ixpCount: number;
276 529 projectCount: number;
530 + projectPlannedMw?: number | null;
531 + aiCount?: number;
532 + mwCoverage?: number | null;
533 +}
534 +
535 +/** Operator concentration, computed on facility counts and (separately, when coverage permits) on known MW. */
536 +export interface MarketConcentration {
537 + operatorCount: number;
538 + facilities: { hhi: number; top3Share: number; top5Share: number; top: Array<{ id: string; slug: string; name: string; count: number; share: number }> };
539 + knownMw: { hhi: number; top3Share: number; top5Share: number; coverage: number; top: Array<{ id: string; slug: string; name: string; mw: number; share: number }> } | null;
540 + note: string;
541 +}
542 +
543 +/** Transparent momentum components — shown separately, never collapsed into a score. */
544 +export interface MarketMomentum {
545 + window: "12m";
546 + projectsAnnounced: number;
547 + projectsEnteredConstruction: number;
548 + constructionMw: number | null;
549 + announcedMw: number | null;
550 + newEntrants: Array<{ id: string; slug: string; name: string }>;
551 + facilitiesOpened: number;
552 + cloudRegionsAdded: number;
553 + gridEvents: number;
554 + eventsTotal: number;
277 555 }
278 556
279 557 export interface MetroDetail extends MetroSummary {
558 + concentration: MarketConcentration;
559 + momentum: MarketMomentum;
560 + pipeline: PipelineBreakdown;
561 + gridConstraints: GridConstraintDTO[];
562 + aiFacilities: FacilitySummary[];
563 + openingTimeline: Array<{ year: number; opened: number; openedMw: number | null }>;
564 + coverage: CoverageRow;
565 + claims: ClaimDTO[];
280 566 aliases: string[];
281 567 description: string | null;
282 568 operators: OperatorSummary[];
@@ -301,8 +587,38 @@ export interface IxpSummary {
301 587 website: string | null;
302 588 networkCount: number | null;
303 589 facilityCount: number;
590 + metro?: { id: string; slug: string; name: string } | null;
591 + lat?: number | null;
592 + lng?: number | null;
593 +}
594 +
595 +export interface IxpDetail extends IxpSummary {
596 + facilities: FacilitySummary[];
597 + operators: Array<{ id: string; slug: string; name: string; facilityCount: number }>;
598 + nearbyFacilities: Array<FacilitySummary & Distance>;
599 + trafficNote: string | null;
600 + externalIds: Record<string, string | number>;
601 + sourceHistory: SourceHistoryItem[];
602 +}
603 +
604 +export interface CloudRegionDetail extends CloudRegionSummary {
605 + metro: { id: string; slug: string; name: string } | null;
606 + /** facilities publicly verified to host the region — usually empty; otherwise "region associated with market" */
607 + hostFacilities: FacilitySummary[];
608 + marketFacilities: FacilitySummary[];
609 + siblingRegions: CloudRegionSummary[];
610 + announcedOn: string | null;
611 + events: EventDTO[];
612 + provenance: ProvenanceDTO[];
613 + externalIds: Record<string, string | number>;
614 + note: string;
304 615 }
305 616
617 +export const MAP_LAYERS = ["facilities", "capacity", "projects", "ai", "cloud", "ixps", "connectivity", "power", "pipeline"] as const;
618 +export type MapLayer = (typeof MAP_LAYERS)[number];
619 +export const DENSITY_VIEWS = ["facilities", "known_mw", "pipeline_mw", "ai", "operators", "cloud", "ixps"] as const;
620 +export type DensityView = (typeof DENSITY_VIEWS)[number];
621 +
306 622 export interface MapPoint {
307 623 id: string;
308 624 slug: string;
@@ -317,6 +633,12 @@ export interface MapPoint {
317 633 ai?: 1;
318 634 hs?: 1;
319 635 c?: string | null; // country iso2
636 + /** point kind for multi-layer maps (default facility) */
637 + k?: "facility" | "project" | "ixp" | "cloud_region" | "landing_station" | "substation" | "power_plant";
638 + /** opening year (time machine) */
639 + y?: number | null;
640 + /** record scope: campus rows carry their buildings */
641 + rs?: "building" | "facility" | "campus";
320 642 }
321 643
322 644 export interface MapCluster {
@@ -329,16 +651,40 @@ export interface MapCluster {
329 651 label?: string | null; // metro/country label at low zoom
330 652 }
331 653
654 +export interface DensityCell { lat: number; lng: number; w: number; n: number }
655 +
332 656 export interface MapResponse {
333 657 zoom: number;
334 − mode: "clusters" | "points";
658 + mode: "clusters" | "points" | "density";
659 + layer?: MapLayer;
335 660 clusters?: MapCluster[];
336 661 points?: MapPoint[];
662 + /** overlay points for non-facility layers (projects, IXPs, cloud regions…) */
663 + overlay?: MapPoint[];
664 + density?: { view: DensityView; cells: DensityCell[]; max: number };
337 665 total: number;
666 + /** true when the point count exceeded the cap and clusters were returned instead */
667 + degraded?: boolean;
668 + /** time machine: the year filter applied (facilities known open by that year) and coverage of opening dates */
669 + year?: number | null;
670 + yearCoverage?: number | null;
671 +}
672 +
673 +export interface MapFilters extends FacilityFilters {
674 + layer?: MapLayer;
675 + density?: DensityView;
676 + year?: number;
677 + project_status?: string;
678 + expected_from?: number;
679 + expected_to?: number;
680 + location_precision?: string;
681 + ai_evidence?: string;
682 + zoom?: number;
683 + bbox?: string;
338 684 }
339 685
340 686 export interface SearchHit {
341 − type: "facility" | "operator" | "country" | "metro" | "city" | "cloud_region" | "project" | "ixp";
687 + type: "facility" | "operator" | "country" | "metro" | "city" | "cloud_region" | "project" | "ixp" | "action";
342 688 id: string;
343 689 slug: string;
344 690 title: string;
@@ -350,7 +696,7 @@ export interface SearchHit {
350 696
351 697 export interface SearchResponse {
352 698 query: string;
353 − interpreted?: { countryIso2?: string; operator?: string; status?: FacilityStatus; minMw?: number; facilityType?: FacilityType; text?: string };
699 + interpreted?: { countryIso2?: string; operator?: string; status?: FacilityStatus; minMw?: number; maxMw?: number; facilityType?: FacilityType; text?: string; ai?: boolean; entity?: "facility" | "project"; year?: number; yearOp?: "before" | "after" | "in"; metro?: string };
354 700 hits: SearchHit[];
355 701 facilities?: { total: number; items: FacilitySummary[] };
356 702 }
@@ -416,6 +762,217 @@ export interface Dashboard {
416 762 latestEvents: EventDTO[];
417 763 newProjects: ProjectSummary[];
418 764 recentlyVerified: FacilitySummary[];
765 + /** what changed today / this week — deterministic counts (see Pulse) */
766 + pulse: Pulse;
767 + fastestGrowingMetros: Array<{ id: string; slug: string; name: string; countryIso2: string; momentum: MarketMomentum }>;
768 + majorProjects: ProjectSummary[];
769 + operatorExpansion: Array<{ operator: { id: string; slug: string; name: string }; newCountries: string[]; newMetros: string[]; projects12m: number }>;
770 + powerEvents: EventDTO[];
771 + gridConstraints: GridConstraintDTO[];
772 + coverage: CoverageReport["global"];
773 + ingestion: { lastRunAt: string | null; runs24h: number; documents24h: number; events24h: number; connectorsHealthy: number; connectorsTotal: number };
774 +}
775 +
776 +/** Global infrastructure pulse — measurable changes over a window, computed deterministically from events / entities. */
777 +export interface Pulse {
778 + window: "24h" | "7d" | "30d";
779 + since: string;
780 + newProjects: number;
781 + projectsEnteredConstruction: number;
782 + facilitiesOpened: number;
783 + newlyAnnouncedMw: number | null;
784 + constructionStartedMw: number | null;
785 + operatorsNewMarkets: Array<{ operator: { id: string; slug: string; name: string }; market: { id: string; slug: string; name: string } | null; countryIso2: string | null }>;
786 + cloudRegionsAnnounced: number;
787 + powerAgreements: number;
788 + gridConstraintEvents: number;
789 + acquisitions: number;
790 + financingEvents: number;
791 + newFacilitiesIndexed: number;
792 + capacityChanges: number;
793 + eventsTotal: number;
794 + majorEvents: EventDTO[];
795 + byCountry: Array<{ iso2: string; name: string; slug: string; events: number; newProjects: number }>;
796 + byOperator: Array<{ id: string; slug: string; name: string; events: number; newProjects: number }>;
797 +}
798 +
799 +export interface CoverageRow {
800 + key: string;
801 + name: string;
802 + slug: string;
803 + facilities: number;
804 + /** share of facilities with a known MW figure (containment-aware) */
805 + capacityCoverage: number;
806 + operatorCoverage: number;
807 + preciseLocationCoverage: number; // exact / parcel / street
808 + anyLocationCoverage: number;
809 + statusCoverage: number;
810 + openingDateCoverage: number;
811 + multiSourceCoverage: number;
812 + projects: number;
813 + projectsWithLocation: number;
814 + connectivityCoverage: number; // facilities with carriers / IXPs known
815 + primarySourceShare: number;
816 +}
817 +
818 +export interface CoverageReport {
819 + global: CoverageRow;
820 + countries: CoverageRow[];
821 + metros: CoverageRow[];
822 + operators: CoverageRow[];
823 + fields: Array<{ field: string; label: string; coverage: number; count: number }>;
824 + bySourceKind: Array<{ kind: SourceKind; facilities: number; fields: number }>;
825 + generatedAt: string;
826 +}
827 +
828 +export interface SourceCoverage {
829 + id: string;
830 + name: string;
831 + kind: SourceKind;
832 + recordsContributed: number;
833 + fieldsContributed: number;
834 + uniqueRecords: number;
835 + lastSuccessfulCrawl: string | null;
836 + freshnessDays: number | null;
837 + authority: AuthorityTier;
838 + failureRate: number | null;
839 + license: string | null;
840 + redistribution: string | null;
841 +}
842 +
843 +export interface OperatorComparison {
844 + operators: Array<OperatorSummary & { pipeline: PipelineBreakdown; velocity: ExpansionVelocity; aiCount: number; newMarkets12m: Array<{ id: string; slug: string; name: string }>; recentProjects: ProjectSummary[]; countriesList: string[]; coverage: number }>;
845 + generatedAt: string;
846 +}
847 +
848 +export interface AiIndex {
849 + stats: { facilities: number; confirmed: number; likely: number; associated: number; projects: number; plannedMw: number | null; constructionMw: number | null; countries: number; operators: number; mwCoverage: number };
850 + topOperators: Array<{ id: string; slug: string; name: string; facilities: number; projects: number; plannedMw: number | null }>;
851 + topMetros: Array<{ id: string; slug: string; name: string; countryIso2: string; facilities: number; projects: number; plannedMw: number | null }>;
852 + topCountries: Array<{ iso2: string; slug: string; name: string; facilities: number; projects: number; plannedMw: number | null }>;
853 + pipelineByStage: Array<{ status: FacilityStatus; count: number; mw: number | null }>;
854 + recentAnnouncements: EventDTO[];
855 + recentProjects: ProjectSummary[];
856 + facilities: FacilitySummary[];
857 + evidenceNote: string;
858 +}
859 +
860 +export interface PowerOverview {
861 + gridConstraints: GridConstraintDTO[];
862 + powerEvents: EventDTO[];
863 + largeLoads: Array<{ facility: { id: string; slug: string; name: string } | null; project: { id: string; slug: string; name: string } | null; kind: "utility_capacity" | "grid_connection" | "planned_load"; mw: number; countryIso2: string | null; metro: { id: string; slug: string; name: string } | null; sourceName: string | null; url: string | null }>;
864 + countryEnergy: Array<{ iso2: string; name: string; slug: string; energy: EnergyContext; knownMw: number | null; facilities: number }>;
865 + utilities: Array<{ id: string; slug: string; name: string; facilityCount: number; kind: string | null }>;
866 + note: string;
867 +}
868 +
869 +export interface ConnectivityOverview {
870 + ixps: Array<IxpSummary & { operatorCount: number }>;
871 + cloudRegions: CloudRegionSummary[];
872 + carrierHotels: FacilitySummary[];
873 + landingStations: Array<{ id: string; name: string; lat: number; lng: number; countryIso2: string | null; sourceName: string }>;
874 + byMetro: Array<{ id: string; slug: string; name: string; countryIso2: string; ixps: number; cloudRegions: number; facilitiesWithCarriers: number; facilities: number }>;
875 + note: string;
876 +}
877 +
878 +export interface TimeMachineFrame { year: number; facilities: number; knownMw: number | null; announced: number; construction: number; opened: number }
879 +export interface TimeMachine { frames: TimeMachineFrame[]; earliestReliableYear: number; openingDateCoverage: number; note: string }
880 +
881 +/** /explore structured query. Every key is optional; unknown keys are rejected. */
882 +export interface ExploreQuery {
883 + entity?: "facilities" | "projects";
884 + status?: string;
885 + project_status?: string;
886 + country?: string;
887 + metro?: string;
888 + operator?: string;
889 + type?: string;
890 + min_mw?: number;
891 + max_mw?: number;
892 + ai?: "confirmed" | "likely" | "associated" | "any";
893 + hyperscale?: boolean;
894 + expected_before?: number;
895 + expected_after?: number;
896 + opened_from?: number;
897 + opened_to?: number;
898 + announced_since?: string;
899 + confidence?: string;
900 + location_precision?: string;
901 + has_mw?: boolean;
902 + project_class?: string;
903 + q?: string;
904 + sort?: string;
905 + order?: "asc" | "desc";
906 + page?: number;
907 + per_page?: number;
908 + view?: "table" | "map" | "charts";
909 +}
910 +
911 +export interface ExploreResponse {
912 + query: ExploreQuery;
913 + total: number;
914 + items: Array<FacilitySummary | ProjectSummary>;
915 + facets: { status: Array<{ key: string; count: number }>; country: Array<{ key: string; name: string; count: number }>; operator: Array<{ key: string; name: string; count: number }>; type?: Array<{ key: string; count: number }>; ai: Array<{ key: string; count: number }> };
916 + charts: { byStatus: Array<{ key: string; count: number; mw: number | null }>; byCountry: Array<{ key: string; name: string; count: number; mw: number | null }>; byYear: Array<{ year: number; count: number; mw: number | null }> };
917 + map: { points: MapPoint[]; total: number; degraded: boolean };
918 + mwCoverage: number;
919 +}
920 +
921 +export interface DownloadDataset {
922 + key: "facilities" | "projects" | "operators" | "events" | "cloud-regions" | "ixps" | "countries" | "markets";
923 + label: string;
924 + description: string;
925 + formats: Array<"csv" | "json" | "geojson">;
926 + filters: string[];
927 + rows: number;
928 + license: string;
929 + attribution: string[];
930 + /** sources excluded from redistribution (their rows are omitted or their fields blanked) */
931 + excludedSources: Array<{ id: string; name: string; reason: string }>;
932 +}
933 +
934 +export interface WatchlistItem { id: string; entityType: "operator" | "metro" | "country" | "project" | "facility"; entityId: string; slug: string; name: string; createdAt: string }
935 +
936 +export interface ApiEndpointDoc { method: "GET" | "POST" | "DELETE"; path: string; summary: string; group: string; params: Array<{ name: string; in: "query" | "path"; type: string; description: string; example?: string }>; example: { curl: string; js: string; python: string }; responseSchema: string }
937 +
938 +export interface QualityFlagDTO {
939 + id: string;
940 + entityType: string;
941 + entityId: string;
942 + entity: { slug: string; name: string } | null;
943 + claimId: string | null;
944 + code: string;
945 + severity: "info" | "warn" | "critical";
946 + field: string | null;
947 + message: string;
948 + details: Record<string, unknown> | null;
949 + priority: number;
950 + status: "open" | "resolved" | "dismissed";
951 + resolution: string | null;
952 + resolvedBy: string | null;
953 + resolvedAt: string | null;
954 + runId: string | null;
955 + createdAt: string;
956 + updatedAt: string;
957 +}
958 +
959 +export interface QualityOverview {
960 + openFlags: number;
961 + bySeverity: Record<string, number>;
962 + byCode: Array<{ code: string; count: number; label: string }>;
963 + possibleDuplicates: number;
964 + suspiciousMw: number;
965 + suspiciousInvestment: number;
966 + projectFalsePositiveCandidates: number;
967 + unscopedClaims: number;
968 + claimsInReview: number;
969 + unverifiedLargeProjects: number;
970 + facilitiesMissingCountry: number;
971 + projectsWithoutFacility: number;
972 + orphanOperators: number;
973 + largest: { operationalFacilityMw: Array<{ id: string; slug: string; name: string; mw: number; scope: string | null; semantics: string | null }>; constructionFacilityMw: Array<{ id: string; slug: string; name: string; mw: number }>; projectMw: Array<{ id: string; slug: string; name: string; mw: number; scope: string | null }>; investment: Array<{ id: string; slug: string; name: string; investmentUsd: number; scope: string | null }>; operatorPipelineMw: Array<{ id: string; slug: string; name: string; mw: number }>; metroPipelineMw: Array<{ id: string; slug: string; name: string; mw: number }> };
974 + reviewQueue: QualityFlagDTO[];
975 + alerts: Array<{ id: string; level: string; component: string; message: string; createdAt: string }>;
419 976 }
420 977
421 978 export interface ConnectorHealthDTO {
@@ -425,9 +982,11 @@ export interface ConnectorHealthDTO {
425 982 kind: SourceKind;
426 983 mode: string;
427 984 enabled: boolean;
428 − health: "ok" | "degraded" | "failing" | "paused" | "never_run";
985 + health: "ok" | "degraded" | "failing" | "paused" | "never_run" | "blocked" | "schema_change" | "no_new_content" | "quarantine";
429 986 parserVersion: string;
430 987 lastRunAt: string | null;
988 + lastSuccessAt?: string | null;
989 + lastFailureAt?: string | null;
431 990 nextRunAt: string | null;
432 991 lastStatus: string | null;
433 992 discovered: number;
@@ -439,6 +998,23 @@ export interface ConnectorHealthDTO {
439 998 avgResponseMs: number | null;
440 999 cost: { direct: number; firecrawl: number; scrapfly: number; credits: number };
441 1000 schedule: Record<string, string>;
1001 + quarantine?: boolean;
1002 + consecutiveFailures?: number;
1003 + blockedSince?: string | null;
1004 + urlsDiscovered?: number;
1005 + urlsFetched?: number;
1006 + newDocs?: number;
1007 + changedDocs?: number;
1008 + recordsCreated?: number;
1009 + recordsModified?: number;
1010 + rejectedClaims?: number;
1011 + httpErrors?: number;
1012 + antiBotEscalations?: number;
1013 + scrapflyRequests?: number;
1014 + firecrawlRequests?: number;
1015 + priorityScore?: number | null;
1016 + license?: string | null;
1017 + redistribution?: string | null;
442 1018 }
443 1019
444 1020 export interface FacilityFilters {
added packages/core/src/claims.test.ts +138 −0
@@ -0,0 +1,138 @@
1 +import { describe, expect, it } from "vitest";
2 +import { authorityTier, capacitySanity, classifyAiEvidence, classifyCapacitySemantics, classifyInvestmentSemantics, classifyProjectEvent, classifyScope, findEvidence, investmentSanity, moneyMwCollision, projectTransition, sentenceSpans } from "./claims.js";
3 +
4 +describe("classifyScope", () => {
5 + it("detects portfolio / company / country figures", () => {
6 + expect(classifyScope("bringing its total capacity to over 410MW in Japan").scope).toBe("portfolio");
7 + expect(classifyScope("STACK now operates more than 1.2 GW across its global portfolio").scope).toBe("portfolio");
8 + expect(classifyScope("Microsoft plans to spend $80 billion in capex in fiscal 2025").scope).toBe("company");
9 + expect(classifyScope("the national programme will add 5 GW across the country").scope).toBe("country");
10 + expect(classifyScope("Hyperscale Data Center Market to Surpass USD 624.2 Billion by 2035, CAGR of 9%").scope).toBe("company");
11 + });
12 + it("detects campus / building / facility figures", () => {
13 + expect(classifyScope("the campus will deliver 300 MW at full build-out").scope).toBe("campus");
14 + expect(classifyScope("the first building offers 24 MW of critical IT load").scope).toBe("building");
15 + expect(classifyScope("the facility will provide 36 MW").scope).toBe("facility");
16 + });
17 + it("falls back to the subject hint only without any signal", () => {
18 + expect(classifyScope("36 MW", "facility").scope).toBe("facility");
19 + expect(classifyScope("36 MW").scope).toBe("unknown");
20 + });
21 +});
22 +
23 +describe("classifyCapacitySemantics", () => {
24 + it("separates IT load, grid, utility, planned, ultimate and phase figures", () => {
25 + expect(classifyCapacitySemantics("0.779MW Critical IT Load").predicate).toBe("critical_power_mw");
26 + expect(classifyCapacitySemantics("offers 24 MW of IT capacity").predicate).toBe("it_capacity_mw");
27 + expect(classifyCapacitySemantics("secured a 300 MW grid connection agreement with Dominion").predicate).toBe("grid_connection_mw");
28 + expect(classifyCapacitySemantics("dual utility feeds totaling 60 MW of utility power").predicate).toBe("utility_capacity_mw");
29 + expect(classifyCapacitySemantics("the campus will scale up to 1 GW at full build-out").predicate).toBe("ultimate_campus_mw");
30 + expect(classifyCapacitySemantics("the first phase will deliver 48 MW").predicate).toBe("phase_mw");
31 + expect(classifyCapacitySemantics("the planned 200 MW facility").predicate).toBe("planned_power_mw");
32 + expect(classifyCapacitySemantics("currently operating 12 MW").predicate).toBe("current_power_mw");
33 + expect(classifyCapacitySemantics("36 MW").predicate).toBeNull();
34 + });
35 +});
36 +
37 +describe("classifyInvestmentSemantics", () => {
38 + it("keeps deal values and company capex away from project investment", () => {
39 + expect(classifyInvestmentSemantics("Microsoft will invest $80 billion through 2030 across its global fleet").predicate).toBe("multi_year_capex_usd");
40 + expect(classifyInvestmentSemantics("acquires the portfolio for $2.3 billion").predicate).toBe("deal_value_usd");
41 + expect(classifyInvestmentSemantics("the $1.2 billion campus in Abilene").predicate).toBe("campus_investment_usd");
42 + expect(classifyInvestmentSemantics("a $500 million investment in the new facility", "facility").predicate).toBe("project_investment_usd");
43 + });
44 +});
45 +
46 +describe("sanity engines", () => {
47 + it("blocks unscoped and market-statistic MW", () => {
48 + const flags = capacitySanity({ value: 12_000, predicate: null, scope: "company", recordScope: "project" });
49 + expect(flags.some((f) => f.code === "scope_company" && f.blocks)).toBe(true);
50 + expect(capacitySanity({ value: 25_000, predicate: "planned_power_mw", scope: "campus", recordScope: "project" }).some((f) => f.code === "mw_market_statistic")).toBe(true);
51 + });
52 + it("flags single facilities above 1 GW, 5x changes and money collisions", () => {
53 + expect(capacitySanity({ value: 1_500, predicate: "it_capacity_mw", scope: "facility", recordScope: "facility" }).map((f) => f.code)).toContain("mw_single_site_gt_1000");
54 + expect(capacitySanity({ value: 1_500, predicate: "ultimate_campus_mw", scope: "campus", recordScope: "campus", campusDesignation: true }).map((f) => f.code)).not.toContain("mw_single_site_gt_1000");
55 + expect(capacitySanity({ value: 779, predicate: "it_capacity_mw", scope: "facility", recordScope: "facility", previous: 0.779 }).map((f) => f.code)).toContain("mw_change_5x");
56 + expect(moneyMwCollision("a $300 million, 300 MW campus")).toBe(true);
57 + expect(moneyMwCollision("a $300 million, 48 MW campus")).toBe(false);
58 + });
59 + it("flags implausible investments", () => {
60 + expect(investmentSanity({ value: 175e9, scope: "facility", predicate: "project_investment_usd", recordScope: "project" }).map((f) => f.code)).toContain("inv_single_site_gt_50b");
61 + expect(investmentSanity({ value: 624e9, scope: "company", predicate: "multi_year_capex_usd", recordScope: "project" }).some((f) => f.blocks)).toBe(true);
62 + });
63 +});
64 +
65 +describe("findEvidence", () => {
66 + const text = "STACK Infrastructure today announced the appointment of Matt VanderZanden as CEO. The company operates more than 13 GW across its portfolio. Its newest campus in Lancaster will deliver 500 MW at full build-out.";
67 + it("returns the sentence quoting the figure with offsets", () => {
68 + const ev = findEvidence(text, 500, "mw");
69 + expect(ev?.text).toMatch(/Lancaster will deliver 500 MW/);
70 + expect(text.slice(ev!.start, ev!.end)).toBe(ev!.text);
71 + expect(findEvidence(text, 13_000, "mw")?.text).toMatch(/13 GW/);
72 + expect(findEvidence(text, 42, "mw")).toBeNull();
73 + });
74 + it("finds money in billions and millions", () => {
75 + expect(findEvidence("Meta will spend $10 billion on the Richland Parish site.", 10e9, "usd")?.text).toMatch(/\$10 billion/);
76 + expect(findEvidence("A US$450 million facility.", 450e6, "usd")).not.toBeNull();
77 + });
78 + it("splits sentences with offsets", () => {
79 + const spans = sentenceSpans("One. Two two. Three?");
80 + expect(spans.map((s) => s.text)).toEqual(["One.", "Two two.", "Three?"]);
81 + });
82 +});
83 +
84 +describe("classifyProjectEvent", () => {
85 + it("vetoes appointments, market research, PPAs, HQs and financing", () => {
86 + expect(classifyProjectEvent({ title: "STACK Infrastructure Appoints Matt VanderZanden as Chief Executive Officer, STACK Americas" }).class).toBe("EXECUTIVE_APPOINTMENT");
87 + expect(classifyProjectEvent({ title: "Hyperscale Data Center Market to Surpass USD 624.2 Billion by 2035" }).class).toBe("GENERAL_COMPANY_NEWS");
88 + expect(classifyProjectEvent({ title: "Ormat Technologies Signs 20-Year PPA with Switch for ~13 MW of Carbon-Free Geothermal Capacity to Power Data Centers" }).class).toBe("POWER_AGREEMENT");
89 + expect(classifyProjectEvent({ title: "STACK Secures $1.3B Financing for Development" }).class).toBe("FINANCING");
90 + expect(classifyProjectEvent({ title: "AirTrunk unveils new global headquarters in Sydney" }).mayCreateProject).toBe(false);
91 + expect(classifyProjectEvent({ title: "Element Critical Enters Chicago Market Acquiring Two Data Centers in the Suburbs" }).class).toBe("ACQUISITION");
92 + expect(classifyProjectEvent({ title: "Deepening our investment in Richland Parish, Louisiana" }).mayCreateProject).toBe(false);
93 + });
94 + it("recognises physical development with strong evidence", () => {
95 + const c = classifyProjectEvent({ title: "Vantage breaks ground on 192 MW campus in Frederick, Maryland", hasOperator: true, hasLocation: true, hasExplicitName: false });
96 + expect(c.class).toBe("CONSTRUCTION_START");
97 + expect(c.mayCreateProject).toBe(true);
98 + expect(classifyProjectEvent({ title: "Microsoft to build $3.3 billion data center campus in Mount Pleasant, Wisconsin", hasOperator: true, hasLocation: true }).class).toBe("NEW_BUILD");
99 + expect(classifyProjectEvent({ title: "Loudoun County approves rezoning for 300 MW data center campus", hasLocation: true, hasExplicitName: true }).class).toBe("PERMIT");
100 + expect(classifyProjectEvent({ title: "Compass acquires 400-acre site in Red Oak for data center campus", hasOperator: true, hasLocation: true }).class).toBe("LAND_ACQUISITION");
101 + expect(classifyProjectEvent({ title: "STACK Grows Northern Virginia Data Center Campus", hasOperator: true, hasLocation: true }).class).toBe("EXPANSION");
102 + });
103 + it("needs name-or-operator + location + verb to create a project", () => {
104 + const weak = classifyProjectEvent({ title: "Massive data center investment announced", hasOperator: false, hasLocation: false });
105 + expect(weak.mayCreateProject).toBe(false);
106 + const build = classifyProjectEvent({ title: "Data center campus planned", hasOperator: false, hasLocation: true, hasExplicitName: false });
107 + expect(build.evidence.strength).toBe("weak");
108 + });
109 + it("lets a build headline override a financing keyword", () => {
110 + const c = classifyProjectEvent({ title: "Aligned secures $2B financing to build 400 MW campus in Phoenix", hasOperator: true, hasLocation: true });
111 + expect(c.physical).toBe(true);
112 + });
113 +});
114 +
115 +describe("authority and transitions", () => {
116 + it("uses field-level authority tiers", () => {
117 + expect(authorityTier({ field: "capacity", sourceKind: "utility" })).toBe("A");
118 + expect(authorityTier({ field: "capacity", sourceKind: "community" })).toBe("E");
119 + expect(authorityTier({ field: "geometry", sourceKind: "community" })).toBe("A");
120 + expect(authorityTier({ field: "interconnection", sourceKind: "registry" })).toBe("A");
121 + expect(authorityTier({ field: "capacity", sourceKind: "operator", isEstimate: true })).toBe("C");
122 + expect(authorityTier({ field: "capacity", sourceKind: "news", method: "llm:extract" })).toBe("E");
123 + });
124 + it("validates project lifecycle transitions", () => {
125 + expect(projectTransition("announced", "permitting")).toBe("forward");
126 + expect(projectTransition("under_construction", "delayed")).toBe("side");
127 + expect(projectTransition("delayed", "under_construction")).toBe("resume");
128 + expect(projectTransition("operational", "announced")).toBe("backward");
129 + expect(projectTransition("operational", "cancelled")).toBe("invalid");
130 + expect(projectTransition(null, "announced")).toBe("forward");
131 + });
132 + it("grades AI evidence", () => {
133 + expect(classifyAiEvidence("a purpose-built AI campus with NVIDIA GB200 systems").level).toBe("confirmed");
134 + expect(classifyAiEvidence("AI-ready, liquid-cooled halls at 130 kW per rack").level).toBe("likely");
135 + expect(classifyAiEvidence("leased to CoreWeave").level).toBe("associated");
136 + expect(classifyAiEvidence("a colocation facility").level).toBe("unknown");
137 + });
138 +});
added packages/core/src/claims.ts +459 −0
@@ -0,0 +1,459 @@
1 +/**
2 + * Claim-first data architecture — shared vocabulary and deterministic classifiers.
3 + *
4 + * A CLAIM is one figure or value asserted by one document about one subject, with a SCOPE (what the number describes:
5 + * a building, a facility, a campus, a metro, a country, a company portfolio…), the supporting sentence, the parser that
6 + * produced it and the field-level authority of its source. Reconciliation (apps/worker/src/ingest/claims.ts) decides
7 + * which claim becomes the displayed column value; nothing here touches the database.
8 + *
9 + * Rules of the house:
10 + * - a figure whose scope is not the subject's own scope (portfolio / company / country / metro / unknown) is stored
11 + * as a claim but NEVER written to the subject's capacity or investment column ("unscoped");
12 + * - capacity semantics are kept apart: IT load ≠ utility power ≠ grid connection ≠ planned ≠ ultimate build-out;
13 + * - every numerical claim needs a supporting sentence — no sentence, no claim.
14 + */
15 +import type { SourceKind } from "./types.js";
16 +
17 +// ─── scopes ────────────────────────────────────────────────────────────────────────────────────────
18 +export const CLAIM_SCOPES = ["building", "facility", "campus", "metro", "country", "portfolio", "company", "unknown"] as const;
19 +export type ClaimScope = (typeof CLAIM_SCOPES)[number];
20 +/** Scopes that may populate a facility / campus / project record's own figures. */
21 +export const SITE_SCOPES: ReadonlySet<ClaimScope> = new Set<ClaimScope>(["building", "facility", "campus"]);
22 +export function isSiteScope(s: ClaimScope | string | null | undefined): boolean { return !!s && SITE_SCOPES.has(s as ClaimScope); }
23 +
24 +// ─── predicates ────────────────────────────────────────────────────────────────────────────────────
25 +export const CAPACITY_PREDICATES = ["it_capacity_mw", "critical_power_mw", "utility_capacity_mw", "grid_connection_mw", "current_power_mw", "planned_power_mw", "ultimate_campus_mw", "phase_mw"] as const;
26 +export type CapacityPredicate = (typeof CAPACITY_PREDICATES)[number];
27 +export const INVESTMENT_PREDICATES = ["project_investment_usd", "campus_investment_usd", "company_investment_usd", "country_program_usd", "multi_year_capex_usd", "deal_value_usd"] as const;
28 +export type InvestmentPredicate = (typeof INVESTMENT_PREDICATES)[number];
29 +export const CLAIM_STATUSES = ["current", "superseded", "rejected", "review", "unscoped"] as const;
30 +export type ClaimStatus = (typeof CLAIM_STATUSES)[number];
31 +
32 +export const CAPACITY_PREDICATE_LABEL: Record<CapacityPredicate, string> = {
33 + it_capacity_mw: "IT capacity",
34 + critical_power_mw: "Critical power",
35 + utility_capacity_mw: "Utility capacity",
36 + grid_connection_mw: "Grid connection",
37 + current_power_mw: "Current power",
38 + planned_power_mw: "Planned capacity",
39 + ultimate_campus_mw: "Ultimate campus build-out",
40 + phase_mw: "Phase capacity",
41 +};
42 +export const INVESTMENT_PREDICATE_LABEL: Record<InvestmentPredicate, string> = {
43 + project_investment_usd: "Project investment",
44 + campus_investment_usd: "Campus investment",
45 + company_investment_usd: "Company investment",
46 + country_program_usd: "Country program",
47 + multi_year_capex_usd: "Multi-year capex",
48 + deal_value_usd: "Deal value",
49 +};
50 +
51 +/** Which facility column a capacity predicate feeds (null = stored as a claim only). */
52 +export const CAPACITY_COLUMN: Record<CapacityPredicate, "itCapacityMw" | "totalPowerMw" | "plannedPowerMw" | "utilityCapacityMw" | "gridConnectionMw" | "ultimateCampusMw" | null> = {
53 + it_capacity_mw: "itCapacityMw",
54 + critical_power_mw: "itCapacityMw",
55 + utility_capacity_mw: "utilityCapacityMw",
56 + grid_connection_mw: "gridConnectionMw",
57 + current_power_mw: "totalPowerMw",
58 + planned_power_mw: "plannedPowerMw",
59 + ultimate_campus_mw: "ultimateCampusMw",
60 + phase_mw: null,
61 +};
62 +
63 +// ─── field-level source authority ──────────────────────────────────────────────────────────────────
64 +export type AuthorityTier = "A" | "B" | "C" | "D" | "E";
65 +export const TIER_RANK: Record<AuthorityTier, number> = { A: 5, B: 4, C: 3, D: 2, E: 1 };
66 +export const AUTHORITY_FIELDS = ["capacity", "investment", "geometry", "interconnection", "status", "ownership", "dates", "identity", "planning"] as const;
67 +export type AuthorityField = (typeof AUTHORITY_FIELDS)[number];
68 +
69 +/**
70 + * Field-level authority hierarchy. One universal rank is wrong: OpenStreetMap is excellent for geometry and useless for
71 + * capacity; PeeringDB (registry) is the reference for interconnection; planning filings beat press releases for approved
72 + * capacity; utility filings beat everything for grid requirements.
73 + */
74 +export const FIELD_AUTHORITY: Record<AuthorityField, Record<SourceKind, AuthorityTier>> = {
75 + capacity: { utility: "A", government: "A", filing: "A", operator: "B", cloud_provider: "B", registry: "C", secondary: "C", news: "D", dataset: "E", community: "E" },
76 + investment: { filing: "A", government: "A", operator: "B", cloud_provider: "B", utility: "C", secondary: "C", news: "D", registry: "E", dataset: "E", community: "E" },
77 + geometry: { government: "A", community: "A", registry: "B", operator: "B", dataset: "B", cloud_provider: "C", utility: "C", filing: "C", secondary: "D", news: "E" },
78 + interconnection: { registry: "A", operator: "B", community: "C", dataset: "C", secondary: "D", news: "D", government: "D", filing: "D", utility: "E", cloud_provider: "E" },
79 + status: { operator: "A", government: "A", filing: "A", cloud_provider: "A", utility: "B", registry: "B", secondary: "C", news: "C", dataset: "D", community: "D" },
80 + ownership: { filing: "A", operator: "A", government: "B", registry: "B", cloud_provider: "B", secondary: "C", news: "C", dataset: "D", community: "D", utility: "D" },
81 + dates: { operator: "A", government: "A", filing: "A", cloud_provider: "A", secondary: "B", news: "B", registry: "C", dataset: "C", utility: "C", community: "D" },
82 + identity: { operator: "A", registry: "A", government: "B", filing: "B", cloud_provider: "B", dataset: "C", community: "C", secondary: "D", news: "D", utility: "D" },
83 + planning: { government: "A", filing: "A", utility: "B", operator: "B", secondary: "C", news: "C", cloud_provider: "C", registry: "D", dataset: "E", community: "E" },
84 +};
85 +
86 +/** Field a predicate belongs to (for the authority lookup). */
87 +export function authorityFieldFor(predicate: string): AuthorityField {
88 + if ((CAPACITY_PREDICATES as readonly string[]).includes(predicate) || /mw$/i.test(predicate)) return "capacity";
89 + if ((INVESTMENT_PREDICATES as readonly string[]).includes(predicate) || /usd$|investment/i.test(predicate)) return "investment";
90 + if (/^(geo|lat|lng|address|postal)/i.test(predicate)) return "geometry";
91 + if (/^(carriers|ixps|cloudProviders|networks)/i.test(predicate)) return "interconnection";
92 + if (/status/i.test(predicate)) return "status";
93 + if (/(operator|owner|tenant|developer)/i.test(predicate)) return "ownership";
94 + if (/(On|Opening|date)$/i.test(predicate) || /^(opened|announced|construction|expected)/i.test(predicate)) return "dates";
95 + if (/permit|planning|approv/i.test(predicate)) return "planning";
96 + return "identity";
97 +}
98 +
99 +export interface AuthorityInput { field: AuthorityField; sourceKind: SourceKind | string | null | undefined; isEstimate?: boolean | null; /** extraction method: structured / parser / regex on prose / llm */ method?: string | null; reviewed?: boolean }
100 +
101 +/** Authority tier of one observation for one field. Estimates and LLM-only extractions drop a tier; a human review raises one. */
102 +export function authorityTier(i: AuthorityInput): AuthorityTier {
103 + const table = FIELD_AUTHORITY[i.field];
104 + let rank = TIER_RANK[(i.sourceKind && table[i.sourceKind as SourceKind]) || "D"];
105 + if (i.isEstimate) rank -= 1;
106 + if (i.method && /^llm/i.test(i.method)) rank -= 1;
107 + if (i.reviewed) rank += 1;
108 + rank = Math.max(1, Math.min(5, rank));
109 + return (Object.keys(TIER_RANK) as AuthorityTier[]).find((t) => TIER_RANK[t] === rank) ?? "E";
110 +}
111 +
112 +// ─── scope classification ──────────────────────────────────────────────────────────────────────────
113 +const PORTFOLIO_RE = /\b(across (?:its|our|their|the company'?s?|a) (?:global |entire |whole |growing |existing )?(?:portfolio|footprint|platform|network|fleet|estate)|portfolio|footprint|under management|under development globally|(?:global|total|combined|aggregate|cumulative) (?:capacity|footprint|pipeline|portfolio)|pipeline of|development pipeline|(?:brings?|bringing|takes?|taking|lifts?|lifting|grows?|growing) (?:its|the company'?s?|our|their|total) [^.]{0,60}(?:capacity |footprint |portfolio )?to (?:over |more than |nearly |approximately |about |roughly )?\d|now (?:operates|owns|manages|has) (?:over |more than |nearly )?\d|operates (?:over |more than |nearly )?\d+[\d.,]* ?(?:mw|gw|megawatts?|gigawatts?) (?:across|in|of)|(?:more than|over) \d+ (?:data cent(?:er|re)s|facilities|sites|campuses|locations)|globally|worldwide|company-?wide|enterprise-?wide|group-?wide)\b/i;
114 +const COMPANY_RE = /\b(capex|capital expenditure|will invest [^.]{0,40}(?:over the next|through|by) (?:the next )?(?:\d+ years|20\d\d)|(?:through|by|over the next|over the coming) (?:20\d\d|\d+ years|the decade)|(?:total|planned|annual) investment (?:of|in) [^.]{0,30}(?:across|in) (?:the (?:us|uk|country|region)|[A-Z][a-z]+ and [A-Z][a-z]+)|(?:in|for) (?:its|our|their) (?:fiscal|financial) (?:year|quarter)|(?:fiscal|fy) ?20\d\d|guidance|earnings|revenue|quarterly results|annual report|10-k|market (?:to|will|set to|expected to|projected to) (?:surpass|reach|hit|grow|exceed|top)|market size|cagr|forecast period)\b/i;
115 +const COUNTRY_RE = /\b(nationwide|nation-?wide|national (?:program|programme|plan|strategy|investment|pipeline)|across the country|the country'?s|country-?wide|(?:in|across) (?:the )?(?:united states|u\.s\.|us|uk|united kingdom|europe|emea|apac|apj|asia|asia-pacific|latin america|the americas|the middle east|africa|india|japan|germany|france|australia|canada|brazil|malaysia|singapore|the netherlands|ireland|spain|italy|the nordics|scandinavia|the region)\b)/i;
116 +const METRO_RE = /\b((?:in|across) the (?:[A-Z][\w.]+ )?(?:market|metro|region|area|corridor|cluster)|(?:the )?(?:northern virginia|dfw|dallas-fort worth|silicon valley|greater [A-Z][a-z]+|the bay area) (?:market|region|area)|market'?s (?:total|inventory|capacity)|inventory|absorption|vacancy)\b/i;
117 +const CAMPUS_RE = /\b(campus|campuses|park|complex|estate|hub|gigafactory|(?:at|when|upon|once) (?:full(?:y)? )?(?:built[- ]out|build[- ]out|buildout|complete|completed|completion)|full build-?out|ultimate(?:ly)?|(?:total|overall|eventual) (?:site|campus|planned) capacity|master-?plan(?:ned)?|(?:across|over|in) (?:\w+ )?(?:phases|buildings|data halls)|multi-?building|multi-?phase|acre (?:site|campus|development)|the (?:site|development|project) (?:will|would|could) (?:deliver|provide|offer|support|house|host|total|reach)|entire (?:site|development|project))\b/i;
118 +const BUILDING_RE = /\b(building [A-Z0-9]+|(?:first|second|third|fourth|fifth|initial|new|final) (?:building|data hall|hall|structure|phase)|data hall|phase (?:\d+|one|two|three|i|ii|iii)\b|bldg\.?|the (?:building|hall) (?:will|would|is|has|offers|provides|delivers)|single[- ]storey|two-?stor(?:e)?y|three-?stor(?:e)?y|\d+[- ]stor(?:e)?y)\b/i;
119 +const FACILITY_RE = /\b(facility|the data cent(?:er|re)|this data cent(?:er|re)|the site|the new data cent(?:er|re)|the (?:\w+ )?data cent(?:er|re) (?:will|would|is|has|offers|provides|delivers))\b/i;
120 +
121 +export interface ScopeClassification { scope: ClaimScope; reason: string }
122 +
123 +/**
124 + * Deterministic scope of a figure from the sentence (and the few words) around it. Precedence: company → portfolio →
125 + * country → metro → campus → building → facility → unknown. `subjectHint` (what the record itself is) only breaks a tie
126 + * when the sentence says nothing. An UNKNOWN scope must never populate a facility column.
127 + */
128 +export function classifyScope(context: string | null | undefined, subjectHint?: ClaimScope | null): ScopeClassification {
129 + const s = (context ?? "").replace(/\s+/g, " ");
130 + if (!s.trim()) return { scope: subjectHint ?? "unknown", reason: subjectHint ? "context:empty+subject" : "context:empty" };
131 + if (COMPANY_RE.test(s)) return { scope: "company", reason: "regex:company" };
132 + if (PORTFOLIO_RE.test(s)) return { scope: "portfolio", reason: "regex:portfolio" };
133 + if (COUNTRY_RE.test(s)) return { scope: "country", reason: "regex:country" };
134 + if (METRO_RE.test(s)) return { scope: "metro", reason: "regex:metro" };
135 + if (CAMPUS_RE.test(s)) return { scope: "campus", reason: "regex:campus" };
136 + if (BUILDING_RE.test(s)) return { scope: "building", reason: "regex:building" };
137 + if (FACILITY_RE.test(s)) return { scope: "facility", reason: "regex:facility" };
138 + if (subjectHint) return { scope: subjectHint, reason: "subject" };
139 + return { scope: "unknown", reason: "no-signal" };
140 +}
141 +
142 +// ─── capacity semantics ────────────────────────────────────────────────────────────────────────────
143 +const IT_RE = /\b(it (?:load|capacity|power)|critical (?:it )?(?:load|capacity)|of it\b|it-?load|white ?space power|customer (?:power|load)|usable (?:power|capacity)|for it equipment|critical power)\b/i;
144 +const CRITICAL_RE = /\bcritical (?:it )?(?:load|power|capacity)\b/i;
145 +const GRID_RE = /\b(grid connection|grid-?connected|interconnection (?:agreement|capacity|request|queue)|connection (?:agreement|to the grid|capacity)|from the (?:grid|utility|substation)|substation|transmission (?:line|capacity|connection)|(?:secured|reserved|contracted|allocated|committed|approved) (?:\w+ )?(?:mw|gw|megawatts?|gigawatts?) of (?:grid |utility )?(?:power|electricity|capacity|supply|load)|load (?:request|letter|study|agreement)|large[- ]load|energi[sz]ation of)\b/i;
146 +const UTILITY_RE = /\b(utility (?:power|capacity|supply|feed|service)|power (?:supply|feed|service) (?:agreement|contract|capacity)|(?:power|electricity) (?:supply|allocation) (?:of|totaling|totalling)|(?:dual|redundant) (?:utility )?feeds?|incoming power|electrical (?:capacity|supply|service)|total (?:site )?power|power capacity of|available power)\b/i;
147 +const ULTIMATE_RE = /\b((?:at|when|upon|once) (?:full(?:y)? )?(?:built[- ]out|build[- ]out|buildout|complete|completed|completion)|full build-?out|ultimate(?:ly)?|final build-?out|(?:total|overall|eventual|potential|maximum) (?:site|campus|planned|development) (?:capacity|power|load)|(?:scale|scalable|scaling|grow|growing|expand|expandable) (?:up )?to (?:over |more than )?\d|up to (?:over |more than |approximately |nearly )?\d|master-?plan)\b/i;
148 +const PLANNED_RE = /\b(planned|proposed|will (?:deliver|provide|offer|have|house|support|feature|bring|add|total)|would (?:deliver|provide|offer|have|house|support)|expected to (?:deliver|provide|offer|reach|have)|is (?:set|slated|expected|designed) to|designed (?:for|to)|target(?:s|ed|ing)? (?:capacity|of)|to be (?:built|developed|delivered)|future|when complete|on completion|approved (?:for|capacity)|permitted|zoned for|application for|seeks? (?:approval|permission) for)\b/i;
149 +const PHASE_RE = /\b((?:first|second|third|fourth|initial|next|final) phase|phase (?:\d+|one|two|three|four|i|ii|iii|iv)\b|(?:first|initial) (?:building|data hall|tranche)|initial(?:ly)? (?:deliver|provide|offer)\w*)\b/i;
150 +const CURRENT_RE = /\b(currently|existing|in operation|operational|operating|live|online|now (?:offers|provides|delivers|has|houses)|today|present(?:ly)?|commissioned|energi[sz]ed|installed|delivered|is (?:offering|providing|delivering))\b/i;
151 +
152 +export interface CapacitySemantics { predicate: CapacityPredicate | null; reason: string }
153 +
154 +/**
155 + * Which kind of MW a sentence describes. Precedence: grid → utility → IT → phase → ultimate → planned → current.
156 + * Returns null when the sentence gives no semantic cue; the caller decides a default from the record status
157 + * (pipeline record → planned, operational record → total power) and flags the figure as "semantics: default".
158 + */
159 +export function classifyCapacitySemantics(context: string | null | undefined): CapacitySemantics {
160 + const s = (context ?? "").replace(/\s+/g, " ");
161 + if (!s.trim()) return { predicate: null, reason: "context:empty" };
162 + const isIt = IT_RE.test(s);
163 + if (GRID_RE.test(s) && !isIt) return { predicate: "grid_connection_mw", reason: "regex:grid" };
164 + if (UTILITY_RE.test(s) && !isIt) return { predicate: "utility_capacity_mw", reason: "regex:utility" };
165 + if (PHASE_RE.test(s) && !ULTIMATE_RE.test(s)) return { predicate: "phase_mw", reason: "regex:phase" };
166 + if (ULTIMATE_RE.test(s)) return { predicate: "ultimate_campus_mw", reason: "regex:ultimate" };
167 + if (isIt) return { predicate: CRITICAL_RE.test(s) ? "critical_power_mw" : "it_capacity_mw", reason: "regex:it" };
168 + if (PLANNED_RE.test(s)) return { predicate: "planned_power_mw", reason: "regex:planned" };
169 + if (CURRENT_RE.test(s)) return { predicate: "current_power_mw", reason: "regex:current" };
170 + return { predicate: null, reason: "no-signal" };
171 +}
172 +
173 +// ─── investment semantics ──────────────────────────────────────────────────────────────────────────
174 +const MULTI_YEAR_RE = /\b((?:through|by|over the next|over the coming|until|to) (?:20\d\d|\d+ years|the decade|the end of the decade)|multi-?year|\d+-year (?:plan|investment|programme|program|commitment)|(?:annual|yearly) (?:capex|capital expenditure|investment)|capex|capital expenditure)\b/i;
175 +const COUNTRY_PROGRAM_RE = /\b((?:national|country|sovereign|government|federal|state) (?:program|programme|plan|initiative|strategy|fund|investment)|(?:invest|investment|commit\w*|pledge\w*) [^.]{0,60}(?:in|across) (?:the )?(?:united states|u\.s\.|us|uk|united kingdom|europe|india|japan|germany|france|australia|canada|brazil|malaysia|singapore|the country|the nation|the region|emea|apac)\b)/i;
176 +const DEAL_RE = /\b(acqui(?:re|res|red|sition)|takeover|buyout|merger|stake|valuation|valued at|purchase price|sale of|sells?|sold|buys?|bought|financing|refinanc\w+|loan|credit facility|bond|notes|debt|equity|raises?|raised|funding round|series [a-e]|lease(?:d|s)? (?:valued|worth)|contract (?:valued|worth))\b/i;
177 +const CAMPUS_INV_RE = /\b(campus|site|park|complex|development|project|phase|build-?out|the facility|the data cent(?:er|re))\b/i;
178 +
179 +export interface InvestmentSemantics { predicate: InvestmentPredicate; scope: ClaimScope; reason: string }
180 +
181 +/** Which kind of money a sentence describes and which scope it applies to. Deal values and company capex never become a project's investment. */
182 +export function classifyInvestmentSemantics(context: string | null | undefined, subjectHint?: ClaimScope | null): InvestmentSemantics {
183 + const s = (context ?? "").replace(/\s+/g, " ");
184 + if (!s.trim()) return { predicate: "project_investment_usd", scope: subjectHint ?? "unknown", reason: "context:empty" };
185 + if (DEAL_RE.test(s) && !/\b(invest(?:s|ed|ing|ment)? (?:of |about |approximately |over |more than |up to )?[$€£]|to build|construction|will invest|plans to invest)\b/i.test(s)) return { predicate: "deal_value_usd", scope: "company", reason: "regex:deal" };
186 + if (COUNTRY_PROGRAM_RE.test(s)) return { predicate: "country_program_usd", scope: "country", reason: "regex:country-program" };
187 + if (MULTI_YEAR_RE.test(s) || COMPANY_RE.test(s)) return { predicate: "multi_year_capex_usd", scope: "company", reason: "regex:multi-year" };
188 + if (PORTFOLIO_RE.test(s)) return { predicate: "company_investment_usd", scope: "portfolio", reason: "regex:portfolio" };
189 + const sc = classifyScope(s, subjectHint);
190 + if (sc.scope === "campus") return { predicate: "campus_investment_usd", scope: "campus", reason: sc.reason };
191 + if (sc.scope === "building" || sc.scope === "facility") return { predicate: "project_investment_usd", scope: sc.scope, reason: sc.reason };
192 + if (CAMPUS_INV_RE.test(s)) return { predicate: "project_investment_usd", scope: subjectHint ?? "facility", reason: "regex:project" };
193 + return { predicate: "project_investment_usd", scope: sc.scope, reason: sc.reason };
194 +}
195 +
196 +// ─── sanity engines ────────────────────────────────────────────────────────────────────────────────
197 +export type FlagSeverity = "info" | "warn" | "critical";
198 +export interface SanityFlag { code: string; severity: FlagSeverity; message: string; field?: string; value?: number | null; blocks?: boolean }
199 +
200 +export interface CapacitySanityInput {
201 + value: number;
202 + predicate: CapacityPredicate | null;
203 + scope: ClaimScope;
204 + /** the record the figure is being written to */
205 + recordScope: "building" | "facility" | "campus" | "project";
206 + previous?: number | null;
207 + context?: string | null;
208 + /** true when the facility / project name says campus / park / complex */
209 + campusDesignation?: boolean;
210 +}
211 +
212 +const MONEY_NUM_RE = /(?:US\$|USD|\$|€|£|A\$|C\$|S\$)\s?(\d{1,3}(?:[.,]\d{3})*(?:\.\d+)?)\s*(?:million|billion|bn|m\b|b\b)?/gi;
213 +const MW_NUM_RE = /(\d{1,3}(?:[.,]\d{3})*(?:\.\d+)?)\s*(?:mw|gw|megawatts?|gigawatts?)\b/gi;
214 +
215 +/** "$300 million" and "300 MW" in the same sentence share a number: the unit may have been confused. */
216 +export function moneyMwCollision(context: string | null | undefined): boolean {
217 + if (!context) return false;
218 + const money = new Set<string>();
219 + for (const m of context.matchAll(MONEY_NUM_RE)) money.add(m[1]!.replace(/,/g, ""));
220 + if (!money.size) return false;
221 + for (const m of context.matchAll(MW_NUM_RE)) if (money.has(m[1]!.replace(/,/g, ""))) return true;
222 + return false;
223 +}
224 +
225 +/** Deterministic capacity checks. `blocks: true` flags must keep the figure out of the record's columns. */
226 +export function capacitySanity(i: CapacitySanityInput): SanityFlag[] {
227 + const out: SanityFlag[] = [];
228 + const v = i.value;
229 + if (!Number.isFinite(v) || v <= 0) return [{ code: "mw_invalid", severity: "critical", message: `invalid MW ${v}`, value: v, blocks: true }];
230 + if (!isSiteScope(i.scope)) out.push({ code: `scope_${i.scope}`, severity: "critical", message: `figure describes a ${i.scope} (${i.scope === "unknown" ? "scope could not be determined" : "not this site"}) — kept as a claim, not assigned`, value: v, blocks: true });
231 + if (i.scope === "campus" && i.recordScope === "building") out.push({ code: "scope_campus_on_building", severity: "critical", message: "campus-level figure on a building record", value: v, blocks: true });
232 + if (v >= 20_000) out.push({ code: "mw_market_statistic", severity: "critical", message: `${v} MW is an industry / market statistic, not a site`, value: v, blocks: true });
233 + if (v > 1_000 && i.recordScope !== "campus" && !i.campusDesignation && i.predicate !== "ultimate_campus_mw") out.push({ code: "mw_single_site_gt_1000", severity: "critical", message: `single facility above 1 000 MW (${v} MW) without a campus designation`, value: v });
234 + if (v > 500 && i.recordScope === "building") out.push({ code: "mw_building_gt_500", severity: "warn", message: `one building above 500 MW (${v} MW)`, value: v });
235 + if (i.previous != null && i.previous > 0 && (v / i.previous >= 5 || i.previous / v >= 5)) out.push({ code: "mw_change_5x", severity: "warn", message: `figure changed more than 5× (${i.previous} → ${v} MW)`, value: v });
236 + if (i.context && moneyMwCollision(i.context)) out.push({ code: "mw_money_collision", severity: "critical", message: "a money figure with the same number sits in the same sentence — verify the unit", value: v });
237 + if (i.predicate == null) out.push({ code: "mw_semantics_default", severity: "info", message: "sentence does not say what kind of MW this is; semantics defaulted from the record status", value: v });
238 + if (i.predicate === "grid_connection_mw" || i.predicate === "utility_capacity_mw") out.push({ code: "mw_utility_not_it", severity: "info", message: "utility / grid figure — not IT load", value: v });
239 + return out;
240 +}
241 +
242 +export interface InvestmentSanityInput { value: number; scope: ClaimScope; predicate: InvestmentPredicate; recordScope: "facility" | "campus" | "project"; previous?: number | null; context?: string | null }
243 +
244 +export function investmentSanity(i: InvestmentSanityInput): SanityFlag[] {
245 + const out: SanityFlag[] = [];
246 + const v = i.value;
247 + if (!Number.isFinite(v) || v <= 0) return [{ code: "inv_invalid", severity: "critical", message: `invalid investment ${v}`, value: v, blocks: true }];
248 + if (!isSiteScope(i.scope)) out.push({ code: `inv_scope_${i.scope}`, severity: "critical", message: `investment describes a ${i.scope} — kept as a claim, not assigned`, value: v, blocks: true });
249 + if (i.predicate === "deal_value_usd" || i.predicate === "multi_year_capex_usd" || i.predicate === "company_investment_usd" || i.predicate === "country_program_usd") out.push({ code: `inv_${i.predicate}`, severity: "critical", message: `${INVESTMENT_PREDICATE_LABEL[i.predicate]} is not a site investment`, value: v, blocks: true });
250 + if (v > 500e9) out.push({ code: "inv_gt_500b", severity: "critical", message: "above $500B — an industry statistic", value: v, blocks: true });
251 + if (v > 50e9) out.push({ code: "inv_single_site_gt_50b", severity: "critical", message: `single site investment above $50B (${Math.round(v / 1e9)}B) — requires verification`, value: v });
252 + if (i.previous != null && i.previous > 0 && (v / i.previous >= 5 || i.previous / v >= 5)) out.push({ code: "inv_change_5x", severity: "warn", message: `investment changed more than 5× (${i.previous} → ${v})`, value: v });
253 + if (i.context && moneyMwCollision(i.context)) out.push({ code: "inv_money_mw_collision", severity: "warn", message: "same number appears as MW in the sentence — verify", value: v });
254 + return out;
255 +}
256 +
257 +// ─── evidence ──────────────────────────────────────────────────────────────────────────────────────
258 +export interface Evidence { text: string; start: number; end: number }
259 +
260 +/** Split into sentences with absolute offsets (abbreviation-tolerant enough for press prose). */
261 +export function sentenceSpans(text: string): Array<{ text: string; start: number; end: number }> {
262 + const out: Array<{ text: string; start: number; end: number }> = [];
263 + const re = /[^.!?\n]+(?:[.!?]+(?=\s+[A-Z0-9"'“(]|\s*$)|\n|$)/g;
264 + for (const m of text.matchAll(re)) {
265 + const raw = m[0];
266 + const lead = raw.length - raw.trimStart().length;
267 + const t = raw.trim();
268 + if (!t) continue;
269 + out.push({ text: t, start: (m.index ?? 0) + lead, end: (m.index ?? 0) + lead + t.length });
270 + }
271 + return out;
272 +}
273 +
274 +function numberVariants(value: number, unit: "mw" | "usd"): RegExp {
275 + if (unit === "mw") {
276 + const alts = new Set<string>();
277 + const push = (n: number, u: string) => { if (Number.isFinite(n)) alts.add(`${fmtNum(n)}\\+?\\s?(?:${u})`); };
278 + push(value, "MW|megawatts?");
279 + push(value / 1000, "GW|gigawatts?");
280 + push(value * 1000, "kW|kilowatts?");
281 + return new RegExp(`(?<![\\d.])(?:${[...alts].join("|")})\\b`, "i");
282 + }
283 + const alts = new Set<string>();
284 + const push = (n: number, u: string) => { if (Number.isFinite(n) && n >= 1) alts.add(`(?:US\\$|USD ?|\\$|€|£|A\\$|C\\$|S\\$)\\s?${fmtNum(n)}\\s?(?:${u})`); };
285 + push(value / 1e9, "billion|bn|b\\b");
286 + push(value / 1e6, "million|mn|m\\b");
287 + push(value / 1e12, "trillion|tn");
288 + if (value < 1e6) push(value, "");
289 + return new RegExp(`(?:${[...alts].join("|")})`, "i");
290 +}
291 +
292 +function fmtNum(n: number): string {
293 + const r = Math.round(n * 1000) / 1000;
294 + const plain = String(r);
295 + const withCommas = r >= 1000 && Number.isInteger(r) ? r.toLocaleString("en-US") : null;
296 + const variants = [plain, withCommas].filter((x): x is string => !!x).map((x) => x.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
297 + // "1.5" also written "1,5"; integers may carry a trailing ".0"
298 + if (!Number.isInteger(r)) variants.push(plain.replace(".", ",").replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
299 + return `(?:${variants.join("|")})`;
300 +}
301 +
302 +/** The sentence that supports a figure (value in MW or USD). null when no sentence quotes it — then there is no claim. */
303 +export function findEvidence(text: string | null | undefined, value: number, unit: "mw" | "usd"): Evidence | null {
304 + if (!text) return null;
305 + const re = numberVariants(value, unit);
306 + for (const s of sentenceSpans(text)) {
307 + if (re.test(s.text)) return { text: s.text.length > 600 ? `${s.text.slice(0, 597)}…` : s.text, start: s.start, end: s.end };
308 + }
309 + return null;
310 +}
311 +
312 +// ─── project event classification ─────────────────────────────────────────────────────────────────
313 +export const PROJECT_CLASSES = ["NEW_BUILD", "EXPANSION", "CONSTRUCTION_START", "PERMIT", "LAND_ACQUISITION", "POWER_AGREEMENT", "GRID_CONNECTION", "FINANCING", "ACQUISITION", "PARTNERSHIP", "CUSTOMER_AGREEMENT", "EXECUTIVE_APPOINTMENT", "SUSTAINABILITY", "PRODUCT_NEWS", "GENERAL_COMPANY_NEWS", "UNKNOWN"] as const;
314 +export type ProjectClass = (typeof PROJECT_CLASSES)[number];
315 +/** Classes that describe physical development and may create or materially modify a PROJECT. */
316 +export const PHYSICAL_CLASSES: ReadonlySet<ProjectClass> = new Set<ProjectClass>(["NEW_BUILD", "EXPANSION", "CONSTRUCTION_START", "PERMIT", "LAND_ACQUISITION", "GRID_CONNECTION"]);
317 +/** Classes that may attach an EVENT to an identifiable project / company but never create a project. */
318 +export const ASSOCIATED_CLASSES: ReadonlySet<ProjectClass> = new Set<ProjectClass>(["POWER_AGREEMENT", "FINANCING", "ACQUISITION", "PARTNERSHIP", "CUSTOMER_AGREEMENT"]);
319 +
320 +const CLASS_RULES: Array<[ProjectClass, RegExp]> = [
321 + ["EXECUTIVE_APPOINTMENT", /\b(appoint(?:s|ed|ment)?|names? [A-Z][\w'-]+ [A-Z][\w'-]+ (?:as|to)|joins? (?:as|the board|its board)|(?:new|hires?|promotes?|welcomes?|taps?) (?:[A-Z][\w'-]+ [A-Z][\w'-]+ as )?(?:CEO|CFO|COO|CTO|CRO|CIO|CMO|chief|president|head of|director|vice president|VP|managing director|general manager|board member|chairman|chair)|leadership (?:change|team|appointments?)|steps? down|retire(?:s|ment)|passing of|obituary|in memoriam|executive (?:team|hire|appointment))\b/i],
322 + ["GENERAL_COMPANY_NEWS", /\b(market (?:to|will|set to|expected to|projected to) (?:surpass|reach|hit|grow|exceed|top)|market size|cagr|forecast(?:s|ed)? (?:to|that)|report(?: finds| shows| reveals|:)|survey|study (?:finds|shows|reveals)|(?:index|trends?|predictions?|outlook|white ?paper|e-?book|webinar|podcast|interview|q&a|faq|explained|101\b|guide to|tips|best practices|case study|customer spotlight|award|shortlist|finalist|recogni[sz]ed|named (?:a |one of |to )?(?:top|best|leader))\b|quarterly results|half[- ]year results|annual results|earnings|revenue|ebitda|guidance|investor (?:day|presentation)|lawsuit|sues?\b|court|settlement|opinion|op-ed|why |how |what )/i],
323 + ["POWER_AGREEMENT", /\b(ppa\b|power purchase agreement|(?:virtual|corporate|long-term|\d+-year) (?:power|energy) (?:purchase )?agreement|(?:energy|power|electricity) (?:supply|offtake) (?:agreement|deal|contract)|offtake|(?:signs?|signed|inks?|inked|secures?|secured) (?:a )?(?:\d+[ -]?(?:mw|gw) )?(?:of )?(?:solar|wind|nuclear|geothermal|hydro|gas|renewable|clean|carbon-free) (?:power|energy|capacity|supply)|(?:solar|wind|nuclear|geothermal|hydro|gas|battery|fuel cell|smr|reactor)s? (?:to power|will power|powering|deal|plant to supply|farm to supply)|behind-the-meter|on-site (?:generation|power plant|gas plant|solar)|microgrid|small modular reactor)\b/i],
324 + ["SUSTAINABILITY", /\b(sustainability (?:report|goals?|targets?|commitment|strategy)|net[- ]zero|carbon[- ]neutral|carbon[- ]free|esg\b|renewable (?:energy )?(?:certificates?|credits?|target|goal|commitment)|100% renewable|water (?:positive|stewardship|usage|conservation)|emissions? (?:reduction|target)|leed (?:gold|platinum|certif)|green (?:building|certification)|energy efficiency (?:program|initiative)|climate (?:pledge|commitment)|biodiversity|tree planting|heat (?:reuse|recovery) (?:program|scheme|initiative))\b/i],
325 + ["GRID_CONNECTION", /\b(grid connection|grid-?connected|interconnection (?:agreement|request|capacity|queue|study)|connection agreement|(?:new |expanded )?substation|transmission (?:line|upgrade|project|capacity)|energi[sz]ation|energi[sz]ed|large[- ]load (?:tariff|request|agreement|customer)|load (?:request|study|letter)|(?:secured|reserved|allocated|approved) (?:\d+[ -]?(?:mw|gw) of )?(?:grid |utility )?(?:power|capacity) (?:from|with) (?:the utility|[A-Z][\w&]+ (?:energy|power|electric|utility)))\b/i],
326 + ["FINANCING", /\b(financ(?:es|ing|ed)|refinanc\w+|raises? (?:\$|€|£|us\$)|raised (?:\$|€|£|us\$)|series [a-e]\b|funding (?:round|arranged|secured)|loan|credit facility|green bond|bonds?\b|notes? offering|debt (?:facility|package|raise)|equity (?:raise|investment|stake)|ipo\b|capital raise|construction (?:loan|financing)|securitization|abs\b|term loan|(?:secures?|secured|closes?|closed) (?:\$|€|£|us\$)[\d.,]+ ?(?:m|bn|b|million|billion)? (?:in )?(?:financing|loan|facility|debt|funding))\b/i],
327 + ["ACQUISITION", /\b(acqui(?:res?|red|sition) (?:of )?(?:[A-Z][\w&'.-]+ )+(?:group|holdings|inc|ltd|llc|limited|corp|sa|ag|plc|pte|data ?cent(?:er|re)s?|portfolio|platform|company|business)|(?:acquires?|acquired|acquiring|buys?|bought|buying|purchases?|purchased|purchasing|takes? over|took over) (?:a |an |the |its |two |three |four |five |six |\d+ )?(?:\d+[ -]?(?:mw|gw) )?(?:(?:hyperscale|colocation|carrier-neutral|edge|operational|existing) )?(?:data ?cent(?:er|re)s?|facility|facilities|portfolio|platform|operator|business|company|stake|campus from|site from)|acquisition of|takeover|buyout|merger|merges? with|to be acquired|stake in|sells?|sold|divest\w+|sale of)\b/i],
328 + ["CUSTOMER_AGREEMENT", /\b((?:signs?|signed|secures?|secured|inks?|inked|lands?|landed|wins?|won|announces?) (?:a |an |its |the )?(?:multi-year |long-term |major |anchor |hyperscale |\d+[ -]?(?:mw|gw) )?(?:colocation |hosting |capacity |wholesale |pre-?)?(?:lease|leases|tenant|customer|contract|agreement|deal) (?:with|for|from)|(?:lease|leases|leased) (?:\d+[ -]?(?:mw|gw)|[\d,]+ (?:sq|square))|(?:selects?|selected|chooses?|chose|picks?|picked|taps?|tapped) [A-Z][\w&'.-]+ (?:for|as|to)|(?:hosting|colocation|lease|leasing|services?|master|supply|framework|capacity) agreements?|pre-?leas\w+|fully leased|anchor tenant|moves? into|deploys? (?:in|at|with)|expands? (?:its )?(?:footprint|presence) (?:in|at|with) [A-Z][\w&'.-]+(?:'s)? (?:data ?cent(?:er|re)|facility|campus))\b/i],
329 + ["PARTNERSHIP", /\b(partnership|partners? with|joint venture|\bjv\b|join(?:s|ed)? forces|team(?:s|ed)? up|collaborat(?:es?|ion|ing)|alliance|memorandum of understanding|\bmou\b|strategic (?:agreement|relationship|cooperation)|framework agreement)\b/i],
330 + ["PRODUCT_NEWS", /\b(launches? (?:a )?(?:new )?(?:service|platform|product|portal|program|programme|offering|solution|marketplace|tool|api|feature)|introduces? (?:a )?(?:new )?(?:service|platform|product|offering|solution)|now available|available (?:now|in|from)|unveils? (?:a )?(?:new )?(?:service|platform|product|offering|solution|brand|logo|website)|rebrand|new (?:brand|logo|website|identity)|integrat(?:es|ion) with|certified by|(?:earns?|achieves?|receives?) (?:[A-Z]\w+ )?(?:certification|rating|accreditation|iso \d))\b/i],
331 + ["LAND_ACQUISITION", /\b((?:acquires?|acquired|buys?|bought|purchases?|purchased|secures?|secured|closes? on|closed on|options?) (?:a |an |the |its |additional |another )?(?:\d[\d,.]*[ -]?(?:acre|hectare|ha)s?|land|site|parcel|plot|property|farmland|lot)|land (?:purchase|acquisition|deal|bank)|(?:acre|hectare) (?:site|parcel|plot|property) (?:acquired|purchased|bought)|site (?:acquisition|purchase)|zoned land|land for (?:a |its |the )?(?:new )?(?:data ?cent(?:er|re)|campus))\b/i],
332 + ["PERMIT", /\b(planning (?:application|permission|approval|consent|committee|commission|board|inquiry)|permit(?:s|ted|ting)?\b|rezon\w+|zoning (?:approval|change|request|case|board|commission)|(?:files?|filed|submits?|submitted|lodges?|lodged) (?:plans?|an application|a planning|planning|for approval|proposal)|(?:approv(?:es|ed|al)|green-?light(?:s|ed)?|consent(?:s|ed)?|clears?|cleared|greenlit|(?:votes?|voted) to approve|unanimously approved|signs? off|sign-off) (?:for |of |on )?(?:a |the |its |\d+[ -]?(?:mw|gw) )?(?:data ?cent(?:er|re)|campus|project|plans?|development|proposal|application|rezoning|site plan|special use|conditional use)|environmental (?:impact|assessment|review|permit)|eia\b|site plan (?:approval|review)|special (?:use|exception) permit|conditional use permit|comprehensive plan amendment|public hearing|(?:county|city|town|council|board|commission|supervisors) (?:approves?|approved|rejects?|rejected|denies|denied|defers?|deferred|delays?|tables?|tabled))\b/i],
333 + ["CONSTRUCTION_START", /\b(breaks? ground|broke ground|groundbreaking|ground-?breaking|construction (?:has |have )?(?:started|begins?|began|begun|commenced|commences|kicks? off|is under ?way|under ?way|gets under ?way)|(?:begins?|began|starts?|started|commences?|commenced|kicks? off) (?:construction|building|work|site work|vertical construction)|under construction|topped out|topping out|tops out|steel (?:erection|going up)|first concrete|shovels in the ground|site work (?:begins|has begun|under ?way)|construction (?:milestone|update|progress))\b/i],
334 + ["EXPANSION", /\b(expan(?:sion|ds?|ded|ding)|adds? (?:\d+[ -]?(?:mw|gw)|capacity|a (?:second|third|fourth|new) (?:building|data hall|phase))|additional (?:\d+[ -]?(?:mw|gw)|capacity|building|data hall|phase)|(?:second|third|fourth|fifth|next|new) (?:phase|building|data hall|hall|facility (?:at|on|in) (?:its|the) (?:existing )?campus)|phase (?:2|3|4|ii|iii|iv|two|three|four)\b|(?:grows?|growing|scales?|scaling|extends?|extending|enlarges?|doubles?|doubling|triples?|tripling) (?:its |the )?(?:capacity|campus|footprint at|data cent(?:er|re)|site|facility)|(?:grows?|expands?|extends?) [A-Z][\w .-]{0,40}?(?:campus|data cent(?:er|re)|facility|site)\b|more capacity (?:at|in|to)|further (?:capacity|building|facility|phase)|extension (?:of|to) (?:its|the) (?:existing )?(?:data cent(?:er|re)|campus|facility|site))\b/i],
335 + ["NEW_BUILD", /\b((?:to|will|would|could|may|might|plans? to|set to|aims? to|intends? to|proposes? to|wants? to|is|are|is expected to|expected to) (?:build|construct|develop|create|deliver|establish|open|erect|bring)|(?:plans?|planned|planning|proposes?|proposed|proposal|announces?|announced|unveils?|unveiled|reveals?|revealed|confirms?|confirmed|launches?|launched|commits? to|committed to|pledges?|earmarks?|eyes|eyeing|mulls?|weighs?|considering|exploring|in talks) (?:for |to build |to develop |to construct |to invest in |to open |a |an |its |the |new |first |second |two |three |\d+[ -]?(?:mw|gw) |\$[\d.,]+ ?(?:m|bn|b|million|billion)? |€[\d.,]+ ?(?:m|bn|b|million|billion)? |£[\d.,]+ ?(?:m|bn|b|million|billion)? |hyperscale |ai |massive |major |huge |giant |sprawling |\w+-acre )*(?:data ?cent(?:er|re)s?|campus|campuses|facility|facilities|ai factory|ai factories|hub|site|development|project|complex|cluster|supercomputer)|new (?:\d+[ -]?(?:mw|gw) |hyperscale |ai |\$[\d.,]+ ?(?:m|bn|b|million|billion) )?(?:data ?cent(?:er|re)|campus|facility|ai factory|hub) (?:in|near|at|for|to|planned|proposed|coming|slated|set)|(?:\d+[ -]?(?:mw|gw)) (?:ai )?(?:data ?cent(?:er|re)|campus|facility|site|project|development|hub) (?:in|near|at|for|planned|proposed)|(?:build|building|develop|developing|construct|constructing|open|opening) (?:a |an |its |the |new |first |\d+[ -]?(?:mw|gw) )*(?:data ?cent(?:er|re)|campus|facility|ai factory) (?:in|near|at|on|for)|first (?:data ?cent(?:er|re)|campus|facility) in|(?:enters?|entering|entry into) (?:the )?[A-Z][\w.-]+ (?:market|with)|coming to|slated for|(?:to|will|would|could) (?:house|host|feature|include) (?:a |an |up to )?(?:\d+[ -]?(?:mw|gw)|data ?cent(?:er|re)|campus))\b/i],
336 +];
337 +
338 +/** Verbs / nouns that describe physical development (build, construct, break ground, expand, permit, approve, file, acquire land, energize…). */
339 +export const DEVELOPMENT_VERB_RE = /\b(build|builds|building|built|construct\w*|develop\w*|break\w* ground|broke ground|groundbreaking|expan\w+|open|opens|opening|opened|permit\w*|approv\w+|file[sd]?|filing|rezon\w+|acquire[sd]? (?:\d[\d,.]*[ -]?(?:acre|hectare)s?|land|a site|the site|a parcel)|land (?:purchase|acquisition)|energi[sz]\w+|deliver\w*|commission\w*|top\w* out|phase|campus|data ?cent(?:er|re)s?|facility|site plan|planning application|zoning)\b/i;
340 +
341 +export interface ProjectClassInput {
342 + title: string | null | undefined;
343 + /** the lead of the article (first ~600 characters) */
344 + lead?: string | null;
345 + /** an explicit project / facility name was found */
346 + hasExplicitName?: boolean;
347 + /** a recognised operator is the headline subject or the primary operator */
348 + hasOperator?: boolean;
349 + /** a city / region / country was found in the title or lead */
350 + hasLocation?: boolean;
351 + /** the title-derived status (pipeline / operational / null) */
352 + status?: string | null;
353 +}
354 +
355 +export interface ProjectClassification {
356 + class: ProjectClass;
357 + /** true when the class may create or materially modify a project */
358 + physical: boolean;
359 + /** true when the class may attach an event to an identifiable project / company */
360 + associated: boolean;
361 + /** evidence threshold for creating a project: (explicit name or operator) + location + development verb */
362 + evidence: { name: boolean; operator: boolean; location: boolean; verb: boolean; strength: "strong" | "weak" | "none" };
363 + /** true only for physical classes with strong evidence */
364 + mayCreateProject: boolean;
365 + reasons: string[];
366 +}
367 +
368 +/**
369 + * Classify what an announcement is about BEFORE anything becomes a project. The title decides; the lead breaks ties when
370 + * the title is generic ("Company announces …"). Order of the rules = precedence: people, market research and
371 + * sustainability news are vetoed first, then money / deals / customers, then physical development classes.
372 + */
373 +export function classifyProjectEvent(i: ProjectClassInput): ProjectClassification {
374 + const title = (i.title ?? "").replace(/\s+/g, " ").trim();
375 + const lead = (i.lead ?? "").replace(/\s+/g, " ").slice(0, 600);
376 + const reasons: string[] = [];
377 + let cls: ProjectClass = "UNKNOWN";
378 + const physicalTitle = ["CONSTRUCTION_START", "EXPANSION", "NEW_BUILD", "PERMIT", "LAND_ACQUISITION", "GRID_CONNECTION"].some((c) => CLASS_RULES.find(([k]) => k === c)![1].test(title));
379 + for (const [k, re] of CLASS_RULES) {
380 + if (re.test(title)) {
381 + // a money / deal / partnership headline that also says "to build 300MW campus" is a build headline
382 + if (["FINANCING", "ACQUISITION", "CUSTOMER_AGREEMENT", "PARTNERSHIP", "POWER_AGREEMENT", "PRODUCT_NEWS", "SUSTAINABILITY"].includes(k) && physicalTitle && /\b(to build|will build|to develop|to construct|breaks? ground|broke ground|groundbreaking|new (?:\d+[ -]?(?:mw|gw) )?(?:data ?cent(?:er|re)|campus)|\d+[ -]?(?:mw|gw) (?:data ?cent(?:er|re)|campus)|expansion|expands? (?:its|the) (?:campus|data ?cent(?:er|re)))\b/i.test(title)) {
383 + reasons.push(`title:${k}+build→physical`);
384 + continue;
385 + }
386 + cls = k;
387 + reasons.push(`title:${k}`);
388 + break;
389 + }
390 + }
391 + if (cls === "UNKNOWN" && lead) {
392 + for (const [k, re] of CLASS_RULES) {
393 + // the lead may only add physical / associated classes — a people or market veto must be visible in the headline
394 + if (["EXECUTIVE_APPOINTMENT", "GENERAL_COMPANY_NEWS", "SUSTAINABILITY", "PRODUCT_NEWS"].includes(k)) continue;
395 + if (re.test(lead)) { cls = k; reasons.push(`lead:${k}`); break; }
396 + }
397 + }
398 + if (cls === "UNKNOWN" && i.status && /^(announced|proposed|rumored)$/.test(i.status) && /\b(data ?cent(?:er|re)|campus|facility)\b/i.test(`${title} ${lead}`)) { cls = "NEW_BUILD"; reasons.push("status:announced→NEW_BUILD"); }
399 + if (cls === "UNKNOWN" && i.status === "under_construction") { cls = "CONSTRUCTION_START"; reasons.push("status:construction"); }
400 + if (cls === "UNKNOWN" && (i.status === "permitting" || i.status === "approved")) { cls = "PERMIT"; reasons.push(`status:${i.status}`); }
401 + if (cls === "UNKNOWN" && i.status === "expansion") { cls = "EXPANSION"; reasons.push("status:expansion"); }
402 +
403 + const verb = DEVELOPMENT_VERB_RE.test(title) || (!!lead && DEVELOPMENT_VERB_RE.test(lead));
404 + const name = !!i.hasExplicitName;
405 + const operator = !!i.hasOperator;
406 + const location = !!i.hasLocation;
407 + const strength: ProjectClassification["evidence"]["strength"] = (name || operator) && location && verb ? "strong" : (name || operator || location) && verb ? "weak" : "none";
408 + const physical = PHYSICAL_CLASSES.has(cls);
409 + const associated = ASSOCIATED_CLASSES.has(cls);
410 + return { class: cls, physical, associated, evidence: { name, operator, location, verb, strength }, mayCreateProject: physical && strength === "strong", reasons };
411 +}
412 +
413 +// ─── project lifecycle state machine ───────────────────────────────────────────────────────────────
414 +export const PROJECT_STAGES = ["rumored", "proposed", "announced", "permitting", "approved", "under_construction", "partially_operational", "operational", "delayed", "cancelled"] as const;
415 +export type ProjectStage = (typeof PROJECT_STAGES)[number];
416 +const STAGE_ORDER: Record<string, number> = { rumored: 0, proposed: 1, announced: 2, permitting: 3, approved: 4, under_construction: 5, partially_operational: 6, operational: 7, expansion: 7 };
417 +
418 +export type TransitionVerdict = "forward" | "side" | "resume" | "same" | "backward" | "invalid";
419 +
420 +/**
421 + * Allowed lifecycle transitions. Forward moves along the pipeline are always fine; `delayed` / `cancelled` are side
422 + * branches reachable from any pre-operational stage; leaving `delayed` resumes the pipeline; a backward move
423 + * (operational → announced) is rejected unless the incoming source outranks the stored one (caller decides).
424 + */
425 +export function projectTransition(from: string | null | undefined, to: string | null | undefined): TransitionVerdict {
426 + if (!to) return "invalid";
427 + if (!from || from === "unknown") return "forward";
428 + if (from === to) return "same";
429 + if (to === "delayed" || to === "cancelled") return from === "operational" ? "invalid" : "side";
430 + if (from === "cancelled") return to === "announced" || to === "proposed" || to === "rumored" ? "resume" : "invalid";
431 + if (from === "delayed") return "resume";
432 + const a = STAGE_ORDER[from], b = STAGE_ORDER[to];
433 + if (a == null || b == null) return "invalid";
434 + return b > a ? "forward" : "backward";
435 +}
436 +
437 +// ─── AI evidence levels ────────────────────────────────────────────────────────────────────────────
438 +export const AI_EVIDENCE_LEVELS = ["confirmed", "likely", "associated", "unknown"] as const;
439 +export type AiEvidence = (typeof AI_EVIDENCE_LEVELS)[number];
440 +const AI_CONFIRMED_RE = /\b(ai (?:data ?cent(?:er|re)|campus|factory|factories|infrastructure|cluster|supercomputer|compute (?:campus|facility|site)|training (?:facility|site|campus))|(?:purpose-?built|dedicated|designed) (?:for|to) (?:ai|artificial intelligence|accelerated computing|gpu|hpc|high-performance computing|machine learning|large language models?|llms?|training)|hpc (?:data ?cent(?:er|re)|campus|facility|cluster)|supercomput(?:er|ing) (?:cent(?:er|re)|facility|campus)|gpu (?:cluster|campus|farm|cloud|data ?cent(?:er|re)|hosting)|accelerated computing (?:facility|campus|data ?cent(?:er|re)|infrastructure)|exascale|(?:nvidia|blackwell|hopper|gb\d{3}|h\d{3}|b\d{3}) (?:gpus?|systems?|superchips?|clusters?))\b/i;
441 +const AI_LIKELY_RE = /\b(ai-?ready|ai-?optimi[sz]ed|high[- ]density|liquid[- ]cool\w*|direct-to-chip|immersion cool\w*|rear-door heat exchangers?|(?:\d{2,3}|1\d{2}) ?kw (?:per |\/ ?)rack|gpu|accelerated computing|hpc|machine learning|inference|training clusters?|sovereign ai|ai workloads?)\b/i;
442 +const AI_ASSOCIATED_RE = /\b(openai|anthropic|xai|coreweave|lambda labs|crusoe|nebius|nscale|fluidstack|voltage park|stargate|ai (?:tenant|customer|lease|demand|boom|workloads?)|for ai|ai and cloud|cloud and ai)\b/i;
443 +
444 +/** Evidence level that a facility / project is AI/HPC infrastructure. One "AI" keyword never confirms it. */
445 +export function classifyAiEvidence(text: string | null | undefined): { level: AiEvidence; evidence: string | null } {
446 + if (!text) return { level: "unknown", evidence: null };
447 + const s = text.replace(/\s+/g, " ");
448 + const c = AI_CONFIRMED_RE.exec(s);
449 + if (c) return { level: "confirmed", evidence: c[0] };
450 + const l = AI_LIKELY_RE.exec(s);
451 + if (l) return { level: "likely", evidence: l[0] };
452 + const a = AI_ASSOCIATED_RE.exec(s);
453 + if (a) return { level: "associated", evidence: a[0] };
454 + return { level: "unknown", evidence: null };
455 +}
456 +
457 +// ─── event significance ────────────────────────────────────────────────────────────────────────────
458 +export type Significance = "major" | "medium" | "minor";
459 +export function significanceBand(score: number): Significance { return score >= 75 ? "major" : score >= 45 ? "medium" : "minor"; }
modified packages/core/src/ids.ts +5 −0
@@ -18,6 +18,11 @@ export const ID_PREFIXES = {
18 18 match: "mtc",
19 19 provenance: "prv",
20 20 ranking: "rnk",
21 + claim: "clm",
22 + flag: "flg",
23 + constraint: "grd",
24 + watch: "wtc",
25 + cluster: "cls",
21 26 } as const;
22 27 export type IdKind = keyof typeof ID_PREFIXES;
23 28
modified packages/core/src/index.browser.ts +1 −0
@@ -5,3 +5,4 @@ export * from "./geo.js";
5 5 export * from "./confidence.js";
6 6 export * from "./diff.js";
7 7 export * from "./api-types.js";
8 +export * from "./claims.js";
modified packages/core/src/index.ts +1 −0
@@ -7,3 +7,4 @@ export * from "./confidence.js";
7 7 export * from "./diff.js";
8 8 export * from "./ssrf.js";
9 9 export * from "./api-types.js";
10 +export * from "./claims.js";
added packages/core/src/normalize.test.ts +68 −0
@@ -0,0 +1,68 @@
1 +import { describe, expect, it } from "vitest";
2 +import { parseMw, parseAllMw, parseMoney, parsePartialDate, inferStatus, normalizeStatus, countryFromText, parseAreaSqm } from "./normalize.js";
3 +
4 +describe("parseMw", () => {
5 + it("parses plain figures and units", () => {
6 + expect(parseMw("36 MW")).toBe(36);
7 + expect(parseMw("36MW")).toBe(36);
8 + expect(parseMw("120 megawatts")).toBe(120);
9 + expect(parseMw("1.2 GW")).toBe(1200);
10 + expect(parseMw("36,000 kW")).toBe(36);
11 + expect(parseMw("40+MW")).toBe(40);
12 + });
13 + it("treats a single dot followed by three digits as a decimal (DataBank 0.779MW regression)", () => {
14 + expect(parseMw("0.779MW Critical IT Load")).toBe(0.779);
15 + expect(parseMw("1.250 MW")).toBe(1.25);
16 + expect(parseMw("2.500MW")).toBe(2.5);
17 + });
18 + it("still reads European thousands with two or more groups", () => {
19 + expect(parseMw("1.500.000 kW")).toBe(1500);
20 + });
21 + it("reads a comma decimal", () => {
22 + expect(parseMw("1,2 GW")).toBe(1200);
23 + });
24 + it("returns null without a unit", () => {
25 + expect(parseMw("300 million")).toBeNull();
26 + expect(parseMw(null)).toBeNull();
27 + });
28 +});
29 +
30 +describe("parseAllMw", () => {
31 + it("dedupes and sorts descending, drops absurd figures", () => {
32 + expect(parseAllMw("a 36 MW hall in a 300MW campus, 36 MW again, 500,000 MW nonsense")).toEqual([300, 36]);
33 + });
34 +});
35 +
36 +describe("parseMoney", () => {
37 + it("parses currencies and multipliers", () => {
38 + expect(parseMoney("$1.2 billion")).toEqual({ amount: 1_200_000_000, currency: "USD" });
39 + expect(parseMoney("€800m")).toEqual({ amount: 800_000_000, currency: "EUR" });
40 + expect(parseMoney("£2bn")).toEqual({ amount: 2_000_000_000, currency: "GBP" });
41 + expect(parseMoney("US$450 million")).toEqual({ amount: 450_000_000, currency: "USD" });
42 + });
43 +});
44 +
45 +describe("parsePartialDate", () => {
46 + it("handles partial dates", () => {
47 + expect(parsePartialDate("Q2 2027")).toBe("2027-Q2");
48 + expect(parsePartialDate("June 2027")).toBe("2027-06");
49 + expect(parsePartialDate("2027")).toBe("2027");
50 + expect(parsePartialDate("15 June 2027")).toBe("2027-06-15");
51 + });
52 +});
53 +
54 +describe("status", () => {
55 + it("infers and normalizes statuses", () => {
56 + expect(inferStatus("breaks ground on new campus")).toBe("under_construction");
57 + expect(inferStatus("planning application submitted")).toBe("permitting");
58 + expect(normalizeStatus("Under Construction")).toBe("under_construction");
59 + expect(normalizeStatus("coming soon")).toBe("announced");
60 + });
61 +});
62 +
63 +describe("misc", () => {
64 + it("country from text and areas", () => {
65 + expect(countryFromText("a campus in the Netherlands")).toBe("NL");
66 + expect(parseAreaSqm("8,170 Square Feet")).toBe(759.0);
67 + });
68 +});
modified packages/core/src/normalize.ts +4 −2
@@ -123,9 +123,11 @@ export function parseMw(text: string | null | undefined): number | null {
123 123 const m = s.match(/(\d{1,3}(?:[.,]\d{3})+|\d+(?:[.,]\d+)?)\+?\s*(gigawatts?|gw|megawatts?|mw|kilowatts?|kw)\b/i);
124 124 if (!m) return null;
125 125 let num = m[1]!;
126 − // "36,000" → 36000 ; "1,2" (fr decimal) → 1.2 ; "1.2" → 1.2
126 + // "36,000" → 36000 ; "1,2" (fr decimal) → 1.2 ; "1.2" → 1.2 ; "0.779" → 0.779 (a decimal, never 779) ;
127 + // "1.500.000" (two dot groups, European thousands) → 1500000. A single dot followed by three digits is a
128 + // decimal in every English-language capacity figure ("0.779MW", "1.250 MW") — DataBank MSP4 was stored as 779 MW.
127 129 if (/^\d{1,3}(,\d{3})+$/.test(num)) num = num.replace(/,/g, "");
128 − else if (/^\d{1,3}(\.\d{3})+$/.test(num) && !/^\d+\.\d{1,2}$/.test(num)) num = num.replace(/\./g, "");
130 + else if (/^\d{1,3}(\.\d{3}){2,}$/.test(num)) num = num.replace(/\./g, "");
129 131 else num = num.replace(",", ".");
130 132 const v = Number(num);
131 133 if (!Number.isFinite(v)) return null;
modified packages/core/src/types.ts +42 −1
@@ -127,6 +127,17 @@ export const EVENT_TYPES = [
127 127 "incident",
128 128 "page_changed",
129 129 "news",
130 + // infrastructure graph (2026-09)
131 + "land_acquired",
132 + "grid_connection",
133 + "grid_constraint",
134 + "utility_event",
135 + "operator_expansion",
136 + "customer_agreement",
137 + "partnership",
138 + "executive_change",
139 + "project_delayed",
140 + "project_cancelled",
130 141 ] as const;
131 142 export type EventType = (typeof EVENT_TYPES)[number];
132 143
@@ -207,6 +218,16 @@ export interface NormalizedFacility {
207 218 description?: string | null;
208 219 /** per-field provenance overrides; default provenance applies to all other fields */
209 220 facts?: Fact[];
221 + /** supporting sentence per numeric field (itCapacityMw, totalPowerMw, plannedPowerMw…) — required for prose-extracted figures */
222 + claimContext?: Record<string, string>;
223 + /** what this record is: one building, a facility, or a campus (containment-aware aggregation) */
224 + recordScope?: "building" | "facility" | "campus" | null;
225 + /** name / key of the campus-level facility record this building belongs to */
226 + parentFacilityKey?: string | null;
227 + aiEvidence?: "confirmed" | "likely" | "associated" | "unknown" | null;
228 + developerName?: string | null;
229 + landownerName?: string | null;
230 + tenantNames?: string[];
210 231 provenance: Provenance;
211 232 }
212 233
@@ -292,6 +313,24 @@ export interface NormalizedProject {
292 313 sourceUrl?: string | null;
293 314 timeline?: Array<{ date: string; type: EventType | string; description: string; url?: string }>;
294 315 externalIds?: Record<string, string | number>;
316 + /** classification of the announcement (packages/core claims.ts) — non-physical classes never create a project */
317 + projectClass?: string | null;
318 + evidenceLevel?: "strong" | "weak" | "none" | null;
319 + capacityScope?: string | null;
320 + capacitySemantics?: string | null;
321 + investmentScope?: string | null;
322 + investmentSemantics?: string | null;
323 + investmentCurrency?: string | null;
324 + investmentOriginal?: number | null;
325 + /** supporting sentence per numeric field (plannedMw, investmentUsd) */
326 + claimContext?: Record<string, string>;
327 + aiEvidence?: "confirmed" | "likely" | "associated" | "unknown" | null;
328 + developerName?: string | null;
329 + tenantName?: string | null;
330 + campusName?: string | null;
331 + constructionStartedOn?: string | null;
332 + approvedOn?: string | null;
333 + permitFiledOn?: string | null;
295 334 provenance: Provenance;
296 335 }
297 336
@@ -304,7 +343,9 @@ export interface NormalizedNewsEvent {
304 343 summary?: string | null;
305 344 pageType?: PageType;
306 345 eventType?: EventType;
307 − mentions?: { operators?: string[]; countriesIso2?: string[]; cities?: string[]; mw?: number[]; facilities?: string[] };
346 + mentions?: { operators?: string[]; countriesIso2?: string[]; cities?: string[]; mw?: number[]; facilities?: string[]; metro?: string | null };
347 + projectClass?: string | null;
348 + isAi?: boolean | null;
308 349 provenance: Provenance;
309 350 }
310 351
added packages/db/migrations/0003_claims_quality_graph.sql +202 −0
@@ -0,0 +1,202 @@
1 +-- 0003 — claim-first data layer, quality flags, campus containment, capacity ontology, run tracking, snapshots.
2 +-- Idempotent (IF NOT EXISTS everywhere) so it can be re-applied on a database created before this file existed.
3 +
4 +-- ─── claims ────────────────────────────────────────────────────────────────────────────────────────
5 +CREATE TABLE IF NOT EXISTS "claims" (
6 + "id" text PRIMARY KEY,
7 + "subject_type" text NOT NULL, -- facility | campus | project | operator | market | country
8 + "subject_id" text NOT NULL,
9 + "predicate" text NOT NULL, -- it_capacity_mw | planned_power_mw | project_investment_usd | status | …
10 + "value" double precision,
11 + "value_text" text,
12 + "unit" text, -- MW | USD | …
13 + "scope" text NOT NULL DEFAULT 'unknown', -- building | facility | campus | metro | country | portfolio | company | unknown
14 + "scope_reason" text,
15 + "source_id" text NOT NULL,
16 + "connector_id" text NOT NULL,
17 + "document_id" text,
18 + "url" text NOT NULL,
19 + "published_at" text, -- partial date as published
20 + "retrieved_at" timestamp with time zone NOT NULL DEFAULT now(),
21 + "confidence" text NOT NULL DEFAULT 'moderate',
22 + "is_estimate" boolean NOT NULL DEFAULT false,
23 + "authority_tier" text NOT NULL DEFAULT 'D',
24 + "evidence_text" text,
25 + "evidence_start" integer,
26 + "evidence_end" integer,
27 + "parser_name" text,
28 + "parser_version" text,
29 + "run_id" text,
30 + "status" text NOT NULL DEFAULT 'current', -- current | superseded | rejected | review | unscoped
31 + "rejection_reason" text,
32 + "first_observed" timestamp with time zone NOT NULL DEFAULT now(),
33 + "last_observed" timestamp with time zone NOT NULL DEFAULT now(),
34 + "created_at" timestamp with time zone NOT NULL DEFAULT now()
35 +);
36 +CREATE UNIQUE INDEX IF NOT EXISTS "claims_uq" ON "claims" ("subject_type", "subject_id", "predicate", "source_id", "url", COALESCE("value", 0), COALESCE("value_text", ''));
37 +CREATE INDEX IF NOT EXISTS "claims_subject_idx" ON "claims" ("subject_type", "subject_id", "predicate");
38 +CREATE INDEX IF NOT EXISTS "claims_status_idx" ON "claims" ("status");
39 +CREATE INDEX IF NOT EXISTS "claims_run_idx" ON "claims" ("run_id");
40 +CREATE INDEX IF NOT EXISTS "claims_document_idx" ON "claims" ("document_id");
41 +
42 +-- ─── quality flags ─────────────────────────────────────────────────────────────────────────────────
43 +CREATE TABLE IF NOT EXISTS "quality_flags" (
44 + "id" text PRIMARY KEY,
45 + "entity_type" text NOT NULL,
46 + "entity_id" text NOT NULL,
47 + "claim_id" text,
48 + "code" text NOT NULL, -- mw_single_site_gt_1000 | scope_company | inv_single_site_gt_50b | project_false_positive | duplicate | …
49 + "severity" text NOT NULL DEFAULT 'warn', -- info | warn | critical
50 + "field" text,
51 + "message" text NOT NULL,
52 + "details" jsonb,
53 + "priority" integer NOT NULL DEFAULT 0, -- review priority (impact-weighted)
54 + "status" text NOT NULL DEFAULT 'open', -- open | resolved | dismissed
55 + "resolution" text,
56 + "resolved_by" text,
57 + "resolved_at" timestamp with time zone,
58 + "run_id" text,
59 + "dedupe_key" text NOT NULL,
60 + "created_at" timestamp with time zone NOT NULL DEFAULT now(),
61 + "updated_at" timestamp with time zone NOT NULL DEFAULT now()
62 +);
63 +CREATE UNIQUE INDEX IF NOT EXISTS "quality_flags_dedupe_uq" ON "quality_flags" ("dedupe_key");
64 +CREATE INDEX IF NOT EXISTS "quality_flags_entity_idx" ON "quality_flags" ("entity_type", "entity_id");
65 +CREATE INDEX IF NOT EXISTS "quality_flags_open_idx" ON "quality_flags" ("status", "priority" DESC) WHERE "status" = 'open';
66 +CREATE INDEX IF NOT EXISTS "quality_flags_code_idx" ON "quality_flags" ("code");
67 +
68 +-- ─── facilities: containment, capacity ontology, AI evidence ───────────────────────────────────────
69 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "parent_facility_id" text REFERENCES "facilities"("id");
70 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "record_scope" text NOT NULL DEFAULT 'facility'; -- building | facility | campus
71 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "ai_evidence" text NOT NULL DEFAULT 'unknown'; -- confirmed | likely | associated | unknown
72 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "utility_capacity_mw" double precision;
73 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "grid_connection_mw" double precision;
74 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "ultimate_campus_mw" double precision;
75 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "capacity_scope" text; -- scope of the displayed MW figure
76 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "capacity_semantics" text; -- predicate behind the displayed MW figure
77 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "developer_id" text REFERENCES "operators"("id");
78 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "landowner_id" text REFERENCES "operators"("id");
79 +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "review_priority" integer NOT NULL DEFAULT 0;
80 +CREATE INDEX IF NOT EXISTS "facilities_parent_idx" ON "facilities" ("parent_facility_id");
81 +CREATE INDEX IF NOT EXISTS "facilities_campus_idx" ON "facilities" ("campus_id");
82 +CREATE INDEX IF NOT EXISTS "facilities_merged_idx" ON "facilities" ("merged_into") WHERE "merged_into" IS NOT NULL;
83 +CREATE INDEX IF NOT EXISTS "facilities_owner_idx" ON "facilities" ("owner_id");
84 +CREATE INDEX IF NOT EXISTS "facilities_ai_idx" ON "facilities" ("ai_evidence") WHERE "ai_evidence" <> 'unknown';
85 +CREATE INDEX IF NOT EXISTS "facilities_opened_idx" ON "facilities" ("opened_on") WHERE "opened_on" IS NOT NULL;
86 +
87 +-- ─── projects: classification, evidence, lifecycle, geocoding, scopes ─────────────────────────────
88 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "project_class" text; -- NEW_BUILD | EXPANSION | … | UNKNOWN
89 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "evidence_level" text; -- strong | weak | none
90 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "merged_into" text;
91 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "ai_evidence" text NOT NULL DEFAULT 'unknown';
92 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "capacity_scope" text;
93 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "capacity_semantics" text;
94 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "investment_scope" text;
95 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "investment_semantics" text;
96 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "developer_id" text REFERENCES "operators"("id");
97 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "tenant_id" text REFERENCES "operators"("id");
98 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "campus_id" text REFERENCES "campuses"("id");
99 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "construction_started_on" text;
100 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "approved_on" text;
101 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "permit_filed_on" text;
102 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "opened_on" text;
103 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "review_priority" integer NOT NULL DEFAULT 0;
104 +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "hidden" boolean NOT NULL DEFAULT false; -- false positives kept for audit, never listed
105 +CREATE INDEX IF NOT EXISTS "projects_class_idx" ON "projects" ("project_class");
106 +CREATE INDEX IF NOT EXISTS "projects_merged_idx" ON "projects" ("merged_into") WHERE "merged_into" IS NOT NULL;
107 +CREATE INDEX IF NOT EXISTS "projects_live_idx" ON "projects" ("status", "country_iso2") WHERE "merged_into" IS NULL AND NOT "hidden";
108 +CREATE INDEX IF NOT EXISTS "projects_latlng_idx" ON "projects" ("lat", "lng") WHERE "lat" IS NOT NULL;
109 +CREATE INDEX IF NOT EXISTS "projects_metro_idx" ON "projects" ("metro_id");
110 +
111 +-- ─── provenance / versions: run tracking + scope ──────────────────────────────────────────────────
112 +ALTER TABLE "provenance" ADD COLUMN IF NOT EXISTS "run_id" text;
113 +ALTER TABLE "provenance" ADD COLUMN IF NOT EXISTS "scope" text;
114 +ALTER TABLE "provenance" ADD COLUMN IF NOT EXISTS "is_winner" boolean NOT NULL DEFAULT false; -- the observation backing the displayed column value
115 +CREATE INDEX IF NOT EXISTS "provenance_current_idx" ON "provenance" ("entity_type", "entity_id", "field") WHERE "is_current";
116 +CREATE INDEX IF NOT EXISTS "provenance_run_idx" ON "provenance" ("run_id");
117 +ALTER TABLE "document_versions" ADD COLUMN IF NOT EXISTS "run_id" text;
118 +ALTER TABLE "document_versions" ADD COLUMN IF NOT EXISTS "extractor_version" text;
119 +CREATE INDEX IF NOT EXISTS "document_versions_run_idx" ON "document_versions" ("run_id");
120 +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "run_id" text;
121 +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "cluster_id" text; -- same underlying announcement across outlets
122 +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "evidence_count" integer NOT NULL DEFAULT 1;
123 +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "is_ai" boolean NOT NULL DEFAULT false;
124 +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "source_kind" text;
125 +CREATE INDEX IF NOT EXISTS "events_significance_idx" ON "events" ("significance" DESC, "detected_at" DESC);
126 +CREATE INDEX IF NOT EXISTS "events_project_idx" ON "events" ("project_id");
127 +CREATE INDEX IF NOT EXISTS "events_cluster_idx" ON "events" ("cluster_id");
128 +CREATE INDEX IF NOT EXISTS "events_run_idx" ON "events" ("run_id");
129 +CREATE INDEX IF NOT EXISTS "events_metro_idx" ON "events" ("metro_id");
130 +
131 +-- ─── news items ────────────────────────────────────────────────────────────────────────────────────
132 +ALTER TABLE "news_items" ADD COLUMN IF NOT EXISTS "project_class" text;
133 +ALTER TABLE "news_items" ADD COLUMN IF NOT EXISTS "cluster_id" text;
134 +ALTER TABLE "news_items" ADD COLUMN IF NOT EXISTS "metro_id" text;
135 +CREATE INDEX IF NOT EXISTS "news_items_operator_gin" ON "news_items" USING gin ("operator_ids");
136 +CREATE INDEX IF NOT EXISTS "news_items_cluster_idx" ON "news_items" ("cluster_id");
137 +
138 +-- ─── connectors: quarantine + health ───────────────────────────────────────────────────────────────
139 +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "quarantine" boolean NOT NULL DEFAULT false;
140 +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "consecutive_failures" integer NOT NULL DEFAULT 0;
141 +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "blocked_since" timestamp with time zone;
142 +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "last_discovered" integer;
143 +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "priority_score" real;
144 +ALTER TABLE "connector_runs" ADD COLUMN IF NOT EXISTS "quarantined" boolean NOT NULL DEFAULT false;
145 +
146 +-- ─── sources: licensing ────────────────────────────────────────────────────────────────────────────
147 +ALTER TABLE "sources" ADD COLUMN IF NOT EXISTS "redistribution" text; -- allowed | attribution | restricted | unknown
148 +ALTER TABLE "sources" ADD COLUMN IF NOT EXISTS "attribution_required" boolean;
149 +
150 +-- ─── entity keys: scope the primary key by entity type ─────────────────────────────────────────────
151 +DO $$
152 +BEGIN
153 + IF EXISTS (SELECT 1 FROM pg_constraint WHERE conname = 'entity_keys_pkey' AND conrelid = 'entity_keys'::regclass
154 + AND array_length(conkey, 1) = 1) THEN
155 + ALTER TABLE "entity_keys" DROP CONSTRAINT "entity_keys_pkey";
156 + ALTER TABLE "entity_keys" ADD PRIMARY KEY ("key", "entity_type");
157 + END IF;
158 +END $$;
159 +
160 +-- ─── daily snapshots for "as of" views and regression checks ───────────────────────────────────────
161 +CREATE TABLE IF NOT EXISTS "entity_snapshots" (
162 + "day" date NOT NULL,
163 + "kind" text NOT NULL, -- global_totals | ranking | facility_status | project_stage | operator_totals | country_totals
164 + "key" text NOT NULL,
165 + "payload" jsonb NOT NULL,
166 + "created_at" timestamp with time zone NOT NULL DEFAULT now(),
167 + CONSTRAINT "entity_snapshots_pk" PRIMARY KEY ("day", "kind", "key")
168 +);
169 +CREATE INDEX IF NOT EXISTS "entity_snapshots_kind_idx" ON "entity_snapshots" ("kind", "key", "day");
170 +
171 +-- ─── watchlists (private, cookie-scoped) ───────────────────────────────────────────────────────────
172 +CREATE TABLE IF NOT EXISTS "watchlists" (
173 + "id" text PRIMARY KEY,
174 + "owner_token" text NOT NULL,
175 + "entity_type" text NOT NULL,
176 + "entity_id" text NOT NULL,
177 + "created_at" timestamp with time zone NOT NULL DEFAULT now()
178 +);
179 +CREATE UNIQUE INDEX IF NOT EXISTS "watchlists_uq" ON "watchlists" ("owner_token", "entity_type", "entity_id");
180 +
181 +-- ─── grid / power context (events already carry the type; this keeps market-level constraint notes) ─
182 +CREATE TABLE IF NOT EXISTS "grid_constraints" (
183 + "id" text PRIMARY KEY,
184 + "metro_id" text REFERENCES "metros"("id"),
185 + "country_iso2" text REFERENCES "countries"("iso2"),
186 + "kind" text NOT NULL, -- moratorium | grid_delay | capacity_restriction | load_cap | new_transmission | new_substation | regulation | large_load_queue
187 + "title" text NOT NULL,
188 + "summary" text,
189 + "effective_date" text,
190 + "source_id" text,
191 + "document_id" text,
192 + "url" text NOT NULL,
193 + "event_id" text,
194 + "confidence" text NOT NULL DEFAULT 'moderate',
195 + "created_at" timestamp with time zone NOT NULL DEFAULT now()
196 +);
197 +CREATE INDEX IF NOT EXISTS "grid_constraints_metro_idx" ON "grid_constraints" ("metro_id");
198 +CREATE INDEX IF NOT EXISTS "grid_constraints_country_idx" ON "grid_constraints" ("country_iso2");
199 +
200 +-- ─── daily metrics time-series lookups ─────────────────────────────────────────────────────────────
201 +CREATE INDEX IF NOT EXISTS "daily_metrics_series_idx" ON "daily_metrics" ("metric", "dim", "day");
202 +CREATE INDEX IF NOT EXISTS "entity_matches_created_fac_idx" ON "entity_matches" ((candidate->>'createdFacilityId')) WHERE "status" = 'pending';
modified packages/db/src/schema.ts +174 −5
@@ -139,11 +139,24 @@ export const facilities = pgTable(
139 139 firstSeen: ts("first_seen").notNull().defaultNow(),
140 140 lastVerified: ts("last_verified"),
141 141 mergedInto: text("merged_into"),
142 + /** containment: a building row points at its campus row (aggregation never counts both) */
143 + parentFacilityId: text("parent_facility_id"),
144 + recordScope: text("record_scope").notNull().default("facility"), // building | facility | campus
145 + aiEvidence: text("ai_evidence").notNull().default("unknown"), // confirmed | likely | associated | unknown
146 + utilityCapacityMw: doublePrecision("utility_capacity_mw"),
147 + gridConnectionMw: doublePrecision("grid_connection_mw"),
148 + ultimateCampusMw: doublePrecision("ultimate_campus_mw"),
149 + capacityScope: text("capacity_scope"),
150 + capacitySemantics: text("capacity_semantics"),
151 + developerId: text("developer_id"),
152 + landownerId: text("landowner_id"),
153 + reviewPriority: integer("review_priority").notNull().default(0),
142 154 createdAt: createdAt(),
143 155 updatedAt: updatedAt(),
144 156 },
145 157 (t) => [
146 158 index("facilities_country_idx").on(t.countryIso2),
159 + index("facilities_parent_idx").on(t.parentFacilityId),
147 160 index("facilities_operator_idx").on(t.operatorId),
148 161 index("facilities_metro_idx").on(t.metroId),
149 162 index("facilities_status_idx").on(t.status),
@@ -169,13 +182,13 @@ export const facilityAliases = pgTable(
169 182 export const entityKeys = pgTable(
170 183 "entity_keys",
171 184 {
172 − key: text("key").primaryKey(),
185 + key: text("key").notNull(),
173 186 entityType: text("entity_type").notNull(),
174 187 entityId: text("entity_id").notNull(),
175 188 connectorId: text("connector_id").notNull(),
176 189 createdAt: createdAt(),
177 190 },
178 − (t) => [index("entity_keys_entity_idx").on(t.entityType, t.entityId)],
191 + (t) => [primaryKey({ columns: [t.key, t.entityType] }), index("entity_keys_entity_idx").on(t.entityType, t.entityId)],
179 192 );
180 193
181 194 export const cloudRegions = pgTable(
@@ -275,10 +288,28 @@ export const projects = pgTable(
275 288 confidence: text("confidence").notNull().default("moderate"),
276 289 externalIds: jsonb("external_ids").$type<Record<string, string | number>>().notNull().default(sql`'{}'::jsonb`),
277 290 lastUpdate: ts("last_update").notNull().defaultNow(),
291 + projectClass: text("project_class"), // NEW_BUILD | EXPANSION | CONSTRUCTION_START | PERMIT | … | UNKNOWN
292 + evidenceLevel: text("evidence_level"), // strong | weak | none
293 + mergedInto: text("merged_into"),
294 + aiEvidence: text("ai_evidence").notNull().default("unknown"),
295 + capacityScope: text("capacity_scope"),
296 + capacitySemantics: text("capacity_semantics"),
297 + investmentScope: text("investment_scope"),
298 + investmentSemantics: text("investment_semantics"),
299 + developerId: text("developer_id"),
300 + tenantId: text("tenant_id"),
301 + campusId: text("campus_id"),
302 + constructionStartedOn: text("construction_started_on"),
303 + approvedOn: text("approved_on"),
304 + permitFiledOn: text("permit_filed_on"),
305 + openedOn: text("opened_on"),
306 + reviewPriority: integer("review_priority").notNull().default(0),
307 + /** false positives are hidden, never deleted (audit trail) */
308 + hidden: boolean("hidden").notNull().default(false),
278 309 createdAt: createdAt(),
279 310 updatedAt: updatedAt(),
280 311 },
281 − (t) => [index("projects_country_idx").on(t.countryIso2), index("projects_operator_idx").on(t.operatorId), index("projects_status_idx").on(t.status), index("projects_norm_idx").on(t.normalizedName)],
312 + (t) => [index("projects_country_idx").on(t.countryIso2), index("projects_operator_idx").on(t.operatorId), index("projects_status_idx").on(t.status), index("projects_norm_idx").on(t.normalizedName), index("projects_class_idx").on(t.projectClass), index("projects_metro_idx").on(t.metroId)],
282 313 );
283 314
284 315 export const projectTimeline = pgTable(
@@ -310,6 +341,8 @@ export const sources = pgTable("sources", {
310 341 attribution: text("attribution"),
311 342 robotsAllowed: boolean("robots_allowed"),
312 343 notes: text("notes"),
344 + redistribution: text("redistribution"), // allowed | attribution | restricted | unknown
345 + attributionRequired: boolean("attribution_required"),
313 346 createdAt: createdAt(),
314 347 updatedAt: updatedAt(),
315 348 });
@@ -332,6 +365,12 @@ export const connectors = pgTable("connectors", {
332 365 lastError: text("last_error"),
333 366 nextRunAt: ts("next_run_at"),
334 367 stats: jsonb("stats").$type<Record<string, number>>().notNull().default(sql`'{}'::jsonb`),
368 + /** quarantine: runs extract and preview but publish nothing (after parser changes, HTML changes, suspicious spikes) */
369 + quarantine: boolean("quarantine").notNull().default(false),
370 + consecutiveFailures: integer("consecutive_failures").notNull().default(0),
371 + blockedSince: ts("blocked_since"),
372 + lastDiscovered: integer("last_discovered"),
373 + priorityScore: real("priority_score"),
335 374 createdAt: createdAt(),
336 375 updatedAt: updatedAt(),
337 376 });
@@ -348,6 +387,7 @@ export const connectorRuns = pgTable(
348 387 stats: jsonb("stats").$type<Record<string, number>>().notNull().default(sql`'{}'::jsonb`),
349 388 error: text("error"),
350 389 log: jsonb("log").$type<Array<{ t: string; level: string; msg: string }>>().notNull().default(sql`'[]'::jsonb`),
390 + quarantined: boolean("quarantined").notNull().default(false),
351 391 },
352 392 (t) => [index("connector_runs_connector_idx").on(t.connectorId, t.startedAt)],
353 393 );
@@ -414,8 +454,10 @@ export const documentVersions = pgTable(
414 454 diffSummary: jsonb("diff_summary").$type<{ addedCount: number; removedCount: number; ratio: number; added: string[]; removed: string[] }>(),
415 455 detectedChanges: jsonb("detected_changes").$type<Array<Record<string, unknown>>>().notNull().default(sql`'[]'::jsonb`),
416 456 significance: integer("significance").notNull().default(0),
457 + runId: text("run_id"),
458 + extractorVersion: text("extractor_version"),
417 459 },
418 − (t) => [index("document_versions_doc_idx").on(t.documentId, t.fetchedAt)],
460 + (t) => [index("document_versions_doc_idx").on(t.documentId, t.fetchedAt), index("document_versions_run_idx").on(t.runId)],
419 461 );
420 462
421 463 /** Per-field provenance for every entity value. */
@@ -440,9 +482,14 @@ export const provenance = pgTable(
440 482 extractorVersion: text("extractor_version"),
441 483 isCurrent: boolean("is_current").notNull().default(true),
442 484 note: text("note"),
485 + runId: text("run_id"),
486 + scope: text("scope"),
487 + /** the observation currently backing the displayed column value */
488 + isWinner: boolean("is_winner").notNull().default(false),
443 489 },
444 490 (t) => [
445 491 index("provenance_entity_idx").on(t.entityType, t.entityId),
492 + index("provenance_run_idx").on(t.runId),
446 493 index("provenance_source_idx").on(t.sourceId),
447 494 uniqueIndex("provenance_uq").on(t.entityType, t.entityId, t.field, t.sourceId, t.url),
448 495 ],
@@ -472,9 +519,18 @@ export const events = pgTable(
472 519 metroId: text("metro_id"),
473 520 projectId: text("project_id"),
474 521 fingerprint: text("fingerprint").notNull(),
522 + runId: text("run_id"),
523 + /** documents describing the same underlying announcement share a cluster id */
524 + clusterId: text("cluster_id"),
525 + evidenceCount: integer("evidence_count").notNull().default(1),
526 + isAi: boolean("is_ai").notNull().default(false),
527 + sourceKind: text("source_kind"),
475 528 },
476 529 (t) => [
477 530 uniqueIndex("events_fingerprint_uq").on(t.fingerprint),
531 + index("events_significance_idx").on(t.significance, t.detectedAt),
532 + index("events_project_idx").on(t.projectId),
533 + index("events_cluster_idx").on(t.clusterId),
478 534 index("events_detected_idx").on(t.detectedAt),
479 535 index("events_entity_idx").on(t.entityType, t.entityId),
480 536 index("events_country_idx").on(t.countryIso2),
@@ -502,9 +558,12 @@ export const newsItems = pgTable(
502 558 projectId: text("project_id"),
503 559 mw: doublePrecision("mw"),
504 560 significance: integer("significance").notNull().default(20),
561 + projectClass: text("project_class"),
562 + clusterId: text("cluster_id"),
563 + metroId: text("metro_id"),
505 564 createdAt: createdAt(),
506 565 },
507 − (t) => [index("news_items_published_idx").on(t.publishedAt)],
566 + (t) => [index("news_items_published_idx").on(t.publishedAt), index("news_items_cluster_idx").on(t.clusterId)],
508 567 );
509 568
510 569 /** Reconciliation queue: candidate records whose match to an existing facility was ambiguous. */
@@ -585,3 +644,113 @@ export const connectorState = pgTable(
585 644 },
586 645 (t) => [primaryKey({ columns: [t.connectorId, t.key] })],
587 646 );
647 +
648 +/** Claim-first store: one figure asserted by one document about one subject, with scope, evidence and authority (packages/core claims.ts). */
649 +export const claims = pgTable(
650 + "claims",
651 + {
652 + id: text("id").primaryKey(),
653 + subjectType: text("subject_type").notNull(),
654 + subjectId: text("subject_id").notNull(),
655 + predicate: text("predicate").notNull(),
656 + value: doublePrecision("value"),
657 + valueText: text("value_text"),
658 + unit: text("unit"),
659 + scope: text("scope").notNull().default("unknown"),
660 + scopeReason: text("scope_reason"),
661 + sourceId: text("source_id").notNull(),
662 + connectorId: text("connector_id").notNull(),
663 + documentId: text("document_id"),
664 + url: text("url").notNull(),
665 + publishedAt: text("published_at"),
666 + retrievedAt: ts("retrieved_at").notNull().defaultNow(),
667 + confidence: text("confidence").notNull().default("moderate"),
668 + isEstimate: boolean("is_estimate").notNull().default(false),
669 + authorityTier: text("authority_tier").notNull().default("D"),
670 + evidenceText: text("evidence_text"),
671 + evidenceStart: integer("evidence_start"),
672 + evidenceEnd: integer("evidence_end"),
673 + parserName: text("parser_name"),
674 + parserVersion: text("parser_version"),
675 + runId: text("run_id"),
676 + status: text("status").notNull().default("current"), // current | superseded | rejected | review | unscoped
677 + rejectionReason: text("rejection_reason"),
678 + firstObserved: ts("first_observed").notNull().defaultNow(),
679 + lastObserved: ts("last_observed").notNull().defaultNow(),
680 + createdAt: createdAt(),
681 + },
682 + (t) => [index("claims_subject_idx").on(t.subjectType, t.subjectId, t.predicate), index("claims_status_idx").on(t.status), index("claims_run_idx").on(t.runId), index("claims_document_idx").on(t.documentId)],
683 +);
684 +
685 +/** Deterministic data-quality flags (capacity / investment sanity, scope, duplicates, project false positives…). */
686 +export const qualityFlags = pgTable(
687 + "quality_flags",
688 + {
689 + id: text("id").primaryKey(),
690 + entityType: text("entity_type").notNull(),
691 + entityId: text("entity_id").notNull(),
692 + claimId: text("claim_id"),
693 + code: text("code").notNull(),
694 + severity: text("severity").notNull().default("warn"), // info | warn | critical
695 + field: text("field"),
696 + message: text("message").notNull(),
697 + details: jsonb("details").$type<Record<string, unknown>>(),
698 + priority: integer("priority").notNull().default(0),
699 + status: text("status").notNull().default("open"), // open | resolved | dismissed
700 + resolution: text("resolution"),
701 + resolvedBy: text("resolved_by"),
702 + resolvedAt: ts("resolved_at"),
703 + runId: text("run_id"),
704 + dedupeKey: text("dedupe_key").notNull(),
705 + createdAt: createdAt(),
706 + updatedAt: updatedAt(),
707 + },
708 + (t) => [uniqueIndex("quality_flags_dedupe_uq").on(t.dedupeKey), index("quality_flags_entity_idx").on(t.entityType, t.entityId), index("quality_flags_code_idx").on(t.code)],
709 +);
710 +
711 +/** Daily JSON snapshots (global totals, rankings, status / stage counts) for "as of" views and regression checks. */
712 +export const entitySnapshots = pgTable(
713 + "entity_snapshots",
714 + {
715 + day: date("day").notNull(),
716 + kind: text("kind").notNull(),
717 + key: text("key").notNull(),
718 + payload: jsonb("payload").$type<Record<string, unknown>>().notNull(),
719 + createdAt: createdAt(),
720 + },
721 + (t) => [primaryKey({ columns: [t.day, t.kind, t.key] }), index("entity_snapshots_kind_idx").on(t.kind, t.key, t.day)],
722 +);
723 +
724 +/** Private watchlists (owner = opaque cookie token, no accounts). */
725 +export const watchlists = pgTable(
726 + "watchlists",
727 + {
728 + id: text("id").primaryKey(),
729 + ownerToken: text("owner_token").notNull(),
730 + entityType: text("entity_type").notNull(),
731 + entityId: text("entity_id").notNull(),
732 + createdAt: createdAt(),
733 + },
734 + (t) => [uniqueIndex("watchlists_uq").on(t.ownerToken, t.entityType, t.entityId)],
735 +);
736 +
737 +/** Public reports of grid constraints (moratoria, delays, load caps, new transmission…) attached to a market / country. */
738 +export const gridConstraints = pgTable(
739 + "grid_constraints",
740 + {
741 + id: text("id").primaryKey(),
742 + metroId: text("metro_id").references(() => metros.id),
743 + countryIso2: text("country_iso2").references(() => countries.iso2),
744 + kind: text("kind").notNull(),
745 + title: text("title").notNull(),
746 + summary: text("summary"),
747 + effectiveDate: text("effective_date"),
748 + sourceId: text("source_id"),
749 + documentId: text("document_id"),
750 + url: text("url").notNull(),
751 + eventId: text("event_id"),
752 + confidence: text("confidence").notNull().default("moderate"),
753 + createdAt: createdAt(),
754 + },
755 + (t) => [index("grid_constraints_metro_idx").on(t.metroId), index("grid_constraints_country_idx").on(t.countryIso2)],
756 +);
588 757