Phase 1 — claim-first data layer: claims + quality flags + snapshots (migration 0003), scope/semantics/authority classifiers, announcement class veto + evidence threshold, capacity/investment sanity engines, HQ guard, containment-aware aggregation, effective parser versions, connector quarantine + health states, extraction debugger (dci trace), quality sweep + regression checks, parseMw decimal fix, API contract 2.0 types
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
42 changed files +3,188 −163
modified
apps/worker/src/cli.ts
+25 −2
@@ -8,7 +8,9 @@ | ||
| 8 | 8 | * discover <id> [--dry-run] |
| 9 | 9 | * docs <id> [--due] [--limit n] [--group g] |
| 10 | 10 | * inspect <url-or-docId> document row + latest version + provenance rows referencing it |
| 11 | − * reprocess <id> [--limit n] [--group g] re-extract from archived bodies (no network) | |
| 11 | + * reprocess <id> [--limit n] [--group g] [--stale] re-extract from archived bodies (no network); --stale = older parser version only | |
| 12 | + * trace <doc-id|url> [--live] [--json] extraction debugger: every pipeline stage for one document, nothing persisted | |
| 13 | + * quarantine <id> [on|off] preview-only mode for a connector (ingest rolled back) · quality · snapshot · gaps | |
| 12 | 14 | * stats global counts (entities, documents, runs, events, queues) |
| 13 | 15 | * doctor Postgres / Redis / ClickHouse / MinIO, budgets, stale connectors, error rates → system_alerts |
| 14 | 16 | * rank | metrics | refresh-stats maintenance jobs, run inline |
@@ -31,6 +33,8 @@ import { computeRankings } from "./rankings.js"; | ||
| 31 | 33 | import { closeScheduler, enqueueRun, pauseConnector, queueSnapshot, schedulerTick, startSchedulerLoop } from "./scheduler.js"; |
| 32 | 34 | import { parseDiscoveredFrom } from "./scheduling.js"; |
| 33 | 35 | import { getRaw } from "./storage.js"; |
| 36 | +import { traceDocument } from "./trace.js"; | |
| 37 | +import { dataGaps, qualitySweep, snapshotAndCheck } from "./quality.js"; | |
| 34 | 38 | import { ensureClickHouse } from "@dci/db/clickhouse"; |
| 35 | 39 | |
| 36 | 40 | /* ---------- arg parsing ---------- */ |
@@ -137,7 +141,7 @@ async function cmdRun(a: Args): Promise<number> { | ||
| 137 | 141 | await ensureClickHouseQuiet(); |
| 138 | 142 | if (!dryRun) await syncConnectorsToDb([requireConnector(id)]); |
| 139 | 143 | const urls = list(a, "url"); |
| 140 | − const r = await runConnector(id, { task: urls.length ? "crawl" : taskOf(a, "full"), group: str(a, "group"), limit: num(a, "limit"), dryRun, urls: urls.length ? urls : undefined, force: bool(a, "force"), logLevel: bool(a, "verbose") ? "debug" : undefined, sampleEntities: dryRun ? num(a, "samples") ?? 20 : num(a, "samples") ?? 0 }); | |
| 144 | + const r = await runConnector(id, { task: urls.length ? "crawl" : taskOf(a, "full"), group: str(a, "group"), limit: num(a, "limit"), dryRun, urls: urls.length ? urls : undefined, force: bool(a, "force"), staleOnly: bool(a, "stale"), quarantine: bool(a, "quarantine"), logLevel: bool(a, "verbose") ? "debug" : undefined, sampleEntities: dryRun ? num(a, "samples") ?? 20 : num(a, "samples") ?? 0 }); | |
| 141 | 145 | printRun(r, bool(a, "json")); |
| 142 | 146 | return r.status === "failed" ? 1 : 0; |
| 143 | 147 | } |
@@ -254,6 +258,20 @@ async function cmdValidateYaml(a: Args): Promise<number> { | ||
| 254 | 258 | return 0; |
| 255 | 259 | } |
| 256 | 260 | |
| 261 | +function printTrace(t: Awaited<ReturnType<typeof traceDocument>>): void { | |
| 262 | + const d = t.document ?? {}; | |
| 263 | + console.log(paint("bold", `SOURCE DOCUMENT ${String(d.id ?? "?")}`)); console.log(` ${String(d.url ?? "")}\n connector ${String(d.connectorId ?? "")} · pageType ${String(d.pageType ?? "")} · extractor ${String(d.extractorVersion ?? "")} · refs ${JSON.stringify(d.entityRefs ?? [])}`); | |
| 264 | + if (t.fetch) console.log(`\n${paint("bold", "RAW FETCH")} ${t.fetch.source} · HTTP ${t.fetch.status} · ${t.fetch.contentType ?? "?"} · ${t.fetch.bytes} B · L${t.fetch.level} ${t.fetch.fetcher}`); | |
| 265 | + console.log(`\n${paint("bold", "PARSED TEXT")} title: ${t.text.title ?? "—"} · ${t.text.length} chars · classified ${t.text.classification?.pageType ?? "?"} (${t.text.classification?.rule ?? ""}) · MW in text: ${t.text.classification?.mw.join(", ") || "none"}`); | |
| 266 | + if (t.announcement) { const a = t.announcement as Record<string, unknown>; const c = a.classification as { class: string; mayCreateProject: boolean; evidence: Record<string, unknown> }; console.log(`\n${paint("bold", "ANNOUNCEMENT")} class ${paint(c.mayCreateProject ? "green" : "yellow", c.class)} · mayCreateProject ${c.mayCreateProject} · evidence ${JSON.stringify(c.evidence)}\n status ${String(a.status)} · headline MW ${String(a.headlineMw)} (scope ${String(a.capacityScope)}, ${String(a.capacitySemantics)}) · money ${JSON.stringify(a.money)} (${String(a.investmentScope)}/${String(a.investmentSemantics)}) · location ${JSON.stringify(a.location)} · operator ${String(a.operator)} · AI ${String(a.aiEvidence)}${a.hqGuarded ? " · HQ mention stripped" : ""}`); const ev = a.evidence as { plannedMw?: { text: string } | null; investment?: { text: string } | null }; if (ev?.plannedMw) console.log(paint("dim", ` MW evidence: "${ev.plannedMw.text.slice(0, 220)}"`)); if (ev?.investment) console.log(paint("dim", ` $ evidence: "${ev.investment.text.slice(0, 220)}"`)); } | |
| 267 | + console.log(`\n${paint("bold", `STRUCTURED DATA ${t.records.length} record(s)`)}`); for (const r of t.records.slice(0, 20)) console.log(` ${r.kind.padEnd(11)} ${r.key} · certainty ${r.certainty ?? "—"} · ${Object.keys(r.data).filter((k) => r.data[k] != null && !k.startsWith("_")).slice(0, 12).join(", ")}`); | |
| 268 | + console.log(`\n${paint("bold", `EXTRACTED ENTITIES ${t.entities.length} · valid ${t.validation.valid} · rejected ${t.validation.rejected}`)}`); for (const i of t.validation.issues.slice(0, 20)) console.log(paint(i.level === "error" ? "red" : "yellow", ` ${i.level} ${i.key}${i.field ? "." + i.field : ""}: ${i.message}`)); | |
| 269 | + console.log(`\n${paint("bold", `EXTRACTED CLAIMS ${t.claims.length}`)}`); for (const c of t.claims) console.log(` ${c.field.padEnd(14)} ${String(c.value).padStart(10)} ${c.unit} · scope ${paint(["building", "facility", "campus"].includes(c.scope) ? "green" : "yellow", c.scope)} (${c.scopeReason}) · ${c.semantics ?? "—"}${c.evidence ? paint("dim", `\n "${c.evidence.text.slice(0, 200)}"`) : paint("red", "\n no supporting sentence")}`); | |
| 270 | + console.log(`\n${paint("bold", `MATCH CANDIDATES ${t.matches.length} facility record(s)`)}`); for (const m of t.matches) { console.log(` ${m.entityKey} → ${m.how}${m.matchedId ? ` ${m.matchedId}` : ""}`); for (const c of m.candidates.slice(0, 5)) console.log(paint("dim", ` ${c.score.toFixed(3)} ${c.name} (${c.operatorName ?? "—"}, ${c.city ?? "—"}) ${c.reasons.join(" ")}`)); } | |
| 271 | + if (t.reconciliation) { const r = t.reconciliation; console.log(`\n${paint("bold", "RECONCILIATION (dry run)")} created ${r.created} · updated ${r.updated} · unchanged ${r.unchanged} · merged ${r.merged} · pending ${r.pendingMatches} · rejected ${r.rejected} · events ${r.events} · provenance ${r.provenanceRows} · claims ${r.claims} (${r.unscopedClaims} unscoped) · flags ${r.qualityFlags} · projects vetoed ${r.projectsVetoed}`); console.log(`\n${paint("bold", "RESULTING DATABASE CHANGES")}`); for (const c of r.changes.slice(0, 20)) console.log(` ${JSON.stringify(c).slice(0, 220)}`); for (const ref of r.refs.slice(0, 20)) console.log(paint("dim", ` ref ${ref.type} ${ref.id}`)); } | |
| 272 | + if (t.error) console.log(paint("red", `\nERROR ${t.error}`)); | |
| 273 | +} | |
| 274 | + | |
| 257 | 275 | function help(): number { |
| 258 | 276 | console.log(readFileSync(new URL(import.meta.url), "utf8").split("\n").slice(1, 22).map((l) => l.replace(/^ \*\s?/, "")).join("\n")); |
| 259 | 277 | return 0; |
@@ -284,6 +302,11 @@ async function main(): Promise<number> { | ||
| 284 | 302 | case "scheduler": return cmdScheduler(a); |
| 285 | 303 | case "clickhouse-init": await ensureClickHouse(); console.log("clickhouse tables ensured"); return 0; |
| 286 | 304 | case "validate": return cmdValidateYaml(a); |
| 305 | + case "trace": { const key = a.positional[0]; if (!key) throw new Error("usage: dci trace <doc-id-or-url> [--live] [--json]"); await registerAllConnectors(); const t = await traceDocument(key, { live: bool(a, "live") }); if (bool(a, "json")) { console.log(JSON.stringify(t, null, 2)); return t.error ? 1 : 0; } printTrace(t); return t.error ? 1 : 0; } | |
| 306 | + case "quarantine": { const id = a.positional[0]; const on = (a.positional[1] ?? "on") !== "off"; if (!id) throw new Error("usage: dci quarantine <connector-id> [on|off]"); await getDb().execute(sql`update connectors set quarantine = ${on}, updated_at = now() where id = ${id}`); console.log(`${id}: quarantine ${on ? "ON (ingest rolled back, preview only)" : "OFF"}`); return 0; } | |
| 307 | + case "quality": { const r = await qualitySweep(); console.log(JSON.stringify(r, null, 2)); return 0; } | |
| 308 | + case "snapshot": { const r = await snapshotAndCheck(); console.log(JSON.stringify(r, null, 2)); return 0; } | |
| 309 | + case "gaps": { const r = await dataGaps(); console.log(table(Object.entries(r).map(([k, v]) => ({ gap: k, count: v })), ["gap", "count"])); return 0; } | |
| 287 | 310 | case "worker": { const { startWorker } = await import("./main.js"); const w = await startWorker(); await new Promise<void>((resolve) => { const stop = () => void w.stop().then(resolve); process.once("SIGINT", stop); process.once("SIGTERM", stop); }); return 0; } |
| 288 | 311 | case "help": case "--help": case "-h": return help(); |
| 289 | 312 | default: console.error(paint("red", `unknown command "${a.cmd}"`)); help(); return 2; |
modified
apps/worker/src/configs.ts
+38 −8
@@ -3,8 +3,8 @@ | ||
| 3 | 3 | * GenericConnector) and mirror them into the `connectors` + `sources` tables for the admin UI / scheduler. |
| 4 | 4 | */ |
| 5 | 5 | import { existsSync } from "node:fs"; |
| 6 | −import { GenericConnector, getImplementation, loadConnectorConfigs, type Connector, type ConnectorConfig } from "@dci/connectors"; | |
| 7 | −import { stableId } from "@dci/core"; | |
| 6 | +import { GenericConnector, getImplementation, getParser, loadConnectorConfigs, type Connector, type ConnectorConfig } from "@dci/connectors"; | |
| 7 | +import { sha256, stableId } from "@dci/core"; | |
| 8 | 8 | import { getDb, connectors as connectorsTable, sources as sourcesTable, sql, eq, type Db } from "@dci/db"; |
| 9 | 9 | import { getEnv } from "./env.js"; |
| 10 | 10 | |
@@ -13,6 +13,23 @@ export interface LoadedConnector { | ||
| 13 | 13 | connector: Connector; |
| 14 | 14 | /** stableId("source", connectorId) — one source row per connector */ |
| 15 | 15 | sourceId: string; |
| 16 | + /** | |
| 17 | + * Effective extractor version = the YAML `parserVersion` + the names and versions of every parser the extractors | |
| 18 | + * reference (+ the implementation's own version). Bumping a parser's `version` therefore invalidates every cached | |
| 19 | + * extraction of every connector that uses it — `shouldSkipExtraction` compares this, not the YAML string alone. | |
| 20 | + */ | |
| 21 | + extractorVersion: string; | |
| 22 | +} | |
| 23 | + | |
| 24 | +export function effectiveExtractorVersion(cfg: ConnectorConfig, connector: Connector): string { | |
| 25 | + const parts: string[] = []; | |
| 26 | + for (const ex of Object.values(cfg.extractors ?? {})) { | |
| 27 | + if (!ex.parser) continue; | |
| 28 | + try { const p = getParser(ex.parser); parts.push(`${p.name}@${p.version}`); } catch { parts.push(`${ex.parser}@?`); } | |
| 29 | + } | |
| 30 | + if (cfg.implementation) parts.push(`impl:${cfg.implementation}@${connector.parserVersion}`); | |
| 31 | + parts.sort(); | |
| 32 | + return parts.length ? `${cfg.parserVersion}+${sha256(parts.join("|")).slice(0, 10)}` : cfg.parserVersion; | |
| 16 | 33 | } |
| 17 | 34 | |
| 18 | 35 | export function sourceIdFor(connectorId: string): string { return stableId("source", connectorId); } |
@@ -34,7 +51,7 @@ export function loadAllConnectors(opts: { reload?: boolean; dir?: string } = {}) | ||
| 34 | 51 | const dir = opts.dir ?? getEnv().configDir; |
| 35 | 52 | const out = new Map<string, LoadedConnector>(); |
| 36 | 53 | if (existsSync(dir)) { |
| 37 | − for (const cfg of loadConnectorConfigs(dir)) out.set(cfg.id, { cfg, connector: buildConnector(cfg), sourceId: sourceIdFor(cfg.id) }); | |
| 54 | + for (const cfg of loadConnectorConfigs(dir)) { const connector = buildConnector(cfg); out.set(cfg.id, { cfg, connector, sourceId: sourceIdFor(cfg.id), extractorVersion: effectiveExtractorVersion(cfg, connector) }); } | |
| 38 | 55 | } |
| 39 | 56 | cache = out; |
| 40 | 57 | return [...out.values()]; |
@@ -60,22 +77,22 @@ export interface SyncResult { connectors: number; sources: number; disabledInDb: | ||
| 60 | 77 | export async function syncConnectorsToDb(loaded: LoadedConnector[] = loadAllConnectors(), db: Db = getDb()): Promise<SyncResult> { |
| 61 | 78 | let n = 0, s = 0; |
| 62 | 79 | const now = new Date().toISOString(); |
| 63 | − for (const { cfg, connector, sourceId } of loaded) { | |
| 80 | + for (const { cfg, sourceId, extractorVersion } of loaded) { | |
| 64 | 81 | const config = JSON.parse(JSON.stringify(cfg)) as Record<string, unknown>; |
| 65 | 82 | await db |
| 66 | 83 | .insert(connectorsTable) |
| 67 | − .values({ id: cfg.id, sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: connector.parserVersion, schedule: cfg.schedule }) | |
| 84 | + .values({ id: cfg.id, sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: extractorVersion, schedule: cfg.schedule }) | |
| 68 | 85 | .onConflictDoUpdate({ |
| 69 | 86 | target: connectorsTable.id, |
| 70 | − set: { sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: connector.parserVersion, schedule: cfg.schedule, updatedAt: now }, | |
| 87 | + set: { sourceName: cfg.name, domain: cfg.domain, kind: cfg.kind, mode: cfg.mode, enabled: cfg.enabled, config, parserVersion: extractorVersion, schedule: cfg.schedule, updatedAt: now }, | |
| 71 | 88 | }); |
| 72 | 89 | n++; |
| 73 | 90 | await db |
| 74 | 91 | .insert(sourcesTable) |
| 75 | − .values({ id: sourceId, connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null }) | |
| 92 | + .values({ id: sourceId, connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null, redistribution: redistributionOf(cfg.license), attributionRequired: attributionRequiredOf(cfg.license, cfg.attribution) }) | |
| 76 | 93 | .onConflictDoUpdate({ |
| 77 | 94 | target: sourcesTable.id, |
| 78 | − set: { connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null, updatedAt: now }, | |
| 95 | + set: { connectorId: cfg.id, name: cfg.name, domain: cfg.domain, kind: cfg.kind, priority: cfg.priority, url: cfg.homepage ?? `https://${cfg.domain.replace(/^https?:\/\//, "")}`, license: cfg.license ?? null, attribution: cfg.attribution ?? null, notes: cfg.notes ?? null, redistribution: redistributionOf(cfg.license), attributionRequired: attributionRequiredOf(cfg.license, cfg.attribution), updatedAt: now }, | |
| 79 | 96 | }); |
| 80 | 97 | s++; |
| 81 | 98 | } |
@@ -94,3 +111,16 @@ export async function syncConnectorsToDb(loaded: LoadedConnector[] = loadAllConn | ||
| 94 | 111 | } |
| 95 | 112 | return { connectors: n, sources: s, disabledInDb }; |
| 96 | 113 | } |
| 114 | + | |
| 115 | +/** Redistribution policy derived from the declared licence (drives /download and the API `sources` block). */ | |
| 116 | +export function redistributionOf(license: string | null | undefined): "allowed" | "attribution" | "restricted" | "unknown" { | |
| 117 | + const l = (license ?? "").toLowerCase(); | |
| 118 | + if (!l) return "unknown"; | |
| 119 | + if (/cc0|public domain|pddl|open government|ogl|us government|unrestricted/.test(l)) return "allowed"; | |
| 120 | + if (/odbl|cc[- ]by|odc-by|attribution|mit|apache|peeringdb|open data/.test(l)) return "attribution"; | |
| 121 | + if (/all rights reserved|proprietary|copyright|terms of (use|service)|facts only|no redistribution|non-?commercial|nc\b/.test(l)) return "restricted"; | |
| 122 | + return "unknown"; | |
| 123 | +} | |
| 124 | +export function attributionRequiredOf(license: string | null | undefined, attribution: string | null | undefined): boolean { | |
| 125 | + return redistributionOf(license) === "attribution" || !!attribution; | |
| 126 | +} | |
modified
apps/worker/src/connectors/news/article-parser.ts
+7 −2
@@ -131,7 +131,7 @@ export function buildRecords(args: { connectorId: string; url: string; a: Announ | ||
| 131 | 131 | mw: a.mwAll, status: statusForEvent, statusInferred: a.status, money: a.money, |
| 132 | 132 | text: params.keepText ? content.text.slice(0, 20_000) : null, |
| 133 | 133 | operators: a.operators.map((o) => o.name), countries: country ? [country] : [], cities: city ? [city] : [], |
| 134 | − relevance: a.relevance, extractor: EXTRACTOR_VERSION, ...(args.extraEvent ?? {}), | |
| 134 | + relevance: a.relevance, extractor: EXTRACTOR_VERSION, projectClass: a.classification.class, isAi: a.aiEvidence === "confirmed" || a.aiEvidence === "likely", ...(args.extraEvent ?? {}), | |
| 135 | 135 | }, |
| 136 | 136 | methods: { title: content.titleMethod, publishedAt: content.publishedMethod, summary: content.summaryMethod, mw: "regex:mw_v1", eventType: a.methods.pageType ?? "classify", operators: "lexicon:operator", location: a.methods.location ?? "none" }, |
| 137 | 137 | }; |
@@ -147,7 +147,12 @@ export function buildRecords(args: { connectorId: string; url: string; a: Announ | ||
| 147 | 147 | data: { |
| 148 | 148 | name: a.projectName, operatorName: a.operator?.name ?? null, city, regionName: a.location?.region ?? null, countryIso2: country, |
| 149 | 149 | status: a.status, announcedOn: content.published, expectedOpening: a.expectedOpening, plannedMw: a.headlineMw, investmentUsd: a.investmentUsd, |
| 150 | − acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url, ...(args.extraProject ?? {}), | |
| 150 | + acreage: a.acreage, phaseCount: a.phaseCount, description: description || null, sourceUrl: url, | |
| 151 | + projectClass: a.classification.class, evidenceLevel: a.classification.evidence.strength, | |
| 152 | + capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics, | |
| 153 | + investmentCurrency: a.money?.currency ?? null, investmentOriginal: a.money?.amount ?? null, | |
| 154 | + claimContext: { ...(a.evidence.plannedMw ? { plannedMw: a.evidence.plannedMw.text } : {}), ...(a.evidence.investment ? { investmentUsd: a.evidence.investment.text } : {}) }, | |
| 155 | + aiEvidence: a.aiEvidence, ...(args.extraProject ?? {}), | |
| 151 | 156 | }, |
| 152 | 157 | methods: { |
| 153 | 158 | name: a.methods.name ?? "title", operatorName: a.methods.operatorName ?? "none", city: a.methods.location ?? "none", regionName: a.methods.location ?? "none", countryIso2: a.methods.location ?? "none", |
modified
apps/worker/src/connectors/news/extract-project.test.ts
+39 −0
@@ -406,3 +406,42 @@ describe("news_edgar_fts_v1", () => { | ||
| 406 | 406 | expect(r.data).toMatchObject({ title: "Riot Platforms, Inc.: Form 8-K (EX-99.2)", publishedAt: "2026-08-10", operators: ["Riot Platforms"], countries: ["US"], cities: ["Castle Rock"], text: null }); |
| 407 | 407 | }); |
| 408 | 408 | }); |
| 409 | + | |
| 410 | +describe("announcement classifier veto (2026-09 upgrade)", () => { | |
| 411 | + const body = (extra: string) => `${extra} The company operates more than 13 GW of capacity across its global portfolio, with campuses in Northern Virginia, Dallas and Phoenix. Its newest Lancaster campus will deliver 500 MW at full build-out.`; | |
| 412 | + it("never turns an executive appointment into a project, whatever MW the body quotes", () => { | |
| 413 | + const a = extractAnnouncement("STACK Infrastructure Appoints Matt VanderZanden as Chief Executive Officer, STACK Americas", body("STACK Infrastructure today announced the appointment of Matt VanderZanden as CEO of STACK Americas."), { publishedAt: "2025-12-12" }); | |
| 414 | + expect(a.classification.class).toBe("EXECUTIVE_APPOINTMENT"); | |
| 415 | + expect(qualifiesAsProject(a)).toBe(false); | |
| 416 | + }); | |
| 417 | + it("keeps portfolio totals out of the headline figure and scopes the site figure", () => { | |
| 418 | + const a = extractAnnouncement("STACK breaks ground on new Lancaster data center campus", body("STACK Infrastructure broke ground today on its Lancaster, Texas campus."), { publishedAt: "2026-05-01" }); | |
| 419 | + expect(a.classification.class).toBe("CONSTRUCTION_START"); | |
| 420 | + expect(a.headlineMw).toBe(500); | |
| 421 | + expect(a.capacityScope).toBe("campus"); | |
| 422 | + expect(a.evidence.plannedMw?.text).toMatch(/500 MW/); | |
| 423 | + expect(qualifiesAsProject(a)).toBe(true); | |
| 424 | + }); | |
| 425 | + it("does not locate a project at the company's headquarters", () => { | |
| 426 | + const a = extractAnnouncement("Denver-based Vantage plans 192 MW data center campus in Abilene, Texas", "Denver-based Vantage Data Centers said it will build a 192 MW campus in Abilene, Texas, with the first building expected in 2027.", { publishedAt: "2026-03-01" }); | |
| 427 | + expect(a.location?.city).toBe("Abilene"); | |
| 428 | + expect(a.hqGuarded).toBe(true); | |
| 429 | + }); | |
| 430 | + it("classifies PPA, financing and market-research headlines as non-physical", () => { | |
| 431 | + for (const [title, cls] of [ | |
| 432 | + ["Ormat Technologies Signs 20-Year PPA with Switch for ~13 MW of Carbon-Free Geothermal Capacity to Power Data Centers", "POWER_AGREEMENT"], | |
| 433 | + ["STACK Secures $1.3B Financing for Development", "FINANCING"], | |
| 434 | + ["Hyperscale Data Center Market to Surpass USD 624.2 Billion by 2035", "GENERAL_COMPANY_NEWS"], | |
| 435 | + ] as Array<[string, string]>) { | |
| 436 | + const a = extractAnnouncement(title, `${title}. Data center capacity of 300 MW is mentioned in passing.`, { publishedAt: "2026-01-02" }); | |
| 437 | + expect(a.classification.class).toBe(cls); | |
| 438 | + expect(qualifiesAsProject(a)).toBe(false); | |
| 439 | + } | |
| 440 | + }); | |
| 441 | + it("grades AI evidence from explicit wording only", () => { | |
| 442 | + const a = extractAnnouncement("Crusoe to build 1.2 GW AI factory campus in Abilene, Texas", "Crusoe will develop a purpose-built AI campus in Abilene, Texas with NVIDIA GB200 systems.", { publishedAt: "2026-01-02" }); | |
| 443 | + expect(a.aiEvidence).toBe("confirmed"); | |
| 444 | + const b = extractAnnouncement("Equinix opens DA11 colocation data center in Dallas", "Equinix opened DA11, a colocation facility in Dallas, Texas.", { publishedAt: "2026-01-02" }); | |
| 445 | + expect(b.aiEvidence).toBe("unknown"); | |
| 446 | + }); | |
| 447 | +}); | |
modified
apps/worker/src/connectors/news/extract-project.ts
+56 −8
@@ -1,5 +1,5 @@ | ||
| 1 | −import type { EventType, FacilityStatus, PageType } from "@dci/core"; | |
| 2 | −import { PIPELINE_STATUSES, cleanText, parseAllMw, parseAreaHa, parseMoney, parsePartialDate, partialDateSortKey, round } from "@dci/core"; | |
| 1 | +import type { EventType, FacilityStatus, PageType, ClaimScope, Evidence, ProjectClassification, AiEvidence } from "@dci/core"; | |
| 2 | +import { PIPELINE_STATUSES, cleanText, parseAllMw, parseAreaHa, parseMoney, parsePartialDate, partialDateSortKey, round, classifyProjectEvent, classifyScope, classifyCapacitySemantics, classifyInvestmentSemantics, classifyAiEvidence, findEvidence } from "@dci/core"; | |
| 3 | 3 | import { classifyPage, eventTypeFor, isDataCenterRelevant } from "@dci/connectors"; |
| 4 | 4 | import { detectOperators, primaryOperator, type OperatorHit } from "./operators-lexicon.js"; |
| 5 | 5 | import { detectLocation, type LocationHit } from "./locations-lexicon.js"; |
@@ -39,9 +39,33 @@ export interface Announcement { | ||
| 39 | 39 | explicitName: string | null; |
| 40 | 40 | projectName: string; |
| 41 | 41 | methods: Record<string, string>; |
| 42 | + /** what the announcement is about (people / money / deal / build…) — decides whether a project may exist at all */ | |
| 43 | + classification: ProjectClassification; | |
| 44 | + /** supporting sentences for the headline figures (no sentence → no claim) */ | |
| 45 | + evidence: { plannedMw: Evidence | null; investment: Evidence | null }; | |
| 46 | + capacityScope: ClaimScope; | |
| 47 | + capacitySemantics: string | null; | |
| 48 | + investmentScope: ClaimScope; | |
| 49 | + investmentSemantics: string | null; | |
| 50 | + aiEvidence: AiEvidence; | |
| 51 | + /** a "City, State-based" / "headquartered in" mention was removed before locating the project */ | |
| 52 | + hqGuarded: boolean; | |
| 42 | 53 | } |
| 43 | 54 | |
| 44 | −export const EXTRACTOR_VERSION = "news_v1"; | |
| 55 | +export const EXTRACTOR_VERSION = "news_v2"; | |
| 56 | + | |
| 57 | +/** | |
| 58 | + * Company-domicile mentions must not locate a project: "Denver-based Vantage plans a campus in Abilene, Texas" is in | |
| 59 | + * Abilene. Strips "<Place>-based", "headquartered in <Place>" and "<company>, based in <Place>" before location detection. | |
| 60 | + */ | |
| 61 | +export function stripHqMentions(text: string): { text: string; stripped: boolean } { | |
| 62 | + let stripped = false; | |
| 63 | + const out = text | |
| 64 | + .replace(/\b([A-Z][a-zA-Z.'-]+(?:,? [A-Z][a-zA-Z.'-]+){0,3})[- ]based\b/g, () => { stripped = true; return " "; }) | |
| 65 | + .replace(/\b(?:headquartered|HQ'?d|domiciled) in ([A-Z][a-zA-Z.'-]+(?:,? [A-Z][a-zA-Z.'-]+){0,3})/g, () => { stripped = true; return " "; }) | |
| 66 | + .replace(/\b(Inc\.?|LLC|Ltd\.?|Limited|Group|plc|Corp\.?|Corporation|company|firm|developer|operator|provider|startup|REIT|fund)[,]? (?:which is |that is )?based in ([A-Z][a-zA-Z.'-]+(?:,? [A-Z][a-zA-Z.'-]+){0,3})/g, (_m, w: string) => { stripped = true; return `${w} `; }); | |
| 67 | + return { text: out, stripped }; | |
| 68 | +} | |
| 45 | 69 | |
| 46 | 70 | const FULL_MONTHS: Record<string, string> = { january: "Jan", february: "Feb", march: "Mar", april: "Apr", june: "Jun", july: "Jul", august: "Aug", september: "Sep", october: "Oct", november: "Nov", december: "Dec" }; |
| 47 | 71 | /** |
@@ -405,12 +429,17 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op | ||
| 405 | 429 | const exp = parseExpectedOpening(`${title}. ${body}`, { minYear, maxYear: now.getFullYear() + 15, after: opts.publishedAt ?? null }); |
| 406 | 430 | if (exp.value) methods.expectedOpening = exp.method; |
| 407 | 431 | |
| 408 | − // location: title first, then lead, then body; a title hit without a city is completed from the lead when consistent | |
| 409 | − let location = detectLocation(title); | |
| 410 | − const leadLoc = detectLocation(text.slice(0, 1500)); | |
| 411 | − if (!location) location = leadLoc ?? detectLocation(body); | |
| 432 | + // location: title first, then lead, then body; a title hit without a city is completed from the lead when consistent. | |
| 433 | + // Company domiciles ("Denver-based", "headquartered in Ashburn") are removed first — they are not the project's location. | |
| 434 | + const hqTitle = stripHqMentions(title); | |
| 435 | + const hqLead = stripHqMentions(text.slice(0, 1500)); | |
| 436 | + const hqBody = stripHqMentions(body); | |
| 437 | + let location = detectLocation(hqTitle.text); | |
| 438 | + const leadLoc = detectLocation(hqLead.text); | |
| 439 | + if (!location) location = leadLoc ?? detectLocation(hqBody.text); | |
| 412 | 440 | else if (!location.city && leadLoc?.city && (!location.country || leadLoc.country === location.country) && (!location.region || !leadLoc.region || leadLoc.region === location.region)) location = { ...leadLoc, method: `${location.method}+${leadLoc.method}` }; |
| 413 | 441 | if (location) { methods.location = location.method; } |
| 442 | + const hqGuarded = hqTitle.stripped || hqLead.stripped || hqBody.stripped; | |
| 414 | 443 | |
| 415 | 444 | const operators = detectOperators(`${title}\n${body}`); |
| 416 | 445 | let operator = primaryOperator(title, body); |
@@ -429,7 +458,22 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op | ||
| 429 | 458 | // people / deals / money / market headlines are plain news whatever the body's vocabulary says |
| 430 | 459 | if (nonProject && !["acquisition", "closure", "cloud_region", "financial_disclosure"].includes(pageType)) { pageType = "press_release"; eventType = "news"; methods.pageType = `${methods.pageType}+veto:title`; } |
| 431 | 460 | |
| 432 | − return { title, pageType, eventType, relevance, leadRelevance: leadRelevanceOf(title, text), titleSignal: titleSignalOf(title), nonProjectTitle: nonProject, mwAll: mwBody, headlineMw, money, investmentUsd, investmentUsdApprox, acreage, phaseCount, expectedOpening: exp.value, status: st.status, location, operators, operator, explicitName, projectName: gen.name, methods }; | |
| 461 | + // classification BEFORE anything becomes a project: people / money / deal / market headlines never do | |
| 462 | + const classification = classifyProjectEvent({ title, lead: head, hasExplicitName: !!explicitName, hasOperator: !!operator, hasLocation: !!(location?.city || location?.region || location?.country), status: st.status }); | |
| 463 | + // evidence + scope + semantics of the headline figures (the sentence decides; no sentence → the claim is not site-scoped) | |
| 464 | + const campusHint: ClaimScope = /\b(campus|park|complex|hub)\b/i.test(`${explicitName ?? ""} ${title}`) ? "campus" : "facility"; | |
| 465 | + const mwEvidence = headlineMw != null ? findEvidence(`${title}. ${body}`, headlineMw, "mw") : null; | |
| 466 | + const mwScope = headlineMw != null ? (mwEvidence ? classifyScope(mwEvidence.text, campusHint) : { scope: "unknown" as ClaimScope, reason: "no-evidence" }) : { scope: "unknown" as ClaimScope, reason: "none" }; | |
| 467 | + const mwSem = mwEvidence ? classifyCapacitySemantics(mwEvidence.text) : { predicate: null, reason: "none" }; | |
| 468 | + const invEvidence = money ? findEvidence(`${title}. ${body}`, money.amount, "usd") : null; | |
| 469 | + const invSem = money ? classifyInvestmentSemantics(invEvidence?.text ?? title, campusHint) : { predicate: "project_investment_usd" as const, scope: "unknown" as ClaimScope, reason: "none" }; | |
| 470 | + const ai = classifyAiEvidence(`${title}. ${head}`); | |
| 471 | + methods.classification = classification.reasons.join("+") || "none"; | |
| 472 | + if (mwEvidence) methods.plannedMwEvidence = "sentence"; if (invEvidence) methods.investmentEvidence = "sentence"; | |
| 473 | + return { | |
| 474 | + title, pageType, eventType, relevance, leadRelevance: leadRelevanceOf(title, text), titleSignal: titleSignalOf(title), nonProjectTitle: nonProject, mwAll: mwBodySite.length ? mwBodySite : mwBody.slice(0, 3), headlineMw, money, investmentUsd, investmentUsdApprox, acreage, phaseCount, expectedOpening: exp.value, status: st.status, location, operators, operator, explicitName, projectName: gen.name, methods, | |
| 475 | + classification, evidence: { plannedMw: mwEvidence, investment: invEvidence }, capacityScope: mwScope.scope, capacitySemantics: mwSem.predicate ?? (headlineMw != null ? "planned_power_mw" : null), investmentScope: invSem.scope, investmentSemantics: invSem.predicate, aiEvidence: ai.level, hqGuarded, | |
| 476 | + }; | |
| 433 | 477 | } |
| 434 | 478 | |
| 435 | 479 | /** Portfolio / company-wide context around a MW figure — not the size of the announced site. */ |
@@ -467,6 +511,10 @@ const PROJECT_STATUSES: readonly FacilityStatus[] = [...PIPELINE_STATUSES, "expa | ||
| 467 | 511 | export function qualifiesAsProject(a: Announcement, opts: { minMw?: number; minInvestmentUsd?: number; minAcres?: number; lenient?: boolean } = {}): boolean { |
| 468 | 512 | if (a.relevance !== "strong") return false; |
| 469 | 513 | if (a.nonProjectTitle) return false; |
| 514 | + // the announcement classifier: only physical development classes with (name or operator) + location + development verb | |
| 515 | + if (!a.classification.physical) return false; | |
| 516 | + if (!opts.lenient && !a.classification.mayCreateProject) return false; | |
| 517 | + if (opts.lenient && a.classification.evidence.strength === "none") return false; | |
| 470 | 518 | if (!a.status || !PROJECT_STATUSES.includes(a.status)) return false; |
| 471 | 519 | if (!PROJECT_PAGE_TYPES.includes(a.pageType)) return false; |
| 472 | 520 | // An announcement names its subject up front: data-center vocabulary in the lead, and a title that mentions the |
modified
apps/worker/src/context.ts
+1 −1
@@ -148,7 +148,7 @@ export function createRunContext(loaded: LoadedConnector, opts: CreateContextOpt | ||
| 148 | 148 | |
| 149 | 149 | provenance(url: string, extra: Partial<Provenance> = {}): Provenance { |
| 150 | 150 | const now = new Date().toISOString(); |
| 151 | − return { sourceId: loaded.sourceId, connectorId: cfg.id, url, firstObserved: now, lastObserved: now, retrievedAt: now, confidence: SOURCE_KIND_BASE[cfg.kind] ?? "moderate", extractorVersion: loaded.connector.parserVersion, ...extra }; | |
| 151 | + return { sourceId: loaded.sourceId, connectorId: cfg.id, url, firstObserved: now, lastObserved: now, retrievedAt: now, confidence: SOURCE_KIND_BASE[cfg.kind] ?? "moderate", extractorVersion: loaded.extractorVersion, ...extra }; | |
| 152 | 152 | }, |
| 153 | 153 | |
| 154 | 154 | async fetch(url: string, o: FetchOptions = {}): Promise<RawDocument> { |
modified
apps/worker/src/documents.ts
+3 −1
@@ -225,7 +225,7 @@ export interface VersionResult { versionId: string; diff: LineDiff | null; signi | ||
| 225 | 225 | * Record a new content version. `prev` is the document row as it was BEFORE recordFetch (old hash / storage key). |
| 226 | 226 | * The old text is loaded from object storage when available; `detectedChanges` come from the ingest layer. |
| 227 | 227 | */ |
| 228 | −export async function recordVersion(prev: DocumentRow, raw: RawDocument, newHash: string, storageKey: string | null, detectedChanges: DetectedChange[], opts: { dryRun?: boolean; now?: Date; runId?: string } = {}): Promise<VersionResult> { | |
| 228 | +export async function recordVersion(prev: DocumentRow, raw: RawDocument, newHash: string, storageKey: string | null, detectedChanges: DetectedChange[], opts: { dryRun?: boolean; now?: Date; runId?: string; extractorVersion?: string } = {}): Promise<VersionResult> { | |
| 229 | 229 | const now = opts.now ?? new Date(); |
| 230 | 230 | let diff: LineDiff | null = null; |
| 231 | 231 | if (prev.storageKey && prev.contentHash && prev.contentHash !== newHash) { |
@@ -251,6 +251,8 @@ export async function recordVersion(prev: DocumentRow, raw: RawDocument, newHash | ||
| 251 | 251 | diffSummary: diff ? { addedCount: diff.addedCount, removedCount: diff.removedCount, ratio: Math.round(diff.ratio * 10_000) / 10_000, added: diff.added.slice(0, 60), removed: diff.removed.slice(0, 60) } : null, |
| 252 | 252 | detectedChanges: detectedChanges as unknown as Array<Record<string, unknown>>, |
| 253 | 253 | significance, |
| 254 | + runId: opts.runId ?? null, | |
| 255 | + extractorVersion: opts.extractorVersion ?? null, | |
| 254 | 256 | }) |
| 255 | 257 | .onConflictDoNothing(); |
| 256 | 258 | if (prev.contentHash) { |
modified
apps/worker/src/health-http.ts
+10 −0
@@ -15,6 +15,8 @@ export interface HealthHttpOptions { | ||
| 15 | 15 | /** refresh snapshot gauges (queue depth, budgets…) right before rendering */ |
| 16 | 16 | beforeScrape?: () => Promise<void>; |
| 17 | 17 | log?: (msg: string) => void; |
| 18 | + /** extra JSON routes keyed by path prefix ("/trace" matches "/trace/<id>") */ | |
| 19 | + routes?: Record<string, (url: URL) => Promise<{ status: number; body: unknown }>>; | |
| 18 | 20 | } |
| 19 | 21 | |
| 20 | 22 | export function startHealthHttp(o: HealthHttpOptions): Server { |
@@ -35,6 +37,14 @@ export function startHealthHttp(o: HealthHttpOptions): Server { | ||
| 35 | 37 | res.end(renderMetrics()); |
| 36 | 38 | return; |
| 37 | 39 | } |
| 40 | + for (const [prefix, handler] of Object.entries(o.routes ?? {})) { | |
| 41 | + if (url.pathname === prefix || url.pathname.startsWith(`${prefix}/`)) { | |
| 42 | + const r = await handler(url); | |
| 43 | + res.writeHead(r.status, { "content-type": "application/json", "cache-control": "no-store" }); | |
| 44 | + res.end(JSON.stringify(r.body)); | |
| 45 | + return; | |
| 46 | + } | |
| 47 | + } | |
| 38 | 48 | res.writeHead(404, { "content-type": "application/json" }); |
| 39 | 49 | res.end(JSON.stringify({ error: "not found" })); |
| 40 | 50 | } catch (e) { |
modified
apps/worker/src/ingest/campuses.ts
+2 −0
@@ -30,6 +30,8 @@ export async function resolveCampus(tx: Tx, ctx: IngestContext, c: CampusInput): | ||
| 30 | 30 | } |
| 31 | 31 | if (!id && c.externalIds) { |
| 32 | 32 | for (const [k, v] of Object.entries(c.externalIds)) { |
| 33 | + // allowlist: only namespaces that name one campus may link records | |
| 34 | + if (v == null || v === "" || !/^(osm|wikidata|peeringdb_campus|[a-z0-9]+_campus(_slug|_code)?)$/.test(k) || /^(operator|owner|state|region|country|postal|market|source)_/.test(k)) continue; | |
| 33 | 35 | const r = await tx.execute(sql`select id from campuses where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`); |
| 34 | 36 | if (r[0]) { |
| 35 | 37 | id = String(r[0].id); |
added
apps/worker/src/ingest/claims.ts
+218 −0
@@ -0,0 +1,218 @@ | ||
| 1 | +/** | |
| 2 | + * Claim store + quality flags + winner bookkeeping (docs/CLAIMS.md). | |
| 3 | + * | |
| 4 | + * Every numerical figure observed about a subject becomes a `claims` row carrying its scope, semantics, supporting | |
| 5 | + * sentence, parser and field-level authority tier. The ingest layer then asks `resolveCapacityColumn()` which claim | |
| 6 | + * may populate a column: only site-scoped claims (building / facility / campus) that pass the deterministic sanity | |
| 7 | + * engine; everything else is stored with status `unscoped` / `review` and surfaces in the admin quality dashboard. | |
| 8 | + */ | |
| 9 | +import { sql } from "@dci/db"; | |
| 10 | +import { | |
| 11 | + authorityFieldFor, | |
| 12 | + authorityTier, | |
| 13 | + capacitySanity, | |
| 14 | + investmentSanity, | |
| 15 | + isSiteScope, | |
| 16 | + newId, | |
| 17 | + stableId, | |
| 18 | + TIER_RANK, | |
| 19 | + type AuthorityTier, | |
| 20 | + type CapacityPredicate, | |
| 21 | + type ClaimScope, | |
| 22 | + type ClaimStatus, | |
| 23 | + type Evidence, | |
| 24 | + type InvestmentPredicate, | |
| 25 | + type Provenance, | |
| 26 | + type SanityFlag, | |
| 27 | +} from "@dci/core"; | |
| 28 | +import type { IngestContext, Tx } from "./common.js"; | |
| 29 | + | |
| 30 | +export interface ClaimInput { | |
| 31 | + predicate: string; | |
| 32 | + value?: number | null; | |
| 33 | + valueText?: string | null; | |
| 34 | + unit?: string | null; | |
| 35 | + scope: ClaimScope; | |
| 36 | + scopeReason?: string | null; | |
| 37 | + evidence?: Evidence | null; | |
| 38 | + publishedAt?: string | null; | |
| 39 | + provenance: Provenance; | |
| 40 | + parserName?: string | null; | |
| 41 | + /** override the computed status (e.g. "review" after a sanity flag) */ | |
| 42 | + status?: ClaimStatus; | |
| 43 | + rejectionReason?: string | null; | |
| 44 | +} | |
| 45 | + | |
| 46 | +export interface WrittenClaim { id: string; status: ClaimStatus; tier: AuthorityTier } | |
| 47 | + | |
| 48 | +export function claimId(subjectType: string, subjectId: string, c: { predicate: string; value?: number | null; valueText?: string | null; provenance: Provenance }, url: string): string { | |
| 49 | + return stableId("claim", `${subjectType}|${subjectId}|${c.predicate}|${c.provenance.sourceId}|${url}|${c.value ?? ""}|${c.valueText ?? ""}`); | |
| 50 | +} | |
| 51 | + | |
| 52 | +/** Upsert one claim. Same source + url + predicate with a different value supersedes the earlier claim. */ | |
| 53 | +export async function writeClaim(tx: Tx, ctx: IngestContext, subjectType: string, subjectId: string, c: ClaimInput): Promise<WrittenClaim> { | |
| 54 | + const p = c.provenance; | |
| 55 | + const url = p.url || ctx.doc?.url || ""; | |
| 56 | + const sourceId = p.sourceId || ctx.run.sourceId; | |
| 57 | + const tier = authorityTier({ field: authorityFieldFor(c.predicate), sourceKind: ctx.run.sourceKind, isEstimate: p.isEstimate, method: p.method }); | |
| 58 | + const status: ClaimStatus = c.status ?? (isSiteScope(c.scope) ? "current" : "unscoped"); | |
| 59 | + const id = claimId(subjectType, subjectId, c, url); | |
| 60 | + const observed = p.lastObserved || ctx.now; | |
| 61 | + if (!ctx.run.dryRun) { | |
| 62 | + await tx.execute(sql` | |
| 63 | + insert into claims (id, subject_type, subject_id, predicate, value, value_text, unit, scope, scope_reason, source_id, connector_id, document_id, url, published_at, retrieved_at, | |
| 64 | + confidence, is_estimate, authority_tier, evidence_text, evidence_start, evidence_end, parser_name, parser_version, run_id, status, rejection_reason, first_observed, last_observed) | |
| 65 | + values (${id}, ${subjectType}, ${subjectId}, ${c.predicate}, ${c.value ?? null}, ${c.valueText ?? null}, ${c.unit ?? null}, ${c.scope}, ${c.scopeReason ?? null}, ${sourceId}, ${p.connectorId || ctx.run.connectorId}, | |
| 66 | + ${p.documentId ?? ctx.doc?.documentId ?? null}, ${url}, ${c.publishedAt ?? null}, ${p.retrievedAt || ctx.now}, ${p.confidence ?? "moderate"}, ${!!p.isEstimate}, ${tier}, | |
| 67 | + ${c.evidence?.text ?? null}, ${c.evidence?.start ?? null}, ${c.evidence?.end ?? null}, ${c.parserName ?? p.method ?? null}, ${p.extractorVersion ?? null}, ${ctx.run.runId}, ${status}, ${c.rejectionReason ?? null}, ${p.firstObserved || observed}, ${observed}) | |
| 68 | + on conflict (id) do update set last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, confidence = excluded.confidence, authority_tier = excluded.authority_tier, | |
| 69 | + evidence_text = coalesce(excluded.evidence_text, claims.evidence_text), evidence_start = coalesce(excluded.evidence_start, claims.evidence_start), evidence_end = coalesce(excluded.evidence_end, claims.evidence_end), | |
| 70 | + scope = excluded.scope, scope_reason = excluded.scope_reason, run_id = excluded.run_id, parser_version = excluded.parser_version, | |
| 71 | + status = case when claims.status in ('rejected') then claims.status else excluded.status end, rejection_reason = coalesce(excluded.rejection_reason, claims.rejection_reason)`); | |
| 72 | + // the same source, same page, same predicate now says something else → the older claim is superseded | |
| 73 | + await tx.execute(sql`update claims set status = 'superseded' where subject_type = ${subjectType} and subject_id = ${subjectId} and predicate = ${c.predicate} and source_id = ${sourceId} and url = ${url} and id <> ${id} and status in ('current', 'review')`); | |
| 74 | + } | |
| 75 | + ctx.stats.claims = (ctx.stats.claims ?? 0) + 1; | |
| 76 | + return { id, status, tier }; | |
| 77 | +} | |
| 78 | + | |
| 79 | +/** Review priority 0–100: impact-weighted (MW / money size, low confidence, new country, AI, ambiguity, unexpected change). */ | |
| 80 | +export function reviewPriority(i: { mw?: number | null; investmentUsd?: number | null; confidence?: string | null; newCountry?: boolean; ai?: boolean; ambiguous?: boolean; unexpectedChange?: boolean; severity?: "info" | "warn" | "critical"; homepageVisible?: boolean }): number { | |
| 81 | + let p = 0; | |
| 82 | + const mw = i.mw ?? 0; | |
| 83 | + p += mw >= 1000 ? 40 : mw >= 300 ? 30 : mw >= 100 ? 20 : mw >= 20 ? 10 : mw > 0 ? 4 : 0; | |
| 84 | + const inv = i.investmentUsd ?? 0; | |
| 85 | + p += inv >= 10e9 ? 25 : inv >= 1e9 ? 15 : inv >= 100e6 ? 8 : 0; | |
| 86 | + if (i.confidence === "unverified" || i.confidence === "estimated") p += 10; | |
| 87 | + if (i.newCountry) p += 10; | |
| 88 | + if (i.ai) p += 5; | |
| 89 | + if (i.ambiguous) p += 8; | |
| 90 | + if (i.unexpectedChange) p += 12; | |
| 91 | + if (i.severity === "critical") p += 15; else if (i.severity === "warn") p += 5; | |
| 92 | + if (i.homepageVisible) p += 10; | |
| 93 | + return Math.max(0, Math.min(100, Math.round(p))); | |
| 94 | +} | |
| 95 | + | |
| 96 | +/** Upsert deterministic quality flags for an entity (deduped by entity + code + field). Resolved flags are not reopened for the same key. */ | |
| 97 | +export async function writeQualityFlags(tx: Tx, ctx: IngestContext, entityType: string, entityId: string, flags: SanityFlag[], opts: { claimId?: string | null; priority?: number; details?: Record<string, unknown> } = {}): Promise<number> { | |
| 98 | + let n = 0; | |
| 99 | + for (const f of flags) { | |
| 100 | + if (f.severity === "info") continue; | |
| 101 | + const dedupeKey = `${entityType}|${entityId}|${f.code}|${f.field ?? ""}`; | |
| 102 | + const id = stableId("flag", dedupeKey); | |
| 103 | + const priority = opts.priority ?? reviewPriority({ mw: /mw/.test(f.code) ? f.value : null, investmentUsd: /inv/.test(f.code) ? f.value : null, severity: f.severity }); | |
| 104 | + if (!ctx.run.dryRun) { | |
| 105 | + await tx.execute(sql` | |
| 106 | + insert into quality_flags (id, entity_type, entity_id, claim_id, code, severity, field, message, details, priority, status, run_id, dedupe_key) | |
| 107 | + values (${id}, ${entityType}, ${entityId}, ${opts.claimId ?? null}, ${f.code}, ${f.severity}, ${f.field ?? null}, ${f.message.slice(0, 1000)}, ${JSON.stringify({ ...(opts.details ?? {}), value: f.value ?? null })}::jsonb, ${priority}, 'open', ${ctx.run.runId}, ${dedupeKey}) | |
| 108 | + on conflict (dedupe_key) do update set message = excluded.message, details = excluded.details, priority = greatest(quality_flags.priority, excluded.priority), claim_id = coalesce(excluded.claim_id, quality_flags.claim_id), | |
| 109 | + run_id = excluded.run_id, updated_at = now(), status = case when quality_flags.status = 'dismissed' then 'dismissed' else 'open' end`); | |
| 110 | + } | |
| 111 | + n++; | |
| 112 | + } | |
| 113 | + ctx.stats.qualityFlags = (ctx.stats.qualityFlags ?? 0) + n; | |
| 114 | + return n; | |
| 115 | +} | |
| 116 | + | |
| 117 | +/** Close open flags of the given codes for an entity (the condition no longer holds). */ | |
| 118 | +export async function resolveQualityFlags(tx: Tx, ctx: IngestContext, entityType: string, entityId: string, codes: string[], resolution = "auto: condition cleared"): Promise<void> { | |
| 119 | + if (!codes.length || ctx.run.dryRun) return; | |
| 120 | + await tx.execute(sql`update quality_flags set status = 'resolved', resolution = ${resolution}, resolved_by = 'system', resolved_at = now(), updated_at = now() | |
| 121 | + where entity_type = ${entityType} and entity_id = ${entityId} and status = 'open' and code in ${codes}`); | |
| 122 | +} | |
| 123 | + | |
| 124 | +/** | |
| 125 | + * Mark, per field, the current provenance row whose value equals the stored column value as the winner (and unmark | |
| 126 | + * the others). This is what makes "why does the page show 300 MW?" answerable without a diff. | |
| 127 | + */ | |
| 128 | +export async function markWinners(tx: Tx, ctx: IngestContext, entityType: string, entityId: string, fields: Array<{ field: string; value: unknown }>): Promise<void> { | |
| 129 | + if (ctx.run.dryRun) return; | |
| 130 | + for (const f of fields) { | |
| 131 | + if (f.value == null) continue; | |
| 132 | + const valueJson = JSON.stringify(f.value); | |
| 133 | + await tx.execute(sql`update provenance set is_winner = (value = ${valueJson}::jsonb) where entity_type = ${entityType} and entity_id = ${entityId} and field = ${f.field} and is_current and is_winner <> (value = ${valueJson}::jsonb)`); | |
| 134 | + } | |
| 135 | +} | |
| 136 | + | |
| 137 | +export interface CapacityDecision { | |
| 138 | + /** write the figure into the record's column */ | |
| 139 | + assign: boolean; | |
| 140 | + claim: WrittenClaim; | |
| 141 | + flags: SanityFlag[]; | |
| 142 | + blockedBy: string[]; | |
| 143 | +} | |
| 144 | + | |
| 145 | +export interface CapacityClaimInput { | |
| 146 | + field: "itCapacityMw" | "totalPowerMw" | "plannedPowerMw" | "plannedMw" | "utilityCapacityMw" | "gridConnectionMw" | "ultimateCampusMw"; | |
| 147 | + predicate: CapacityPredicate; | |
| 148 | + value: number; | |
| 149 | + scope: ClaimScope; | |
| 150 | + scopeReason?: string | null; | |
| 151 | + evidence?: Evidence | null; | |
| 152 | + context?: string | null; | |
| 153 | + previous?: number | null; | |
| 154 | + recordScope: "building" | "facility" | "campus" | "project"; | |
| 155 | + campusDesignation?: boolean; | |
| 156 | + publishedAt?: string | null; | |
| 157 | + provenance: Provenance; | |
| 158 | + parserName?: string | null; | |
| 159 | + /** the sentence did not say what kind of MW this is; semantics were defaulted from the record status */ | |
| 160 | + semanticsDefaulted?: boolean; | |
| 161 | +} | |
| 162 | + | |
| 163 | +/** Store a capacity claim, run the sanity engine, write flags and decide whether the column may take the value. */ | |
| 164 | +export async function recordCapacityClaim(tx: Tx, ctx: IngestContext, subjectType: "facility" | "project" | "campus", subjectId: string, i: CapacityClaimInput): Promise<CapacityDecision> { | |
| 165 | + const flags = capacitySanity({ value: i.value, predicate: i.semanticsDefaulted ? null : i.predicate, scope: i.scope, recordScope: i.recordScope, previous: i.previous ?? null, context: i.context ?? i.evidence?.text ?? null, campusDesignation: i.campusDesignation }); | |
| 166 | + const blockedBy = flags.filter((f) => f.blocks).map((f) => f.code); | |
| 167 | + const critical = flags.some((f) => f.severity === "critical"); | |
| 168 | + const status: ClaimStatus = blockedBy.length ? (isSiteScope(i.scope) ? "review" : "unscoped") : critical ? "review" : "current"; | |
| 169 | + const claim = await writeClaim(tx, ctx, subjectType, subjectId, { predicate: i.predicate, value: i.value, unit: "MW", scope: i.scope, scopeReason: i.scopeReason ?? null, evidence: i.evidence ?? null, publishedAt: i.publishedAt ?? null, provenance: i.provenance, parserName: i.parserName ?? null, status, rejectionReason: blockedBy.length ? blockedBy.join(",") : null }); | |
| 170 | + await writeQualityFlags(tx, ctx, subjectType, subjectId, flags.map((f) => ({ ...f, field: f.field ?? i.field })), { claimId: claim.id, priority: reviewPriority({ mw: i.value, severity: flags.some((f) => f.severity === "critical") ? "critical" : flags.some((f) => f.severity === "warn") ? "warn" : "info" }) }); | |
| 171 | + if (!blockedBy.length) { | |
| 172 | + const cleared = ["scope_unknown", "scope_company", "scope_portfolio", "scope_country", "scope_metro", "mw_market_statistic", "mw_invalid"].filter((c) => !flags.some((f) => f.code === c)); | |
| 173 | + await resolveQualityFlags(tx, ctx, subjectType, subjectId, cleared.map((c) => c)); | |
| 174 | + } | |
| 175 | + // a critical (non-blocking) flag such as "> 1 000 MW single site" still assigns — the figure is what the source says — but | |
| 176 | + // stays visible for review; a blocking flag never assigns | |
| 177 | + return { assign: blockedBy.length === 0, claim, flags, blockedBy }; | |
| 178 | +} | |
| 179 | + | |
| 180 | +export interface InvestmentClaimInput { | |
| 181 | + value: number; | |
| 182 | + currency?: string | null; | |
| 183 | + predicate: InvestmentPredicate; | |
| 184 | + scope: ClaimScope; | |
| 185 | + scopeReason?: string | null; | |
| 186 | + evidence?: Evidence | null; | |
| 187 | + context?: string | null; | |
| 188 | + previous?: number | null; | |
| 189 | + recordScope: "facility" | "campus" | "project"; | |
| 190 | + publishedAt?: string | null; | |
| 191 | + provenance: Provenance; | |
| 192 | + parserName?: string | null; | |
| 193 | +} | |
| 194 | + | |
| 195 | +export async function recordInvestmentClaim(tx: Tx, ctx: IngestContext, subjectType: "facility" | "project" | "campus" | "operator", subjectId: string, i: InvestmentClaimInput): Promise<CapacityDecision> { | |
| 196 | + const flags = investmentSanity({ value: i.value, scope: i.scope, predicate: i.predicate, recordScope: i.recordScope, previous: i.previous ?? null, context: i.context ?? i.evidence?.text ?? null }); | |
| 197 | + const blockedBy = flags.filter((f) => f.blocks).map((f) => f.code); | |
| 198 | + const status: ClaimStatus = blockedBy.length ? (isSiteScope(i.scope) && i.predicate === "project_investment_usd" ? "review" : "unscoped") : flags.some((f) => f.severity === "critical") ? "review" : "current"; | |
| 199 | + const claim = await writeClaim(tx, ctx, subjectType, subjectId, { predicate: i.predicate, value: i.value, unit: i.currency ?? "USD", scope: i.scope, scopeReason: i.scopeReason ?? null, evidence: i.evidence ?? null, publishedAt: i.publishedAt ?? null, provenance: i.provenance, parserName: i.parserName ?? null, status, rejectionReason: blockedBy.length ? blockedBy.join(",") : null }); | |
| 200 | + await writeQualityFlags(tx, ctx, subjectType, subjectId, flags.map((f) => ({ ...f, field: f.field ?? "investmentUsd" })), { claimId: claim.id, priority: reviewPriority({ investmentUsd: i.value, severity: flags.some((f) => f.severity === "critical") ? "critical" : "warn" }) }); | |
| 201 | + return { assign: blockedBy.length === 0 && (i.currency ?? "USD") === "USD", claim, flags, blockedBy }; | |
| 202 | +} | |
| 203 | + | |
| 204 | +/** Best current claim for a predicate (highest tier, then most recent), for the evidence drawer and reconciliation. */ | |
| 205 | +export async function bestCurrentClaim(tx: Tx, subjectType: string, subjectId: string, predicate: string): Promise<{ id: string; value: number | null; tier: AuthorityTier; url: string; evidence: string | null } | null> { | |
| 206 | + const rows = await tx.execute(sql`select id, value, authority_tier, url, evidence_text from claims where subject_type = ${subjectType} and subject_id = ${subjectId} and predicate = ${predicate} and status = 'current' order by last_observed desc limit 50`); | |
| 207 | + if (!rows.length) return null; | |
| 208 | + const sorted = [...rows].sort((a, b) => (TIER_RANK[String(b.authority_tier) as AuthorityTier] ?? 0) - (TIER_RANK[String(a.authority_tier) as AuthorityTier] ?? 0)); | |
| 209 | + const r = sorted[0]!; | |
| 210 | + return { id: String(r.id), value: r.value == null ? null : Number(r.value), tier: String(r.authority_tier) as AuthorityTier, url: String(r.url), evidence: r.evidence_text == null ? null : String(r.evidence_text) }; | |
| 211 | +} | |
| 212 | + | |
| 213 | +/** A generic non-numeric claim (status, operator, opening date…) — same store, text value. */ | |
| 214 | +export async function writeTextClaim(tx: Tx, ctx: IngestContext, subjectType: string, subjectId: string, predicate: string, value: string, provenance: Provenance, opts: { scope?: ClaimScope; evidence?: Evidence | null; publishedAt?: string | null } = {}): Promise<WrittenClaim> { | |
| 215 | + return writeClaim(tx, ctx, subjectType, subjectId, { predicate, valueText: value, scope: opts.scope ?? "facility", evidence: opts.evidence ?? null, publishedAt: opts.publishedAt ?? null, provenance, status: "current" }); | |
| 216 | +} | |
| 217 | + | |
| 218 | +export { newId as _newClaimId }; | |
modified
apps/worker/src/ingest/contract.ts
+10 −0
@@ -42,6 +42,16 @@ export interface IngestStats { | ||
| 42 | 42 | implausibleDropped?: number; |
| 43 | 43 | /** projects folded into an already-known announcement (same operator, city, MW ±10 %, within 30 days) */ |
| 44 | 44 | projectDedup?: number; |
| 45 | + /** claims rows written */ | |
| 46 | + claims?: number; | |
| 47 | + /** quality flags written */ | |
| 48 | + qualityFlags?: number; | |
| 49 | + /** figures kept as claims but not assigned to a column (portfolio / company / unknown scope, market statistics) */ | |
| 50 | + unscopedClaims?: number; | |
| 51 | + /** project candidates vetoed by the announcement classifier (appointments, financing, PPAs, market research…) */ | |
| 52 | + projectsVetoed?: number; | |
| 53 | + /** building records linked to their campus record instead of being flagged as duplicates */ | |
| 54 | + campusLinks?: number; | |
| 45 | 55 | } |
| 46 | 56 | |
| 47 | 57 | export type IngestFn = (run: IngestRun, entities: NormalizedEntity[], doc: IngestDocRef | null) => Promise<IngestStats>; |
modified
apps/worker/src/ingest/events.ts
+5 −2
@@ -110,6 +110,9 @@ export interface EventInput { | ||
| 110 | 110 | reviewStatus?: "auto" | "pending"; |
| 111 | 111 | /** override the fingerprint day/value (default: newValue + ctx.day) */ |
| 112 | 112 | fingerprint?: string; |
| 113 | + isAi?: boolean; | |
| 114 | + /** documents describing the same announcement share a cluster id */ | |
| 115 | + clusterId?: string | null; | |
| 113 | 116 | } |
| 114 | 117 | |
| 115 | 118 | /** Insert an event unless the same fingerprint already exists today. Returns the id when inserted. */ |
@@ -118,12 +121,12 @@ export async function recordEvent(tx: Tx, ctx: IngestContext, e: EventInput): Pr | ||
| 118 | 121 | const id = newId("event"); |
| 119 | 122 | const rows = await tx.execute(sql` |
| 120 | 123 | insert into events (id, entity_type, entity_id, event_type, detected_at, effective_date, old_value, new_value, source_id, document_id, url, title, summary, |
| 121 | − significance, confidence, review_status, country_iso2, operator_id, metro_id, project_id, fingerprint) | |
| 124 | + significance, confidence, review_status, country_iso2, operator_id, metro_id, project_id, fingerprint, run_id, source_kind, is_ai, cluster_id) | |
| 122 | 125 | values (${id}, ${e.entityType}, ${e.entityId}, ${e.eventType}, ${ctx.now}, ${e.effectiveDate ?? null}, |
| 123 | 126 | ${e.oldValue === undefined ? null : JSON.stringify(e.oldValue)}::jsonb, ${e.newValue === undefined ? null : JSON.stringify(e.newValue)}::jsonb, |
| 124 | 127 | ${ctx.run.sourceId}, ${ctx.doc?.documentId ?? null}, ${e.url}, ${e.title.slice(0, 300)}, ${e.summary ?? null}, |
| 125 | 128 | ${Math.max(0, Math.min(100, Math.round(e.significance)))}, ${e.confidence ?? "moderate"}, ${e.reviewStatus ?? "auto"}, |
| 126 | − ${e.countryIso2 ?? null}, ${e.operatorId ?? null}, ${e.metroId ?? null}, ${e.projectId ?? null}, ${fp}) | |
| 129 | + ${e.countryIso2 ?? null}, ${e.operatorId ?? null}, ${e.metroId ?? null}, ${e.projectId ?? null}, ${fp}, ${ctx.run.runId}, ${ctx.run.sourceKind}, ${!!e.isAi}, ${e.clusterId ?? null}) | |
| 127 | 130 | on conflict (fingerprint) do nothing |
| 128 | 131 | returning id`); |
| 129 | 132 | if (!rows.length) return null; |
modified
apps/worker/src/ingest/facilities.ts
+120 −18
@@ -21,7 +21,9 @@ import { bboxAround, validGeo } from "./geo.js"; | ||
| 21 | 21 | import { countryFromPoint } from "./country-lookup.js"; |
| 22 | 22 | import { facilityIdForKey, upsertKey } from "./keys.js"; |
| 23 | 23 | import { attachIxpsByName } from "./ixps.js"; |
| 24 | −import { bestMatch, bestMw, completenessScore, decide, facilityConfidence, isPipelineStatus, shouldReplace, shouldReplaceGeo, shouldReplaceMw, type FacilityCandidate, type FieldObservation, type MatchDecision, type MatchScore } from "./match.js"; | |
| 24 | +import { bestMatch, bestMw, completenessScore, decide, facilityConfidence, isPipelineStatus, scoreFacilityMatch, shouldReplace, shouldReplaceGeo, shouldReplaceMw, type FacilityCandidate, type FieldObservation, type MatchDecision, type MatchScore } from "./match.js"; | |
| 25 | +import { CAPACITY_COLUMN, classifyAiEvidence, classifyCapacitySemantics, classifyScope, findEvidence, type CapacityPredicate, type ClaimScope } from "@dci/core"; | |
| 26 | +import { markWinners, recordCapacityClaim, writeQualityFlags } from "./claims.js"; | |
| 25 | 27 | import { assignMetro } from "./metros.js"; |
| 26 | 28 | import { operatorNames, resolveOperator } from "./operators.js"; |
| 27 | 29 | import { backingObservation, loadCurrentProvenance, summarizeSources, writeProvenance, type CurrentProvenance, type ObservedField } from "./provenance.js"; |
@@ -70,6 +72,16 @@ const COLUMNS: Record<string, string> = { | ||
| 70 | 72 | externalIds: "external_ids", |
| 71 | 73 | sourceCount: "source_count", |
| 72 | 74 | lastVerified: "last_verified", |
| 75 | + parentFacilityId: "parent_facility_id", | |
| 76 | + recordScope: "record_scope", | |
| 77 | + aiEvidence: "ai_evidence", | |
| 78 | + utilityCapacityMw: "utility_capacity_mw", | |
| 79 | + gridConnectionMw: "grid_connection_mw", | |
| 80 | + ultimateCampusMw: "ultimate_campus_mw", | |
| 81 | + capacityScope: "capacity_scope", | |
| 82 | + capacitySemantics: "capacity_semantics", | |
| 83 | + developerId: "developer_id", | |
| 84 | + landownerId: "landowner_id", | |
| 73 | 85 | }; |
| 74 | 86 | |
| 75 | 87 | const SCALAR_FIELDS = ["name", "city", "regionName", "address", "postalCode", "status", "facilityType", "tier", "buildingSqm", "siteAreaHa", "rackCount", "pue", "coolingType", "renewableClaim", "openedOn", "constructionStartedOn", "announcedOn", "website", "description"] as const; |
@@ -108,14 +120,21 @@ async function followMerged(tx: Tx, id: string): Promise<string> { | ||
| 108 | 120 | * id, a state name, an investment figure, a shared `ref` tag…) is informative but must never link records — a |
| 109 | 121 | * shared `operator_wikidata` once folded 121 AWS OpenStreetMap features into a single facility. |
| 110 | 122 | */ |
| 111 | −export const IDENTIFYING_EXTERNAL_ID_KEYS: ReadonlySet<string> = new Set(["osm", "wikidata", "wikipedia_en", "peeringdb_fac", "peeringdb", "pdb_fac", "dcmap", "geonames", "facebook_page", "meta_info_sheet", "google_detail_page", "cyrusone_slug", "qts_slug", "coresite_code"]); | |
| 112 | −const IDENTIFYING_KEY_RE = /(^|_)(id|slug|code|page|sheet|url|qid|uid)$/; | |
| 123 | +export const IDENTIFYING_EXTERNAL_ID_KEYS: ReadonlySet<string> = new Set([ | |
| 124 | + "osm", "wikidata", "wikipedia_en", "peeringdb_fac", "peeringdb", "pdb_fac", "dcmap", "geonames", | |
| 125 | + "facebook_page", "meta_info_sheet", "meta_location", "google_detail_page", "google_location", | |
| 126 | + "equinix_ibx", "digitalrealty_node", "digitalrealty_site_code", "ntt_slug", "cyrusone_slug", "stack_slug", "qts_slug", "vantage_slug", "coresite_code", "switch_slug", | |
| 127 | + "databank_code", "edgeconnex_slug", "tierpoint_slug", "airtrunk_code", "nextdc_code", "atnorth_code", "cloudhq_slug", "skybox_slug", "odata_code", "telehouse_slug", "coltdcs_slug", "globalswitch_slug", "virtus_slug", "digitaledge_code", | |
| 128 | +]); | |
| 113 | 129 | |
| 114 | −/** true when `key` names one facility (curated namespaces, or `*_id` / `*_slug` / `*_code` / `*_page` keys that are not operator-level). */ | |
| 130 | +/** | |
| 131 | + * ALLOWLIST ONLY. `state_code`, `region_code`, `postal_code`, `source_url`, `osm_ref` (a shared `ref` tag), `stack_campus` | |
| 132 | + * (one campus, many buildings) or `investment_currency` all end in an id-looking suffix and would fold unrelated | |
| 133 | + * facilities into one row (the `*_code` heuristic once folded 121 AWS OpenStreetMap features). A new connector that | |
| 134 | + * needs its key to link records adds it here explicitly. | |
| 135 | + */ | |
| 115 | 136 | export function isIdentifyingExternalId(key: string): boolean { |
| 116 | − if (IDENTIFYING_EXTERNAL_ID_KEYS.has(key)) return true; | |
| 117 | − if (/^(operator|owner|brand|company|parent|network)_/.test(key)) return false; | |
| 118 | − return IDENTIFYING_KEY_RE.test(key); | |
| 137 | + return IDENTIFYING_EXTERNAL_ID_KEYS.has(key); | |
| 119 | 138 | } |
| 120 | 139 | |
| 121 | 140 | async function byExternalIds(tx: Tx, ext: Record<string, string | number> | undefined): Promise<string | null> { |
@@ -141,10 +160,15 @@ async function loadCandidates(tx: Tx, nf: NormalizedFacility, operatorId: string | ||
| 141 | 160 | } |
| 142 | 161 | if (operatorId && countryIso2) conds.push(sql`(f.operator_id = ${operatorId} and f.country_iso2 = ${countryIso2})`); |
| 143 | 162 | if (!conds.length) return []; |
| 163 | + // deterministic and relevance-ordered: same operator first, then same normalized name, then nearest — a dense metro | |
| 164 | + // (Ashburn, Dallas, Singapore) has far more than 400 rows in a 40 km box and the true duplicate must be on the first page | |
| 165 | + const orderGeo = geo ? sql`, ((f.lat - ${geo.lat}) * (f.lat - ${geo.lat}) + (f.lng - ${geo.lng}) * (f.lng - ${geo.lng})) asc nulls last` : sql``; | |
| 144 | 166 | const rows = await tx.execute(sql` |
| 145 | 167 | select f.id, f.name, f.normalized_name, f.operator_id, o.name as operator_name, f.country_iso2, f.city, f.address, f.lat, f.lng, f.geo_precision, f.external_ids, |
| 146 | 168 | (select coalesce(array_agg(alias), '{}'::text[]) from facility_aliases a where a.facility_id = f.id) as aliases |
| 147 | − from facilities f left join operators o on o.id = f.operator_id where f.merged_into is null and (${sql.join(conds, sql` or `)}) limit 400`); | |
| 169 | + from facilities f left join operators o on o.id = f.operator_id where f.merged_into is null and (${sql.join(conds, sql` or `)}) | |
| 170 | + order by (${operatorId ?? null}::text is not null and f.operator_id = ${operatorId ?? null}) desc, (f.normalized_name in ${aliasNorms.length ? aliasNorms : ["__none__"]}) desc${orderGeo}, f.created_at asc | |
| 171 | + limit 400`); | |
| 148 | 172 | return rows.map((r) => ({ |
| 149 | 173 | id: String(r.id), |
| 150 | 174 | name: String(r.name), |
@@ -164,10 +188,23 @@ async function loadCandidates(tx: Tx, nf: NormalizedFacility, operatorId: string | ||
| 164 | 188 | |
| 165 | 189 | interface Resolution { |
| 166 | 190 | id: string | null; |
| 167 | − how: "key" | "external_id" | "merge" | "pending" | "create"; | |
| 191 | + how: "key" | "external_id" | "merge" | "pending" | "create" | "campus_link"; | |
| 168 | 192 | match?: { candidate: FacilityCandidate; match: MatchScore } | null; |
| 169 | 193 | } |
| 170 | 194 | |
| 195 | +/** true when a facility name designates a campus / park / multi-building site rather than one building. */ | |
| 196 | +export function isCampusName(name: string | null | undefined, campusName?: string | null): boolean { | |
| 197 | + return /\b(campus|park|cluster|hub|gigafactory|complex|estate|mega ?site)\b/i.test(`${name ?? ""} ${campusName ?? ""}`); | |
| 198 | +} | |
| 199 | + | |
| 200 | +/** Record scope from the source's statement or the name: campus designation → campus, building code (DC12, Hall 3, Building B) → building, else facility. */ | |
| 201 | +export function inferRecordScope(nf: { name: string; recordScope?: string | null; campusName?: string | null }): "building" | "facility" | "campus" { | |
| 202 | + if (nf.recordScope === "building" || nf.recordScope === "facility" || nf.recordScope === "campus") return nf.recordScope; | |
| 203 | + if (isCampusName(nf.name)) return "campus"; | |
| 204 | + if (/\b(building|bldg|hall|data hall|phase)\s*[A-Z0-9]{1,3}\b/i.test(nf.name) || (nf.campusName && nf.campusName !== nf.name)) return "building"; | |
| 205 | + return "facility"; | |
| 206 | +} | |
| 207 | + | |
| 171 | 208 | async function resolveFacility(tx: Tx, ctx: IngestContext, nf: NormalizedFacility, operatorId: string | null, operatorName: string | null, countryIso2: string | null): Promise<Resolution> { |
| 172 | 209 | const byKey = await facilityIdForKey(tx, ctx, nf.key); |
| 173 | 210 | if (byKey) return { id: byKey, how: "key" }; |
@@ -183,6 +220,8 @@ async function resolveFacility(tx: Tx, ctx: IngestContext, nf: NormalizedFacilit | ||
| 183 | 220 | if (!best) return { id: null, how: "create" }; |
| 184 | 221 | const d: MatchDecision = decide(best.match.score); |
| 185 | 222 | if (d === "merge") return { id: best.candidate.id, how: "merge", match: best }; |
| 223 | + // campus vs one of its buildings: not a duplicate but a containment — create the record and link it to its parent | |
| 224 | + if (best.match.reasons.includes("rule:campus-vs-building")) return { id: null, how: "campus_link", match: best }; | |
| 186 | 225 | if (d === "pending") return { id: null, how: "pending", match: best }; |
| 187 | 226 | return { id: null, how: "create", match: best.match.score >= 0.3 ? best : null }; |
| 188 | 227 | } |
@@ -241,8 +280,21 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF | ||
| 241 | 280 | |
| 242 | 281 | // --- field merge ----------------------------------------------------------------------------------------------- |
| 243 | 282 | const before: Row = existing ? { ...existing } : {}; |
| 244 | − const next: Row = existing ? { ...existing } : { id, geoPrecision: "unknown", status: "unknown", facilityType: "unknown", confidence: "unverified", completeness: 0, sourceCount: 0, isAi: false, isHyperscale: false, mwIsEstimate: false, externalIds: {}, certifications: [], aliases: [] }; | |
| 283 | + const next: Row = existing ? { ...existing } : { id, geoPrecision: "unknown", status: "unknown", facilityType: "unknown", confidence: "unverified", completeness: 0, sourceCount: 0, isAi: false, isHyperscale: false, mwIsEstimate: false, externalIds: {}, certifications: [], aliases: [], recordScope: "facility", aiEvidence: "unknown" }; | |
| 245 | 284 | const observed: ObservedField[] = []; |
| 285 | + // containment: a building that matched its campus record (or names its campus record by key) points at the parent | |
| 286 | + if (res.how === "campus_link" && res.match) { | |
| 287 | + const candIsCampus = isCampusName(res.match.candidate.name) && !isCampusName(name); | |
| 288 | + if (candIsCampus) { next.parentFacilityId = res.match.candidate.id; next.recordScope = "building"; } | |
| 289 | + else if (isCampusName(name) && !isCampusName(res.match.candidate.name) && !ctx.run.dryRun) { | |
| 290 | + // the incoming record is the campus: the stored building becomes its child | |
| 291 | + await tx.execute(sql`update facilities set parent_facility_id = ${id}, record_scope = 'building', updated_at = now() where id = ${res.match.candidate.id} and parent_facility_id is null`); | |
| 292 | + } | |
| 293 | + ctx.stats.campusLinks = (ctx.stats.campusLinks ?? 0) + 1; | |
| 294 | + } | |
| 295 | + if (nf.parentFacilityKey && !next.parentFacilityId) { const pid = await facilityIdForKey(tx, ctx, nf.parentFacilityKey); if (pid && pid !== id) { next.parentFacilityId = pid; next.recordScope = "building"; } } | |
| 296 | + if (nf.developerName) { const dev = await resolveOperator(tx, ctx, { name: nf.developerName }); if (dev) { next.developerId = dev.id; observed.push({ field: "developerName", value: dev.name, provenance: pf("developerName") }); } } | |
| 297 | + if (nf.landownerName) { const lo = await resolveOperator(tx, ctx, { name: nf.landownerName }); if (lo) { next.landownerId = lo.id; observed.push({ field: "landownerName", value: lo.name, provenance: pf("landownerName") }); } } | |
| 246 | 298 | const incoming: Record<string, unknown> = { |
| 247 | 299 | name, |
| 248 | 300 | city: cleanText(nf.city), |
@@ -278,13 +330,38 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF | ||
| 278 | 330 | const dec = shouldReplace(obs(v, p, ctx), storedObs(prov, field, existing?.[field])); |
| 279 | 331 | if (dec.replace) next[field] = v; |
| 280 | 332 | } |
| 333 | + // capacity figures go through the claim store: scope + semantics from the supporting sentence (structured operator | |
| 334 | + // specs have none → the record's own scope), sanity engine, then the authority policy for the column | |
| 335 | + const recordScope = inferRecordScope(nf); | |
| 336 | + const campusDesignation = isCampusName(name, nf.campusName); | |
| 337 | + const MW_PREDICATE: Record<(typeof MW_FIELDS)[number], CapacityPredicate> = { itCapacityMw: "it_capacity_mw", totalPowerMw: "current_power_mw", plannedPowerMw: "planned_power_mw" }; | |
| 281 | 338 | for (const field of MW_FIELDS) { |
| 282 | 339 | const v = nf[field]; |
| 283 | 340 | if (v == null || !Number.isFinite(v) || v <= 0) continue; |
| 284 | 341 | const p = pf(field); |
| 285 | − observed.push({ field, value: v, provenance: p }); | |
| 286 | − const dec = shouldReplaceMw(obs(v, p, ctx), storedObs(prov, field, existing?.[field])); | |
| 287 | − if (dec.replace) next[field] = v; | |
| 342 | + const context = nf.claimContext?.[field] ?? null; | |
| 343 | + const structured = !context; // a parser read a spec table / JSON field: the figure describes the record itself | |
| 344 | + const sc = structured ? { scope: recordScope as ClaimScope, reason: "structured:record" } : classifyScope(context, recordScope as ClaimScope); | |
| 345 | + const sem = structured ? { predicate: MW_PREDICATE[field], reason: "field" } : classifyCapacitySemantics(context); | |
| 346 | + const predicate: CapacityPredicate = sem.predicate ?? MW_PREDICATE[field]; | |
| 347 | + const evidence = context ? findEvidence(context, v, "mw") ?? { text: context.slice(0, 600), start: 0, end: Math.min(600, context.length) } : null; | |
| 348 | + const decision = await recordCapacityClaim(tx, ctx, "facility", id, { field, predicate, value: v, scope: sc.scope, scopeReason: sc.reason, evidence, context, previous: (existing?.[field] as number | null) ?? null, recordScope, campusDesignation, provenance: p, semanticsDefaulted: !structured && !sem.predicate }); | |
| 349 | + if (!decision.assign) { ctx.stats.unscopedClaims = (ctx.stats.unscopedClaims ?? 0) + 1; continue; } | |
| 350 | + // utility / grid / ultimate figures live in their own columns, never in IT / total / planned | |
| 351 | + const column = CAPACITY_COLUMN[predicate]; | |
| 352 | + const target = column && column !== "itCapacityMw" && column !== "totalPowerMw" && column !== "plannedPowerMw" ? column : field; | |
| 353 | + observed.push({ field: target, value: v, provenance: { ...p, note: p.note ?? `${predicate} · ${sc.scope}` } }); | |
| 354 | + const dec = shouldReplaceMw(obs(v, p, ctx), storedObs(prov, target, existing?.[target])); | |
| 355 | + if (dec.replace) { next[target] = v; if (target === field) { next.capacityScope = sc.scope; next.capacitySemantics = predicate; } } | |
| 356 | + } | |
| 357 | + next.recordScope = recordScope; | |
| 358 | + // AI evidence: graded from the source text, never from one keyword | |
| 359 | + { | |
| 360 | + const ai = classifyAiEvidence(`${nf.aiEvidence === "confirmed" ? "AI campus" : ""} ${name} ${nf.description ?? ""} ${nf.facilityType === "ai" || nf.facilityType === "hpc" ? "AI data center" : ""}`); | |
| 361 | + const level = nf.aiEvidence && nf.aiEvidence !== "unknown" ? nf.aiEvidence : ai.level; | |
| 362 | + const rank: Record<string, number> = { unknown: 0, associated: 1, likely: 2, confirmed: 3 }; | |
| 363 | + if ((rank[level] ?? 0) > (rank[String(next.aiEvidence ?? "unknown")] ?? 0)) next.aiEvidence = level; | |
| 364 | + if (level === "confirmed" || level === "likely") { next.isAi = true; if (nf.isAi !== true) observed.push({ field: "isAi", value: true, provenance: { ...P, method: `ai-evidence:${level}${ai.evidence ? `:${ai.evidence}` : ""}` } }); } | |
| 288 | 365 | } |
| 289 | 366 | // operator / owner: authority policy like other fields (a news source never re-assigns an operator set by the operator itself); |
| 290 | 367 | // an operator inferred from the name never replaces one stated by a source |
@@ -352,12 +429,14 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF | ||
| 352 | 429 | next.confidence = res.how === "pending" ? "unverified" : "moderate"; |
| 353 | 430 | await tx.execute(sql`insert into facilities (id, slug, name, normalized_name, operator_id, owner_id, campus_id, metro_id, country_iso2, city, region_name, address, postal_code, lat, lng, geo_precision, geo_source, geohash, |
| 354 | 431 | status, facility_type, tier, building_sqm, site_area_ha, it_capacity_mw, total_power_mw, planned_power_mw, mw_is_estimate, rack_count, pue, cooling_type, renewable_claim, opened_on, construction_started_on, announced_on, |
| 355 | − website, description, is_ai, is_hyperscale, certifications, confidence, completeness, external_ids, source_count, first_seen) | |
| 432 | + website, description, is_ai, is_hyperscale, certifications, confidence, completeness, external_ids, source_count, first_seen, | |
| 433 | + parent_facility_id, record_scope, ai_evidence, utility_capacity_mw, grid_connection_mw, ultimate_campus_mw, capacity_scope, capacity_semantics, developer_id, landowner_id) | |
| 356 | 434 | values (${id}, ${slug}, ${next.name}, ${normalizedName}, ${next.operatorId ?? null}, ${next.ownerId ?? null}, ${next.campusId ?? null}, ${next.metroId ?? null}, ${next.countryIso2 ?? null}, ${next.city ?? null}, ${next.regionName ?? null}, |
| 357 | 435 | ${next.address ?? null}, ${next.postalCode ?? null}, ${next.lat ?? null}, ${next.lng ?? null}, ${next.geoPrecision ?? "unknown"}, ${next.geoSource ?? null}, ${next.geohash ?? null}, |
| 358 | 436 | ${next.status ?? "unknown"}, ${next.facilityType ?? "unknown"}, ${next.tier ?? null}, ${next.buildingSqm ?? null}, ${next.siteAreaHa ?? null}, ${next.itCapacityMw ?? null}, ${next.totalPowerMw ?? null}, ${next.plannedPowerMw ?? null}, false, |
| 359 | 437 | ${next.rackCount ?? null}, ${next.pue ?? null}, ${next.coolingType ?? null}, ${next.renewableClaim ?? null}, ${next.openedOn ?? null}, ${next.constructionStartedOn ?? null}, ${next.announcedOn ?? null}, |
| 360 | − ${next.website ?? null}, ${next.description ?? null}, ${!!next.isAi}, ${!!next.isHyperscale}, ${textArray((next.certifications as string[]) ?? [])}::text[], ${next.confidence}, 0, ${JSON.stringify(next.externalIds ?? {})}::jsonb, 0, ${ctx.now})`); | |
| 438 | + ${next.website ?? null}, ${next.description ?? null}, ${!!next.isAi}, ${!!next.isHyperscale}, ${textArray((next.certifications as string[]) ?? [])}::text[], ${next.confidence}, 0, ${JSON.stringify(next.externalIds ?? {})}::jsonb, 0, ${ctx.now}, | |
| 439 | + ${next.parentFacilityId ?? null}, ${next.recordScope ?? "facility"}, ${next.aiEvidence ?? "unknown"}, ${next.utilityCapacityMw ?? null}, ${next.gridConnectionMw ?? null}, ${next.ultimateCampusMw ?? null}, ${next.capacityScope ?? null}, ${next.capacitySemantics ?? null}, ${next.developerId ?? null}, ${next.landownerId ?? null})`); | |
| 361 | 440 | } else { |
| 362 | 441 | const sets = []; |
| 363 | 442 | for (const [camel, col] of Object.entries(COLUMNS)) { |
@@ -385,8 +464,9 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF | ||
| 385 | 464 | const aliasValues = uniqStrings([...(nf.aliases ?? []), name !== next.name ? name : null, existing && String(existing.name) !== String(next.name) ? String(existing.name) : null]).filter((a) => normalizeName(a) && a !== next.name); |
| 386 | 465 | for (const a of aliasValues) await tx.execute(sql`insert into facility_aliases (facility_id, alias, normalized, source_id) values (${id}, ${a}, ${normalizeName(a)}, ${ctx.run.sourceId}) on conflict do nothing`); |
| 387 | 466 | |
| 388 | − // provenance for every observed field | |
| 467 | + // provenance for every observed field, then mark which observation backs each displayed value | |
| 389 | 468 | await writeProvenance(tx, ctx, "facility", id, observed, nf.key); |
| 469 | + await markWinners(tx, ctx, "facility", id, [...SCALAR_FIELDS, ...MW_FIELDS, "utilityCapacityMw", "gridConnectionMw", "ultimateCampusMw", "countryIso2"].map((f) => ({ field: f, value: next[f] })).concat([{ field: "operatorId", value: next.operatorId }, { field: "ownerId", value: next.ownerId }])); | |
| 390 | 470 | |
| 391 | 471 | // tenants / IXPs |
| 392 | 472 | if (nf.carriers?.length) await attachTenants(tx, ctx, id, { names: nf.carriers, role: "carrier" }); |
@@ -408,6 +488,13 @@ export async function ingestFacility(tx: Tx, ctx: IngestContext, nf: NormalizedF | ||
| 408 | 488 | } else if (res.how === "create" && res.match) { |
| 409 | 489 | await tx.execute(sql`insert into entity_matches (id, connector_id, candidate_key, candidate, matched_facility_id, score, reasons, status, decided_by, decided_at) |
| 410 | 490 | values (${newId("match")}, ${ctx.run.connectorId}, ${nf.key}, ${JSON.stringify(candidateJson(nf, id))}::jsonb, ${res.match.candidate.id}, ${res.match.match.score}, ${textArray(res.match.match.reasons)}::text[], 'auto_created', 'system', ${ctx.now})`); |
| 491 | + } else if (res.how === "campus_link" && res.match) { | |
| 492 | + await tx.execute(sql`insert into entity_matches (id, connector_id, candidate_key, candidate, matched_facility_id, score, reasons, status, decided_by, decided_at) | |
| 493 | + values (${newId("match")}, ${ctx.run.connectorId}, ${nf.key}, ${JSON.stringify(candidateJson(nf, id))}::jsonb, ${res.match.candidate.id}, ${res.match.match.score}, ${textArray(res.match.match.reasons)}::text[], 'related_campus', 'system', ${ctx.now})`); | |
| 494 | + } | |
| 495 | + // duplicate candidates awaiting review are a quality flag too (they inflate every aggregate until decided) | |
| 496 | + if (res.how === "pending" && res.match) { | |
| 497 | + await writeQualityFlags(tx, ctx, "facility", id, [{ code: "possible_duplicate", severity: "warn", message: `possible duplicate of ${res.match.candidate.name} (score ${res.match.match.score})`, field: "name" }], { details: { matchedFacilityId: res.match.candidate.id, reasons: res.match.match.reasons } }); | |
| 411 | 498 | } |
| 412 | 499 | |
| 413 | 500 | // --- events ---------------------------------------------------------------------------------------------------- |
@@ -485,7 +572,7 @@ export async function refreshDerived(tx: Tx, id: string, forceConfidence: Confid | ||
| 485 | 572 | // a facility created from an ambiguous match stays "unverified" until the admin decides |
| 486 | 573 | const pending = forceConfidence ? [] : await tx.execute(sql`select 1 from entity_matches where status = 'pending' and candidate->>'createdFacilityId' = ${id} limit 1`); |
| 487 | 574 | const confidence = forceConfidence ?? (pending.length ? "unverified" : facilityConfidence({ sourceKinds: kinds, sourceCount: sourceIds.length, lastVerifiedIso: lastPrimaryObserved, onlyEstimates })); |
| 488 | − const counts = (await tx.execute(sql`select carriers_count, ixp_count from facilities where id = ${id}`))[0]; | |
| 575 | + const counts = (await tx.execute(sql`select carriers_count, ixp_count, (select coalesce(max(priority), 0) from quality_flags q where q.entity_type = 'facility' and q.entity_id = ${id} and q.status = 'open') as review_priority from facilities where id = ${id}`))[0]; | |
| 489 | 576 | const completeness = completenessScore({ |
| 490 | 577 | geoPrecision: row.geoPrecision as string, |
| 491 | 578 | hasCoords: validLatLng(row.lat, row.lng), |
@@ -503,7 +590,7 @@ export async function refreshDerived(tx: Tx, id: string, forceConfidence: Confid | ||
| 503 | 590 | carriersCount: (counts?.carriers_count as number | null) ?? null, |
| 504 | 591 | ixpCount: (counts?.ixp_count as number | null) ?? null, |
| 505 | 592 | }); |
| 506 | − await tx.execute(sql`update facilities set mw_is_estimate = ${mwIsEstimate}, source_count = ${sourceIds.length}, last_verified = ${lastPrimaryObserved}, confidence = ${confidence}, completeness = ${completeness} where id = ${id}`); | |
| 593 | + await tx.execute(sql`update facilities set mw_is_estimate = ${mwIsEstimate}, source_count = ${sourceIds.length}, last_verified = ${lastPrimaryObserved}, confidence = ${confidence}, completeness = ${completeness}, review_priority = ${Number(counts?.review_priority ?? 0)} where id = ${id}`); | |
| 507 | 594 | } |
| 508 | 595 | |
| 509 | 596 | /** Admin approval of a pending match: fold `fromId` into `intoId` (keys, aliases, provenance, links, projects). */ |
@@ -528,3 +615,18 @@ export async function mergeFacilities(tx: Tx, fromId: string, intoId: string, de | ||
| 528 | 615 | await refreshDerived(tx, intoId); |
| 529 | 616 | } |
| 530 | 617 | |
| 618 | + | |
| 619 | +/** Extraction debugger: how would this normalized facility resolve right now (candidates, scores, decision)? Read-only. */ | |
| 620 | +export async function previewFacilityResolution(tx: Tx, ctx: IngestContext, nf: NormalizedFacility): Promise<{ how: string; matchedId: string | null; candidates: Array<{ id: string; name: string; operatorName: string | null; city: string | null; score: number; reasons: string[]; distanceKm: number | null }> }> { | |
| 621 | + const operator = nf.operatorName ? await resolveOperator(tx, ctx, { name: nf.operatorName, key: nf.operatorKey ?? null }) : null; | |
| 622 | + const country = await safeCountry(tx, ctx, nf.countryIso2); | |
| 623 | + const byKey = await facilityIdForKey(tx, ctx, nf.key); | |
| 624 | + const byExt = byKey ? null : await byExternalIds(tx, nf.externalIds); | |
| 625 | + const candidates = await loadCandidates(tx, nf, operator?.id ?? null, country); | |
| 626 | + const geo = validGeo(nf.geo) ? nf.geo : null; | |
| 627 | + const probe = { name: nf.name, aliases: nf.aliases, operatorId: operator?.id ?? null, operatorName: operator?.name ?? nf.operatorName ?? null, countryIso2: country, city: nf.city ?? null, address: nf.address ?? null, lat: geo?.lat ?? null, lng: geo?.lng ?? null, geoPrecision: geo?.precision ?? null }; | |
| 628 | + const scored = candidates.map((c) => { const m = scoreFacilityMatch(probe, c); return { id: c.id, name: c.name, operatorName: c.operatorName ?? null, city: c.city ?? null, score: m.score, reasons: m.reasons, distanceKm: m.distanceKm ?? null }; }).sort((a, b) => b.score - a.score).slice(0, 10); | |
| 629 | + const best = scored[0]; | |
| 630 | + const how = byKey ? "key" : byExt ? "external_id" : !best ? "create" : best.reasons.includes("rule:campus-vs-building") ? "campus_link" : decide(best.score); | |
| 631 | + return { how, matchedId: byKey ?? byExt ?? (best && (how === "merge") ? best.id : null), candidates: scored }; | |
| 632 | +} | |
added
apps/worker/src/ingest/geocode.ts
+43 −0
@@ -0,0 +1,43 @@ | ||
| 1 | +/** | |
| 2 | + * Offline city-level geocoding for projects (and facilities without coordinates): a metro seed (name / alias) or the | |
| 3 | + * curated city table gives CITY-level coordinates, never more precise. Nothing is guessed: unknown city → null. | |
| 4 | + */ | |
| 5 | +import { normalizeName, type GeoPoint } from "@dci/core"; | |
| 6 | +import { CITIES } from "../connectors/cloud/city-coords.js"; | |
| 7 | +import type { Tx } from "./common.js"; | |
| 8 | +import { loadMetros } from "./metros.js"; | |
| 9 | + | |
| 10 | +const cityIndex = new Map<string, { lat: number; lng: number; iso: string; precision: GeoPoint["precision"]; name: string }>(); | |
| 11 | +for (const c of CITIES) { | |
| 12 | + for (const a of [c.city, ...(c.aliases ?? [])]) { | |
| 13 | + const k = `${c.iso}|${normalizeName(a)}`; | |
| 14 | + if (!cityIndex.has(k)) cityIndex.set(k, { lat: c.lat, lng: c.lng, iso: c.iso, precision: c.precision ?? "city", name: c.city }); | |
| 15 | + } | |
| 16 | +} | |
| 17 | + | |
| 18 | +export interface GeocodeInput { city?: string | null; regionName?: string | null; countryIso2?: string | null } | |
| 19 | + | |
| 20 | +/** City-level point for a place name, or null. `source` records how it was derived so provenance stays honest. */ | |
| 21 | +export async function geocodeCity(tx: Tx, i: GeocodeInput): Promise<(GeoPoint & { metroId?: string | null; label: string }) | null> { | |
| 22 | + const city = i.city?.trim(); | |
| 23 | + if (!city) return null; | |
| 24 | + const norm = normalizeName(city); | |
| 25 | + if (!norm) return null; | |
| 26 | + const metros = await loadMetros(tx); | |
| 27 | + const metroHit = metros.find((m) => (!i.countryIso2 || m.countryIso2 === i.countryIso2) && m.normalizedAliases.includes(norm)); | |
| 28 | + if (metroHit) return { lat: metroHit.lat, lng: metroHit.lng, precision: "metro", source: `geocoder:metro:${metroHit.slug}`, metroId: metroHit.id, label: metroHit.name }; | |
| 29 | + if (i.countryIso2) { | |
| 30 | + const hit = cityIndex.get(`${i.countryIso2}|${norm}`); | |
| 31 | + if (hit) return { lat: hit.lat, lng: hit.lng, precision: hit.precision === "exact" ? "city" : hit.precision, source: "geocoder:city-table", label: hit.name }; | |
| 32 | + } else { | |
| 33 | + const hits = [...cityIndex.entries()].filter(([k]) => k.endsWith(`|${norm}`)); | |
| 34 | + if (hits.length === 1) { const h = hits[0]![1]; return { lat: h.lat, lng: h.lng, precision: h.precision === "exact" ? "city" : h.precision, source: "geocoder:city-table", label: h.name }; } | |
| 35 | + } | |
| 36 | + // a county / parish in the lexicon may be a metro alias without the suffix | |
| 37 | + const stripped = normalizeName(city.replace(/\b(county|parish|township|borough)\b/gi, "")); | |
| 38 | + if (stripped && stripped !== norm) { | |
| 39 | + const m2 = metros.find((m) => (!i.countryIso2 || m.countryIso2 === i.countryIso2) && m.normalizedAliases.includes(stripped)); | |
| 40 | + if (m2) return { lat: m2.lat, lng: m2.lng, precision: "approximate", source: `geocoder:metro:${m2.slug}`, metroId: m2.id, label: m2.name }; | |
| 41 | + } | |
| 42 | + return null; | |
| 43 | +} | |
modified
apps/worker/src/ingest/index.ts
+4 −1
@@ -98,7 +98,10 @@ export const ingestEntities: IngestFn = async (run: IngestRun, entities: Normali | ||
| 98 | 98 | }; |
| 99 | 99 | |
| 100 | 100 | export type { IngestDocRef, IngestFn, IngestRun, IngestStats } from "./contract.js"; |
| 101 | −export { mergeFacilities, refreshDerived } from "./facilities.js"; | |
| 101 | +export { mergeFacilities, refreshDerived, previewFacilityResolution, isCampusName, inferRecordScope } from "./facilities.js"; | |
| 102 | +export { hideProject, mergeProjects } from "./projects.js"; | |
| 103 | +export { writeClaim, writeQualityFlags, recordCapacityClaim, recordInvestmentClaim, reviewPriority, bestCurrentClaim } from "./claims.js"; | |
| 104 | +export { geocodeCity } from "./geocode.js"; | |
| 102 | 105 | export { resolveOperator, lookupOperatorId, inferOperatorKind, mergeOperators } from "./operators.js"; |
| 103 | 106 | export { findCanonicalOperator, CANONICAL_OPERATORS, HYPERSCALERS } from "./canonical-operators.js"; |
| 104 | 107 | export { assignMetro, loadMetros, nearestMetro, invalidateMetroCache } from "./metros.js"; |
modified
apps/worker/src/ingest/ixps.ts
+1 −0
@@ -29,6 +29,7 @@ export async function resolveIxp(tx: Tx, ctx: IngestContext, i: IxpInput): Promi | ||
| 29 | 29 | } |
| 30 | 30 | if (!id && i.externalIds) { |
| 31 | 31 | for (const [k, v] of Object.entries(i.externalIds)) { |
| 32 | + if (v == null || v === "" || !/^(peeringdb_ix|peeringdb|pdb_ix|wikidata|ixpdb|pch_id|euro_ix)$/.test(k)) continue; | |
| 32 | 33 | const r = await tx.execute(sql`select id from ixps where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`); |
| 33 | 34 | if (r[0]) { |
| 34 | 35 | id = String(r[0].id); |
modified
apps/worker/src/ingest/keys.ts
+1 −1
@@ -25,6 +25,6 @@ export async function facilityIdForKey(tx: Tx, ctx: IngestContext, key: string): | ||
| 25 | 25 | |
| 26 | 26 | export async function upsertKey(tx: Tx, ctx: IngestContext, key: string, entityType: string, entityId: string): Promise<void> { |
| 27 | 27 | await tx.execute(sql`insert into entity_keys (key, entity_type, entity_id, connector_id) values (${key}, ${entityType}, ${entityId}, ${ctx.run.connectorId}) |
| 28 | − on conflict (key) do update set entity_id = excluded.entity_id, connector_id = excluded.connector_id`); | |
| 28 | + on conflict (key, entity_type) do update set entity_id = excluded.entity_id, connector_id = excluded.connector_id`); | |
| 29 | 29 | if (entityType === "facility") ctx.caches.facilityIdByKey.set(key, entityId); |
| 30 | 30 | } |
modified
apps/worker/src/ingest/news.ts
+22 −7
@@ -3,7 +3,8 @@ | ||
| 3 | 3 | * Items carrying an eventType with significance ≥ 40 also produce an `events` row (entityType news_event). |
| 4 | 4 | */ |
| 5 | 5 | import { sql, textArray } from "@dci/db"; |
| 6 | −import { cleanText, countryFromText, normalizeName, parseAllMw, stableId, type EventType, type NormalizedNewsEvent } from "@dci/core"; | |
| 6 | +import { cleanText, classifyAiEvidence, normalizeName, parseAllMw, stableId, type EventType, type NormalizedNewsEvent } from "@dci/core"; | |
| 7 | +import { countryFromTextSafe } from "../connectors/news/locations-lexicon.js"; | |
| 7 | 8 | import { addRef, bump, isoOrNull, knownCountries, type IngestContext, type Tx } from "./common.js"; |
| 8 | 9 | import { CANONICAL_OPERATORS, findCanonicalOperator } from "./canonical-operators.js"; |
| 9 | 10 | import { recordEvent } from "./events.js"; |
@@ -24,6 +25,16 @@ const EVENT_SIGNIFICANCE: Partial<Record<EventType, number>> = { | ||
| 24 | 25 | cloud_region_launched: 55, |
| 25 | 26 | incident: 70, |
| 26 | 27 | phase_announced: 50, |
| 28 | + land_acquired: 60, | |
| 29 | + grid_connection: 65, | |
| 30 | + grid_constraint: 70, | |
| 31 | + utility_event: 50, | |
| 32 | + operator_expansion: 65, | |
| 33 | + customer_agreement: 45, | |
| 34 | + partnership: 35, | |
| 35 | + executive_change: 20, | |
| 36 | + project_delayed: 70, | |
| 37 | + project_cancelled: 80, | |
| 27 | 38 | news: 30, |
| 28 | 39 | page_changed: 10, |
| 29 | 40 | }; |
@@ -98,7 +109,8 @@ export async function linkNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEven | ||
| 98 | 109 | } |
| 99 | 110 | const known = await knownCountries(tx, ctx); |
| 100 | 111 | const countries = new Set<string>(); |
| 101 | − const fromText = countryFromText(text); | |
| 112 | + // the hardened matcher: "North America" is not the US, "Georgia" is a US state, "… Jordan" is a person | |
| 113 | + const fromText = countryFromTextSafe(text); | |
| 102 | 114 | if (fromText && known.has(fromText)) countries.add(fromText); |
| 103 | 115 | for (const c of n.mentions?.countriesIso2 ?? []) if (c && known.has(c.toUpperCase())) countries.add(c.toUpperCase()); |
| 104 | 116 | const facilityIds: string[] = []; |
@@ -110,8 +122,9 @@ export async function linkNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEven | ||
| 110 | 122 | : await tx.execute(sql`select id from facilities where merged_into is null and normalized_name = ${norm} limit 2`); |
| 111 | 123 | if (rows.length === 1) facilityIds.push(String(rows[0]!.id)); |
| 112 | 124 | } |
| 113 | − const mwList = n.mentions?.mw?.filter((v) => typeof v === "number" && v > 0) ?? []; | |
| 114 | − const mw = mwList.length ? Math.max(...mwList) : (parseAllMw(text)[0] ?? null); | |
| 125 | + // `mentions.mw` is already portfolio-filtered by the news parser; the fallback only reads the headline + summary | |
| 126 | + const mwList = n.mentions?.mw?.filter((v) => typeof v === "number" && v > 0 && v <= 20_000) ?? []; | |
| 127 | + const mw = mwList.length ? Math.max(...mwList) : (parseAllMw(text).filter((v) => v <= 20_000)[0] ?? null); | |
| 115 | 128 | return { operatorIds, operatorNames: [...operatorNames], countryIso2s: [...countries], facilityIds, mw }; |
| 116 | 129 | } |
| 117 | 130 | |
@@ -127,12 +140,13 @@ export async function ingestNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEv | ||
| 127 | 140 | const mentions = { ...(n.mentions ?? {}), linkedOperators: links.operatorNames }; |
| 128 | 141 | // third-party publishers (kind = news): title, date, link and a ≤ 400-character summary only — never more text |
| 129 | 142 | const summary = ctx.run.sourceKind === "news" ? shortSummary(n.summary, THIRD_PARTY_SUMMARY_MAX) : cleanText(n.summary); |
| 130 | − const inserted = await tx.execute(sql`insert into news_items (id, source_id, connector_id, url, title, published_at, summary, page_type, event_type, mentions, operator_ids, country_iso2s, facility_ids, mw, significance) | |
| 143 | + const ai = n.isAi ?? ["confirmed", "likely"].includes(classifyAiEvidence(`${title} ${n.summary ?? ""}`).level); | |
| 144 | + const inserted = await tx.execute(sql`insert into news_items (id, source_id, connector_id, url, title, published_at, summary, page_type, event_type, mentions, operator_ids, country_iso2s, facility_ids, mw, significance, project_class) | |
| 131 | 145 | values (${id}, ${ctx.run.sourceId}, ${ctx.run.connectorId}, ${url}, ${title.slice(0, 500)}, ${publishedAt}, ${summary}, ${n.pageType ?? "unknown"}, ${eventType}, ${JSON.stringify(mentions)}::jsonb, |
| 132 | − ${textArray(links.operatorIds)}::text[], ${textArray(links.countryIso2s)}::text[], ${textArray(links.facilityIds)}::text[], ${links.mw}, ${significance}) | |
| 146 | + ${textArray(links.operatorIds)}::text[], ${textArray(links.countryIso2s)}::text[], ${textArray(links.facilityIds)}::text[], ${links.mw}, ${significance}, ${n.projectClass ?? null}) | |
| 133 | 147 | on conflict (url) do update set title = excluded.title, summary = coalesce(excluded.summary, news_items.summary), published_at = coalesce(excluded.published_at, news_items.published_at), |
| 134 | 148 | event_type = coalesce(excluded.event_type, news_items.event_type), mentions = excluded.mentions, operator_ids = excluded.operator_ids, country_iso2s = excluded.country_iso2s, |
| 135 | − facility_ids = excluded.facility_ids, mw = coalesce(excluded.mw, news_items.mw), significance = greatest(excluded.significance, news_items.significance) | |
| 149 | + facility_ids = excluded.facility_ids, mw = coalesce(excluded.mw, news_items.mw), significance = greatest(excluded.significance, news_items.significance), project_class = coalesce(excluded.project_class, news_items.project_class) | |
| 136 | 150 | returning (xmax = 0) as inserted`); |
| 137 | 151 | const isNew = Boolean(inserted[0]?.inserted); |
| 138 | 152 | if (isNew) ctx.stats.created++; |
@@ -156,6 +170,7 @@ export async function ingestNews(tx: Tx, ctx: IngestContext, n: NormalizedNewsEv | ||
| 156 | 170 | countryIso2: links.countryIso2s[0] ?? null, |
| 157 | 171 | operatorId: links.operatorIds[0] ?? null, |
| 158 | 172 | fingerprint: `news:${id}:${eventType}`, |
| 173 | + isAi: ai, | |
| 159 | 174 | }); |
| 160 | 175 | } |
| 161 | 176 | } |
modified
apps/worker/src/ingest/operators.ts
+3 −1
@@ -76,10 +76,12 @@ async function byKey(tx: Tx, key: string): Promise<OperatorRow | null> { | ||
| 76 | 76 | return rows[0] ? rowFrom(rows[0]) : null; |
| 77 | 77 | } |
| 78 | 78 | |
| 79 | +/** External-id namespaces that identify ONE operator (allowlist — a shared `hq_country` or `stock_exchange` must never fold two companies). */ | |
| 80 | +export const OPERATOR_IDENTIFYING_KEYS: ReadonlySet<string> = new Set(["wikidata", "wikipedia_en", "peeringdb_org", "peeringdb_net", "lei", "cik", "asn", "crunchbase", "linkedin", "gleif"]); | |
| 79 | 81 | async function byExternalIds(tx: Tx, ext: Record<string, string | number> | undefined): Promise<OperatorRow | null> { |
| 80 | 82 | if (!ext) return null; |
| 81 | 83 | for (const [k, v] of Object.entries(ext)) { |
| 82 | − if (v == null || v === "") continue; | |
| 84 | + if (v == null || v === "" || !OPERATOR_IDENTIFYING_KEYS.has(k)) continue; | |
| 83 | 85 | const rows = await tx.execute(sql`select ${COLS} from operators where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`); |
| 84 | 86 | if (rows[0]) return rowFrom(rows[0]); |
| 85 | 87 | } |
modified
apps/worker/src/ingest/projects.ts
+257 −58
@@ -1,17 +1,47 @@ | ||
| 1 | 1 | /** |
| 2 | − * Project reconciliation: key → external ids → same sourceUrl → (operator, trigram(normalized name) ≥ 0.85, same country). | |
| 3 | − * Timeline rows are deduplicated by (date, type, sha(description)); status changes become project_status_changed events. | |
| 2 | + * Project reconciliation + persistence (docs/PROJECT-EXTRACTION.md, docs/CLAIMS.md). | |
| 3 | + * | |
| 4 | + * - The announcement CLASS decides first: appointments, financing, PPAs, partnerships, customer deals, market research | |
| 5 | + * never create a project. Non-physical classes attach a timeline row + event to an existing, identifiable project. | |
| 6 | + * - Evidence threshold: a new project needs (explicit name or operator) + location + a development verb. | |
| 7 | + * - Capacity and investment go through the claim store (scope + sanity engine); a company-wide or portfolio figure is | |
| 8 | + * kept as a claim and never written to the project's columns. | |
| 9 | + * - Lifecycle transitions follow the state machine; a backward move needs a source that outranks the stored one. | |
| 10 | + * - Projects without coordinates are geocoded at CITY / METRO level (never more precise) from the metro seeds. | |
| 4 | 11 | */ |
| 5 | 12 | import { sql } from "@dci/db"; |
| 6 | −import { cleanText, newId, normalizeName, sha256, validLatLng, type ConfidenceLevel, type NormalizedProject } from "@dci/core"; | |
| 7 | −import { addRef, bump, provenanceFor, safeCountry, uniqueSlug, type IngestContext, type Tx } from "./common.js"; | |
| 13 | +import { | |
| 14 | + ASSOCIATED_CLASSES, | |
| 15 | + PHYSICAL_CLASSES, | |
| 16 | + cleanText, | |
| 17 | + classifyAiEvidence, | |
| 18 | + classifyInvestmentSemantics, | |
| 19 | + classifyScope, | |
| 20 | + findEvidence, | |
| 21 | + newId, | |
| 22 | + normalizeName, | |
| 23 | + projectTransition, | |
| 24 | + sha256, | |
| 25 | + validLatLng, | |
| 26 | + type CapacityPredicate, | |
| 27 | + type ClaimScope, | |
| 28 | + type ConfidenceLevel, | |
| 29 | + type EventType, | |
| 30 | + type InvestmentPredicate, | |
| 31 | + type NormalizedProject, | |
| 32 | + type ProjectClass, | |
| 33 | +} from "@dci/core"; | |
| 34 | +import { addRef, authority, bump, provenanceFor, safeCountry, uniqueSlug, type IngestContext, type Tx } from "./common.js"; | |
| 8 | 35 | import { emitDiffEvents, recordEvent, TRACKED_PROJECT_FIELDS } from "./events.js"; |
| 9 | 36 | import { validGeo } from "./geo.js"; |
| 37 | +import { geocodeCity } from "./geocode.js"; | |
| 10 | 38 | import { entityIdForKey, facilityIdForKey, upsertKey } from "./keys.js"; |
| 11 | −import { isPipelineStatus, shouldReplace, shouldReplaceGeo, shouldReplaceMw, type FieldObservation } from "./match.js"; | |
| 39 | +import { isPipelineStatus, shouldReplace, shouldReplaceGeo, type FieldObservation } from "./match.js"; | |
| 12 | 40 | import { assignMetro } from "./metros.js"; |
| 13 | 41 | import { operatorNames, resolveOperator } from "./operators.js"; |
| 14 | 42 | import { backingObservation, loadCurrentProvenance, writeProvenance, type CurrentProvenance, type ObservedField } from "./provenance.js"; |
| 43 | +import { markWinners, recordCapacityClaim, recordInvestmentClaim, reviewPriority, writeQualityFlags, writeTextClaim } from "./claims.js"; | |
| 44 | +import { resolveCampus } from "./campuses.js"; | |
| 15 | 45 | |
| 16 | 46 | type Row = Record<string, unknown>; |
| 17 | 47 | |
@@ -31,6 +61,8 @@ const COLS: Record<string, string> = { | ||
| 31 | 61 | expectedOpening: "expected_opening", |
| 32 | 62 | plannedMw: "planned_mw", |
| 33 | 63 | investmentUsd: "investment_usd", |
| 64 | + investmentCurrency: "investment_currency", | |
| 65 | + investmentOriginal: "investment_original", | |
| 34 | 66 | acreage: "acreage", |
| 35 | 67 | phaseCount: "phase_count", |
| 36 | 68 | isAi: "is_ai", |
@@ -38,8 +70,30 @@ const COLS: Record<string, string> = { | ||
| 38 | 70 | sourceUrl: "source_url", |
| 39 | 71 | confidence: "confidence", |
| 40 | 72 | externalIds: "external_ids", |
| 73 | + projectClass: "project_class", | |
| 74 | + evidenceLevel: "evidence_level", | |
| 75 | + aiEvidence: "ai_evidence", | |
| 76 | + capacityScope: "capacity_scope", | |
| 77 | + capacitySemantics: "capacity_semantics", | |
| 78 | + investmentScope: "investment_scope", | |
| 79 | + investmentSemantics: "investment_semantics", | |
| 80 | + developerId: "developer_id", | |
| 81 | + tenantId: "tenant_id", | |
| 82 | + campusId: "campus_id", | |
| 83 | + constructionStartedOn: "construction_started_on", | |
| 84 | + approvedOn: "approved_on", | |
| 85 | + permitFiledOn: "permit_filed_on", | |
| 86 | + openedOn: "opened_on", | |
| 87 | + reviewPriority: "review_priority", | |
| 88 | + hidden: "hidden", | |
| 41 | 89 | }; |
| 42 | 90 | |
| 91 | +/** External-id namespaces that identify ONE project (allowlist). */ | |
| 92 | +export const PROJECT_IDENTIFYING_KEYS: ReadonlySet<string> = new Set(["wikidata", "planning_ref", "permit_id", "case_number", "application_id", "docket", "planning_application", "rezoning_case", "eia_ref"]); | |
| 93 | + | |
| 94 | +/** Timeline / event type for a non-physical (associated) class. */ | |
| 95 | +const ASSOCIATED_EVENT: Record<string, EventType> = { POWER_AGREEMENT: "power_agreement", FINANCING: "investment_announced", ACQUISITION: "acquisition", PARTNERSHIP: "partnership", CUSTOMER_AGREEMENT: "customer_agreement", GRID_CONNECTION: "grid_connection", LAND_ACQUISITION: "land_acquired", PERMIT: "planning_filed", CONSTRUCTION_START: "construction_started", EXPANSION: "expansion_announced", NEW_BUILD: "project_announced" }; | |
| 96 | + | |
| 43 | 97 | async function loadProject(tx: Tx, id: string): Promise<Row | null> { |
| 44 | 98 | const r = (await tx.execute(sql`select * from projects where id = ${id}`))[0]; |
| 45 | 99 | if (!r) return null; |
@@ -48,25 +102,38 @@ async function loadProject(tx: Tx, id: string): Promise<Row | null> { | ||
| 48 | 102 | return out; |
| 49 | 103 | } |
| 50 | 104 | |
| 105 | +async function followMerged(tx: Tx, id: string): Promise<string> { | |
| 106 | + let cur = id; | |
| 107 | + for (let i = 0; i < 5; i++) { | |
| 108 | + const r = await tx.execute(sql`select merged_into from projects where id = ${cur}`); | |
| 109 | + const m = r[0]?.merged_into; | |
| 110 | + if (!m) break; | |
| 111 | + cur = String(m); | |
| 112 | + } | |
| 113 | + return cur; | |
| 114 | +} | |
| 115 | + | |
| 51 | 116 | async function resolveProject(tx: Tx, ctx: IngestContext, p: NormalizedProject, operatorId: string | null, country: string | null): Promise<string | null> { |
| 52 | 117 | const byKey = await entityIdForKey(tx, p.key, "project"); |
| 53 | − if (byKey) return byKey; | |
| 118 | + if (byKey) return followMerged(tx, byKey); | |
| 54 | 119 | if (p.externalIds) { |
| 55 | 120 | for (const [k, v] of Object.entries(p.externalIds)) { |
| 56 | − const r = await tx.execute(sql`select id from projects where external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`); | |
| 121 | + if (v == null || v === "" || !PROJECT_IDENTIFYING_KEYS.has(k)) continue; | |
| 122 | + const r = await tx.execute(sql`select id from projects where merged_into is null and external_ids @> ${JSON.stringify({ [k]: v })}::jsonb limit 1`); | |
| 57 | 123 | if (r[0]) return String(r[0].id); |
| 58 | 124 | } |
| 59 | 125 | } |
| 60 | 126 | const src = p.sourceUrl ?? null; |
| 61 | 127 | if (src) { |
| 62 | − const r = await tx.execute(sql`select id from projects where source_url = ${src} limit 1`); | |
| 63 | − if (r[0]) return String(r[0].id); | |
| 128 | + const r = await tx.execute(sql`select id from projects where merged_into is null and source_url = ${src} limit 1`); | |
| 129 | + if (r[0]) return followMerged(tx, String(r[0].id)); | |
| 64 | 130 | } |
| 65 | 131 | const norm = normalizeName(p.name); |
| 66 | 132 | if (!norm) return null; |
| 67 | 133 | const rows = await tx.execute(sql` |
| 68 | 134 | select id, similarity(normalized_name, ${norm}) as sim from projects |
| 69 | − where (${country}::text is null or country_iso2 is null or country_iso2 = ${country}) | |
| 135 | + where merged_into is null | |
| 136 | + and (${country}::text is null or country_iso2 is null or country_iso2 = ${country}) | |
| 70 | 137 | and (${operatorId}::text is null or operator_id is null or operator_id = ${operatorId}) |
| 71 | 138 | and (normalized_name = ${norm} or similarity(normalized_name, ${norm}) >= 0.85) |
| 72 | 139 | order by (operator_id = ${operatorId}) desc nulls last, sim desc limit 1`); |
@@ -78,24 +145,25 @@ async function resolveProject(tx: Tx, ctx: IngestContext, p: NormalizedProject, | ||
| 78 | 145 | export const ANNOUNCEMENT_WINDOW_DAYS = 30; |
| 79 | 146 | |
| 80 | 147 | /** |
| 81 | − * The same announcement covered by several outlets: same operator, same city (or same country when neither side | |
| 82 | − * names a city), planned MW within ±10 % and announced within 30 days → one project; the other articles become | |
| 83 | − * timeline rows / provenance. Requires an operator AND a size: without both, two "Ohio data center project" | |
| 84 | − * rows may well be different sites. | |
| 148 | + * The same announcement covered by several outlets: same operator (or, without an operator, the same city), planned MW | |
| 149 | + * within ±10 % and announced within 30 days → one project. Without an operator the city is mandatory. | |
| 85 | 150 | */ |
| 86 | 151 | async function sameAnnouncement(tx: Tx, ctx: IngestContext, p: NormalizedProject, operatorId: string | null, country: string | null): Promise<string | null> { |
| 87 | − if (!operatorId || p.plannedMw == null || p.plannedMw <= 0 || !p.announcedOn) return null; | |
| 152 | + if (p.plannedMw == null || p.plannedMw <= 0 || !p.announcedOn) return null; | |
| 88 | 153 | const day = p.announcedOn.length >= 10 ? p.announcedOn.slice(0, 10) : null; |
| 89 | 154 | if (!day) return null; |
| 90 | 155 | const city = p.city ? normalizeName(p.city) : null; |
| 156 | + if (!operatorId && !city) return null; | |
| 91 | 157 | const rows = await tx.execute(sql` |
| 92 | 158 | select id from projects |
| 93 | − where operator_id = ${operatorId} and planned_mw is not null | |
| 159 | + where merged_into is null and planned_mw is not null | |
| 94 | 160 | and planned_mw between ${p.plannedMw * 0.9} and ${p.plannedMw * 1.1} |
| 95 | 161 | and announced_on is not null and length(announced_on) >= 10 |
| 96 | 162 | and abs(announced_on::date - ${day}::date) <= ${ANNOUNCEMENT_WINDOW_DAYS} |
| 97 | 163 | and (${country}::text is null or country_iso2 is null or country_iso2 = ${country}) |
| 98 | − and ((${city}::text is not null and city is not null and lower(regexp_replace(city, '[^A-Za-z0-9]+', ' ', 'g')) = ${city}) or (${city}::text is null and city is null)) | |
| 164 | + and (${operatorId}::text is null or operator_id is null or operator_id = ${operatorId}) | |
| 165 | + and (${city}::text is null or city is null or lower(regexp_replace(city, '[^A-Za-z0-9]+', ' ', 'g')) = ${city}) | |
| 166 | + and (${operatorId}::text is not null or (${city}::text is not null and city is not null)) | |
| 99 | 167 | order by created_at asc limit 1`); |
| 100 | 168 | if (!rows[0]) return null; |
| 101 | 169 | ctx.stats.projectDedup = (ctx.stats.projectDedup ?? 0) + 1; |
@@ -112,85 +180,189 @@ function stored(prov: CurrentProvenance[], field: string, current: unknown): Fie | ||
| 112 | 180 | return { value: current, sourceKind: b?.sourceKind ?? null, confidence: b?.confidence ?? null, isEstimate: b?.isEstimate ?? false, observedAt: b?.lastObserved ?? null, sourceId: b?.sourceId ?? null, url: b?.url ?? null }; |
| 113 | 181 | } |
| 114 | 182 | |
| 183 | +const isCampusName = (s: string | null | undefined) => /\b(campus|park|complex|hub|cluster|gigafactory|estate)\b/i.test(s ?? ""); | |
| 184 | + | |
| 115 | 185 | export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedProject): Promise<void> { |
| 116 | 186 | const name = cleanText(p.name); |
| 117 | 187 | if (!name) throw new Error(`project ${p.key}: name is required`); |
| 118 | 188 | const url = p.provenance.url || p.sourceUrl || ctx.doc?.url || ""; |
| 119 | 189 | if (!url) throw new Error(`project ${p.key}: provenance url is required`); |
| 190 | + | |
| 191 | + const cls = (p.projectClass ?? null) as ProjectClass | null; | |
| 192 | + const physical = cls == null || PHYSICAL_CLASSES.has(cls); | |
| 193 | + const associated = cls != null && ASSOCIATED_CLASSES.has(cls); | |
| 194 | + | |
| 120 | 195 | const operator = p.operatorName ? await resolveOperator(tx, ctx, { name: p.operatorName }) : null; |
| 121 | 196 | const country0 = await safeCountry(tx, ctx, p.countryIso2); |
| 122 | − const geo = validGeo(p.geo) ? p.geo : null; | |
| 197 | + let geo = validGeo(p.geo) ? p.geo : null; | |
| 198 | + let geoMethod = geo ? "source" : null; | |
| 199 | + if (!geo && p.city) { | |
| 200 | + const g = await geocodeCity(tx, { city: p.city, regionName: p.regionName, countryIso2: country0 }); | |
| 201 | + if (g) { geo = { lat: g.lat, lng: g.lng, precision: g.precision, source: g.source }; geoMethod = g.source; } | |
| 202 | + } | |
| 123 | 203 | const metro = await assignMetro(tx, { lat: geo?.lat, lng: geo?.lng, city: p.city, countryIso2: country0 }); |
| 124 | 204 | const country = country0 ?? metro.countryIso2; |
| 125 | 205 | const facilityId = p.facilityKey ? await facilityIdForKey(tx, ctx, p.facilityKey) : null; |
| 126 | 206 | |
| 127 | 207 | const existingId = await resolveProject(tx, ctx, p, operator?.id ?? null, country); |
| 128 | 208 | const existing = existingId ? await loadProject(tx, existingId) : null; |
| 209 | + | |
| 210 | + // ─── veto: non-physical announcements never create a project; weak evidence never creates a project ───────────── | |
| 211 | + if (!existing) { | |
| 212 | + if (!physical) { ctx.stats.projectsVetoed = (ctx.stats.projectsVetoed ?? 0) + 1; return; } | |
| 213 | + if (p.evidenceLevel === "none") { ctx.stats.projectsVetoed = (ctx.stats.projectsVetoed ?? 0) + 1; return; } | |
| 214 | + } | |
| 215 | + // ─── associated class on an existing project: timeline row + event only, no field changes ───────────────────────── | |
| 216 | + if (existing && !physical) { | |
| 217 | + const evType = (cls && ASSOCIATED_EVENT[cls]) || "news"; | |
| 218 | + const date = p.announcedOn ?? ctx.day; | |
| 219 | + const desc = cleanText(p.description) ?? name; | |
| 220 | + const tid = `ptl_${sha256(`${existing.id}|${date}|${evType}|${desc.toLowerCase()}`).slice(0, 20)}`; | |
| 221 | + const ins = await tx.execute(sql`insert into project_timeline (id, project_id, event_date, event_type, description, source_id, document_id, url) | |
| 222 | + values (${tid}, ${existing.id}, ${date}, ${evType}, ${desc.slice(0, 400)}, ${ctx.run.sourceId}, ${ctx.doc?.documentId ?? null}, ${url}) on conflict (id) do nothing returning id`); | |
| 223 | + if (ins.length && associated) { | |
| 224 | + await recordEvent(tx, ctx, { entityType: "project", entityId: String(existing.id), eventType: evType, title: `${String(existing.name)}: ${cls!.toLowerCase().replace(/_/g, " ")} — ${name}`.slice(0, 300), summary: desc.slice(0, 400), newValue: { class: cls, title: name, mw: p.plannedMw ?? null, investmentUsd: p.investmentUsd ?? null }, significance: cls === "POWER_AGREEMENT" || cls === "GRID_CONNECTION" ? 60 : cls === "FINANCING" ? 55 : 45, confidence: p.provenance.confidence, effectiveDate: p.announcedOn ?? null, url, countryIso2: existing.countryIso2 as string | null, operatorId: existing.operatorId as string | null, metroId: existing.metroId as string | null, projectId: String(existing.id) }); | |
| 225 | + await tx.execute(sql`update projects set last_update = ${ctx.now} where id = ${existing.id}`); | |
| 226 | + } | |
| 227 | + // money attached to a deal / financing headline is a claim about the project, never its investment column | |
| 228 | + if (p.investmentUsd != null && p.investmentUsd > 0) { | |
| 229 | + const sem = classifyInvestmentSemantics(p.claimContext?.investmentUsd ?? name); | |
| 230 | + await recordInvestmentClaim(tx, ctx, "project", String(existing.id), { value: p.investmentUsd, currency: p.investmentCurrency ?? "USD", predicate: cls === "FINANCING" || cls === "ACQUISITION" ? "deal_value_usd" : (sem.predicate as InvestmentPredicate), scope: cls === "FINANCING" || cls === "ACQUISITION" ? "company" : sem.scope, evidence: p.claimContext?.investmentUsd ? findEvidence(p.claimContext.investmentUsd, p.investmentUsd, "usd") : null, recordScope: "project", publishedAt: p.announcedOn ?? null, provenance: p.provenance }); | |
| 231 | + } | |
| 232 | + addRef(ctx, "project", String(existing.id)); | |
| 233 | + bump(ctx, "project"); | |
| 234 | + return; | |
| 235 | + } | |
| 236 | + | |
| 129 | 237 | const prov = existing ? await loadCurrentProvenance(tx, "project", existingId!) : []; |
| 130 | 238 | const id = existing ? String(existing.id) : newId("project"); |
| 131 | 239 | const before: Row = existing ? { ...existing } : {}; |
| 132 | − const next: Row = existing ? { ...existing } : { id, status: "announced", geoPrecision: "unknown", isAi: false, confidence: "moderate", externalIds: {} }; | |
| 240 | + const next: Row = existing ? { ...existing } : { id, status: "announced", geoPrecision: "unknown", isAi: false, confidence: "moderate", externalIds: {}, aiEvidence: "unknown", hidden: false, reviewPriority: 0 }; | |
| 133 | 241 | const observed: ObservedField[] = []; |
| 242 | + const campusDesignation = isCampusName(name) || isCampusName(p.campusName); | |
| 134 | 243 | |
| 135 | 244 | const scalars: Record<string, unknown> = { |
| 136 | 245 | name, |
| 137 | 246 | city: cleanText(p.city), |
| 138 | 247 | regionName: cleanText(p.regionName), |
| 139 | − status: p.status && p.status !== "unknown" ? p.status : null, | |
| 140 | 248 | announcedOn: p.announcedOn ?? null, |
| 141 | 249 | expectedOpening: p.expectedOpening ?? null, |
| 142 | − investmentUsd: p.investmentUsd ?? null, | |
| 143 | 250 | acreage: p.acreage ?? null, |
| 144 | 251 | phaseCount: p.phaseCount ?? null, |
| 145 | 252 | description: cleanText(p.description), |
| 146 | 253 | sourceUrl: p.sourceUrl ?? null, |
| 254 | + constructionStartedOn: p.constructionStartedOn ?? null, | |
| 255 | + approvedOn: p.approvedOn ?? null, | |
| 256 | + permitFiledOn: p.permitFiledOn ?? null, | |
| 257 | + countryIso2: country, | |
| 147 | 258 | }; |
| 148 | 259 | for (const [field, v] of Object.entries(scalars)) { |
| 149 | 260 | if (v == null) continue; |
| 150 | 261 | observed.push({ field, value: v, provenance: p.provenance }); |
| 151 | 262 | if (shouldReplace(obs(v, field, p, ctx), stored(prov, field, existing?.[field])).replace) next[field] = v; |
| 152 | 263 | } |
| 264 | + | |
| 265 | + // ─── status: lifecycle state machine ──────────────────────────────────────────────────────────────────────────────── | |
| 266 | + const incomingStatus = p.status && p.status !== "unknown" ? p.status : null; | |
| 267 | + if (incomingStatus) { | |
| 268 | + observed.push({ field: "status", value: incomingStatus, provenance: p.provenance }); | |
| 269 | + const verdict = projectTransition(existing?.status as string | null, incomingStatus); | |
| 270 | + if (verdict === "forward" || verdict === "side" || verdict === "resume" || (verdict === "same" && !existing)) next.status = incomingStatus; | |
| 271 | + else if (verdict === "backward") { | |
| 272 | + // a backward move is accepted only from a source that outranks the stored one (a government filing correcting a news story) | |
| 273 | + const st = stored(prov, "status", existing?.status); | |
| 274 | + if (authority(ctx.run.sourceKind, p.provenance.confidence, false) > authority(st?.sourceKind, st?.confidence, false)) next.status = incomingStatus; | |
| 275 | + else await writeQualityFlags(tx, ctx, "project", id, [{ code: "status_backward", severity: "warn", message: `source says ${incomingStatus} but project is ${String(existing?.status)} — kept, lower-authority source`, field: "status" }]); | |
| 276 | + } else if (verdict === "invalid") { | |
| 277 | + await writeQualityFlags(tx, ctx, "project", id, [{ code: "status_invalid_transition", severity: "warn", message: `invalid lifecycle transition ${String(existing?.status)} → ${incomingStatus}`, field: "status" }]); | |
| 278 | + } | |
| 279 | + // stage dates derived from the announcement that moved the stage | |
| 280 | + const day = p.announcedOn ?? null; | |
| 281 | + if (day) { | |
| 282 | + if (incomingStatus === "under_construction" && !next.constructionStartedOn) next.constructionStartedOn = day; | |
| 283 | + if (incomingStatus === "approved" && !next.approvedOn) next.approvedOn = day; | |
| 284 | + if (incomingStatus === "permitting" && !next.permitFiledOn) next.permitFiledOn = day; | |
| 285 | + if ((incomingStatus === "operational" || incomingStatus === "partially_operational") && !next.openedOn) next.openedOn = day; | |
| 286 | + } | |
| 287 | + await writeTextClaim(tx, ctx, "project", id, "status", incomingStatus, p.provenance, { scope: campusDesignation ? "campus" : "facility", publishedAt: p.announcedOn ?? null }); | |
| 288 | + } | |
| 289 | + | |
| 290 | + // ─── capacity through the claim store ─────────────────────────────────────────────────────────────────────────────── | |
| 153 | 291 | if (p.plannedMw != null && p.plannedMw > 0) { |
| 154 | − observed.push({ field: "plannedMw", value: p.plannedMw, provenance: p.provenance }); | |
| 155 | − if (shouldReplaceMw(obs(p.plannedMw, "plannedMw", p, ctx), stored(prov, "plannedMw", existing?.plannedMw)).replace) next.plannedMw = p.plannedMw; | |
| 292 | + const context = p.claimContext?.plannedMw ?? null; | |
| 293 | + const scopeHint: ClaimScope = campusDesignation ? "campus" : "facility"; | |
| 294 | + const sc = p.capacityScope ? { scope: p.capacityScope as ClaimScope, reason: "extractor" } : classifyScope(context, context ? null : scopeHint); | |
| 295 | + const predicate = (p.capacitySemantics ?? "planned_power_mw") as CapacityPredicate; | |
| 296 | + const evidence = context ? findEvidence(context, p.plannedMw, "mw") ?? { text: context.slice(0, 600), start: 0, end: Math.min(600, context.length) } : null; | |
| 297 | + const decision = await recordCapacityClaim(tx, ctx, "project", id, { field: "plannedMw", predicate, value: p.plannedMw, scope: sc.scope, scopeReason: sc.reason, evidence, context, previous: (existing?.plannedMw as number | null) ?? null, recordScope: "project", campusDesignation, publishedAt: p.announcedOn ?? null, provenance: p.provenance, semanticsDefaulted: !p.capacitySemantics }); | |
| 298 | + if (decision.assign) { | |
| 299 | + observed.push({ field: "plannedMw", value: p.plannedMw, provenance: p.provenance, scope: sc.scope }); | |
| 300 | + if (shouldReplace(obs(p.plannedMw, "plannedMw", p, ctx), stored(prov, "plannedMw", existing?.plannedMw)).replace) { next.plannedMw = p.plannedMw; next.capacityScope = sc.scope; next.capacitySemantics = predicate; } | |
| 301 | + } else ctx.stats.unscopedClaims = (ctx.stats.unscopedClaims ?? 0) + 1; | |
| 156 | 302 | } |
| 303 | + // ─── investment through the claim store ───────────────────────────────────────────────────────────────────────────── | |
| 304 | + const money = p.investmentOriginal != null && p.investmentOriginal > 0 ? { amount: p.investmentOriginal, currency: p.investmentCurrency ?? "USD" } : p.investmentUsd != null && p.investmentUsd > 0 ? { amount: p.investmentUsd, currency: "USD" } : null; | |
| 305 | + if (money) { | |
| 306 | + const context = p.claimContext?.investmentUsd ?? null; | |
| 307 | + const sem = p.investmentSemantics && p.investmentScope ? { predicate: p.investmentSemantics as InvestmentPredicate, scope: p.investmentScope as ClaimScope, reason: "extractor" } : classifyInvestmentSemantics(context ?? name, campusDesignation ? "campus" : "facility"); | |
| 308 | + const evidence = context ? findEvidence(context, money.amount, "usd") ?? { text: context.slice(0, 600), start: 0, end: Math.min(600, context.length) } : null; | |
| 309 | + const decision = await recordInvestmentClaim(tx, ctx, "project", id, { value: money.amount, currency: money.currency, predicate: sem.predicate, scope: sem.scope, scopeReason: sem.reason, evidence, context, previous: (existing?.investmentUsd as number | null) ?? null, recordScope: "project", publishedAt: p.announcedOn ?? null, provenance: p.provenance }); | |
| 310 | + if (decision.assign && p.investmentUsd != null) { | |
| 311 | + observed.push({ field: "investmentUsd", value: p.investmentUsd, provenance: p.provenance, scope: sem.scope }); | |
| 312 | + if (shouldReplace(obs(p.investmentUsd, "investmentUsd", p, ctx), stored(prov, "investmentUsd", existing?.investmentUsd)).replace) { next.investmentUsd = p.investmentUsd; next.investmentScope = sem.scope; next.investmentSemantics = sem.predicate; } | |
| 313 | + } else ctx.stats.unscopedClaims = (ctx.stats.unscopedClaims ?? 0) + 1; | |
| 314 | + if (money.currency !== "USD") { next.investmentCurrency = money.currency; next.investmentOriginal = money.amount; } | |
| 315 | + } | |
| 316 | + | |
| 317 | + // ─── related entities ─────────────────────────────────────────────────────────────────────────────────────────────── | |
| 157 | 318 | if (operator) { |
| 158 | 319 | observed.push({ field: "operatorName", value: operator.name, provenance: p.provenance }); |
| 159 | 320 | if (shouldReplace(obs(operator.id, "operatorName", p, ctx), stored(prov, "operatorId", existing?.operatorId)).replace) next.operatorId = operator.id; |
| 160 | 321 | } |
| 322 | + if (p.developerName) { const d = await resolveOperator(tx, ctx, { name: p.developerName }); if (d) { next.developerId = d.id; observed.push({ field: "developerName", value: d.name, provenance: p.provenance }); } } | |
| 323 | + if (p.tenantName) { const t = await resolveOperator(tx, ctx, { name: p.tenantName }); if (t) { next.tenantId = t.id; observed.push({ field: "tenantName", value: t.name, provenance: p.provenance }); } } | |
| 161 | 324 | if (facilityId) next.facilityId = facilityId; |
| 325 | + if (p.campusName) { const c = await resolveCampus(tx, ctx, { name: p.campusName, operatorId: operator?.id ?? null, countryIso2: country, city: p.city ?? null, lat: geo?.lat ?? null, lng: geo?.lng ?? null }); if (c) next.campusId = c.id; } | |
| 162 | 326 | if (geo) { |
| 163 | − observed.push({ field: "geo", value: { lat: geo.lat, lng: geo.lng, precision: geo.precision, source: geo.source }, provenance: p.provenance }); | |
| 164 | − if (shouldReplaceGeo(geo.precision, existing?.geoPrecision as string | null, validLatLng(existing?.lat, existing?.lng))) { | |
| 165 | − next.lat = geo.lat; | |
| 166 | − next.lng = geo.lng; | |
| 167 | − next.geoPrecision = geo.precision; | |
| 168 | − } | |
| 169 | − } | |
| 170 | − if (country) { | |
| 171 | − observed.push({ field: "countryIso2", value: country, provenance: p.provenance }); | |
| 172 | − next.countryIso2 = next.countryIso2 ?? country; | |
| 327 | + observed.push({ field: "geo", value: { lat: geo.lat, lng: geo.lng, precision: geo.precision, source: geo.source }, provenance: { ...p.provenance, method: geoMethod ?? p.provenance.method } }); | |
| 328 | + if (shouldReplaceGeo(geo.precision, existing?.geoPrecision as string | null, validLatLng(existing?.lat, existing?.lng))) { next.lat = geo.lat; next.lng = geo.lng; next.geoPrecision = geo.precision; } | |
| 173 | 329 | } |
| 174 | 330 | { |
| 175 | 331 | const m = validLatLng(next.lat, next.lng) ? await assignMetro(tx, { lat: next.lat as number, lng: next.lng as number, city: next.city as string | null, countryIso2: next.countryIso2 as string | null }) : metro; |
| 176 | 332 | if (m.metroId) next.metroId = m.metroId; |
| 333 | + if (!next.countryIso2 && m.countryIso2) next.countryIso2 = m.countryIso2; | |
| 334 | + } | |
| 335 | + // ─── AI evidence: graded, never from one keyword ────────────────────────────────────────────────────────────────── | |
| 336 | + { | |
| 337 | + const rank: Record<string, number> = { unknown: 0, associated: 1, likely: 2, confirmed: 3 }; | |
| 338 | + const graded = classifyAiEvidence(`${name} ${p.description ?? ""}`); | |
| 339 | + const level = p.aiEvidence && p.aiEvidence !== "unknown" ? p.aiEvidence : graded.level; | |
| 340 | + if ((rank[level] ?? 0) > (rank[String(next.aiEvidence ?? "unknown")] ?? 0)) next.aiEvidence = level; | |
| 341 | + next.isAi = next.aiEvidence === "confirmed" || next.aiEvidence === "likely"; | |
| 177 | 342 | } |
| 178 | − if (/\b(ai|gpu|hpc|accelerated|inference|training)\b/i.test(`${name} ${p.description ?? ""}`)) next.isAi = true; | |
| 343 | + if (cls) { next.projectClass = cls; if (p.evidenceLevel) next.evidenceLevel = p.evidenceLevel; } | |
| 179 | 344 | if (p.externalIds && Object.keys(p.externalIds).length) next.externalIds = { ...((next.externalIds as Record<string, unknown>) ?? {}), ...p.externalIds }; |
| 180 | 345 | |
| 346 | + // ─── persist ──────────────────────────────────────────────────────────────────────────────────────────────────────── | |
| 181 | 347 | const normalizedName = normalizeName(String(next.name)); |
| 182 | 348 | if (!existing) { |
| 183 | 349 | const slug = await uniqueSlug(tx, "projects", `${operator && !normalizeName(name).includes(normalizeName(operator.name)) ? `${operator.name} ` : ""}${name}`, (next.city as string | null) ?? (next.countryIso2 as string | null)); |
| 350 | + next.confidence = p.evidenceLevel === "weak" ? "unverified" : (p.provenance.confidence ?? "moderate"); | |
| 184 | 351 | await tx.execute(sql`insert into projects (id, slug, name, normalized_name, operator_id, facility_id, metro_id, country_iso2, city, region_name, lat, lng, geo_precision, status, announced_on, expected_opening, planned_mw, |
| 185 | − investment_usd, acreage, phase_count, is_ai, description, source_url, confidence, external_ids, last_update) | |
| 352 | + investment_usd, investment_currency, investment_original, acreage, phase_count, is_ai, description, source_url, confidence, external_ids, last_update, | |
| 353 | + project_class, evidence_level, ai_evidence, capacity_scope, capacity_semantics, investment_scope, investment_semantics, developer_id, tenant_id, campus_id, construction_started_on, approved_on, permit_filed_on, opened_on, hidden) | |
| 186 | 354 | values (${id}, ${slug}, ${next.name}, ${normalizedName}, ${next.operatorId ?? null}, ${next.facilityId ?? null}, ${next.metroId ?? null}, ${next.countryIso2 ?? null}, ${next.city ?? null}, ${next.regionName ?? null}, |
| 187 | 355 | ${next.lat ?? null}, ${next.lng ?? null}, ${next.geoPrecision ?? "unknown"}, ${next.status ?? "announced"}, ${next.announcedOn ?? null}, ${next.expectedOpening ?? null}, ${next.plannedMw ?? null}, |
| 188 | − ${next.investmentUsd ?? null}, ${next.acreage ?? null}, ${next.phaseCount ?? null}, ${!!next.isAi}, ${next.description ?? null}, ${next.sourceUrl ?? null}, ${p.provenance.confidence ?? "moderate"}, ${JSON.stringify(next.externalIds ?? {})}::jsonb, ${ctx.now})`); | |
| 356 | + ${next.investmentUsd ?? null}, ${next.investmentCurrency ?? null}, ${next.investmentOriginal ?? null}, ${next.acreage ?? null}, ${next.phaseCount ?? null}, ${!!next.isAi}, ${next.description ?? null}, ${next.sourceUrl ?? null}, ${next.confidence}, ${JSON.stringify(next.externalIds ?? {})}::jsonb, ${ctx.now}, | |
| 357 | + ${next.projectClass ?? null}, ${next.evidenceLevel ?? null}, ${next.aiEvidence ?? "unknown"}, ${next.capacityScope ?? null}, ${next.capacitySemantics ?? null}, ${next.investmentScope ?? null}, ${next.investmentSemantics ?? null}, ${next.developerId ?? null}, ${next.tenantId ?? null}, ${next.campusId ?? null}, | |
| 358 | + ${next.constructionStartedOn ?? null}, ${next.approvedOn ?? null}, ${next.permitFiledOn ?? null}, ${next.openedOn ?? null}, false)`); | |
| 189 | 359 | ctx.stats.created++; |
| 360 | + if (p.evidenceLevel === "weak") await writeQualityFlags(tx, ctx, "project", id, [{ code: "project_weak_evidence", severity: "warn", message: "created from weak evidence (missing operator/name or location) — review", field: "name" }]); | |
| 361 | + if (/\s[|]\s|\s[–—]\s| - /.test(name) || name.length > 90) await writeQualityFlags(tx, ctx, "project", id, [{ code: "project_title_like_name", severity: "warn", message: "project name looks like an article headline", field: "name" }]); | |
| 190 | 362 | } else { |
| 191 | 363 | const sets = []; |
| 192 | 364 | for (const [camel, col] of Object.entries(COLS)) { |
| 193 | − if (camel === "confidence") continue; | |
| 365 | + if (camel === "confidence" || camel === "reviewPriority" || camel === "hidden") continue; | |
| 194 | 366 | if (JSON.stringify(before[camel] ?? null) === JSON.stringify(next[camel] ?? null)) continue; |
| 195 | 367 | if (camel === "externalIds") sets.push(sql`${sql.identifier(col)} = ${JSON.stringify(next[camel] ?? {})}::jsonb`); |
| 196 | 368 | else sets.push(sql`${sql.identifier(col)} = ${next[camel] as string | number | boolean | null}`); |
@@ -206,10 +378,21 @@ export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedPro | ||
| 206 | 378 | addRef(ctx, "project", id); |
| 207 | 379 | bump(ctx, "project"); |
| 208 | 380 | await writeProvenance(tx, ctx, "project", id, observed, p.key); |
| 381 | + await markWinners(tx, ctx, "project", id, ["status", "plannedMw", "investmentUsd", "expectedOpening", "announcedOn", "countryIso2", "city", "name"].map((f) => ({ field: f, value: next[f] })).concat([{ field: "operatorId", value: next.operatorId }])); | |
| 382 | + // review priority from open flags + size | |
| 383 | + { | |
| 384 | + const r = (await tx.execute(sql`select coalesce(max(priority), 0) as p from quality_flags where entity_type = 'project' and entity_id = ${id} and status = 'open'`))[0]; | |
| 385 | + const rp = Math.max(Number(r?.p ?? 0), reviewPriority({ mw: next.plannedMw as number | null, investmentUsd: next.investmentUsd as number | null, confidence: String(next.confidence ?? ""), ai: !!next.isAi }) - 30); | |
| 386 | + if (!ctx.run.dryRun) await tx.execute(sql`update projects set review_priority = ${Math.max(0, rp)} where id = ${id}`); | |
| 387 | + } | |
| 209 | 388 | |
| 210 | − // timeline (deduplicated); another outlet's coverage of an existing project is kept as a "reported" row | |
| 389 | + // ─── timeline (deduplicated) ──────────────────────────────────────────────────────────────────────────────────────── | |
| 211 | 390 | let timelineAdded = 0; |
| 212 | 391 | const timeline = [...(p.timeline ?? [])]; |
| 392 | + if (!timeline.length && p.announcedOn && cls) { | |
| 393 | + const evType = ASSOCIATED_EVENT[cls] ?? "project_announced"; | |
| 394 | + timeline.push({ date: p.announcedOn, type: incomingStatus === "under_construction" ? "construction_started" : incomingStatus === "approved" ? "planning_approved" : incomingStatus === "permitting" ? "planning_filed" : evType, description: cleanText(p.description)?.slice(0, 300) ?? name, url }); | |
| 395 | + } | |
| 213 | 396 | if (existing && p.sourceUrl && existing.sourceUrl && p.sourceUrl !== existing.sourceUrl && p.announcedOn && p.description) { |
| 214 | 397 | timeline.push({ date: p.announcedOn, type: "reported", description: `Also reported: ${String(p.description).slice(0, 300)}`, url: p.sourceUrl }); |
| 215 | 398 | } |
@@ -222,18 +405,19 @@ export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedPro | ||
| 222 | 405 | } |
| 223 | 406 | if (timelineAdded && existing) await tx.execute(sql`update projects set last_update = ${ctx.now} where id = ${id}`); |
| 224 | 407 | |
| 225 | − // events | |
| 226 | − const confidence = (existing ? String(existing.confidence) : (p.provenance.confidence ?? "moderate")) as ConfidenceLevel; | |
| 408 | + // ─── events ───────────────────────────────────────────────────────────────────────────────────────────────────────── | |
| 409 | + const confidence = (existing ? String(existing.confidence) : String(next.confidence ?? "moderate")) as ConfidenceLevel; | |
| 227 | 410 | if (!existing) { |
| 228 | 411 | const mw = next.plannedMw as number | null; |
| 229 | 412 | const where = [next.city, next.countryIso2].filter(Boolean).join(", "); |
| 413 | + const evType: EventType = incomingStatus === "under_construction" ? "construction_started" : incomingStatus === "approved" ? "planning_approved" : incomingStatus === "permitting" ? "planning_filed" : cls === "EXPANSION" ? "expansion_announced" : cls === "LAND_ACQUISITION" ? "land_acquired" : cls === "GRID_CONNECTION" ? "grid_connection" : "project_announced"; | |
| 230 | 414 | await recordEvent(tx, ctx, { |
| 231 | 415 | entityType: "project", |
| 232 | 416 | entityId: id, |
| 233 | − eventType: "project_announced", | |
| 417 | + eventType: evType, | |
| 234 | 418 | title: `New project: ${next.name}${operator && !String(next.name).toLowerCase().includes(operator.name.toLowerCase()) ? ` (${operator.name})` : ""}${where ? ` — ${where}` : ""}`, |
| 235 | − summary: [mw != null ? `${mw} MW planned.` : null, next.status ? `Status: ${String(next.status).replace(/_/g, " ")}.` : null, next.expectedOpening ? `Expected ${String(next.expectedOpening)}.` : null].filter(Boolean).join(" ") || null, | |
| 236 | − newValue: { name: next.name, status: next.status, plannedMw: mw }, | |
| 419 | + summary: [mw != null ? `${mw} MW planned (${String(next.capacityScope ?? "site")} scope).` : null, next.status ? `Status: ${String(next.status).replace(/_/g, " ")}.` : null, next.expectedOpening ? `Expected ${String(next.expectedOpening)}.` : null].filter(Boolean).join(" ") || null, | |
| 420 | + newValue: { name: next.name, status: next.status, plannedMw: mw, class: cls }, | |
| 237 | 421 | significance: mw != null && mw >= 100 ? 75 : isPipelineStatus(next.status as string) ? 60 : 50, |
| 238 | 422 | confidence, |
| 239 | 423 | effectiveDate: (next.announcedOn as string | null) ?? null, |
@@ -242,23 +426,38 @@ export async function ingestProject(tx: Tx, ctx: IngestContext, p: NormalizedPro | ||
| 242 | 426 | operatorId: next.operatorId as string | null, |
| 243 | 427 | metroId: next.metroId as string | null, |
| 244 | 428 | projectId: id, |
| 429 | + isAi: !!next.isAi, | |
| 430 | + reviewStatus: p.evidenceLevel === "weak" ? "pending" : "auto", | |
| 245 | 431 | }); |
| 246 | 432 | } else { |
| 247 | 433 | const names = await operatorNames(tx, ctx, [before.operatorId as string | null, next.operatorId as string | null]); |
| 248 | − await emitDiffEvents(tx, ctx, { | |
| 249 | − entityType: "project", | |
| 250 | − entityId: id, | |
| 251 | − entityName: String(next.name), | |
| 252 | − before, | |
| 253 | − after: next, | |
| 254 | − specs: TRACKED_PROJECT_FIELDS, | |
| 255 | − url, | |
| 256 | − confidence, | |
| 257 | − countryIso2: next.countryIso2 as string | null, | |
| 258 | − operatorId: next.operatorId as string | null, | |
| 259 | − metroId: next.metroId as string | null, | |
| 260 | − projectId: id, | |
| 261 | − operatorNames: names, | |
| 262 | − }); | |
| 434 | + await emitDiffEvents(tx, ctx, { entityType: "project", entityId: id, entityName: String(next.name), before, after: next, specs: TRACKED_PROJECT_FIELDS, url, confidence, countryIso2: next.countryIso2 as string | null, operatorId: next.operatorId as string | null, metroId: next.metroId as string | null, projectId: id, operatorNames: names }); | |
| 435 | + // delayed / cancelled get their own, louder event types on top of the generic status change | |
| 436 | + if (before.status !== next.status && (next.status === "delayed" || next.status === "cancelled")) { | |
| 437 | + await recordEvent(tx, ctx, { entityType: "project", entityId: id, eventType: next.status === "delayed" ? "project_delayed" : "project_cancelled", title: `${String(next.name)}: ${next.status === "delayed" ? "delayed" : "cancelled"}`, summary: cleanText(p.description)?.slice(0, 400) ?? null, oldValue: before.status, newValue: next.status, significance: next.status === "cancelled" ? 85 : 70, confidence, effectiveDate: p.announcedOn ?? null, url, countryIso2: next.countryIso2 as string | null, operatorId: next.operatorId as string | null, metroId: next.metroId as string | null, projectId: id, isAi: !!next.isAi }); | |
| 438 | + } | |
| 263 | 439 | } |
| 264 | 440 | } |
| 441 | + | |
| 442 | +/** Admin: hide a false-positive project (kept for audit, removed from every listing and aggregate). */ | |
| 443 | +export async function hideProject(tx: Tx, id: string, reason: string, decidedBy = "admin"): Promise<void> { | |
| 444 | + await tx.execute(sql`update projects set hidden = true, updated_at = now() where id = ${id}`); | |
| 445 | + await tx.execute(sql`update events set review_status = 'rejected' where project_id = ${id}`); | |
| 446 | + await tx.execute(sql`insert into quality_flags (id, entity_type, entity_id, code, severity, field, message, priority, status, resolution, resolved_by, resolved_at, dedupe_key) | |
| 447 | + values (${`flg_${sha256(`project|${id}|hidden`).slice(0, 16)}`}, 'project', ${id}, 'project_false_positive', 'critical', 'name', ${reason.slice(0, 1000)}, 0, 'resolved', 'hidden', ${decidedBy}, now(), ${`project|${id}|project_false_positive|`}) | |
| 448 | + on conflict (dedupe_key) do update set status = 'resolved', resolution = 'hidden', resolved_by = ${decidedBy}, resolved_at = now(), message = excluded.message`); | |
| 449 | +} | |
| 450 | + | |
| 451 | +/** Admin: fold `fromId` into `intoId` (keys, provenance, claims, timeline, events, news). */ | |
| 452 | +export async function mergeProjects(tx: Tx, fromId: string, intoId: string, decidedBy = "admin"): Promise<void> { | |
| 453 | + if (fromId === intoId) return; | |
| 454 | + await tx.execute(sql`update entity_keys set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${fromId}`); | |
| 455 | + await tx.execute(sql`update provenance set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${fromId} and not exists (select 1 from provenance q where q.entity_type = 'project' and q.entity_id = ${intoId} and q.field = provenance.field and q.source_id = provenance.source_id and q.url = provenance.url)`); | |
| 456 | + await tx.execute(sql`delete from provenance where entity_type = 'project' and entity_id = ${fromId}`); | |
| 457 | + await tx.execute(sql`update claims set subject_id = ${intoId} where subject_type = 'project' and subject_id = ${fromId}`); | |
| 458 | + await tx.execute(sql`update project_timeline set project_id = ${intoId} where project_id = ${fromId}`); | |
| 459 | + await tx.execute(sql`update events set project_id = ${intoId}, entity_id = case when entity_type = 'project' and entity_id = ${fromId} then ${intoId} else entity_id end where project_id = ${fromId} or (entity_type = 'project' and entity_id = ${fromId})`); | |
| 460 | + await tx.execute(sql`update news_items set project_id = ${intoId} where project_id = ${fromId}`); | |
| 461 | + await tx.execute(sql`update projects set merged_into = ${intoId}, hidden = true, updated_at = now() where id = ${fromId}`); | |
| 462 | + await tx.execute(sql`update quality_flags set status = 'resolved', resolution = ${`merged into ${intoId}`}, resolved_by = ${decidedBy}, resolved_at = now() where entity_type = 'project' and entity_id = ${fromId} and status = 'open'`); | |
| 463 | +} | |
modified
apps/worker/src/ingest/provenance.ts
+5 −3
@@ -51,6 +51,8 @@ export interface ObservedField { | ||
| 51 | 51 | value: unknown; |
| 52 | 52 | provenance: Provenance; |
| 53 | 53 | entityKey?: string; |
| 54 | + /** claim scope of the observation (building / facility / campus / …) when known */ | |
| 55 | + scope?: string | null; | |
| 54 | 56 | } |
| 55 | 57 | |
| 56 | 58 | export function provenanceId(entityType: string, entityId: string, field: string, sourceId: string, url: string): string { |
@@ -71,13 +73,13 @@ export async function writeProvenance(tx: Tx, ctx: IngestContext, entityType: st | ||
| 71 | 73 | const observed = p.lastObserved || ctx.now; |
| 72 | 74 | await tx.execute(sql` |
| 73 | 75 | insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, document_id, url, first_observed, last_observed, retrieved_at, |
| 74 | − confidence, is_estimate, method, extractor_version, is_current, note) | |
| 76 | + confidence, is_estimate, method, extractor_version, is_current, note, run_id, scope) | |
| 75 | 77 | values (${id}, ${entityType}, ${entityId}, ${f.field}, ${valueJson}::jsonb, ${sourceId}, ${p.connectorId || ctx.run.connectorId}, ${p.documentId ?? ctx.doc?.documentId ?? null}, ${url}, |
| 76 | − ${p.firstObserved || observed}, ${observed}, ${p.retrievedAt || ctx.now}, ${p.confidence ?? "moderate"}, ${!!p.isEstimate}, ${p.method ?? null}, ${p.extractorVersion ?? null}, true, ${p.note ?? null}) | |
| 78 | + ${p.firstObserved || observed}, ${observed}, ${p.retrievedAt || ctx.now}, ${p.confidence ?? "moderate"}, ${!!p.isEstimate}, ${p.method ?? null}, ${p.extractorVersion ?? null}, true, ${p.note ?? null}, ${ctx.run.runId}, ${f.scope ?? null}) | |
| 77 | 79 | on conflict (entity_type, entity_id, field, source_id, url) do update set |
| 78 | 80 | value = excluded.value, last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, confidence = excluded.confidence, |
| 79 | 81 | is_estimate = excluded.is_estimate, method = excluded.method, extractor_version = excluded.extractor_version, document_id = coalesce(excluded.document_id, provenance.document_id), |
| 80 | − is_current = true, note = excluded.note`); | |
| 82 | + is_current = true, note = excluded.note, run_id = excluded.run_id, scope = coalesce(excluded.scope, provenance.scope)`); | |
| 81 | 83 | // the same source reporting the field from another URL earlier → superseded |
| 82 | 84 | await tx.execute(sql`update provenance set is_current = false where entity_type = ${entityType} and entity_id = ${entityId} and field = ${f.field} and source_id = ${sourceId} and id <> ${id} and is_current = true`); |
| 83 | 85 | n++; |
modified
apps/worker/src/main.ts
+17 −1
@@ -28,6 +28,8 @@ import { computeRankings } from "./rankings.js"; | ||
| 28 | 28 | import { CRAWL_QUEUE, MAINTENANCE_QUEUE, QUEUE_PREFIX, RUN_LOCK_TTL_SECONDS, WORKER_STATUS_KEY, closeScheduler, getRedis, publishQueueMetrics, queueSnapshot, runLockKey, startSchedulerLoop, type CrawlJobData, type MaintenanceJobData } from "./scheduler.js"; |
| 29 | 29 | import { ensureBucket } from "./storage.js"; |
| 30 | 30 | import { cleanupMaintenance, reconcileOrphanRuns } from "./maintenance.js"; |
| 31 | +import { snapshotAndCheck, qualitySweep, dataGaps } from "./quality.js"; | |
| 32 | +import { traceDocument } from "./trace.js"; | |
| 31 | 33 | |
| 32 | 34 | export interface WorkerStatus { |
| 33 | 35 | startedAt: string; |
@@ -156,6 +158,8 @@ export async function startWorker(opts: { withScheduler?: boolean } = {}): Promi | ||
| 156 | 158 | case "metrics": return await computeDailyMetrics(); |
| 157 | 159 | case "refresh-stats": return await refreshStats(); |
| 158 | 160 | case "cleanup": return await cleanupMaintenance(); |
| 161 | + case "snapshot": return await snapshotAndCheck(); | |
| 162 | + case "quality": return await qualitySweep(); | |
| 159 | 163 | default: throw new Error(`unknown maintenance kind ${String((job.data as { kind?: string }).kind)}`); |
| 160 | 164 | } |
| 161 | 165 | } finally { running.delete(String(job.id)); } |
@@ -176,7 +180,19 @@ export async function startWorker(opts: { withScheduler?: boolean } = {}): Promi | ||
| 176 | 180 | onTick: (r) => { lastScheduler = { at: new Date().toISOString(), enqueued: r.enqueued.length, checked: r.checked }; if (r.enqueued.length) console.error(`[scheduler] enqueued ${r.enqueued.join(", ")}`); }, |
| 177 | 181 | onError: (e) => console.error(`[scheduler] tick failed: ${e.message}`), |
| 178 | 182 | }); |
| 179 | − const http = startHealthHttp({ port: env.workerPort, role: "worker", health: async () => ({ ...(await status()), queueDepth: await queueSnapshot().catch(() => []) }), ready: () => !shuttingDown, beforeScrape: refreshGauges, log: (m) => console.error(`[worker] ${m}`) }); | |
| 183 | + const http = startHealthHttp({ | |
| 184 | + port: env.workerPort, | |
| 185 | + role: "worker", | |
| 186 | + health: async () => ({ ...(await status()), queueDepth: await queueSnapshot().catch(() => []) }), | |
| 187 | + ready: () => !shuttingDown, | |
| 188 | + beforeScrape: refreshGauges, | |
| 189 | + log: (m) => console.error(`[worker] ${m}`), | |
| 190 | + routes: { | |
| 191 | + // extraction debugger (internal network only; the API proxies it behind the admin token) | |
| 192 | + "/trace": async (url) => { const id = url.pathname.split("/")[2]; if (!id) return { status: 400, body: { error: "usage: /trace/<document-id>" } }; return { status: 200, body: await traceDocument(decodeURIComponent(id), { live: url.searchParams.get("live") === "1" }) }; }, | |
| 193 | + "/data-gaps": async () => ({ status: 200, body: await dataGaps() }), | |
| 194 | + }, | |
| 195 | + }); | |
| 180 | 196 | await heartbeat(); |
| 181 | 197 | const hb = setInterval(() => void heartbeat(), 15_000); |
| 182 | 198 | console.error(`[worker] ready · queues ${env.queues.join(",")} · crawl concurrency ${env.crawlConcurrency} · scheduler ${withScheduler ? "embedded" : "external"} · ${loaded.length} connectors · host ${env.hostname}`); |
modified
apps/worker/src/pipeline.ts
+43 −13
@@ -10,14 +10,14 @@ import type { DiscoveredUrl, RawDocument } from "@dci/connectors"; | ||
| 10 | 10 | import { entityIsValid, pageTitle } from "@dci/connectors"; |
| 11 | 11 | import type { NormalizedEntity, ValidationIssue } from "@dci/core"; |
| 12 | 12 | import { contentFingerprint, sha256, newId } from "@dci/core"; |
| 13 | −import { getDb, connectorRuns, connectors as connectorsTable, eq, inArray } from "@dci/db"; | |
| 13 | +import { getDb, connectorRuns, connectors as connectorsTable, eq, inArray, sql } from "@dci/db"; | |
| 14 | 14 | import { loadAllConnectors, requireConnector, type LoadedConnector } from "./configs.js"; |
| 15 | 15 | import { createRunContext, type LogLevel, type LogLine, type RunContext } from "./context.js"; |
| 16 | 16 | import { connectorDocStats, documentIdFor, dueDocuments, nextCheckByGroup, recordFetch, recordVersion, registerDiscovered, textProjection, updateExtraction, type DocumentRow } from "./documents.js"; |
| 17 | 17 | import { ingestEntities } from "./ingest/index.js"; |
| 18 | 18 | import type { IngestDocRef, IngestRun, IngestStats } from "./ingest/contract.js"; |
| 19 | 19 | import { metrics } from "./prom.js"; |
| 20 | −import { healthFrom, minIntervalMs, parseDiscoveredFrom, shouldSkipExtraction } from "./scheduling.js"; | |
| 20 | +import { connectorHealthFrom, healthFrom, minIntervalMs, parseDiscoveredFrom, shouldSkipExtraction } from "./scheduling.js"; | |
| 21 | 21 | import { getRaw, putRaw } from "./storage.js"; |
| 22 | 22 | |
| 23 | 23 | export type RunTask = "discover" | "crawl" | "full" | "reprocess"; |
@@ -35,6 +35,10 @@ export interface RunOptions { | ||
| 35 | 35 | logLevel?: LogLevel; |
| 36 | 36 | /** collect up to N normalized entities for display (dry runs) */ |
| 37 | 37 | sampleEntities?: number; |
| 38 | + /** quarantine: fetch, archive and extract normally but roll the ingest back (set from connectors.quarantine) */ | |
| 39 | + quarantine?: boolean; | |
| 40 | + /** reprocess only documents whose stored extractor version differs from the current one */ | |
| 41 | + staleOnly?: boolean; | |
| 38 | 42 | } |
| 39 | 43 | |
| 40 | 44 | export interface RunStats extends Record<string, number> { |
@@ -137,6 +141,7 @@ function docToDiscovered(doc: DocumentRow): DiscoveredUrl { | ||
| 137 | 141 | |
| 138 | 142 | function sumIngest(stats: RunStats, s: IngestStats): void { |
| 139 | 143 | stats.created += s.created; stats.updated += s.updated; stats.merged += s.merged; stats.pendingMatches += s.pendingMatches; stats.events += s.events; stats.provenanceRows += s.provenanceRows; |
| 144 | + stats.claims = (stats.claims ?? 0) + (s.claims ?? 0); stats.qualityFlags = (stats.qualityFlags ?? 0) + (s.qualityFlags ?? 0); stats.unscopedClaims = (stats.unscopedClaims ?? 0) + (s.unscopedClaims ?? 0); stats.projectsVetoed = (stats.projectsVetoed ?? 0) + (s.projectsVetoed ?? 0); stats.campusLinks = (stats.campusLinks ?? 0) + (s.campusLinks ?? 0); | |
| 140 | 145 | } |
| 141 | 146 | |
| 142 | 147 | interface DocOutcome { kind: "fetched" | "notModified" | "failed" | "unchanged" | "changed" | "aborted"; ms: number } |
@@ -162,11 +167,11 @@ async function extractAndIngest(ctx: RunContext, loaded: LoadedConnector, doc: D | ||
| 162 | 167 | const classifier = (raw.meta?.classifier as string | undefined) ?? null; |
| 163 | 168 | metrics.ingest.inc({ connector: loaded.cfg.id, result: "rejected" }, report.rejected); |
| 164 | 169 | if (!valid.length) return { ingest: null, valid: 0, rejected: report.rejected, classifier, pageType }; |
| 165 | − const run: IngestRun = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId: ctx.runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: Boolean(opts.dryRun) }; | |
| 170 | + const run: IngestRun = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId: ctx.runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: Boolean(opts.dryRun) || Boolean(opts.quarantine) }; | |
| 166 | 171 | const ref: IngestDocRef = { documentId: doc.id, url: raw.finalUrl || doc.url, pageType, fetchedAt: raw.fetchedAt }; |
| 167 | 172 | const ingest = await ingestEntities(run, valid, ref); |
| 168 | 173 | sumIngest(stats, ingest); |
| 169 | − if (!opts.dryRun) { | |
| 174 | + if (!opts.dryRun && !opts.quarantine) { | |
| 170 | 175 | metrics.ingest.inc({ connector: loaded.cfg.id, result: "created" }, ingest.created); |
| 171 | 176 | metrics.ingest.inc({ connector: loaded.cfg.id, result: "updated" }, ingest.updated); |
| 172 | 177 | metrics.ingest.inc({ connector: loaded.cfg.id, result: "merged" }, ingest.merged); |
@@ -226,7 +231,7 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do | ||
| 226 | 231 | try { storageKey = (await putRaw(cfg.id, doc.id, hash, raw.body, raw.contentType, raw.finalUrl)).key; } catch (e) { ctx.log("warn", `archive failed for ${doc.url}: ${(e as Error).message}`); } |
| 227 | 232 | } |
| 228 | 233 | |
| 229 | − const skip = markupOnly || shouldSkipExtraction({ newHash: hash, storedHash: doc.contentHash, force, extractorVersion: connector.parserVersion, storedExtractorVersion: doc.extractorVersion, notModified: false }); | |
| 234 | + const skip = markupOnly || shouldSkipExtraction({ newHash: hash, storedHash: doc.contentHash, force, extractorVersion: loaded.extractorVersion, storedExtractorVersion: doc.extractorVersion, notModified: false }); | |
| 230 | 235 | // markup-only: adopt the new fingerprint (so a stabilised page re-syncs after one fetch) but keep the archived |
| 231 | 236 | // body — its text projection is identical, which is what versions/diffs are built from. |
| 232 | 237 | const plan = await recordFetch(doc, raw, { contentHash: hash, changed, storageKey, title }, cfg.schedule, { dryRun }); |
@@ -250,9 +255,9 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do | ||
| 250 | 255 | extractError = `extract: ${(e as Error).message}`.slice(0, 500); |
| 251 | 256 | ctx.log("error", `${doc.url} ${extractError}`); |
| 252 | 257 | } |
| 253 | − await updateExtraction(doc.id, { entityRefs: ingestStats?.refs ?? [], extractOk: extractError === null, extractCount: validCount, extractorVersion: connector.parserVersion, error: extractError, pageType, classifier }, dryRun); | |
| 258 | + await updateExtraction(doc.id, { entityRefs: ingestStats?.refs ?? [], extractOk: extractError === null, extractCount: validCount, extractorVersion: loaded.extractorVersion, error: extractError, pageType, classifier }, dryRun); | |
| 254 | 259 | if (changed || !doc.contentHash || force) { |
| 255 | − await recordVersion(doc, raw, hash, storageKey, ingestStats?.changes ?? [], { dryRun, runId: ctx.runId }); | |
| 260 | + await recordVersion(doc, raw, hash, storageKey, ingestStats?.changes ?? [], { dryRun, runId: ctx.runId, extractorVersion: loaded.extractorVersion }); | |
| 256 | 261 | } |
| 257 | 262 | ctx.log("info", `${doc.url} → ${raw.status} ${raw.fetcher} L${raw.level} ${raw.durationMs}ms ${changed ? "CHANGED" : "same"} · ${validCount} entities${ingestStats ? ` (+${ingestStats.created} new, ${ingestStats.updated} upd, ${ingestStats.events} events)` : ""} next=${plan.nextCheck ?? "never"}`); |
| 258 | 263 | return { kind: changed ? "changed" : "fetched", ms: Date.now() - t0 }; |
@@ -262,7 +267,7 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do | ||
| 262 | 267 | stats.failed++; |
| 263 | 268 | ctx.log("error", `${doc.url} unexpected: ${(e as Error).stack ?? (e as Error).message}`); |
| 264 | 269 | try { |
| 265 | − if (!dryRun) await updateExtraction(doc.id, { entityRefs: [], extractOk: false, extractCount: 0, extractorVersion: connector.parserVersion, error: `pipeline: ${(e as Error).message}`.slice(0, 500) }); | |
| 270 | + if (!dryRun) await updateExtraction(doc.id, { entityRefs: [], extractOk: false, extractCount: 0, extractorVersion: loaded.extractorVersion, error: `pipeline: ${(e as Error).message}`.slice(0, 500) }); | |
| 266 | 271 | } catch { /* ignore */ } |
| 267 | 272 | return { kind: "failed", ms: Date.now() - t0 }; |
| 268 | 273 | } |
@@ -281,7 +286,7 @@ async function reprocessDocument(ctx: RunContext, loaded: LoadedConnector, doc: | ||
| 281 | 286 | const { group } = parseDiscoveredFrom(doc.discoveredFrom); |
| 282 | 287 | const raw: RawDocument = { url: doc.url, finalUrl: doc.url, fetchedAt: doc.lastFetched ?? new Date().toISOString(), status: doc.statusCode ?? 200, contentType, body: stored.body, text: isText ? stored.body.toString("utf8") : "", headers: {}, etag: doc.etag, lastModified: doc.lastModified, notModified: false, fetcher: "cache", level: doc.fetchLevel as RawDocument["level"], durationMs: 0, credits: 0, group, pageType: doc.pageType !== "unknown" ? (doc.pageType as RawDocument["pageType"]) : undefined }; |
| 283 | 288 | const r = await extractAndIngest(ctx, loaded, doc, raw, stats, result, opts); |
| 284 | − await updateExtraction(doc.id, { entityRefs: r.ingest?.refs ?? [], extractOk: true, extractCount: r.valid, extractorVersion: loaded.connector.parserVersion, error: null, pageType: r.pageType, classifier: r.classifier }, dryRun); | |
| 289 | + await updateExtraction(doc.id, { entityRefs: r.ingest?.refs ?? [], extractOk: true, extractCount: r.valid, extractorVersion: loaded.extractorVersion, error: null, pageType: r.pageType, classifier: r.classifier }, dryRun); | |
| 285 | 290 | ctx.log("info", `${doc.url} reprocessed → ${r.valid} entities${r.ingest ? ` (+${r.ingest.created} new, ${r.ingest.updated} upd, ${r.ingest.events} events)` : ""}`); |
| 286 | 291 | return { kind: "fetched", ms: Date.now() - t0 }; |
| 287 | 292 | } catch (e) { |
@@ -306,8 +311,17 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi | ||
| 306 | 311 | let fatal: string | null = null; |
| 307 | 312 | const durations: number[] = []; |
| 308 | 313 | |
| 309 | − if (!dryRun) await db.insert(connectorRuns).values({ id: runId, connectorId, task: opts.task, startedAt, status: "running" }); | |
| 310 | − ctx.log("info", `run ${runId} task=${opts.task}${opts.group ? ` group=${opts.group}` : ""}${opts.limit ? ` limit=${opts.limit}` : ""}${dryRun ? " DRY-RUN" : ""}${opts.force ? " force" : ""}`); | |
| 314 | + // quarantine mode (connectors.quarantine): everything runs, nothing is published | |
| 315 | + let quarantine = Boolean(opts.quarantine); | |
| 316 | + let previouslyDiscovered: number | null = null; | |
| 317 | + if (!dryRun) { | |
| 318 | + const row = (await db.execute<{ quarantine: boolean; last_discovered: number | null }>(sql`select quarantine, last_discovered from connectors where id = ${connectorId}`))[0]; | |
| 319 | + if (row?.quarantine) quarantine = true; | |
| 320 | + previouslyDiscovered = row?.last_discovered == null ? null : Number(row.last_discovered); | |
| 321 | + } | |
| 322 | + opts = { ...opts, quarantine }; | |
| 323 | + if (!dryRun) await db.insert(connectorRuns).values({ id: runId, connectorId, task: opts.task, startedAt, status: "running", quarantined: quarantine }); | |
| 324 | + ctx.log("info", `run ${runId} task=${opts.task}${opts.group ? ` group=${opts.group}` : ""}${opts.limit ? ` limit=${opts.limit}` : ""}${dryRun ? " DRY-RUN" : ""}${quarantine ? " QUARANTINE (ingest rolled back)" : ""}${opts.force ? " force" : ""}`); | |
| 311 | 325 | |
| 312 | 326 | try { |
| 313 | 327 | // ---- discover |
@@ -335,7 +349,9 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi | ||
| 335 | 349 | |
| 336 | 350 | // ---- crawl / reprocess |
| 337 | 351 | if (opts.task === "crawl" || opts.task === "reprocess" || (opts.task === "full" && !dryRun) || (opts.urls?.length && opts.task !== "discover")) { |
| 338 | − const docs = await dueDocuments(connectorId, { group: opts.group, limit: opts.limit, force: Boolean(opts.force) || Boolean(targetIds) || opts.task === "reprocess", ids: targetIds }); | |
| 352 | + let docs = await dueDocuments(connectorId, { group: opts.group, limit: opts.limit, force: Boolean(opts.force) || Boolean(targetIds) || opts.task === "reprocess", ids: targetIds }); | |
| 353 | + // `reprocess --stale`: only documents extracted with an older parser version (parser fix → controlled re-extraction) | |
| 354 | + if (opts.task === "reprocess" && opts.staleOnly) docs = docs.filter((d) => d.extractorVersion !== loaded.extractorVersion); | |
| 339 | 355 | stats.selected = docs.length; |
| 340 | 356 | ctx.log("info", `${docs.length} document(s) selected${opts.task === "reprocess" ? " for reprocessing" : ""}`); |
| 341 | 357 | const limit = pLimit(cfg.fetch.concurrency); |
@@ -383,10 +399,24 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi | ||
| 383 | 399 | const nexts = groups.map((g) => g.nextCheck).filter((x): x is string => Boolean(x)).map((x) => Date.parse(x)); |
| 384 | 400 | const nextRunAt = new Date(nexts.length ? Math.min(...nexts) : Date.now() + minIntervalMs(cfg.schedule)).toISOString(); |
| 385 | 401 | const docStats = await connectorDocStats(connectorId); |
| 402 | + const blocked = Object.values(ctx.costByFetcher).reduce((n, c) => n + (c.errors ?? 0), 0); | |
| 403 | + const health = connectorHealthFrom({ runHealth: result.health, status: result.status, fetched: stats.fetched, blocked: Math.min(blocked, stats.failed), discovered: stats.discovered, previouslyDiscovered: opts.task === "discover" || opts.task === "full" ? previouslyDiscovered : null, extracted: stats.extracted, entities: stats.entities, docsTotal: docStats.total }); | |
| 404 | + const failedRun = result.status === "failed" || health === "blocked"; | |
| 405 | + const prev = (await db.execute<{ consecutive_failures: number; blocked_since: string | null }>(sql`select consecutive_failures, blocked_since from connectors where id = ${connectorId}`))[0]; | |
| 406 | + const consecutive = failedRun ? Number(prev?.consecutive_failures ?? 0) + 1 : 0; | |
| 407 | + const blockedSince = health === "blocked" ? (prev?.blocked_since ?? finishedAt) : null; | |
| 408 | + // guard rails: a connector failing 5 runs in a row, or a non-full run that suddenly creates hundreds of records, is quarantined | |
| 409 | + const suspiciousSpike = opts.task !== "full" && !quarantine && stats.created >= 500; | |
| 410 | + const autoQuarantine = consecutive >= 5 || suspiciousSpike; | |
| 386 | 411 | await db |
| 387 | 412 | .update(connectorsTable) |
| 388 | − .set({ health: result.status === "failed" ? "failing" : result.health, lastRunAt: finishedAt, lastStatus: result.status, lastError: result.error, nextRunAt, stats: { ...stats, docsTotal: docStats.total, docsDue: docStats.due, docsQuarantined: docStats.quarantined, docsErrors: docStats.errors, docsExtracted: docStats.extracted, lastRunMs: result.durationMs }, updatedAt: finishedAt }) | |
| 413 | + .set({ health, lastRunAt: finishedAt, lastStatus: result.status, lastError: result.error, nextRunAt, consecutiveFailures: consecutive, blockedSince, lastDiscovered: opts.task === "discover" || opts.task === "full" ? stats.discovered : undefined, ...(autoQuarantine ? { quarantine: true } : {}), stats: { ...stats, docsTotal: docStats.total, docsDue: docStats.due, docsQuarantined: docStats.quarantined, docsErrors: docStats.errors, docsExtracted: docStats.extracted, lastRunMs: result.durationMs, blockedFetches: blocked, unscopedClaims: stats.unscopedClaims ?? 0, projectsVetoed: stats.projectsVetoed ?? 0 }, updatedAt: finishedAt }) | |
| 389 | 414 | .where(eq(connectorsTable.id, connectorId)); |
| 415 | + if (autoQuarantine) { | |
| 416 | + const why = suspiciousSpike ? `run ${runId} created ${stats.created} records in a ${opts.task} run — quarantined pending review` : `${consecutive} consecutive failed/blocked runs — quarantined`; | |
| 417 | + ctx.log("error", why); | |
| 418 | + await db.execute(sql`insert into system_alerts (id, level, component, message, details) values (${`alr_${Date.now().toString(36)}_q_${connectorId.replace(/[^a-z0-9]/gi, "")}`}, 'error', ${`connector:${connectorId}`}, ${why}, ${JSON.stringify({ runId, stats })}::jsonb)`).catch(() => undefined); | |
| 419 | + } | |
| 390 | 420 | } catch (e) { |
| 391 | 421 | console.error(`[${connectorId}] finishing run ${runId} failed: ${(e as Error).message}`); |
| 392 | 422 | } |
added
apps/worker/src/quality.ts
+189 −0
@@ -0,0 +1,189 @@ | ||
| 1 | +/** | |
| 2 | + * Data-quality maintenance: | |
| 3 | + * - `snapshotAndCheck()` — daily JSON snapshots (global totals, per-operator / per-country totals, project stages, | |
| 4 | + * facility statuses) into `entity_snapshots` + automated regression checks against the previous snapshot | |
| 5 | + * (facilities drop > 5 %, known MW jumps > 20 %, one operator gains > 10 GW, one connector creates > 500 projects, | |
| 6 | + * one source changes hundreds of locations) → `system_alerts`. | |
| 7 | + * - `qualitySweep()` — deterministic flags over the live database (largest values, scope, duplicates, orphans, | |
| 8 | + * stale entities, project false-positive candidates…) into `quality_flags`, plus review priorities. | |
| 9 | + * - `dataGaps()` — counts for the admin data-gaps page (facilities without operator / coordinates / capacity…). | |
| 10 | + */ | |
| 11 | +import { getDb, sql, type Db } from "@dci/db"; | |
| 12 | +import { sha256 } from "@dci/core"; | |
| 13 | +import { reviewPriority } from "./ingest/claims.js"; | |
| 14 | + | |
| 15 | +const day = () => new Date().toISOString().slice(0, 10); | |
| 16 | + | |
| 17 | +export interface SnapshotResult { day: string; snapshots: number; alerts: string[] } | |
| 18 | + | |
| 19 | +async function alert(db: Db, component: string, level: "warn" | "error", message: string, details: Record<string, unknown>): Promise<void> { | |
| 20 | + await db.execute(sql`insert into system_alerts (id, level, component, message, details) values (${`alr_${Date.now().toString(36)}_${Math.random().toString(36).slice(2, 6)}`}, ${level}, ${component}, ${message.slice(0, 1000)}, ${JSON.stringify(details)}::jsonb)`); | |
| 21 | +} | |
| 22 | + | |
| 23 | +export async function snapshotAndCheck(db: Db = getDb(), today = day()): Promise<SnapshotResult> { | |
| 24 | + const alerts: string[] = []; | |
| 25 | + const g = (await db.execute(sql` | |
| 26 | + select | |
| 27 | + (select count(*) from facilities where merged_into is null)::int as facilities, | |
| 28 | + (select count(*) from operators)::int as operators, | |
| 29 | + (select count(*) from projects where merged_into is null and not hidden)::int as projects, | |
| 30 | + (select count(distinct country_iso2) from facilities where merged_into is null and country_iso2 is not null)::int as countries, | |
| 31 | + coalesce((select sum(coalesce(it_capacity_mw, total_power_mw)) from facilities where merged_into is null and status in ('operational','partially_operational','expansion') and not exists (select 1 from facilities c where c.parent_facility_id = facilities.id and c.merged_into is null and coalesce(c.it_capacity_mw, c.total_power_mw) is not null)), 0)::double precision as known_mw, | |
| 32 | + coalesce((select sum(coalesce(planned_power_mw, it_capacity_mw, total_power_mw)) from facilities where merged_into is null and status = 'under_construction'), 0)::double precision as construction_mw, | |
| 33 | + coalesce((select sum(planned_mw) from projects where merged_into is null and not hidden and status in ('rumored','proposed','announced','permitting','approved','under_construction','delayed')), 0)::double precision as project_mw, | |
| 34 | + (select count(*) from events where detected_at > now() - interval '24 hours')::int as events_24h, | |
| 35 | + (select count(*) from claims)::int as claims, | |
| 36 | + (select count(*) from quality_flags where status = 'open')::int as open_flags`))[0]!; | |
| 37 | + const totals = Object.fromEntries(Object.entries(g).map(([k, v]) => [k, Number(v)])); | |
| 38 | + const statuses = await db.execute(sql`select status, count(*)::int as n from facilities where merged_into is null group by 1`); | |
| 39 | + const stages = await db.execute(sql`select status, count(*)::int as n, coalesce(sum(planned_mw), 0)::double precision as mw from projects where merged_into is null and not hidden group by 1`); | |
| 40 | + const ops = await db.execute(sql`select o.id, o.slug, count(f.id)::int as facilities, coalesce(sum(coalesce(f.it_capacity_mw, f.total_power_mw)), 0)::double precision as known_mw, | |
| 41 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden)::double precision as project_mw | |
| 42 | + from operators o left join facilities f on f.operator_id = o.id and f.merged_into is null group by o.id, o.slug having count(f.id) > 0 or exists (select 1 from projects p where p.operator_id = o.id and not p.hidden)`); | |
| 43 | + const countries = await db.execute(sql`select country_iso2 as iso2, count(*)::int as facilities, coalesce(sum(coalesce(it_capacity_mw, total_power_mw)), 0)::double precision as known_mw from facilities where merged_into is null and country_iso2 is not null group by 1`); | |
| 44 | + const rankings = await db.execute(sql`select key, rows from rankings where is_current`); | |
| 45 | + const connectors = await db.execute(sql`select connector_id, coalesce(sum((stats->>'created')::int), 0)::int as created, count(*)::int as runs from connector_runs where started_at > now() - interval '24 hours' and status <> 'running' and task <> 'full' group by 1`); | |
| 46 | + const locChanges = await db.execute(sql`select p.connector_id, count(*)::int as n from provenance p where p.field = 'geo' and p.last_observed > now() - interval '24 hours' and p.first_observed < p.last_observed - interval '1 hour' group by 1`); | |
| 47 | + | |
| 48 | + const previous = (await db.execute(sql`select payload from entity_snapshots where kind = 'global_totals' and key = 'global' and day < ${today}::date order by day desc limit 1`))[0]?.payload as Record<string, number> | undefined; | |
| 49 | + const prevOps = new Map<string, Record<string, unknown>>(); | |
| 50 | + for (const r of await db.execute(sql`select key, payload from entity_snapshots where kind = 'operator_totals' and day = (select max(day) from entity_snapshots where kind = 'operator_totals' and day < ${today}::date)`)) prevOps.set(String(r.key), r.payload as Record<string, unknown>); | |
| 51 | + | |
| 52 | + // ── regression checks | |
| 53 | + if (previous) { | |
| 54 | + const pf = Number(previous.facilities ?? 0), nf = totals.facilities!; | |
| 55 | + if (pf > 100 && nf < pf * 0.95) { const m = `facilities dropped ${pf} → ${nf} (${(((nf - pf) / pf) * 100).toFixed(1)} %)`; alerts.push(m); await alert(db, "regression:facilities", "error", m, { previous: pf, now: nf }); } | |
| 56 | + const pm = Number(previous.known_mw ?? 0), nm = totals.known_mw!; | |
| 57 | + if (pm > 500 && nm > pm * 1.2) { const m = `known MW jumped ${Math.round(pm)} → ${Math.round(nm)} (+${(((nm - pm) / pm) * 100).toFixed(1)} %)`; alerts.push(m); await alert(db, "regression:known_mw", "warn", m, { previous: pm, now: nm }); } | |
| 58 | + const pp = Number(previous.projects ?? 0), np = totals.projects!; | |
| 59 | + if (pp > 50 && np > pp * 1.5) { const m = `projects jumped ${pp} → ${np}`; alerts.push(m); await alert(db, "regression:projects", "warn", m, { previous: pp, now: np }); } | |
| 60 | + const pc = Number(previous.countries ?? 0), nc = totals.countries!; | |
| 61 | + if (pc > 20 && nc < pc - 3) { const m = `countries dropped ${pc} → ${nc}`; alerts.push(m); await alert(db, "regression:countries", "warn", m, { previous: pc, now: nc }); } | |
| 62 | + } | |
| 63 | + for (const o of ops) { | |
| 64 | + const prev = prevOps.get(String(o.id)); | |
| 65 | + const gain = Number(o.known_mw) + Number(o.project_mw) - (prev ? Number(prev.known_mw ?? 0) + Number(prev.project_mw ?? 0) : 0); | |
| 66 | + if (prev && gain > 10_000) { const m = `operator ${String(o.slug)} gained ${Math.round(gain)} MW in one day`; alerts.push(m); await alert(db, "regression:operator_mw", "error", m, { operator: o.slug, gainMw: gain }); } | |
| 67 | + } | |
| 68 | + for (const c of connectors) if (Number(c.created) > 500) { const m = `connector ${String(c.connector_id)} created ${Number(c.created)} records in 24 h`; alerts.push(m); await alert(db, `regression:connector:${String(c.connector_id)}`, "warn", m, { created: Number(c.created), runs: Number(c.runs) }); } | |
| 69 | + for (const l of locChanges) if (Number(l.n) > 200) { const m = `connector ${String(l.connector_id)} changed ${Number(l.n)} locations in 24 h`; alerts.push(m); await alert(db, `regression:locations:${String(l.connector_id)}`, "warn", m, { changed: Number(l.n) }); } | |
| 70 | + | |
| 71 | + // ── snapshots | |
| 72 | + let n = 0; | |
| 73 | + await db.transaction(async (tx) => { | |
| 74 | + const put = async (kind: string, key: string, payload: unknown) => { await tx.execute(sql`insert into entity_snapshots (day, kind, key, payload) values (${today}::date, ${kind}, ${key}, ${JSON.stringify(payload)}::jsonb) on conflict (day, kind, key) do update set payload = excluded.payload, created_at = now()`); n++; }; | |
| 75 | + await put("global_totals", "global", totals); | |
| 76 | + await put("facility_status", "global", Object.fromEntries(statuses.map((r) => [String(r.status), Number(r.n)]))); | |
| 77 | + await put("project_stage", "global", Object.fromEntries(stages.map((r) => [String(r.status), { count: Number(r.n), mw: Number(r.mw) }]))); | |
| 78 | + for (const o of ops) await put("operator_totals", String(o.id), { slug: o.slug, facilities: Number(o.facilities), known_mw: Number(o.known_mw), project_mw: Number(o.project_mw) }); | |
| 79 | + for (const c of countries) await put("country_totals", String(c.iso2), { facilities: Number(c.facilities), known_mw: Number(c.known_mw) }); | |
| 80 | + for (const r of rankings) await put("ranking", String(r.key), { rows: (r.rows as unknown[]).slice(0, 100) }); | |
| 81 | + }); | |
| 82 | + return { day: today, snapshots: n, alerts }; | |
| 83 | +} | |
| 84 | + | |
| 85 | +export interface SweepResult { flagged: number; resolved: number; byCode: Record<string, number> } | |
| 86 | + | |
| 87 | +/** Batch quality sweep over the live database. Idempotent (dedupe keys). */ | |
| 88 | +export async function qualitySweep(db: Db = getDb()): Promise<SweepResult> { | |
| 89 | + const byCode: Record<string, number> = {}; | |
| 90 | + let flagged = 0, resolved = 0; | |
| 91 | + const flag = async (entityType: string, entityId: string, code: string, severity: "warn" | "critical", message: string, field: string | null, details: Record<string, unknown>, priority: number) => { | |
| 92 | + const dedupeKey = `${entityType}|${entityId}|${code}|${field ?? ""}`; | |
| 93 | + await db.execute(sql`insert into quality_flags (id, entity_type, entity_id, code, severity, field, message, details, priority, status, dedupe_key) | |
| 94 | + values (${`flg_${sha256(dedupeKey).slice(0, 20)}`}, ${entityType}, ${entityId}, ${code}, ${severity}, ${field}, ${message.slice(0, 1000)}, ${JSON.stringify(details)}::jsonb, ${priority}, 'open', ${dedupeKey}) | |
| 95 | + on conflict (dedupe_key) do update set message = excluded.message, details = excluded.details, priority = excluded.priority, updated_at = now(), status = case when quality_flags.status = 'dismissed' then 'dismissed' when quality_flags.status = 'resolved' and quality_flags.resolved_by <> 'system' then 'resolved' else 'open' end`); | |
| 96 | + byCode[code] = (byCode[code] ?? 0) + 1; flagged++; | |
| 97 | + }; | |
| 98 | + const clear = async (code: string, keepIds: string[], entityType: string) => { | |
| 99 | + const r = await db.execute(sql`update quality_flags set status = 'resolved', resolution = 'auto: condition cleared', resolved_by = 'system', resolved_at = now(), updated_at = now() where code = ${code} and entity_type = ${entityType} and status = 'open' ${keepIds.length ? sql`and entity_id not in ${keepIds}` : sql``} returning id`); | |
| 100 | + resolved += r.length; | |
| 101 | + }; | |
| 102 | + | |
| 103 | + // suspicious MW on facilities (single site > 1 GW without campus designation, building > 500 MW, IT > total) | |
| 104 | + const bigFac = await db.execute(sql`select id, name, slug, coalesce(it_capacity_mw, total_power_mw) as mw, record_scope, confidence from facilities where merged_into is null and coalesce(it_capacity_mw, total_power_mw) > 1000 and record_scope <> 'campus' and name !~* '(campus|park|complex|hub|cluster|mega ?site)'`); | |
| 105 | + for (const r of bigFac) await flag("facility", String(r.id), "mw_single_site_gt_1000", "critical", `${String(r.name)}: ${Number(r.mw)} MW on a single ${String(r.record_scope)} record without a campus designation`, "itCapacityMw", { slug: r.slug, mw: Number(r.mw) }, reviewPriority({ mw: Number(r.mw), confidence: String(r.confidence), severity: "critical", homepageVisible: true })); | |
| 106 | + await clear("mw_single_site_gt_1000", bigFac.map((r) => String(r.id)), "facility"); | |
| 107 | + const itGtTotal = await db.execute(sql`select id, name, slug, it_capacity_mw, total_power_mw from facilities where merged_into is null and it_capacity_mw is not null and total_power_mw is not null and it_capacity_mw > total_power_mw * 1.05`); | |
| 108 | + for (const r of itGtTotal) await flag("facility", String(r.id), "mw_it_gt_total", "warn", `${String(r.name)}: IT capacity ${Number(r.it_capacity_mw)} MW exceeds total power ${Number(r.total_power_mw)} MW`, "itCapacityMw", { slug: r.slug }, reviewPriority({ mw: Number(r.it_capacity_mw), severity: "warn" })); | |
| 109 | + await clear("mw_it_gt_total", itGtTotal.map((r) => String(r.id)), "facility"); | |
| 110 | + // tiny buildings with huge MW (the 0.779 → 779 class of unit errors) | |
| 111 | + const dense = await db.execute(sql`select id, name, slug, it_capacity_mw, building_sqm from facilities where merged_into is null and it_capacity_mw is not null and building_sqm is not null and building_sqm > 0 and it_capacity_mw / building_sqm > 0.05`); | |
| 112 | + for (const r of dense) await flag("facility", String(r.id), "mw_density_implausible", "critical", `${String(r.name)}: ${Number(r.it_capacity_mw)} MW in ${Number(r.building_sqm)} m² (${(Number(r.it_capacity_mw) * 1000 / Number(r.building_sqm)).toFixed(0)} kW/m²) — unit error?`, "itCapacityMw", { slug: r.slug }, reviewPriority({ mw: Number(r.it_capacity_mw), severity: "critical" })); | |
| 113 | + await clear("mw_density_implausible", dense.map((r) => String(r.id)), "facility"); | |
| 114 | + | |
| 115 | + // projects: false-positive candidates, headline names, huge figures, missing location, unknown scope | |
| 116 | + const fp = await db.execute(sql`select id, name, slug, planned_mw, investment_usd, project_class, evidence_level from projects where merged_into is null and not hidden and ( | |
| 117 | + name ~* '\\m(appoint|names? [A-Z]\\w+ [A-Z]\\w+ as|joins|welcomes|promot|retire|award|shortlist|finalist|webinar|podcast|interview|market to (surpass|reach|hit|grow)|market size|cagr|report|survey|headquarters|head office|academy|workforce|partner page|homepage|bubble|comparing|why |how |what |ppa\\M|power purchase|sustainab|net.zero|carbon|financ|refinanc|raises \\$|series [a-e]\\M|bond|loan|credit facility|earnings|results|revenue|acquires [A-Z]\\w+ (group|holdings|inc|ltd|llc)|merger|takeover)' | |
| 118 | + or (project_class is not null and project_class not in ('NEW_BUILD','EXPANSION','CONSTRUCTION_START','PERMIT','LAND_ACQUISITION','GRID_CONNECTION')))`); | |
| 119 | + for (const r of fp) await flag("project", String(r.id), "project_false_positive_candidate", "critical", `${String(r.name)}: headline / class (${String(r.project_class ?? "n/a")}) does not describe physical development`, "name", { slug: r.slug, plannedMw: r.planned_mw, investmentUsd: r.investment_usd }, reviewPriority({ mw: Number(r.planned_mw ?? 0), investmentUsd: Number(r.investment_usd ?? 0), severity: "critical", homepageVisible: true })); | |
| 120 | + await clear("project_false_positive_candidate", fp.map((r) => String(r.id)), "project"); | |
| 121 | + const headline = await db.execute(sql`select id, name, slug, planned_mw from projects where merged_into is null and not hidden and (name ~ '\\s[|]\\s|\\s[–—]\\s| - ' or length(name) > 90)`); | |
| 122 | + for (const r of headline) await flag("project", String(r.id), "project_title_like_name", "warn", `${String(r.name)}: project name looks like an article headline`, "name", { slug: r.slug }, reviewPriority({ mw: Number(r.planned_mw ?? 0), severity: "warn" })); | |
| 123 | + await clear("project_title_like_name", headline.map((r) => String(r.id)), "project"); | |
| 124 | + const bigPrj = await db.execute(sql`select id, name, slug, planned_mw, capacity_scope from projects where merged_into is null and not hidden and planned_mw >= 2000 and name !~* '(campus|park|complex|hub|cluster|gigafactory)'`); | |
| 125 | + for (const r of bigPrj) await flag("project", String(r.id), "mw_project_gt_2000", "critical", `${String(r.name)}: ${Number(r.planned_mw)} MW planned (scope ${String(r.capacity_scope ?? "unknown")}) — verify it is one site`, "plannedMw", { slug: r.slug }, reviewPriority({ mw: Number(r.planned_mw), severity: "critical", homepageVisible: true })); | |
| 126 | + await clear("mw_project_gt_2000", bigPrj.map((r) => String(r.id)), "project"); | |
| 127 | + const bigInv = await db.execute(sql`select id, name, slug, investment_usd from projects where merged_into is null and not hidden and investment_usd > 50e9`); | |
| 128 | + for (const r of bigInv) await flag("project", String(r.id), "inv_single_site_gt_50b", "critical", `${String(r.name)}: $${(Number(r.investment_usd) / 1e9).toFixed(1)}B on one project — verify scope`, "investmentUsd", { slug: r.slug }, reviewPriority({ investmentUsd: Number(r.investment_usd), severity: "critical" })); | |
| 129 | + await clear("inv_single_site_gt_50b", bigInv.map((r) => String(r.id)), "project"); | |
| 130 | + const noLoc = await db.execute(sql`select id, name, slug from projects where merged_into is null and not hidden and country_iso2 is null and lat is null`); | |
| 131 | + for (const r of noLoc) await flag("project", String(r.id), "project_no_location", "warn", `${String(r.name)}: no country and no coordinates`, "countryIso2", { slug: r.slug }, 15); | |
| 132 | + await clear("project_no_location", noLoc.map((r) => String(r.id)), "project"); | |
| 133 | + const unknownScope = await db.execute(sql`select id, name, slug, planned_mw from projects where merged_into is null and not hidden and planned_mw is not null and (capacity_scope is null or capacity_scope not in ('building','facility','campus'))`); | |
| 134 | + for (const r of unknownScope) await flag("project", String(r.id), "project_unknown_scope", "warn", `${String(r.name)}: ${Number(r.planned_mw)} MW with no site-scoped evidence`, "plannedMw", { slug: r.slug }, reviewPriority({ mw: Number(r.planned_mw), severity: "warn" })); | |
| 135 | + await clear("project_unknown_scope", unknownScope.map((r) => String(r.id)), "project"); | |
| 136 | + | |
| 137 | + // facilities: location / country mismatch, missing country, orphan operators, stale | |
| 138 | + const noCountry = await db.execute(sql`select id, name, slug from facilities where merged_into is null and country_iso2 is null`); | |
| 139 | + for (const r of noCountry) await flag("facility", String(r.id), "facility_no_country", "warn", `${String(r.name)}: no country`, "countryIso2", { slug: r.slug }, 8); | |
| 140 | + await clear("facility_no_country", noCountry.map((r) => String(r.id)), "facility"); | |
| 141 | + const stale = await db.execute(sql`select id, name, slug, last_verified from facilities where merged_into is null and status in ('announced','under_construction','permitting','approved') and coalesce(last_verified, first_seen) < now() - interval '400 days'`); | |
| 142 | + for (const r of stale) await flag("facility", String(r.id), "facility_stale_pipeline", "warn", `${String(r.name)}: pipeline status not verified for over 400 days`, "status", { slug: r.slug, lastVerified: r.last_verified }, 12); | |
| 143 | + await clear("facility_stale_pipeline", stale.map((r) => String(r.id)), "facility"); | |
| 144 | + const orphanOps = await db.execute(sql`select o.id, o.name, o.slug from operators o where not exists (select 1 from facilities f where f.operator_id = o.id or f.owner_id = o.id) and not exists (select 1 from projects p where p.operator_id = o.id) and not exists (select 1 from cloud_regions c where c.provider_id = o.id) and not exists (select 1 from facility_tenants t where t.operator_id = o.id) and o.created_at < now() - interval '7 days'`); | |
| 145 | + for (const r of orphanOps) await flag("operator", String(r.id), "operator_orphan", "warn", `${String(r.name)}: operator with no facility, project, region or tenancy`, null, { slug: r.slug }, 5); | |
| 146 | + await clear("operator_orphan", orphanOps.map((r) => String(r.id)), "operator"); | |
| 147 | + const badSlug = await db.execute(sql`select id, name, slug from operators where slug ~ '^(item|op|operator)(-\\d+)?$' or slug = ''`); | |
| 148 | + for (const r of badSlug) await flag("operator", String(r.id), "operator_bad_slug", "warn", `${String(r.name)}: slug "${String(r.slug)}" is not derived from the name (non-Latin script?)`, "slug", {}, 6); | |
| 149 | + await clear("operator_bad_slug", badSlug.map((r) => String(r.id)), "operator"); | |
| 150 | + | |
| 151 | + // review priority roll-up | |
| 152 | + await db.execute(sql`update facilities f set review_priority = coalesce((select max(priority) from quality_flags q where q.entity_type = 'facility' and q.entity_id = f.id and q.status = 'open'), 0) where merged_into is null`); | |
| 153 | + await db.execute(sql`update projects p set review_priority = coalesce((select max(priority) from quality_flags q where q.entity_type = 'project' and q.entity_id = p.id and q.status = 'open'), 0) where merged_into is null`); | |
| 154 | + return { flagged, resolved, byCode }; | |
| 155 | +} | |
| 156 | + | |
| 157 | +export interface DataGaps { [key: string]: number } | |
| 158 | + | |
| 159 | +/** Counts for /admin/data-gaps — what we do not know, ranked for enrichment. */ | |
| 160 | +export async function dataGaps(db: Db = getDb()): Promise<DataGaps> { | |
| 161 | + const r = (await db.execute(sql` | |
| 162 | + select | |
| 163 | + (select count(*) from facilities where merged_into is null and operator_id is null)::int as facilities_without_operator, | |
| 164 | + (select count(*) from facilities where merged_into is null and lat is null)::int as facilities_without_coordinates, | |
| 165 | + (select count(*) from facilities where merged_into is null and geo_precision in ('city','metro','approximate') )::int as facilities_imprecise_coordinates, | |
| 166 | + (select count(*) from facilities where merged_into is null and coalesce(it_capacity_mw, total_power_mw, planned_power_mw) is null and parent_facility_id is null)::int as facilities_without_capacity, | |
| 167 | + (select count(*) from facilities where merged_into is null and status = 'unknown')::int as facilities_unknown_status, | |
| 168 | + (select count(*) from facilities where merged_into is null and facility_type = 'unknown')::int as facilities_unknown_type, | |
| 169 | + (select count(*) from facilities where merged_into is null and opened_on is null)::int as facilities_without_opening_date, | |
| 170 | + (select count(*) from facilities where merged_into is null and source_count <= 1)::int as facilities_single_source, | |
| 171 | + (select count(*) from facilities where merged_into is null and coalesce(last_verified, first_seen) < now() - interval '180 days')::int as facilities_stale_sources, | |
| 172 | + (select count(*) from facilities where merged_into is null and country_iso2 is null)::int as facilities_without_country, | |
| 173 | + (select count(*) from projects where merged_into is null and not hidden and lat is null)::int as projects_without_coordinates, | |
| 174 | + (select count(*) from projects where merged_into is null and not hidden and country_iso2 is null)::int as projects_without_country, | |
| 175 | + (select count(*) from projects where merged_into is null and not hidden and operator_id is null)::int as projects_without_operator, | |
| 176 | + (select count(*) from projects where merged_into is null and not hidden and planned_mw is null)::int as projects_without_capacity, | |
| 177 | + (select count(*) from projects where merged_into is null and not hidden and expected_opening is null)::int as projects_without_expected_opening, | |
| 178 | + (select count(*) from projects where merged_into is null and not hidden and source_url is null)::int as projects_without_source, | |
| 179 | + (select count(*) from projects where merged_into is null and not hidden and facility_id is null)::int as projects_without_facility_link, | |
| 180 | + (select count(*) from operators where hq_country_iso2 is null)::int as operators_without_hq, | |
| 181 | + (select count(*) from operators where website is null)::int as operators_without_website, | |
| 182 | + (select count(*) from entity_matches where status = 'pending')::int as pending_matches, | |
| 183 | + (select count(*) from quality_flags where status = 'open')::int as open_quality_flags, | |
| 184 | + (select count(*) from claims where status = 'unscoped')::int as unscoped_claims, | |
| 185 | + (select count(*) from claims where status = 'review')::int as claims_in_review, | |
| 186 | + (select count(*) from cloud_regions where lat is null)::int as cloud_regions_without_coordinates, | |
| 187 | + (select count(*) from ixps where metro_id is null)::int as ixps_without_metro`))[0]!; | |
| 188 | + return Object.fromEntries(Object.entries(r).map(([k, v]) => [k, Number(v)])); | |
| 189 | +} | |
modified
apps/worker/src/rankings.ts
+71 −20
@@ -34,6 +34,11 @@ export interface Aggregate { | ||
| 34 | 34 | metros: number; |
| 35 | 35 | cloudRegions: number; |
| 36 | 36 | projects: number; |
| 37 | + /** planned MW published on project records (pipeline statuses) — kept apart from facility figures */ | |
| 38 | + projectPlannedMw: number; | |
| 39 | + projectConstructionMw: number; | |
| 40 | + expansionMw: number; | |
| 41 | + aiConfirmed: number; | |
| 37 | 42 | ixps: number; |
| 38 | 43 | population: number | null; |
| 39 | 44 | gdpUsd: number | null; |
@@ -47,18 +52,41 @@ function recentYear(): string { | ||
| 47 | 52 | return String(new Date().getUTCFullYear() - 3); |
| 48 | 53 | } |
| 49 | 54 | |
| 55 | +/** | |
| 56 | + * Containment-aware facility view used by every aggregate: | |
| 57 | + * - `agg_mw` is a facility's own known MW (IT, else total) unless it is a campus whose buildings publish their own | |
| 58 | + * figures (then the campus row contributes nothing — the buildings do); a building without a figure under a campus | |
| 59 | + * with one contributes nothing either (the campus figure already covers it). Never both. | |
| 60 | + * - `counted` excludes campus rows that have building rows (a campus with 4 buildings is 4 facilities, not 5) and | |
| 61 | + * hidden / merged rows. | |
| 62 | + */ | |
| 63 | +export const FACILITY_VIEW = sql` | |
| 64 | + select f.*, | |
| 65 | + exists (select 1 from facilities c where c.parent_facility_id = f.id and c.merged_into is null) as has_children, | |
| 66 | + exists (select 1 from facilities c where c.parent_facility_id = f.id and c.merged_into is null and coalesce(c.it_capacity_mw, c.total_power_mw, c.planned_power_mw) is not null) as children_have_mw, | |
| 67 | + (f.parent_facility_id is not null and exists (select 1 from facilities pp where pp.id = f.parent_facility_id and pp.merged_into is null and coalesce(pp.it_capacity_mw, pp.total_power_mw, pp.planned_power_mw) is not null) | |
| 68 | + and coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is null) as covered_by_parent | |
| 69 | + from facilities f where f.merged_into is null`; | |
| 70 | + | |
| 71 | +/** MW expressions honouring containment (campus vs buildings never double count). */ | |
| 72 | +const KNOWN_MW_EXPR = sql`case when f.children_have_mw then null else coalesce(f.it_capacity_mw, f.total_power_mw) end`; | |
| 73 | +const PIPELINE_MW_EXPR = sql`case when f.children_have_mw then null else coalesce(f.planned_power_mw, f.it_capacity_mw, f.total_power_mw) end`; | |
| 74 | +const COUNTED = sql`(not f.has_children)`; | |
| 75 | + | |
| 50 | 76 | const FACILITY_AGG = (dim: ReturnType<typeof sql>) => sql` |
| 51 | − count(f.id)::int as facilities, | |
| 52 | − count(f.id) filter (where f.status in ${OPERATIONAL})::int as operational, | |
| 53 | − count(f.id) filter (where f.status = 'under_construction')::int as construction, | |
| 54 | − count(f.id) filter (where f.status in ${PLANNED})::int as planned, | |
| 55 | − coalesce(sum(coalesce(f.it_capacity_mw, f.total_power_mw)) filter (where f.status in ${OPERATIONAL}), 0) as known_mw, | |
| 56 | − coalesce(sum(coalesce(f.planned_power_mw, f.it_capacity_mw, f.total_power_mw)) filter (where f.status = 'under_construction'), 0) as construction_mw, | |
| 57 | − coalesce(sum(coalesce(f.planned_power_mw, f.it_capacity_mw, f.total_power_mw)) filter (where f.status in ${PLANNED}), 0) as planned_mw, | |
| 58 | − count(f.id) filter (where f.is_hyperscale)::int as hyperscale, | |
| 59 | − count(f.id) filter (where f.is_ai)::int as ai, | |
| 60 | − count(f.id) filter (where coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is not null)::int as with_mw, | |
| 61 | − count(f.id) filter (where f.opened_on >= ${recentYear()})::int as opened_recent, | |
| 77 | + count(f.id) filter (where ${COUNTED})::int as facilities, | |
| 78 | + count(f.id) filter (where ${COUNTED} and f.status in ${OPERATIONAL})::int as operational, | |
| 79 | + count(f.id) filter (where ${COUNTED} and f.status = 'under_construction')::int as construction, | |
| 80 | + count(f.id) filter (where ${COUNTED} and f.status in ${PLANNED})::int as planned, | |
| 81 | + coalesce(sum(${KNOWN_MW_EXPR}) filter (where f.status in ${OPERATIONAL}), 0) as known_mw, | |
| 82 | + coalesce(sum(${PIPELINE_MW_EXPR}) filter (where f.status = 'under_construction'), 0) as construction_mw, | |
| 83 | + coalesce(sum(${PIPELINE_MW_EXPR}) filter (where f.status in ${PLANNED}), 0) as planned_mw, | |
| 84 | + coalesce(sum(case when f.children_have_mw then null else f.planned_power_mw end) filter (where f.status = 'expansion'), 0) as expansion_mw, | |
| 85 | + count(f.id) filter (where ${COUNTED} and f.is_hyperscale)::int as hyperscale, | |
| 86 | + count(f.id) filter (where ${COUNTED} and f.is_ai)::int as ai, | |
| 87 | + count(f.id) filter (where ${COUNTED} and f.ai_evidence = 'confirmed')::int as ai_confirmed, | |
| 88 | + count(f.id) filter (where ${COUNTED} and (coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is not null or f.covered_by_parent))::int as with_mw, | |
| 89 | + count(f.id) filter (where ${COUNTED} and f.opened_on ~ '^\d{4}' and substring(f.opened_on from 1 for 4) >= ${recentYear()} and substring(f.opened_on from 1 for 4) <= ${String(new Date().getUTCFullYear())})::int as opened_recent, | |
| 62 | 90 | count(distinct f.operator_id)::int as operators, |
| 63 | 91 | count(distinct f.country_iso2)::int as countries, |
| 64 | 92 | count(distinct f.metro_id)::int as metros, |
@@ -87,6 +115,10 @@ function toAgg(r: Record<string, unknown>): Aggregate { | ||
| 87 | 115 | metros: num(r.metros), |
| 88 | 116 | cloudRegions: num(r.cloud_regions), |
| 89 | 117 | projects: num(r.projects), |
| 118 | + projectPlannedMw: num(r.project_planned_mw), | |
| 119 | + projectConstructionMw: num(r.project_construction_mw), | |
| 120 | + expansionMw: num(r.expansion_mw), | |
| 121 | + aiConfirmed: num(r.ai_confirmed), | |
| 90 | 122 | ixps: num(r.ixps), |
| 91 | 123 | population: numOrNull(r.population), |
| 92 | 124 | gdpUsd: numOrNull(r.gdp_usd), |
@@ -98,10 +130,12 @@ export async function aggregateCountries(db: Db): Promise<Aggregate[]> { | ||
| 98 | 130 | const rows = await db.execute(sql` |
| 99 | 131 | select c.iso2 as id, c.slug, c.name, c.iso2 as country_iso2, c.population, c.gdp_usd, |
| 100 | 132 | (select count(*) from cloud_regions cr where cr.country_iso2 = c.iso2 and cr.status <> 'retired')::int as cloud_regions, |
| 101 | − (select count(*) from projects p where p.country_iso2 = c.iso2)::int as projects, | |
| 133 | + (select count(*) from projects p where p.country_iso2 = c.iso2 and p.merged_into is null and not p.hidden)::int as projects, | |
| 134 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.country_iso2 = c.iso2 and p.merged_into is null and not p.hidden and p.status in ${PLANNED})::double precision as project_planned_mw, | |
| 135 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.country_iso2 = c.iso2 and p.merged_into is null and not p.hidden and p.status = 'under_construction')::double precision as project_construction_mw, | |
| 102 | 136 | (select count(*) from ixps x where x.country_iso2 = c.iso2)::int as ixps, |
| 103 | 137 | ${FACILITY_AGG(sql`1 as _`)} |
| 104 | − from countries c left join facilities f on f.country_iso2 = c.iso2 and f.merged_into is null | |
| 138 | + from countries c left join (${FACILITY_VIEW}) f on f.country_iso2 = c.iso2 | |
| 105 | 139 | group by c.iso2, c.slug, c.name, c.population, c.gdp_usd`); |
| 106 | 140 | return rows.map(toAgg); |
| 107 | 141 | } |
@@ -110,10 +144,12 @@ export async function aggregateMetros(db: Db): Promise<Aggregate[]> { | ||
| 110 | 144 | const rows = await db.execute(sql` |
| 111 | 145 | select m.id, m.slug, m.name, m.country_iso2, null::bigint as population, null::double precision as gdp_usd, |
| 112 | 146 | (select count(*) from cloud_regions cr where cr.metro_id = m.id and cr.status <> 'retired')::int as cloud_regions, |
| 113 | − (select count(*) from projects p where p.metro_id = m.id)::int as projects, | |
| 147 | + (select count(*) from projects p where p.metro_id = m.id and p.merged_into is null and not p.hidden)::int as projects, | |
| 148 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.metro_id = m.id and p.merged_into is null and not p.hidden and p.status in ${PLANNED})::double precision as project_planned_mw, | |
| 149 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.metro_id = m.id and p.merged_into is null and not p.hidden and p.status = 'under_construction')::double precision as project_construction_mw, | |
| 114 | 150 | (select count(*) from ixps x where x.metro_id = m.id)::int as ixps, |
| 115 | 151 | ${FACILITY_AGG(sql`1 as _`)} |
| 116 | − from metros m left join facilities f on f.metro_id = m.id and f.merged_into is null | |
| 152 | + from metros m left join (${FACILITY_VIEW}) f on f.metro_id = m.id | |
| 117 | 153 | group by m.id, m.slug, m.name, m.country_iso2`); |
| 118 | 154 | return rows.map(toAgg); |
| 119 | 155 | } |
@@ -122,10 +158,12 @@ export async function aggregateOperators(db: Db, limit = 5000): Promise<Aggregat | ||
| 122 | 158 | const rows = await db.execute(sql` |
| 123 | 159 | select o.id, o.slug, o.name, o.hq_country_iso2 as country_iso2, null::bigint as population, null::double precision as gdp_usd, |
| 124 | 160 | (select count(*) from cloud_regions cr where cr.provider_id = o.id and cr.status <> 'retired')::int as cloud_regions, |
| 125 | − (select count(*) from projects p where p.operator_id = o.id)::int as projects, | |
| 161 | + (select count(*) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden)::int as projects, | |
| 162 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden and p.status in ${PLANNED})::double precision as project_planned_mw, | |
| 163 | + (select coalesce(sum(p.planned_mw), 0) from projects p where p.operator_id = o.id and p.merged_into is null and not p.hidden and p.status = 'under_construction')::double precision as project_construction_mw, | |
| 126 | 164 | 0 as ixps, |
| 127 | 165 | ${FACILITY_AGG(sql`1 as _`)} |
| 128 | − from operators o left join facilities f on f.operator_id = o.id and f.merged_into is null | |
| 166 | + from operators o left join (${FACILITY_VIEW}) f on f.operator_id = o.id | |
| 129 | 167 | group by o.id, o.slug, o.name, o.hq_country_iso2 |
| 130 | 168 | order by count(f.id) desc, o.name asc |
| 131 | 169 | limit ${limit}`); |
@@ -148,7 +186,7 @@ interface RankingSpec { | ||
| 148 | 186 | |
| 149 | 187 | const mwGate = (a: Aggregate) => a.facilities >= MIN_FACILITIES && a.coverage >= MIN_MW_COVERAGE; |
| 150 | 188 | const OPERATIONAL_TXT = "operational, partially operational or expanding"; |
| 151 | −const KNOWN_MW_TXT = `Known MW = sum of IT capacity (or total power when IT capacity is not published) over ${OPERATIONAL_TXT} facilities with a published or filed figure. Estimates are included only when no measured figure exists and are flagged on the facility. Coverage = share of the entity's facilities with any MW figure.`; | |
| 189 | +const KNOWN_MW_TXT = `Known MW = sum of IT capacity (or total power when IT capacity is not published) over ${OPERATIONAL_TXT} facilities with a published or filed figure. A campus and its buildings are never both counted: when buildings publish their own figures the campus figure is ignored, otherwise the campus figure stands for its buildings. Estimates are included only when no measured figure exists and are flagged on the facility. Coverage = share of the entity's facilities with any MW figure (a building covered by its campus figure counts as covered). Figures are published lower bounds, never extrapolated.`; | |
| 152 | 190 | const GATE_TXT = `Entities with fewer than ${MIN_FACILITIES} facilities or MW coverage below ${Math.round(MIN_MW_COVERAGE * 100)} % are excluded to avoid ranking on sparse data.`; |
| 153 | 191 | |
| 154 | 192 | export const RANKING_SPECS: RankingSpec[] = [ |
@@ -167,14 +205,23 @@ export const RANKING_SPECS: RankingSpec[] = [ | ||
| 167 | 205 | // metros |
| 168 | 206 | { key: "metros_facilities", scope: "metros", label: "Metros by facility count", unit: "facilities", methodology: "Facilities assigned to the metro: within the metro radius of its reference point, or in a town listed as part of the market when no coordinates are known.", value: (a) => a.facilities, secondary: (a) => a.knownMw }, |
| 169 | 207 | { key: "metros_known_mw", scope: "metros", label: "Metros by known operational MW", unit: "MW", methodology: `${KNOWN_MW_TXT} ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.knownMw, secondary: (a) => a.operational, eligible: mwGate }, |
| 170 | − { key: "metros_pipeline_mw", scope: "metros", label: "Metros by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction (planned power, or IT/total power when no planned figure exists) over facilities not yet operational. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate }, | |
| 208 | + { key: "metros_pipeline_mw", scope: "metros", label: "Metros by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction (planned power, or IT/total power when no planned figure exists) over facilities not yet operational, plus the planned power of expanding facilities. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw + a.expansionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate }, | |
| 209 | + { key: "metros_ai", scope: "metros", label: "Metros by AI/HPC facilities", unit: "facilities", methodology: "Facilities whose AI evidence is confirmed or likely (source explicitly describes AI/HPC/accelerated computing or GPU / high-density infrastructure). Keyword mentions alone never qualify.", value: (a) => a.ai, secondary: (a) => a.facilities }, | |
| 210 | + { key: "metros_cloud_regions", scope: "metros", label: "Metros by cloud regions", unit: "regions", methodology: "Public cloud regions associated with the metro (announced or operational).", value: (a) => a.cloudRegions, secondary: (a) => a.facilities }, | |
| 211 | + { key: "metros_projects", scope: "metros", label: "Metros by project count", unit: "projects", methodology: "Tracked infrastructure projects (announced, permitting, approved, under construction, delayed) located in the metro; false positives hidden by review are excluded.", value: (a) => a.projects, secondary: (a) => a.projectPlannedMw + a.projectConstructionMw }, | |
| 171 | 212 | { key: "metros_operators", scope: "metros", label: "Metros by operator count", unit: "operators", methodology: "Distinct operators with at least one facility assigned to the metro.", value: (a) => a.operators, secondary: (a) => a.facilities }, |
| 172 | 213 | // operators |
| 173 | 214 | { key: "operators_facilities", scope: "operators", label: "Operators by facility count", unit: "facilities", methodology: "Facilities operated (not merely owned or tenanted) by the operator, subsidiaries folded into the parent brand where the curated alias table says so.", value: (a) => a.facilities, secondary: (a) => a.countries }, |
| 174 | 215 | { key: "operators_countries", scope: "operators", label: "Operators by country footprint", unit: "countries", methodology: "Distinct countries with at least one facility operated by the operator.", value: (a) => a.countries, secondary: (a) => a.facilities, eligible: (a) => a.facilities > 0 }, |
| 175 | 216 | { key: "operators_metros", scope: "operators", label: "Operators by metro footprint", unit: "metros", methodology: "Distinct metros with at least one facility operated by the operator (facilities outside any defined metro are not counted).", value: (a) => a.metros, secondary: (a) => a.facilities, eligible: (a) => a.facilities > 0 }, |
| 176 | 217 | { key: "operators_known_mw", scope: "operators", label: "Operators by known operational MW", unit: "MW", methodology: `${KNOWN_MW_TXT} ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.knownMw, secondary: (a) => a.operational, eligible: mwGate }, |
| 177 | − { key: "operators_pipeline_mw", scope: "operators", label: "Operators by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction over the operator's facilities not yet operational. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate }, | |
| 218 | + { key: "operators_pipeline_mw", scope: "operators", label: "Operators by pipeline MW", unit: "MW", methodology: `Planned MW plus MW under construction over the operator's facilities not yet operational, plus the planned power of expanding facilities. ${GATE_TXT}`, minCoverage: MIN_MW_COVERAGE, value: (a) => a.plannedMw + a.constructionMw + a.expansionMw, secondary: (a) => a.planned + a.construction, eligible: mwGate }, | |
| 219 | + { key: "operators_ai", scope: "operators", label: "Operators by AI/HPC facilities", unit: "facilities", methodology: "Facilities whose AI evidence is confirmed or likely. Keyword mentions alone never qualify.", value: (a) => a.ai, secondary: (a) => a.facilities }, | |
| 220 | + { key: "operators_projects", scope: "operators", label: "Operators by tracked projects", unit: "projects", methodology: "Infrastructure projects (announced, permitting, approved, under construction, delayed) attributed to the operator; the secondary figure is the published planned MW on those records.", value: (a) => a.projects, secondary: (a) => a.projectPlannedMw + a.projectConstructionMw }, | |
| 221 | + { key: "operators_project_mw", scope: "operators", label: "Operators by published project MW", unit: "MW", methodology: "Sum of planned MW published on the operator's project records (pipeline statuses), site-scoped figures only — company-wide or portfolio totals are stored as claims and never summed here.", value: (a) => a.projectPlannedMw + a.projectConstructionMw, secondary: (a) => a.projects, eligible: (a) => a.projects >= 2 }, | |
| 222 | + { key: "countries_project_mw", scope: "countries", label: "Countries by published project MW", unit: "MW", methodology: "Sum of planned MW published on project records located in the country (pipeline statuses), site-scoped figures only.", value: (a) => a.projectPlannedMw + a.projectConstructionMw, secondary: (a) => a.projects, eligible: (a) => a.projects >= 2 }, | |
| 223 | + { key: "countries_projects", scope: "countries", label: "Countries by tracked projects", unit: "projects", methodology: "Infrastructure projects located in the country (announced, permitting, approved, under construction, delayed).", value: (a) => a.projects, secondary: (a) => a.projectPlannedMw + a.projectConstructionMw }, | |
| 224 | + { key: "countries_ixps", scope: "countries", label: "Countries by internet exchanges", unit: "IXPs", methodology: "Internet exchange points indexed in the country.", value: (a) => a.ixps, secondary: (a) => a.facilities }, | |
| 178 | 225 | ]; |
| 179 | 226 | |
| 180 | 227 | const HREF: Record<Ranking["scope"], (slug: string) => string> = { countries: (s) => `/countries/${s}`, metros: (s) => `/metros/${s}`, operators: (s) => `/operators/${s}`, facilities: (s) => `/facilities/${s}` }; |
@@ -253,6 +300,10 @@ function statsJson(a: Aggregate, breakdown: Record<string, number> | undefined, | ||
| 253 | 300 | metroCount: a.metros, |
| 254 | 301 | cloudRegionCount: a.cloudRegions, |
| 255 | 302 | projectCount: a.projects, |
| 303 | + projectPlannedMw: a.projectPlannedMw || null, | |
| 304 | + projectConstructionMw: a.projectConstructionMw || null, | |
| 305 | + expansionMw: a.expansionMw || null, | |
| 306 | + aiConfirmedCount: a.aiConfirmed, | |
| 256 | 307 | ixpCount: a.ixps, |
| 257 | 308 | openedLast3Years: a.openedRecent, |
| 258 | 309 | mwCoverage: a.coverage, |
modified
apps/worker/src/scheduler.ts
+4 −1
@@ -39,7 +39,7 @@ export const RUN_LOCK_PREFIX = "dci:run-lock"; | ||
| 39 | 39 | export const RUN_LOCK_TTL_SECONDS = 4 * 3600; |
| 40 | 40 | |
| 41 | 41 | export interface CrawlJobData { connectorId: string; task: RunTask; group?: string; limit?: number; force?: boolean; urls?: string[]; requestedBy?: string } |
| 42 | −export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup"; | |
| 42 | +export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup" | "snapshot" | "quality"; | |
| 43 | 43 | export interface MaintenanceJobData { kind: MaintenanceKind; requestedBy?: string } |
| 44 | 44 | |
| 45 | 45 | let _redis: Redis | null = null; |
@@ -114,6 +114,9 @@ export async function ensureMaintenanceSchedulers(): Promise<void> { | ||
| 114 | 114 | await q.upsertJobScheduler("daily-rankings", { pattern: "30 0 * * *", tz }, { name: "rankings", data: { kind: "rankings", requestedBy: "scheduler" } }); |
| 115 | 115 | await q.upsertJobScheduler("hourly-refresh-stats", { pattern: "0 * * * *", tz }, { name: "refresh-stats", data: { kind: "refresh-stats", requestedBy: "scheduler" } }); |
| 116 | 116 | await q.upsertJobScheduler("daily-cleanup", { pattern: "0 1 * * *", tz }, { name: "cleanup", data: { kind: "cleanup", requestedBy: "scheduler" } }); |
| 117 | + // daily snapshots + regression checks after rankings; quality sweep every 6 hours | |
| 118 | + await q.upsertJobScheduler("daily-snapshot", { pattern: "45 0 * * *", tz }, { name: "snapshot", data: { kind: "snapshot", requestedBy: "scheduler" } }); | |
| 119 | + await q.upsertJobScheduler("quality-sweep", { pattern: "20 */6 * * *", tz }, { name: "quality", data: { kind: "quality", requestedBy: "scheduler" } }); | |
| 117 | 120 | } |
| 118 | 121 | |
| 119 | 122 | export interface TickResult { checked: number; enqueued: string[]; skipped: string[]; lockHeldElsewhere: boolean } |
modified
apps/worker/src/scheduling.test.ts
+11 −0
@@ -195,3 +195,14 @@ describe("pLimit", () => { | ||
| 195 | 195 | expect(await limit(async () => 42)).toBe(42); |
| 196 | 196 | }); |
| 197 | 197 | }); |
| 198 | + | |
| 199 | +describe("connectorHealthFrom", () => { | |
| 200 | + it("labels blocked, schema change and no-new-content connectors", async () => { | |
| 201 | + const { connectorHealthFrom } = await import("./scheduling.js"); | |
| 202 | + expect(connectorHealthFrom({ runHealth: "failing", status: "partial", fetched: 10, blocked: 8, discovered: 0, previouslyDiscovered: null, extracted: 0, entities: 0, docsTotal: 50 })).toBe("blocked"); | |
| 203 | + expect(connectorHealthFrom({ runHealth: "ok", status: "ok", fetched: 12, blocked: 0, discovered: 0, previouslyDiscovered: null, extracted: 12, entities: 0, docsTotal: 200 })).toBe("schema_change"); | |
| 204 | + expect(connectorHealthFrom({ runHealth: "ok", status: "ok", fetched: 0, blocked: 0, discovered: 0, previouslyDiscovered: 40, extracted: 0, entities: 0, docsTotal: 40 })).toBe("no_new_content"); | |
| 205 | + expect(connectorHealthFrom({ runHealth: "ok", status: "ok", fetched: 8, blocked: 0, discovered: 3, previouslyDiscovered: 40, extracted: 8, entities: 8, docsTotal: 40 })).toBe("ok"); | |
| 206 | + expect(connectorHealthFrom({ runHealth: "degraded", status: "failed", fetched: 8, blocked: 0, discovered: 3, previouslyDiscovered: 40, extracted: 8, entities: 8, docsTotal: 40 })).toBe("failing"); | |
| 207 | + }); | |
| 208 | +}); | |
modified
apps/worker/src/scheduling.ts
+16 −1
@@ -126,7 +126,11 @@ export function shouldSkipExtraction(i: { newHash: string; storedHash: string | | ||
| 126 | 126 | return (i.storedExtractorVersion ?? i.extractorVersion) === i.extractorVersion; |
| 127 | 127 | } |
| 128 | 128 | |
| 129 | −/** Health label from a run: failure rate < 20 % ok, < 50 % degraded, else failing. */ | |
| 129 | +/** | |
| 130 | + * Health label from a run: failure rate < 20 % ok, < 50 % degraded, else failing. A run that attempted nothing is | |
| 131 | + * `ok` only when it had nothing to do (no due documents); a run whose discovery yielded nothing for a connector that | |
| 132 | + * used to discover is reported by the caller as `no_new_content` / `blocked` (see pipeline connector health). | |
| 133 | + */ | |
| 130 | 134 | export function healthFrom(fetched: number, failed: number): "ok" | "degraded" | "failing" { |
| 131 | 135 | if (fetched <= 0) return failed > 0 ? "failing" : "ok"; |
| 132 | 136 | const rate = failed / fetched; |
@@ -155,3 +159,14 @@ export function premiumAllowedAfterErrors(errorCount: number): boolean { | ||
| 155 | 159 | if (!Number.isFinite(errorCount) || errorCount < 2) return true; |
| 156 | 160 | return errorCount % 4 === 0; |
| 157 | 161 | } |
| 162 | + | |
| 163 | +/** Connector health label (admin health center): HEALTHY | DEGRADED | BLOCKED | SCHEMA_CHANGE | NO_NEW_CONTENT | FAILED. */ | |
| 164 | +export type ConnectorHealth = "ok" | "degraded" | "failing" | "blocked" | "schema_change" | "no_new_content" | "never_run"; | |
| 165 | +export function connectorHealthFrom(i: { runHealth: "ok" | "degraded" | "failing"; status: string; fetched: number; blocked: number; discovered: number; previouslyDiscovered: number | null; extracted: number; entities: number; docsTotal: number }): ConnectorHealth { | |
| 166 | + if (i.status === "failed") return "failing"; | |
| 167 | + if (i.fetched > 0 && i.blocked >= Math.max(1, Math.ceil(i.fetched * 0.5))) return "blocked"; | |
| 168 | + // pages fetched fine but the parsers found nothing on pages that used to yield entities → the site's HTML changed | |
| 169 | + if (i.fetched >= 5 && i.extracted >= 5 && i.entities === 0 && i.docsTotal > 0) return "schema_change"; | |
| 170 | + if (i.previouslyDiscovered != null && i.previouslyDiscovered > 0 && i.discovered === 0 && i.fetched === 0) return "no_new_content"; | |
| 171 | + return i.runHealth; | |
| 172 | +} | |
added
apps/worker/src/trace.ts
+149 −0
@@ -0,0 +1,149 @@ | ||
| 1 | +/** | |
| 2 | + * Extraction debugger — the full pipeline for ONE document, stage by stage, without publishing anything: | |
| 3 | + * | |
| 4 | + * SOURCE DOCUMENT → RAW FETCH (archived body or live) → PARSED TEXT → STRUCTURED DATA (extracted records) → | |
| 5 | + * EXTRACTED ENTITIES (normalized) → EXTRACTED CLAIMS (figures with scope / semantics / evidence sentence) → | |
| 6 | + * NORMALIZED VALUES → MATCH CANDIDATES (facility resolution preview) → RECONCILIATION (dry-run ingest: created / | |
| 7 | + * updated / events / rejected) → RESULTING DATABASE CHANGES (what a real run would write). | |
| 8 | + * | |
| 9 | + * Exposed by `dci trace <doc-id|url>` and by the worker HTTP endpoint `GET /trace/<doc-id>` (internal network; the API | |
| 10 | + * proxies it as `/api/admin/documents/:id/trace`). | |
| 11 | + */ | |
| 12 | +import type { RawDocument } from "@dci/connectors"; | |
| 13 | +import { entityIsValid, mainText, pageTitle, isPdf, pdfText, classifyPage } from "@dci/connectors"; | |
| 14 | +import type { NormalizedEntity, NormalizedFacility, NormalizedProject, ValidationIssue } from "@dci/core"; | |
| 15 | +import { classifyAiEvidence, classifyCapacitySemantics, classifyInvestmentSemantics, classifyProjectEvent, classifyScope, findEvidence, newId, parseAllMw } from "@dci/core"; | |
| 16 | +import { getDb } from "@dci/db"; | |
| 17 | +import { requireConnector } from "./configs.js"; | |
| 18 | +import { createRunContext } from "./context.js"; | |
| 19 | +import { getDocument } from "./documents.js"; | |
| 20 | +import { parseDiscoveredFrom } from "./scheduling.js"; | |
| 21 | +import { ingestEntities } from "./ingest/index.js"; | |
| 22 | +import { newContext } from "./ingest/common.js"; | |
| 23 | +import { previewFacilityResolution } from "./ingest/facilities.js"; | |
| 24 | +import { extractAnnouncement } from "./connectors/news/extract-project.js"; | |
| 25 | +import { getRaw } from "./storage.js"; | |
| 26 | + | |
| 27 | +export interface TraceClaim { entityKey: string; field: string; value: number; unit: "MW" | "USD"; scope: string; scopeReason: string; semantics: string | null; evidence: { text: string; start: number; end: number } | null; } | |
| 28 | + | |
| 29 | +export interface TraceResult { | |
| 30 | + document: Record<string, unknown> | null; | |
| 31 | + fetch: { source: "archive" | "live"; status: number; contentType: string | null; bytes: number; storageKey: string | null; level: number; fetcher: string } | null; | |
| 32 | + text: { title: string | null; length: number; excerpt: string; classification: { pageType: string; rule: string; eventType: string | null; mw: number[] } | null }; | |
| 33 | + announcement: Record<string, unknown> | null; | |
| 34 | + records: Array<{ kind: string; key: string; certainty: number | null; pageType: string | null; data: Record<string, unknown>; methods: Record<string, string> }>; | |
| 35 | + entities: NormalizedEntity[]; | |
| 36 | + validation: { total: number; valid: number; rejected: number; issues: ValidationIssue[] }; | |
| 37 | + claims: TraceClaim[]; | |
| 38 | + matches: Array<{ entityKey: string; how: string; matchedId: string | null; candidates: Array<{ id: string; name: string; operatorName: string | null; city: string | null; score: number; reasons: string[]; distanceKm: number | null }> }>; | |
| 39 | + reconciliation: { created: number; updated: number; unchanged: number; merged: number; pendingMatches: number; rejected: number; events: number; provenanceRows: number; claims: number; qualityFlags: number; unscopedClaims: number; projectsVetoed: number; changes: unknown[]; refs: Array<{ type: string; id: string }> } | null; | |
| 40 | + logs: string[]; | |
| 41 | + error: string | null; | |
| 42 | +} | |
| 43 | + | |
| 44 | +/** Highlight offsets of every MW / money figure in the text (for the debugger UI). */ | |
| 45 | +export function highlightFigures(text: string): Array<{ start: number; end: number; kind: "mw" | "usd"; raw: string }> { | |
| 46 | + const out: Array<{ start: number; end: number; kind: "mw" | "usd"; raw: string }> = []; | |
| 47 | + for (const m of text.matchAll(/(\d{1,3}(?:[.,]\d{3})+|\d+(?:[.,]\d+)?)\+?\s?(gigawatts?|gw|megawatts?|mw)\b/gi)) out.push({ start: m.index ?? 0, end: (m.index ?? 0) + m[0].length, kind: "mw", raw: m[0] }); | |
| 48 | + for (const m of text.matchAll(/(?:US\$|USD|\$|€|£|A\$|C\$|S\$)\s?\d[\d.,]*\s*(?:trillion|billion|million|bn|m\b|b\b|k\b)?/gi)) out.push({ start: m.index ?? 0, end: (m.index ?? 0) + m[0].length, kind: "usd", raw: m[0] }); | |
| 49 | + return out.sort((a, b) => a.start - b.start); | |
| 50 | +} | |
| 51 | + | |
| 52 | +export async function traceDocument(idOrUrl: string, opts: { live?: boolean } = {}): Promise<TraceResult> { | |
| 53 | + const logs: string[] = []; | |
| 54 | + const result: TraceResult = { document: null, fetch: null, text: { title: null, length: 0, excerpt: "", classification: null }, announcement: null, records: [], entities: [], validation: { total: 0, valid: 0, rejected: 0, issues: [] }, claims: [], matches: [], reconciliation: null, logs, error: null }; | |
| 55 | + const doc = await getDocument(idOrUrl); | |
| 56 | + if (!doc) { result.error = `no document for ${idOrUrl}`; return result; } | |
| 57 | + result.document = { id: doc.id, connectorId: doc.connectorId, url: doc.url, pageType: doc.pageType, classifier: doc.classifier, contentHash: doc.contentHash, extractorVersion: doc.extractorVersion, extractOk: doc.extractOk, extractCount: doc.extractCount, error: doc.error, lastFetched: doc.lastFetched, lastChanged: doc.lastChanged, fetchLevel: doc.fetchLevel, storageKey: doc.storageKey, title: doc.title, entityRefs: doc.entityRefs, quarantined: doc.quarantined }; | |
| 58 | + const loaded = requireConnector(doc.connectorId); | |
| 59 | + const runId = newId("run"); | |
| 60 | + const ctx = createRunContext(loaded, { runId, dryRun: true, onLog: (l) => logs.push(`${l.level} ${l.msg}`), logLevel: "debug" }); | |
| 61 | + try { | |
| 62 | + // ── raw | |
| 63 | + let raw: RawDocument | null = null; | |
| 64 | + if (!opts.live && doc.storageKey) { | |
| 65 | + const stored = await getRaw(doc.storageKey); | |
| 66 | + if (stored) { | |
| 67 | + const contentType = stored.contentType ?? doc.contentType; | |
| 68 | + const isText = !contentType || /text|json|xml|javascript|html|csv|markdown/i.test(contentType); | |
| 69 | + const { group } = parseDiscoveredFrom(doc.discoveredFrom); | |
| 70 | + raw = { url: doc.url, finalUrl: doc.url, fetchedAt: doc.lastFetched ?? new Date().toISOString(), status: doc.statusCode ?? 200, contentType, body: stored.body, text: isText ? stored.body.toString("utf8") : "", headers: {}, etag: doc.etag, lastModified: doc.lastModified, notModified: false, fetcher: "cache", level: doc.fetchLevel as RawDocument["level"], durationMs: 0, credits: 0, group, pageType: doc.pageType !== "unknown" ? (doc.pageType as RawDocument["pageType"]) : undefined }; | |
| 71 | + result.fetch = { source: "archive", status: raw.status, contentType, bytes: stored.body.length, storageKey: doc.storageKey, level: doc.fetchLevel, fetcher: "cache" }; | |
| 72 | + } | |
| 73 | + } | |
| 74 | + if (!raw) { | |
| 75 | + const { group } = parseDiscoveredFrom(doc.discoveredFrom); | |
| 76 | + raw = await loaded.connector.fetch(ctx, { url: doc.url, group, pageType: doc.pageType !== "unknown" ? (doc.pageType as RawDocument["pageType"]) : undefined, priority: 100 }); | |
| 77 | + result.fetch = { source: "live", status: raw.status, contentType: raw.contentType, bytes: raw.body.length, storageKey: null, level: raw.level, fetcher: raw.fetcher }; | |
| 78 | + if (raw.error) { result.error = `fetch: ${raw.error.code} ${raw.error.message}`; return result; } | |
| 79 | + } | |
| 80 | + // ── text | |
| 81 | + let title: string | null = null; | |
| 82 | + let text = ""; | |
| 83 | + if (isPdf(raw)) { const p = await pdfText(raw.body); text = p.text; title = p.title ?? null; } | |
| 84 | + else if (/html|xml/.test(raw.contentType ?? "") || /<html/i.test(raw.text.slice(0, 2000))) { title = pageTitle(raw.text); text = mainText(raw.text); } | |
| 85 | + else text = raw.markdown ?? raw.text; | |
| 86 | + const cls = classifyPage(raw.finalUrl, title, text); | |
| 87 | + result.text = { title, length: text.length, excerpt: text.slice(0, 6000), classification: { pageType: cls.pageType, rule: cls.rule, eventType: cls.eventType ?? null, mw: cls.mw ?? [] } }; | |
| 88 | + // ── announcement analysis (news-like pages) | |
| 89 | + if (text.length > 80 && /news|press|announcement|planning|construction|expansion|acquisition|power|financial|project|unknown/.test(cls.pageType)) { | |
| 90 | + try { | |
| 91 | + const a = extractAnnouncement(title, text); | |
| 92 | + result.announcement = { title: a.title, pageType: a.pageType, eventType: a.eventType ?? null, relevance: a.relevance, leadRelevance: a.leadRelevance, titleSignal: a.titleSignal, nonProjectTitle: a.nonProjectTitle, classification: a.classification, status: a.status, headlineMw: a.headlineMw, mwAll: a.mwAll, money: a.money, investmentUsd: a.investmentUsd, acreage: a.acreage, phaseCount: a.phaseCount, expectedOpening: a.expectedOpening, location: a.location, operator: a.operator?.name ?? null, operators: a.operators.map((o) => o.name), explicitName: a.explicitName, projectName: a.projectName, capacityScope: a.capacityScope, capacitySemantics: a.capacitySemantics, investmentScope: a.investmentScope, investmentSemantics: a.investmentSemantics, aiEvidence: a.aiEvidence, hqGuarded: a.hqGuarded, evidence: a.evidence, methods: a.methods, figures: highlightFigures(`${a.title}\n${text.slice(0, 6000)}`) }; | |
| 93 | + } catch (e) { logs.push(`warn announcement analysis failed: ${(e as Error).message}`); } | |
| 94 | + } | |
| 95 | + // ── structured data | |
| 96 | + const records = await loaded.connector.extract(ctx, raw); | |
| 97 | + result.records = records.map((r) => ({ kind: r.kind, key: r.key, certainty: r.certainty ?? null, pageType: r.pageType ?? null, data: r.data, methods: r.methods ?? {} })); | |
| 98 | + // ── entities + validation | |
| 99 | + const entities = await loaded.connector.normalize(ctx, records); | |
| 100 | + const report = await loaded.connector.validate(ctx, entities); | |
| 101 | + result.entities = entities; | |
| 102 | + result.validation = { total: report.total, valid: report.valid, rejected: report.rejected, issues: report.issues }; | |
| 103 | + // ── claims (what the claim store would receive) | |
| 104 | + for (const e of entities) { | |
| 105 | + if (e.entityType === "facility") { | |
| 106 | + const f = e as NormalizedFacility; | |
| 107 | + for (const field of ["itCapacityMw", "totalPowerMw", "plannedPowerMw"] as const) { | |
| 108 | + const v = f[field]; | |
| 109 | + if (v == null || v <= 0) continue; | |
| 110 | + const context = f.claimContext?.[field] ?? null; | |
| 111 | + const sc = context ? classifyScope(context, "facility") : { scope: "facility", reason: "structured:record" }; | |
| 112 | + const sem = context ? classifyCapacitySemantics(context) : { predicate: field === "itCapacityMw" ? "it_capacity_mw" : field === "totalPowerMw" ? "current_power_mw" : "planned_power_mw", reason: "field" }; | |
| 113 | + result.claims.push({ entityKey: f.key, field, value: v, unit: "MW", scope: sc.scope, scopeReason: sc.reason, semantics: sem.predicate, evidence: context ? findEvidence(context, v, "mw") : null }); | |
| 114 | + } | |
| 115 | + } else if (e.entityType === "project") { | |
| 116 | + const p = e as NormalizedProject; | |
| 117 | + if (p.plannedMw != null && p.plannedMw > 0) { const c = p.claimContext?.plannedMw ?? null; result.claims.push({ entityKey: p.key, field: "plannedMw", value: p.plannedMw, unit: "MW", scope: p.capacityScope ?? classifyScope(c, "facility").scope, scopeReason: p.capacityScope ? "extractor" : classifyScope(c, "facility").reason, semantics: p.capacitySemantics ?? null, evidence: c ? findEvidence(c, p.plannedMw, "mw") : null }); } | |
| 118 | + if (p.investmentUsd != null && p.investmentUsd > 0) { const c = p.claimContext?.investmentUsd ?? null; const sem = classifyInvestmentSemantics(c ?? p.name); result.claims.push({ entityKey: p.key, field: "investmentUsd", value: p.investmentUsd, unit: "USD", scope: p.investmentScope ?? sem.scope, scopeReason: p.investmentScope ? "extractor" : sem.reason, semantics: p.investmentSemantics ?? sem.predicate, evidence: c ? findEvidence(c, p.investmentUsd, "usd") : null }); } | |
| 119 | + } else if (e.entityType === "news_event") { | |
| 120 | + for (const mw of e.mentions?.mw ?? []) result.claims.push({ entityKey: e.key, field: "mw", value: mw, unit: "MW", scope: "unknown", scopeReason: "news mention (never assigned)", semantics: null, evidence: findEvidence(text, mw, "mw") }); | |
| 121 | + } | |
| 122 | + } | |
| 123 | + // ── match candidates (facilities) — read-only, inside a rolled-back transaction | |
| 124 | + const valid = entities.filter((x) => entityIsValid(report, x.key)); | |
| 125 | + const db = getDb(); | |
| 126 | + const run = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: true }; | |
| 127 | + await db.transaction(async (tx) => { | |
| 128 | + const ictx = newContext(run, { documentId: doc.id, url: raw!.finalUrl, pageType: cls.pageType, fetchedAt: raw!.fetchedAt }); | |
| 129 | + for (const e of valid) { | |
| 130 | + if (e.entityType !== "facility") continue; | |
| 131 | + try { const m = await previewFacilityResolution(tx, ictx, e as NormalizedFacility); result.matches.push({ entityKey: e.key, ...m }); } catch (err) { logs.push(`warn match preview ${e.key}: ${(err as Error).message}`); } | |
| 132 | + } | |
| 133 | + throw new Error("__trace_rollback__"); | |
| 134 | + }).catch((e: Error) => { if (e.message !== "__trace_rollback__") throw e; }); | |
| 135 | + // ── reconciliation (dry-run ingest: everything runs, nothing is committed) | |
| 136 | + if (valid.length) { | |
| 137 | + const st = await ingestEntities(run, valid, { documentId: doc.id, url: raw.finalUrl, pageType: cls.pageType, fetchedAt: raw.fetchedAt }); | |
| 138 | + result.reconciliation = { created: st.created, updated: st.updated, unchanged: st.unchanged, merged: st.merged, pendingMatches: st.pendingMatches, rejected: st.rejected, events: st.events, provenanceRows: st.provenanceRows, claims: st.claims ?? 0, qualityFlags: st.qualityFlags ?? 0, unscopedClaims: st.unscopedClaims ?? 0, projectsVetoed: st.projectsVetoed ?? 0, changes: st.changes, refs: st.refs }; | |
| 139 | + } else result.reconciliation = { created: 0, updated: 0, unchanged: 0, merged: 0, pendingMatches: 0, rejected: report.rejected, events: 0, provenanceRows: 0, claims: 0, qualityFlags: 0, unscopedClaims: 0, projectsVetoed: 0, changes: [], refs: [] }; | |
| 140 | + // AI grading of the whole text for the debugger | |
| 141 | + logs.push(`debug ai-evidence: ${JSON.stringify(classifyAiEvidence(`${title ?? ""} ${text.slice(0, 3000)}`))}`); | |
| 142 | + logs.push(`debug mw figures in text: ${parseAllMw(text.slice(0, 6000)).join(", ") || "none"}`); | |
| 143 | + // the announcement classifier verdict for non-news pages too | |
| 144 | + if (!result.announcement && title) logs.push(`debug classifier: ${JSON.stringify(classifyProjectEvent({ title, lead: text.slice(0, 600) }))}`); | |
| 145 | + } catch (e) { | |
| 146 | + result.error = (e as Error).stack ?? (e as Error).message; | |
| 147 | + } | |
| 148 | + return result; | |
| 149 | +} | |
added
docs/AUDIT-2026-09-11.md
+140 −0
@@ -0,0 +1,140 @@ | ||
| 1 | +# DataCenterIndex — full repository & data audit (2026-09-11) | |
| 2 | + | |
| 3 | +Internal audit written before the "infrastructure intelligence graph" upgrade. Sections A–K follow the brief. Figures come | |
| 4 | +from the production database on BHS128 (`deploy/bin/psql.sh`) on 2026-09-11 evening, one day after launch. | |
| 5 | + | |
| 6 | +## A. Current architecture | |
| 7 | + | |
| 8 | +- **Monorepo** pnpm + TypeScript strict/ESM: `packages/core` (types, ids, normalize, geo, confidence, ssrf, api contract), | |
| 9 | + `packages/db` (Drizzle schema, plain SQL migrations, ClickHouse DDL), `packages/connectors` (connector SDK), `apps/worker` | |
| 10 | + (runtime, ingest, scheduler, rankings, CLI), `apps/api` (Fastify 5, `/api/v1` public + `/api/admin` token), `apps/web` | |
| 11 | + (Next 16 App Router, Tailwind v4, MapLibre + OpenFreeMap, hand-rolled SVG charts). | |
| 12 | +- **Rendering**: every public page is ISR (`revalidate` 30–300 s) on top of a typed API client that never throws; map page | |
| 13 | + is a static shell with client fetches to `/api/v1/map` (server-side clustering, ≤ 5 000 points). | |
| 14 | +- **Deployment**: two OVH servers joined by WireGuard. BHS128 = data/public (Caddy edge :8300 → web :8310 / api :8311, | |
| 15 | + Postgres 17, ClickHouse, Redis, MinIO, Prometheus, Grafana). BHS64b = crawl worker (concurrency 6). Public route via the | |
| 16 | + MacLustr Tunnel gateway BHS64. Nightly backup timer (pg_dump, config tar, ClickHouse backup best-effort, MinIO mirror, | |
| 17 | + rsync off-node to BHS64b). No CI, no Alertmanager. | |
| 18 | +- **Finding (ops)**: the production `scheduler` container is crash-looping (851 restarts, exit 0, no log) because the deployed | |
| 19 | + image predates commit `ff67cee` (standalone scheduler entry). Crawling continues because the BHS64b worker embeds the | |
| 20 | + scheduler loop. Fixed by the redeploy at the end of this upgrade. | |
| 21 | + | |
| 22 | +## B. Current database schema (Postgres) | |
| 23 | + | |
| 24 | +28 tables. Core: `facilities` (2 912 rows; `it_capacity_mw`, `total_power_mw`, `planned_power_mw`, `mw_is_estimate`, | |
| 25 | +`geo_precision`, `status`, `confidence`, `completeness`, `campus_id`, `merged_into`), `operators` (470), `projects` (324; | |
| 26 | +`planned_mw`, `investment_usd`, `status`), `campuses` (114), `metros` (214), `countries` (250), `cloud_regions` (307), | |
| 27 | +`ixps` (12), `facility_ixps`, `facility_tenants`, `facility_aliases`, `entity_keys` (connector-scoped keys, **global PK on | |
| 28 | +`key`**), `provenance` (38 533 rows, one row per entity×field×source×url, `is_current`, `method`, `extractor_version`), | |
| 29 | +`events` (5 549, fingerprint-deduplicated, `significance` 0–100), `news_items` (4 404), `documents` (7 092) + | |
| 30 | +`document_versions` (8 785, content-hash versions with `diff_summary`), `entity_matches` (1 730: 1 200 auto_created, 81 | |
| 31 | +auto_merged, **449 pending never reviewed**), `rankings` (snapshots with `min_coverage`), `daily_metrics`, `sources`, | |
| 32 | +`connectors`, `connector_runs`, `connector_state`, `project_timeline`, `system_alerts`, `admin_sessions`. | |
| 33 | +ClickHouse holds append-only `observations`, `crawl_log`, `page_changes`, `entity_daily`, `api_requests` (best-effort, silently | |
| 34 | +dropped on failure). | |
| 35 | + | |
| 36 | +Indexes: btree on the obvious FKs/status, trigram GIN on names/cities, generated tsvector on facilities, partial `(lat,lng)`. | |
| 37 | +**No PostGIS** (haversine + bbox pre-filter + geohash). No `claims`/observation table in Postgres; no entity-version table; | |
| 38 | +no `run_id` on `provenance`/`document_versions`. | |
| 39 | + | |
| 40 | +## C. Current crawler architecture | |
| 41 | + | |
| 42 | +Declarative YAML per source (`config/connectors/*.yaml`) → `GenericConnector` or registered parser/implementation → | |
| 43 | +`discover → fetch (L1 direct → L2 browser identity → L3 Firecrawl → L4 Scrapfly) → archive (MinIO, zstd, per content hash) | |
| 44 | +→ extract → normalize → validate → ingest (reconcile, provenance, events)`. Robots.txt honoured, per-host token bucket, | |
| 45 | +conditional GET, content-hash change detection, extraction skipped when hash and `extractor_version` unchanged, adaptive | |
| 46 | +`next_check` with EMA change score, per-document quarantine after 3 × 404/410. BullMQ `crawl` + `maintenance` queues, | |
| 47 | +Redis run locks, per-run/per-connector/per-provider premium budgets, Prometheus metrics, `dci doctor`. | |
| 48 | + | |
| 49 | +## D. Current connectors | |
| 50 | + | |
| 51 | +104 YAML configs (87 enabled): 60 operator directories, 20 industry/hyperscaler newsrooms (RSS), 10 cloud-region JSON, | |
| 52 | +6 government, 3 utilities, 3 datasets (PeeringDB — disabled pending AUP approval —, Wikidata, World Bank), SEC EDGAR | |
| 53 | +full-text, OSM Overpass. Facility rows by connector: OSM 1 346, Equinix 262, Digital Realty 259, Wikidata 190, STACK 78, | |
| 54 | +DataBank 75, NTT 72, EdgeConneX 59, CyrusOne 52, Google 47, QTS 47 … 58 named parsers; **zero configs use the declarative | |
| 55 | +extractor** (dead code path); **zero HTML fixtures**; `packages/core` and `packages/connectors` have no tests. | |
| 56 | + | |
| 57 | +## E. Current data-quality issues (measured) | |
| 58 | + | |
| 59 | +Facilities: 2 640 operational / 154 unknown / 76 announced / 37 under construction; only **536 (18 %) have any MW**; | |
| 60 | +797 have no coordinates; 1 377 have `facility_type = unknown`; only 4 flagged AI; 109 duplicate normalized-name pairs and | |
| 61 | +469 coordinate pairs within ~100 m (OSM building-vs-campus, e.g. atNorth ICE02 vs buildings M01…M16). | |
| 62 | + | |
| 63 | +Confirmed extraction errors: | |
| 64 | +- **Decimal MW parsed as thousands**: DataBank MSP4 shows 779 MW; source says "0.779MW Critical IT Load". Same bug hits | |
| 65 | + DFW6 (675), AUS1 (405) and any "0.xxx MW" figure (`parseMw` strips the dot when three decimals follow). | |
| 66 | +- **Headlines stored as project names**: 182/324 projects are named after an article title. | |
| 67 | +- **Executive appointments as projects**: "STACK Appoints Matt VanderZanden as CEO" = 13 000 MW project; | |
| 68 | + "AirTrunk appoints Laura Coad" = 1 400 MW; VIRTUS, 365 Data Centers, Element Critical, Cassava appointments likewise. | |
| 69 | +- **Portfolio/company figures on one project**: "Microsoft Georgia data center project" 12 GW / $175 B (company capex); | |
| 70 | + "Google Alabama" 17 GW (Southern Company pipeline); Nvidia "Washington" 20 GW (Starcloud funding story). | |
| 71 | +- **Market-research articles as projects**: "Hyperscale Data Center Market to Surpass USD 624.2 Billion" ($80.9 B). | |
| 72 | +- **Wrong operator**: "Meta's Canadian AI Data Center" assigned to Google. | |
| 73 | +- **PPA / sustainability / HQ / workforce stories as projects** (Ormat–Switch PPA 3 400 MW, AirTrunk HQ 1 400 MW, | |
| 74 | + Meta Workforce Academy 5 000 MW). | |
| 75 | +- **Projects have no coordinates at all** (0/324), 158 have a country, 0 link to a facility. | |
| 76 | +- **Broken operator slug** `item` for 中華電信數據通信分公司 (26 facilities). | |
| 77 | +- Ranking coverage gate leaves 3 countries in `countries_known_mw` — the MW coverage story must be visible, not hidden. | |
| 78 | + | |
| 79 | +Structural causes (code): one-row-per-entity with no claim scope; `isIdentifyingExternalId` folds unrelated facilities on | |
| 80 | +`*_code/*_url` keys; `loadCandidates` truncates at 400 rows without ordering; news country via unsafe `countryFromText` | |
| 81 | +("North America" → US); no HQ guard on project location; `projects.country_iso2` write-once; campus/building double counting | |
| 82 | +in every MW aggregate; equal-authority ping-pong after 30 days creates event churn; `entity_keys.key` global PK; | |
| 83 | +`parserVersion` is `v1` everywhere so parser fixes never invalidate cached extractions; `healthFrom(0,0) = ok`. | |
| 84 | + | |
| 85 | +## F. Current frontend strengths | |
| 86 | + | |
| 87 | +Unified status/confidence/precision visual language (CSS variables shared by badges, charts and MapLibre paint); | |
| 88 | +per-field provenance popovers + provenance table + source history + sources footer with licences; dense typography with | |
| 89 | +mono tabular figures; URL-as-state everywhere; ⌘K/`/` command palette with API-side query interpretation; server-clustered | |
| 90 | +map with precision rings; full SEO (canonical, per-entity OG images, JSON-LD, sharded sitemaps); a11y depth; an automated | |
| 91 | +UX sweep harness; a substantial `/admin` console (connectors, documents, matches, events, dev tool). | |
| 92 | + | |
| 93 | +## G. Current frontend weaknesses | |
| 94 | + | |
| 95 | +No explore/query builder, no compare, no evidence drawer (only popover + bottom table), no export, no watchlist, no | |
| 96 | +coverage page, no AI index, no pulse, no power/connectivity layers, no map density/time modes, no virtualization, | |
| 97 | +long single-scroll detail pages without active-section tracking, hand-rolled charts without brushing, keyboard story stops | |
| 98 | +at the palette, project pages lack a state machine/timeline funnel, MapLibre controls < 40 px on touch. | |
| 99 | + | |
| 100 | +## H. Current API strengths | |
| 101 | + | |
| 102 | +27 public GET endpoints, envelope `{data, meta, sources}`, zod-validated params, weak ETags + `s-maxage`, rate limits, | |
| 103 | +consistent 404/400 JSON, OpenAPI + Swagger UI, `/map` zoom-tiered clustering, `/search` with interpretation, admin API with | |
| 104 | +constant-time token compare and same-origin proxy with CSRF guard. Weaknesses: no provenance/history/nearby/coverage/export | |
| 105 | +endpoints; list responses carry no `sources`; OpenAPI has no response schemas; placeholder-token check is a no-op. | |
| 106 | + | |
| 107 | +## I. Schema changes proposed (migration 0003) | |
| 108 | + | |
| 109 | +1. `claims` — claim-first store: subject (type,id), predicate, value/unit, **scope** (building/facility/campus/metro/ | |
| 110 | + country/portfolio/company/unknown), source/document/url, published/retrieved, confidence, is_estimate, evidence text + | |
| 111 | + offsets, parser name/version, `run_id`, status (current/superseded/rejected/review/unscoped), rejection reason. | |
| 112 | +2. `quality_flags` — deterministic sanity flags (capacity, investment, scope, location, operator, duplicate, project | |
| 113 | + false-positive) with severity, priority score, status, resolution, linking to entity + claim. | |
| 114 | +3. `facilities.parent_facility_id`, `facilities.record_scope` (building/facility/campus) for containment-aware aggregation; | |
| 115 | + `facilities.ai_evidence` (confirmed/likely/associated/unknown); `facilities.utility_capacity_mw`, `grid_connection_mw`, | |
| 116 | + `ultimate_campus_mw` (capacity ontology columns; IT/total/planned kept). | |
| 117 | +4. `projects.project_class` (NEW_BUILD … EXECUTIVE_APPOINTMENT), `projects.evidence_level`, `projects.merged_into`, | |
| 118 | + `projects.lat/lng` geocoded from metro/city with `geo_precision = city`, `projects.ai_evidence`, `projects.investment_scope`. | |
| 119 | +5. `document_versions.run_id`, `provenance.run_id`, `provenance.scope`; `connectors.quarantine` (bool) + | |
| 120 | + `consecutive_failures`, `blocked_since`. | |
| 121 | +6. `field_authority` (per-field source-kind tiers) as code table in `packages/core` + methodology page (no table needed). | |
| 122 | +7. `entity_snapshots` (daily JSON snapshot of global totals, rankings, facility status counts, project stage counts) for | |
| 123 | + "as of" views and regression checks. | |
| 124 | +8. Indexes: `provenance(is_current)`, `facilities(merged_into)`, `facilities(campus_id)`, `facilities(parent_facility_id)`, | |
| 125 | + `events(significance, detected_at)`, `daily_metrics(metric, dim, day)`, GIN on `news_items.operator_ids`. | |
| 126 | + | |
| 127 | +## J. Components to preserve | |
| 128 | + | |
| 129 | +Everything in section F/H, the connector SDK and escalation ladder, the content-hash/archive pipeline, entity keys, | |
| 130 | +provenance table, event fingerprinting, the matcher (with fixes), rankings coverage gates, the design tokens, the map | |
| 131 | +architecture, the admin console, deploy scripts, backups, the QA sweep. | |
| 132 | + | |
| 133 | +## K. Implementation order | |
| 134 | + | |
| 135 | +Phase 1 data quality (parseMw fix, claim/scope layer, project classifier + evidence threshold, capacity/investment sanity | |
| 136 | +engine, external-id allowlist, HQ/country guards, campus containment aggregation, extraction debugger, connector health, | |
| 137 | +fixtures, cleanup of current bad records) → Phase 2 core UX (home 2.0, map 2.0, facility 3.0, operator/market/country/project | |
| 138 | +2.0, brand, mobile) → Phase 3 intelligence (AI index, pulse, live feed 2.0, compare, explore, coverage, quality dashboards) | |
| 139 | +→ Phase 4 graph (connectivity, power, cloud, IXP pages) → Phase 5 historical (time machine, capacity history, announced vs | |
| 140 | +delivered, velocity) → API 2.0 + docs + download → regression checks, security fixes, deploy, final data audit. | |
modified
packages/connectors/src/generic.ts
+4 −2
@@ -122,7 +122,7 @@ export class GenericConnector implements Connector { | ||
| 122 | 122 | // when no explicit geo but page has JSON-LD/meta geo, the extractor may already have set lat/lng; otherwise leave null (never fake coordinates) |
| 123 | 123 | out.push(f); |
| 124 | 124 | } else if (r.kind === "news_event") { |
| 125 | − const n: NormalizedNewsEvent = { entityType: "news_event", key: r.key, title: str("title") ?? r.url, url: (str("url") ?? r.url)!, publishedAt: str("publishedAt"), summary: str("summary"), pageType: (str("pageType") as PageType | null) ?? r.pageType, eventType: (str("eventType") as NormalizedNewsEvent["eventType"]) ?? undefined, mentions: { mw: Array.isArray(x.mw) ? (x.mw as number[]) : [], operators: [...(Array.isArray(x.operators) ? (x.operators as string[]) : []), ...(d.operatorName ? [d.operatorName] : [])], countriesIso2: [...(Array.isArray(x.countries) ? (x.countries as string[]) : []), countryFromText(`${str("title")} ${str("summary")}`) ?? ""].filter((c, i, a) => c && a.indexOf(c) === i), cities: Array.isArray(x.cities) ? (x.cities as string[]) : [] }, provenance: prov }; | |
| 125 | + const n: NormalizedNewsEvent = { entityType: "news_event", key: r.key, title: str("title") ?? r.url, url: (str("url") ?? r.url)!, publishedAt: str("publishedAt"), summary: str("summary"), pageType: (str("pageType") as PageType | null) ?? r.pageType, eventType: (str("eventType") as NormalizedNewsEvent["eventType"]) ?? undefined, mentions: { mw: Array.isArray(x.mw) ? (x.mw as number[]) : [], operators: [...(Array.isArray(x.operators) ? (x.operators as string[]) : []), ...(d.operatorName ? [d.operatorName] : [])], countriesIso2: [...(Array.isArray(x.countries) ? (x.countries as string[]) : []), countryFromText(`${str("title")} ${str("summary")}`) ?? ""].filter((c, i, a) => c && a.indexOf(c) === i), cities: Array.isArray(x.cities) ? (x.cities as string[]) : [] }, projectClass: str("projectClass"), isAi: typeof x.isAi === "boolean" ? x.isAi : null, provenance: prov }; | |
| 126 | 126 | out.push(n); |
| 127 | 127 | // Project candidate from a fallback news_event (no dedicated parser). Third-party publishers (kind = news) go |
| 128 | 128 | // through `news_article_v1`, which has the real gate — the fallback never creates projects for them. For |
@@ -139,7 +139,9 @@ export class GenericConnector implements Connector { | ||
| 139 | 139 | } else if (r.kind === "project") { |
| 140 | 140 | const name = str("name") ?? str("_title"); if (!name) continue; |
| 141 | 141 | const lat = num("lat"), lng = num("lng"); |
| 142 | − out.push({ entityType: "project", key: r.key, name, operatorName: str("operatorName") ?? d.operatorName ?? null, city: str("city"), regionName: str("regionName"), countryIso2: str("countryIso2") ?? d.countryIso2 ?? null, geo: validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: "approximate", source: this.id } : null, status: (str("status") as FacilityStatus | null) ?? "announced", announcedOn: str("announcedOn"), expectedOpening: str("expectedOpening"), plannedMw: num("plannedMw"), investmentUsd: num("investmentUsd"), acreage: num("acreage"), phaseCount: num("phaseCount"), description: str("description"), sourceUrl: r.url, provenance: prov }); | |
| 142 | + const claimContext = x.claimContext && typeof x.claimContext === "object" ? (x.claimContext as Record<string, string>) : undefined; | |
| 143 | + out.push({ entityType: "project", key: r.key, name, operatorName: str("operatorName") ?? d.operatorName ?? null, city: str("city"), regionName: str("regionName"), countryIso2: str("countryIso2") ?? d.countryIso2 ?? null, geo: validLatLng(lat, lng) ? { lat: lat!, lng: lng!, precision: "approximate", source: this.id } : null, status: (str("status") as FacilityStatus | null) ?? "announced", announcedOn: str("announcedOn"), expectedOpening: str("expectedOpening"), plannedMw: num("plannedMw"), investmentUsd: num("investmentUsd"), acreage: num("acreage"), phaseCount: num("phaseCount"), description: str("description"), sourceUrl: r.url, | |
| 144 | + projectClass: str("projectClass"), evidenceLevel: (str("evidenceLevel") as NormalizedProject["evidenceLevel"]) ?? null, capacityScope: str("capacityScope"), capacitySemantics: str("capacitySemantics"), investmentScope: str("investmentScope"), investmentSemantics: str("investmentSemantics"), investmentCurrency: str("investmentCurrency"), investmentOriginal: num("investmentOriginal"), claimContext, aiEvidence: (str("aiEvidence") as NormalizedProject["aiEvidence"]) ?? null, developerName: str("developerName"), tenantName: str("tenantName"), campusName: str("campusName"), constructionStartedOn: str("constructionStartedOn"), approvedOn: str("approvedOn"), permitFiledOn: str("permitFiledOn"), provenance: prov }); | |
| 143 | 145 | } else if (r.kind === "operator") { |
| 144 | 146 | const name = str("name"); if (!name) continue; |
| 145 | 147 | out.push({ entityType: "operator", key: r.key, name, website: str("website"), kind: (str("kind") as NormalizedEntity extends { kind?: infer K } ? K : never) ?? null, hqCountryIso2: str("hqCountryIso2"), description: str("description"), provenance: prov }); |
modified
packages/core/src/api-types.ts
+580 −4
@@ -3,6 +3,7 @@ | ||
| 3 | 3 | * Every response is wrapped: { data, meta?, sources? }. Dates are ISO strings; partial dates use "YYYY", "YYYY-MM", "YYYY-Qn". |
| 4 | 4 | */ |
| 5 | 5 | import type { ConfidenceLevel, EntityType, EventType, FacilityStatus, FacilityType, GeoPrecision, OperatorKind, ReviewStatus, SourceKind } from "./types.js"; |
| 6 | +import type { AiEvidence, AuthorityTier, ClaimScope, ClaimStatus, ProjectClass, Significance } from "./claims.js"; | |
| 6 | 7 | |
| 7 | 8 | export interface ApiEnvelope<T> { |
| 8 | 9 | data: T; |
@@ -18,6 +19,9 @@ export interface SourceRef { | ||
| 18 | 19 | url?: string | null; |
| 19 | 20 | license?: string | null; |
| 20 | 21 | attribution?: string | null; |
| 22 | + /** allowed | attribution | restricted | unknown — drives /download and re-use guidance */ | |
| 23 | + redistribution?: "allowed" | "attribution" | "restricted" | "unknown" | null; | |
| 24 | + attributionRequired?: boolean | null; | |
| 21 | 25 | } |
| 22 | 26 | |
| 23 | 27 | export interface ProvenanceDTO { |
@@ -33,6 +37,82 @@ export interface ProvenanceDTO { | ||
| 33 | 37 | confidence: ConfidenceLevel; |
| 34 | 38 | isEstimate: boolean; |
| 35 | 39 | method?: string | null; |
| 40 | + /** true for the observation backing the displayed column value */ | |
| 41 | + isWinner?: boolean; | |
| 42 | + scope?: string | null; | |
| 43 | + runId?: string | null; | |
| 44 | + documentId?: string | null; | |
| 45 | +} | |
| 46 | + | |
| 47 | +/** One figure asserted by one document about one subject (claim-first architecture). */ | |
| 48 | +export interface ClaimDTO { | |
| 49 | + id: string; | |
| 50 | + predicate: string; | |
| 51 | + label: string; | |
| 52 | + value: number | null; | |
| 53 | + valueText: string | null; | |
| 54 | + unit: string | null; | |
| 55 | + scope: ClaimScope; | |
| 56 | + scopeReason: string | null; | |
| 57 | + sourceId: string; | |
| 58 | + sourceName: string; | |
| 59 | + sourceKind: SourceKind; | |
| 60 | + url: string; | |
| 61 | + documentId: string | null; | |
| 62 | + publishedAt: string | null; | |
| 63 | + retrievedAt: string; | |
| 64 | + confidence: ConfidenceLevel; | |
| 65 | + isEstimate: boolean; | |
| 66 | + authorityTier: AuthorityTier; | |
| 67 | + evidenceText: string | null; | |
| 68 | + parserName: string | null; | |
| 69 | + parserVersion: string | null; | |
| 70 | + status: ClaimStatus; | |
| 71 | + rejectionReason: string | null; | |
| 72 | + /** true when this claim backs the displayed value */ | |
| 73 | + isWinner: boolean; | |
| 74 | + firstObserved: string; | |
| 75 | + lastObserved: string; | |
| 76 | +} | |
| 77 | + | |
| 78 | +/** A dated value of one field, for capacity history charts and change diffs. */ | |
| 79 | +export interface HistoryPoint { | |
| 80 | + date: string; | |
| 81 | + field: string; | |
| 82 | + predicate?: string | null; | |
| 83 | + value: unknown; | |
| 84 | + oldValue?: unknown; | |
| 85 | + sourceId: string | null; | |
| 86 | + sourceName: string | null; | |
| 87 | + sourceKind: SourceKind | null; | |
| 88 | + url: string | null; | |
| 89 | + claimId?: string | null; | |
| 90 | + eventId?: string | null; | |
| 91 | + kind: "observed" | "changed" | "claim"; | |
| 92 | +} | |
| 93 | + | |
| 94 | +export interface EntityHistory { | |
| 95 | + entityType: "facility" | "project" | "operator"; | |
| 96 | + entityId: string; | |
| 97 | + fields: Record<string, HistoryPoint[]>; | |
| 98 | + /** field value changes with old → new and the evidence behind each */ | |
| 99 | + changes: HistoryPoint[]; | |
| 100 | +} | |
| 101 | + | |
| 102 | +export interface Distance { distanceKm: number } | |
| 103 | +export interface NearbyInfrastructure { | |
| 104 | + center: { lat: number; lng: number }; | |
| 105 | + radiusKm: number; | |
| 106 | + facilities: Array<FacilitySummary & Distance>; | |
| 107 | + projects: Array<ProjectSummary & Distance>; | |
| 108 | + ixps: Array<IxpSummary & Distance & { lat: number | null; lng: number | null }>; | |
| 109 | + cloudRegions: Array<CloudRegionSummary & Distance>; | |
| 110 | + metros: Array<{ id: string; slug: string; name: string; countryIso2: string; distanceKm: number }>; | |
| 111 | + /** cable landing stations / substations / power plants — empty until their connectors exist (never faked) */ | |
| 112 | + landingStations: Array<{ id: string; name: string; lat: number; lng: number; distanceKm: number; sourceName: string }>; | |
| 113 | + substations: Array<{ id: string; name: string; lat: number; lng: number; distanceKm: number; sourceName: string }>; | |
| 114 | + powerPlants: Array<{ id: string; name: string; lat: number; lng: number; distanceKm: number; kind: string | null; capacityMw: number | null; sourceName: string }>; | |
| 115 | + note: string; | |
| 36 | 116 | } |
| 37 | 117 | |
| 38 | 118 | export interface OperatorSummary { |
@@ -47,10 +127,44 @@ export interface OperatorSummary { | ||
| 47 | 127 | metroCount?: number; |
| 48 | 128 | knownMw?: number | null; |
| 49 | 129 | plannedMw?: number | null; |
| 130 | + constructionMw?: number | null; | |
| 50 | 131 | projectCount?: number; |
| 132 | + projectPlannedMw?: number | null; | |
| 133 | + aiCount?: number; | |
| 134 | + cloudRegionCount?: number; | |
| 135 | + mwCoverage?: number | null; | |
| 136 | + isCloudProvider?: boolean; | |
| 137 | + isCarrier?: boolean; | |
| 138 | +} | |
| 139 | + | |
| 140 | +export interface PipelineBreakdown { | |
| 141 | + /** facility counts + known MW by lifecycle bucket (containment-aware, published figures only) */ | |
| 142 | + operational: { count: number; mw: number | null }; | |
| 143 | + construction: { count: number; mw: number | null }; | |
| 144 | + approved: { count: number; mw: number | null }; | |
| 145 | + announced: { count: number; mw: number | null }; | |
| 146 | + /** project records (not facilities) by stage */ | |
| 147 | + projects: Array<{ status: FacilityStatus; count: number; mw: number | null }>; | |
| 148 | + mwCoverage: number; | |
| 149 | +} | |
| 150 | + | |
| 151 | +export interface ExpansionVelocity { | |
| 152 | + windows: Array<{ label: "12m" | "3y" | "5y"; newFacilities: number; newProjects: number; newCountries: number; newMetros: number; openedMw: number | null; announcedMw: number | null }>; | |
| 153 | + /** cumulative countries / metros entered by year (from opened_on / announced_on / first_seen — labelled per point) */ | |
| 154 | + countriesOverTime: Array<{ year: number; countries: number; metros: number; facilities: number; basis: "opened" | "first_seen" }>; | |
| 51 | 155 | } |
| 52 | 156 | |
| 53 | 157 | export interface OperatorDetail extends OperatorSummary { |
| 158 | + pipeline: PipelineBreakdown; | |
| 159 | + velocity: ExpansionVelocity; | |
| 160 | + topCountries: Array<{ iso2: string; name: string; slug: string; facilityCount: number; knownMw: number | null; share: number }>; | |
| 161 | + topMetros: Array<{ id: string; slug: string; name: string; countryIso2: string; facilityCount: number; knownMw: number | null; share: number }>; | |
| 162 | + aiFacilities: FacilitySummary[]; | |
| 163 | + cloudRegions: CloudRegionSummary[]; | |
| 164 | + /** acquisitions, financing, partnerships, executive changes — never mixed with physical projects */ | |
| 165 | + corporateEvents: EventDTO[]; | |
| 166 | + claims: ClaimDTO[]; | |
| 167 | + dataQuality: DataQualitySummary; | |
| 54 | 168 | aliases: string[]; |
| 55 | 169 | description?: string | null; |
| 56 | 170 | parent?: { id: string; slug: string; name: string } | null; |
@@ -93,9 +207,43 @@ export interface FacilitySummary { | ||
| 93 | 207 | carriersCount?: number | null; |
| 94 | 208 | ixpCount?: number | null; |
| 95 | 209 | networksCount?: number | null; |
| 210 | + /** building | facility | campus (containment) */ | |
| 211 | + recordScope?: "building" | "facility" | "campus"; | |
| 212 | + parentFacility?: { id: string; slug: string; name: string } | null; | |
| 213 | + aiEvidence?: AiEvidence; | |
| 214 | + /** what the displayed MW figure means and describes */ | |
| 215 | + capacityScope?: ClaimScope | null; | |
| 216 | + capacitySemantics?: string | null; | |
| 217 | + utilityCapacityMw?: number | null; | |
| 218 | + gridConnectionMw?: number | null; | |
| 219 | + ultimateCampusMw?: number | null; | |
| 220 | + sourceCount?: number; | |
| 221 | +} | |
| 222 | + | |
| 223 | +export interface DataQualitySummary { | |
| 224 | + sourceCount: number; | |
| 225 | + primarySourceCount: number; | |
| 226 | + lastVerified: string | null; | |
| 227 | + completeness: number; | |
| 228 | + fieldsWithProvenance: number; | |
| 229 | + claimsTotal: number; | |
| 230 | + claimsCurrent: number; | |
| 231 | + claimsUnscoped: number; | |
| 232 | + claimsInReview: number; | |
| 233 | + openFlags: Array<{ code: string; severity: "info" | "warn" | "critical"; message: string; field: string | null }>; | |
| 234 | + pendingDuplicate: boolean; | |
| 96 | 235 | } |
| 97 | 236 | |
| 98 | 237 | export interface FacilityDetail extends FacilitySummary { |
| 238 | + buildings: FacilitySummary[]; | |
| 239 | + developer: { id: string; slug: string; name: string } | null; | |
| 240 | + landowner: { id: string; slug: string; name: string } | null; | |
| 241 | + tenants: Array<{ id: string; slug: string; name: string; role: string }>; | |
| 242 | + claims: ClaimDTO[]; | |
| 243 | + capacityHistory: HistoryPoint[]; | |
| 244 | + nearbyInfrastructure: NearbyInfrastructure | null; | |
| 245 | + powerContext: { utilityCapacityMw: number | null; gridConnectionMw: number | null; powerEvents: EventDTO[]; gridConstraints: GridConstraintDTO[]; countryEnergy: { renewableShare: number | null; electricityTwh: number | null; statsYear: number | null } | null; note: string }; | |
| 246 | + dataQuality: DataQualitySummary; | |
| 99 | 247 | aliases: string[]; |
| 100 | 248 | owner: { id: string; slug: string; name: string } | null; |
| 101 | 249 | campus: { id: string; slug: string; name: string } | null; |
@@ -177,13 +325,50 @@ export interface ProjectSummary { | ||
| 177 | 325 | phaseCount: number | null; |
| 178 | 326 | confidence: ConfidenceLevel; |
| 179 | 327 | lastUpdate: string; |
| 328 | + geoPrecision?: GeoPrecision; | |
| 329 | + projectClass?: ProjectClass | null; | |
| 330 | + evidenceLevel?: "strong" | "weak" | "none" | null; | |
| 331 | + isAi?: boolean; | |
| 332 | + aiEvidence?: AiEvidence; | |
| 333 | + capacityScope?: ClaimScope | null; | |
| 334 | + capacitySemantics?: string | null; | |
| 335 | + investmentScope?: ClaimScope | null; | |
| 336 | + investmentSemantics?: string | null; | |
| 337 | + investmentCurrency?: string | null; | |
| 338 | + investmentOriginal?: number | null; | |
| 339 | + developer?: { id: string; slug: string; name: string } | null; | |
| 340 | + tenant?: { id: string; slug: string; name: string } | null; | |
| 341 | + constructionStartedOn?: string | null; | |
| 342 | + approvedOn?: string | null; | |
| 343 | + permitFiledOn?: string | null; | |
| 344 | + openedOn?: string | null; | |
| 345 | +} | |
| 346 | + | |
| 347 | +/** Lifecycle stage view: one row per stage, dated when known, with the evidence that moved the project there. */ | |
| 348 | +export interface ProjectStageDTO { | |
| 349 | + stage: "rumored" | "proposed" | "announced" | "permitting" | "approved" | "under_construction" | "partially_operational" | "operational" | "delayed" | "cancelled"; | |
| 350 | + date: string | null; | |
| 351 | + reached: boolean; | |
| 352 | + current: boolean; | |
| 353 | + url: string | null; | |
| 354 | + sourceName: string | null; | |
| 355 | + eventId: string | null; | |
| 180 | 356 | } |
| 181 | 357 | |
| 182 | 358 | export interface ProjectDetail extends ProjectSummary { |
| 183 | 359 | description: string | null; |
| 184 | 360 | sourceUrl: string | null; |
| 361 | + campus: { id: string; slug: string; name: string } | null; | |
| 362 | + stages: ProjectStageDTO[]; | |
| 363 | + /** days between reached stages (announcement → permitting → approval → construction → opening), null when unknown */ | |
| 364 | + velocityDays: { announcedToPermitting: number | null; permittingToApproval: number | null; approvalToConstruction: number | null; constructionToOpening: number | null; announcedToConstruction: number | null }; | |
| 185 | 365 | timeline: Array<{ date: string; type: string; description: string; url: string | null; sourceName: string | null }>; |
| 186 | 366 | statusHistory: Array<{ date: string; from: FacilityStatus | null; to: FacilityStatus; url: string | null }>; |
| 367 | + claims: ClaimDTO[]; | |
| 368 | + history: HistoryPoint[]; | |
| 369 | + nearbyInfrastructure: NearbyInfrastructure | null; | |
| 370 | + relatedEvents: EventDTO[]; | |
| 371 | + dataQuality: DataQualitySummary; | |
| 187 | 372 | provenance: ProvenanceDTO[]; |
| 188 | 373 | events: EventDTO[]; |
| 189 | 374 | sourceHistory: SourceHistoryItem[]; |
@@ -210,6 +395,52 @@ export interface EventDTO { | ||
| 210 | 395 | reviewStatus: ReviewStatus; |
| 211 | 396 | countryIso2: string | null; |
| 212 | 397 | operator: { id: string; slug: string; name: string } | null; |
| 398 | + metro?: { id: string; slug: string; name: string } | null; | |
| 399 | + project?: { id: string; slug: string; name: string } | null; | |
| 400 | + isAi?: boolean; | |
| 401 | + significanceBand?: Significance; | |
| 402 | + /** documents describing the same announcement (deduplicated feed): count + the other sources */ | |
| 403 | + evidenceCount?: number; | |
| 404 | + clusterId?: string | null; | |
| 405 | + otherSources?: Array<{ sourceName: string; sourceKind: SourceKind; url: string }>; | |
| 406 | + documentId?: string | null; | |
| 407 | +} | |
| 408 | + | |
| 409 | +export interface EventFilters { | |
| 410 | + type?: string; // comma list of EventType | |
| 411 | + country?: string; | |
| 412 | + metro?: string; | |
| 413 | + operator?: string; | |
| 414 | + project?: string; | |
| 415 | + entity_type?: string; | |
| 416 | + entity_id?: string; | |
| 417 | + min_significance?: number; | |
| 418 | + significance?: "major" | "medium" | "minor"; | |
| 419 | + confidence?: string; | |
| 420 | + source_kind?: string; | |
| 421 | + ai?: boolean; | |
| 422 | + since?: string; | |
| 423 | + until?: string; | |
| 424 | + q?: string; | |
| 425 | + /** collapse events sharing a cluster id into one row (default true) */ | |
| 426 | + dedupe?: boolean; | |
| 427 | + page?: number; | |
| 428 | + per_page?: number; | |
| 429 | +} | |
| 430 | + | |
| 431 | +export interface GridConstraintDTO { | |
| 432 | + id: string; | |
| 433 | + kind: "moratorium" | "grid_delay" | "capacity_restriction" | "load_cap" | "new_transmission" | "new_substation" | "regulation" | "large_load_queue" | string; | |
| 434 | + title: string; | |
| 435 | + summary: string | null; | |
| 436 | + effectiveDate: string | null; | |
| 437 | + url: string; | |
| 438 | + sourceName: string | null; | |
| 439 | + sourceKind: SourceKind | null; | |
| 440 | + metro: { id: string; slug: string; name: string } | null; | |
| 441 | + countryIso2: string | null; | |
| 442 | + confidence: ConfidenceLevel; | |
| 443 | + eventId: string | null; | |
| 213 | 444 | } |
| 214 | 445 | |
| 215 | 446 | export interface CountrySummary { |
@@ -233,12 +464,34 @@ export interface CountrySummary { | ||
| 233 | 464 | hyperscaleCount: number; |
| 234 | 465 | aiCount: number; |
| 235 | 466 | projectCount: number; |
| 467 | + projectPlannedMw?: number | null; | |
| 468 | + projectConstructionMw?: number | null; | |
| 469 | + ixpCount?: number; | |
| 236 | 470 | mwCoverage: number; // share of facilities with a known MW figure, 0..1 |
| 237 | 471 | lat: number | null; |
| 238 | 472 | lng: number | null; |
| 239 | 473 | } |
| 240 | 474 | |
| 475 | +export interface EnergyContext { | |
| 476 | + renewableShare: number | null; | |
| 477 | + electricityTwh: number | null; | |
| 478 | + statsYear: number | null; | |
| 479 | + /** grams CO2/kWh where a reliable public source exists, else null */ | |
| 480 | + gridCarbonIntensity: number | null; | |
| 481 | + sourceName: string | null; | |
| 482 | + sourceUrl: string | null; | |
| 483 | + note: string; | |
| 484 | +} | |
| 485 | + | |
| 241 | 486 | export interface CountryDetail extends CountrySummary { |
| 487 | + ixps: IxpSummary[]; | |
| 488 | + gridConstraints: GridConstraintDTO[]; | |
| 489 | + energy: EnergyContext; | |
| 490 | + aiFacilities: FacilitySummary[]; | |
| 491 | + aiProjects: ProjectSummary[]; | |
| 492 | + pipeline: PipelineBreakdown; | |
| 493 | + coverage: CoverageRow; | |
| 494 | + claims: ClaimDTO[]; | |
| 242 | 495 | topOperators: OperatorSummary[]; |
| 243 | 496 | metros: MetroSummary[]; |
| 244 | 497 | cloudRegions: CloudRegionSummary[]; |
@@ -274,9 +527,42 @@ export interface MetroSummary { | ||
| 274 | 527 | cloudRegionCount: number; |
| 275 | 528 | ixpCount: number; |
| 276 | 529 | projectCount: number; |
| 530 | + projectPlannedMw?: number | null; | |
| 531 | + aiCount?: number; | |
| 532 | + mwCoverage?: number | null; | |
| 533 | +} | |
| 534 | + | |
| 535 | +/** Operator concentration, computed on facility counts and (separately, when coverage permits) on known MW. */ | |
| 536 | +export interface MarketConcentration { | |
| 537 | + operatorCount: number; | |
| 538 | + facilities: { hhi: number; top3Share: number; top5Share: number; top: Array<{ id: string; slug: string; name: string; count: number; share: number }> }; | |
| 539 | + knownMw: { hhi: number; top3Share: number; top5Share: number; coverage: number; top: Array<{ id: string; slug: string; name: string; mw: number; share: number }> } | null; | |
| 540 | + note: string; | |
| 541 | +} | |
| 542 | + | |
| 543 | +/** Transparent momentum components — shown separately, never collapsed into a score. */ | |
| 544 | +export interface MarketMomentum { | |
| 545 | + window: "12m"; | |
| 546 | + projectsAnnounced: number; | |
| 547 | + projectsEnteredConstruction: number; | |
| 548 | + constructionMw: number | null; | |
| 549 | + announcedMw: number | null; | |
| 550 | + newEntrants: Array<{ id: string; slug: string; name: string }>; | |
| 551 | + facilitiesOpened: number; | |
| 552 | + cloudRegionsAdded: number; | |
| 553 | + gridEvents: number; | |
| 554 | + eventsTotal: number; | |
| 277 | 555 | } |
| 278 | 556 | |
| 279 | 557 | export interface MetroDetail extends MetroSummary { |
| 558 | + concentration: MarketConcentration; | |
| 559 | + momentum: MarketMomentum; | |
| 560 | + pipeline: PipelineBreakdown; | |
| 561 | + gridConstraints: GridConstraintDTO[]; | |
| 562 | + aiFacilities: FacilitySummary[]; | |
| 563 | + openingTimeline: Array<{ year: number; opened: number; openedMw: number | null }>; | |
| 564 | + coverage: CoverageRow; | |
| 565 | + claims: ClaimDTO[]; | |
| 280 | 566 | aliases: string[]; |
| 281 | 567 | description: string | null; |
| 282 | 568 | operators: OperatorSummary[]; |
@@ -301,8 +587,38 @@ export interface IxpSummary { | ||
| 301 | 587 | website: string | null; |
| 302 | 588 | networkCount: number | null; |
| 303 | 589 | facilityCount: number; |
| 590 | + metro?: { id: string; slug: string; name: string } | null; | |
| 591 | + lat?: number | null; | |
| 592 | + lng?: number | null; | |
| 593 | +} | |
| 594 | + | |
| 595 | +export interface IxpDetail extends IxpSummary { | |
| 596 | + facilities: FacilitySummary[]; | |
| 597 | + operators: Array<{ id: string; slug: string; name: string; facilityCount: number }>; | |
| 598 | + nearbyFacilities: Array<FacilitySummary & Distance>; | |
| 599 | + trafficNote: string | null; | |
| 600 | + externalIds: Record<string, string | number>; | |
| 601 | + sourceHistory: SourceHistoryItem[]; | |
| 602 | +} | |
| 603 | + | |
| 604 | +export interface CloudRegionDetail extends CloudRegionSummary { | |
| 605 | + metro: { id: string; slug: string; name: string } | null; | |
| 606 | + /** facilities publicly verified to host the region — usually empty; otherwise "region associated with market" */ | |
| 607 | + hostFacilities: FacilitySummary[]; | |
| 608 | + marketFacilities: FacilitySummary[]; | |
| 609 | + siblingRegions: CloudRegionSummary[]; | |
| 610 | + announcedOn: string | null; | |
| 611 | + events: EventDTO[]; | |
| 612 | + provenance: ProvenanceDTO[]; | |
| 613 | + externalIds: Record<string, string | number>; | |
| 614 | + note: string; | |
| 304 | 615 | } |
| 305 | 616 | |
| 617 | +export const MAP_LAYERS = ["facilities", "capacity", "projects", "ai", "cloud", "ixps", "connectivity", "power", "pipeline"] as const; | |
| 618 | +export type MapLayer = (typeof MAP_LAYERS)[number]; | |
| 619 | +export const DENSITY_VIEWS = ["facilities", "known_mw", "pipeline_mw", "ai", "operators", "cloud", "ixps"] as const; | |
| 620 | +export type DensityView = (typeof DENSITY_VIEWS)[number]; | |
| 621 | + | |
| 306 | 622 | export interface MapPoint { |
| 307 | 623 | id: string; |
| 308 | 624 | slug: string; |
@@ -317,6 +633,12 @@ export interface MapPoint { | ||
| 317 | 633 | ai?: 1; |
| 318 | 634 | hs?: 1; |
| 319 | 635 | c?: string | null; // country iso2 |
| 636 | + /** point kind for multi-layer maps (default facility) */ | |
| 637 | + k?: "facility" | "project" | "ixp" | "cloud_region" | "landing_station" | "substation" | "power_plant"; | |
| 638 | + /** opening year (time machine) */ | |
| 639 | + y?: number | null; | |
| 640 | + /** record scope: campus rows carry their buildings */ | |
| 641 | + rs?: "building" | "facility" | "campus"; | |
| 320 | 642 | } |
| 321 | 643 | |
| 322 | 644 | export interface MapCluster { |
@@ -329,16 +651,40 @@ export interface MapCluster { | ||
| 329 | 651 | label?: string | null; // metro/country label at low zoom |
| 330 | 652 | } |
| 331 | 653 | |
| 654 | +export interface DensityCell { lat: number; lng: number; w: number; n: number } | |
| 655 | + | |
| 332 | 656 | export interface MapResponse { |
| 333 | 657 | zoom: number; |
| 334 | − mode: "clusters" | "points"; | |
| 658 | + mode: "clusters" | "points" | "density"; | |
| 659 | + layer?: MapLayer; | |
| 335 | 660 | clusters?: MapCluster[]; |
| 336 | 661 | points?: MapPoint[]; |
| 662 | + /** overlay points for non-facility layers (projects, IXPs, cloud regions…) */ | |
| 663 | + overlay?: MapPoint[]; | |
| 664 | + density?: { view: DensityView; cells: DensityCell[]; max: number }; | |
| 337 | 665 | total: number; |
| 666 | + /** true when the point count exceeded the cap and clusters were returned instead */ | |
| 667 | + degraded?: boolean; | |
| 668 | + /** time machine: the year filter applied (facilities known open by that year) and coverage of opening dates */ | |
| 669 | + year?: number | null; | |
| 670 | + yearCoverage?: number | null; | |
| 671 | +} | |
| 672 | + | |
| 673 | +export interface MapFilters extends FacilityFilters { | |
| 674 | + layer?: MapLayer; | |
| 675 | + density?: DensityView; | |
| 676 | + year?: number; | |
| 677 | + project_status?: string; | |
| 678 | + expected_from?: number; | |
| 679 | + expected_to?: number; | |
| 680 | + location_precision?: string; | |
| 681 | + ai_evidence?: string; | |
| 682 | + zoom?: number; | |
| 683 | + bbox?: string; | |
| 338 | 684 | } |
| 339 | 685 | |
| 340 | 686 | export interface SearchHit { |
| 341 | − type: "facility" | "operator" | "country" | "metro" | "city" | "cloud_region" | "project" | "ixp"; | |
| 687 | + type: "facility" | "operator" | "country" | "metro" | "city" | "cloud_region" | "project" | "ixp" | "action"; | |
| 342 | 688 | id: string; |
| 343 | 689 | slug: string; |
| 344 | 690 | title: string; |
@@ -350,7 +696,7 @@ export interface SearchHit { | ||
| 350 | 696 | |
| 351 | 697 | export interface SearchResponse { |
| 352 | 698 | query: string; |
| 353 | − interpreted?: { countryIso2?: string; operator?: string; status?: FacilityStatus; minMw?: number; facilityType?: FacilityType; text?: string }; | |
| 699 | + interpreted?: { countryIso2?: string; operator?: string; status?: FacilityStatus; minMw?: number; maxMw?: number; facilityType?: FacilityType; text?: string; ai?: boolean; entity?: "facility" | "project"; year?: number; yearOp?: "before" | "after" | "in"; metro?: string }; | |
| 354 | 700 | hits: SearchHit[]; |
| 355 | 701 | facilities?: { total: number; items: FacilitySummary[] }; |
| 356 | 702 | } |
@@ -416,6 +762,217 @@ export interface Dashboard { | ||
| 416 | 762 | latestEvents: EventDTO[]; |
| 417 | 763 | newProjects: ProjectSummary[]; |
| 418 | 764 | recentlyVerified: FacilitySummary[]; |
| 765 | + /** what changed today / this week — deterministic counts (see Pulse) */ | |
| 766 | + pulse: Pulse; | |
| 767 | + fastestGrowingMetros: Array<{ id: string; slug: string; name: string; countryIso2: string; momentum: MarketMomentum }>; | |
| 768 | + majorProjects: ProjectSummary[]; | |
| 769 | + operatorExpansion: Array<{ operator: { id: string; slug: string; name: string }; newCountries: string[]; newMetros: string[]; projects12m: number }>; | |
| 770 | + powerEvents: EventDTO[]; | |
| 771 | + gridConstraints: GridConstraintDTO[]; | |
| 772 | + coverage: CoverageReport["global"]; | |
| 773 | + ingestion: { lastRunAt: string | null; runs24h: number; documents24h: number; events24h: number; connectorsHealthy: number; connectorsTotal: number }; | |
| 774 | +} | |
| 775 | + | |
| 776 | +/** Global infrastructure pulse — measurable changes over a window, computed deterministically from events / entities. */ | |
| 777 | +export interface Pulse { | |
| 778 | + window: "24h" | "7d" | "30d"; | |
| 779 | + since: string; | |
| 780 | + newProjects: number; | |
| 781 | + projectsEnteredConstruction: number; | |
| 782 | + facilitiesOpened: number; | |
| 783 | + newlyAnnouncedMw: number | null; | |
| 784 | + constructionStartedMw: number | null; | |
| 785 | + operatorsNewMarkets: Array<{ operator: { id: string; slug: string; name: string }; market: { id: string; slug: string; name: string } | null; countryIso2: string | null }>; | |
| 786 | + cloudRegionsAnnounced: number; | |
| 787 | + powerAgreements: number; | |
| 788 | + gridConstraintEvents: number; | |
| 789 | + acquisitions: number; | |
| 790 | + financingEvents: number; | |
| 791 | + newFacilitiesIndexed: number; | |
| 792 | + capacityChanges: number; | |
| 793 | + eventsTotal: number; | |
| 794 | + majorEvents: EventDTO[]; | |
| 795 | + byCountry: Array<{ iso2: string; name: string; slug: string; events: number; newProjects: number }>; | |
| 796 | + byOperator: Array<{ id: string; slug: string; name: string; events: number; newProjects: number }>; | |
| 797 | +} | |
| 798 | + | |
| 799 | +export interface CoverageRow { | |
| 800 | + key: string; | |
| 801 | + name: string; | |
| 802 | + slug: string; | |
| 803 | + facilities: number; | |
| 804 | + /** share of facilities with a known MW figure (containment-aware) */ | |
| 805 | + capacityCoverage: number; | |
| 806 | + operatorCoverage: number; | |
| 807 | + preciseLocationCoverage: number; // exact / parcel / street | |
| 808 | + anyLocationCoverage: number; | |
| 809 | + statusCoverage: number; | |
| 810 | + openingDateCoverage: number; | |
| 811 | + multiSourceCoverage: number; | |
| 812 | + projects: number; | |
| 813 | + projectsWithLocation: number; | |
| 814 | + connectivityCoverage: number; // facilities with carriers / IXPs known | |
| 815 | + primarySourceShare: number; | |
| 816 | +} | |
| 817 | + | |
| 818 | +export interface CoverageReport { | |
| 819 | + global: CoverageRow; | |
| 820 | + countries: CoverageRow[]; | |
| 821 | + metros: CoverageRow[]; | |
| 822 | + operators: CoverageRow[]; | |
| 823 | + fields: Array<{ field: string; label: string; coverage: number; count: number }>; | |
| 824 | + bySourceKind: Array<{ kind: SourceKind; facilities: number; fields: number }>; | |
| 825 | + generatedAt: string; | |
| 826 | +} | |
| 827 | + | |
| 828 | +export interface SourceCoverage { | |
| 829 | + id: string; | |
| 830 | + name: string; | |
| 831 | + kind: SourceKind; | |
| 832 | + recordsContributed: number; | |
| 833 | + fieldsContributed: number; | |
| 834 | + uniqueRecords: number; | |
| 835 | + lastSuccessfulCrawl: string | null; | |
| 836 | + freshnessDays: number | null; | |
| 837 | + authority: AuthorityTier; | |
| 838 | + failureRate: number | null; | |
| 839 | + license: string | null; | |
| 840 | + redistribution: string | null; | |
| 841 | +} | |
| 842 | + | |
| 843 | +export interface OperatorComparison { | |
| 844 | + operators: Array<OperatorSummary & { pipeline: PipelineBreakdown; velocity: ExpansionVelocity; aiCount: number; newMarkets12m: Array<{ id: string; slug: string; name: string }>; recentProjects: ProjectSummary[]; countriesList: string[]; coverage: number }>; | |
| 845 | + generatedAt: string; | |
| 846 | +} | |
| 847 | + | |
| 848 | +export interface AiIndex { | |
| 849 | + stats: { facilities: number; confirmed: number; likely: number; associated: number; projects: number; plannedMw: number | null; constructionMw: number | null; countries: number; operators: number; mwCoverage: number }; | |
| 850 | + topOperators: Array<{ id: string; slug: string; name: string; facilities: number; projects: number; plannedMw: number | null }>; | |
| 851 | + topMetros: Array<{ id: string; slug: string; name: string; countryIso2: string; facilities: number; projects: number; plannedMw: number | null }>; | |
| 852 | + topCountries: Array<{ iso2: string; slug: string; name: string; facilities: number; projects: number; plannedMw: number | null }>; | |
| 853 | + pipelineByStage: Array<{ status: FacilityStatus; count: number; mw: number | null }>; | |
| 854 | + recentAnnouncements: EventDTO[]; | |
| 855 | + recentProjects: ProjectSummary[]; | |
| 856 | + facilities: FacilitySummary[]; | |
| 857 | + evidenceNote: string; | |
| 858 | +} | |
| 859 | + | |
| 860 | +export interface PowerOverview { | |
| 861 | + gridConstraints: GridConstraintDTO[]; | |
| 862 | + powerEvents: EventDTO[]; | |
| 863 | + largeLoads: Array<{ facility: { id: string; slug: string; name: string } | null; project: { id: string; slug: string; name: string } | null; kind: "utility_capacity" | "grid_connection" | "planned_load"; mw: number; countryIso2: string | null; metro: { id: string; slug: string; name: string } | null; sourceName: string | null; url: string | null }>; | |
| 864 | + countryEnergy: Array<{ iso2: string; name: string; slug: string; energy: EnergyContext; knownMw: number | null; facilities: number }>; | |
| 865 | + utilities: Array<{ id: string; slug: string; name: string; facilityCount: number; kind: string | null }>; | |
| 866 | + note: string; | |
| 867 | +} | |
| 868 | + | |
| 869 | +export interface ConnectivityOverview { | |
| 870 | + ixps: Array<IxpSummary & { operatorCount: number }>; | |
| 871 | + cloudRegions: CloudRegionSummary[]; | |
| 872 | + carrierHotels: FacilitySummary[]; | |
| 873 | + landingStations: Array<{ id: string; name: string; lat: number; lng: number; countryIso2: string | null; sourceName: string }>; | |
| 874 | + byMetro: Array<{ id: string; slug: string; name: string; countryIso2: string; ixps: number; cloudRegions: number; facilitiesWithCarriers: number; facilities: number }>; | |
| 875 | + note: string; | |
| 876 | +} | |
| 877 | + | |
| 878 | +export interface TimeMachineFrame { year: number; facilities: number; knownMw: number | null; announced: number; construction: number; opened: number } | |
| 879 | +export interface TimeMachine { frames: TimeMachineFrame[]; earliestReliableYear: number; openingDateCoverage: number; note: string } | |
| 880 | + | |
| 881 | +/** /explore structured query. Every key is optional; unknown keys are rejected. */ | |
| 882 | +export interface ExploreQuery { | |
| 883 | + entity?: "facilities" | "projects"; | |
| 884 | + status?: string; | |
| 885 | + project_status?: string; | |
| 886 | + country?: string; | |
| 887 | + metro?: string; | |
| 888 | + operator?: string; | |
| 889 | + type?: string; | |
| 890 | + min_mw?: number; | |
| 891 | + max_mw?: number; | |
| 892 | + ai?: "confirmed" | "likely" | "associated" | "any"; | |
| 893 | + hyperscale?: boolean; | |
| 894 | + expected_before?: number; | |
| 895 | + expected_after?: number; | |
| 896 | + opened_from?: number; | |
| 897 | + opened_to?: number; | |
| 898 | + announced_since?: string; | |
| 899 | + confidence?: string; | |
| 900 | + location_precision?: string; | |
| 901 | + has_mw?: boolean; | |
| 902 | + project_class?: string; | |
| 903 | + q?: string; | |
| 904 | + sort?: string; | |
| 905 | + order?: "asc" | "desc"; | |
| 906 | + page?: number; | |
| 907 | + per_page?: number; | |
| 908 | + view?: "table" | "map" | "charts"; | |
| 909 | +} | |
| 910 | + | |
| 911 | +export interface ExploreResponse { | |
| 912 | + query: ExploreQuery; | |
| 913 | + total: number; | |
| 914 | + items: Array<FacilitySummary | ProjectSummary>; | |
| 915 | + facets: { status: Array<{ key: string; count: number }>; country: Array<{ key: string; name: string; count: number }>; operator: Array<{ key: string; name: string; count: number }>; type?: Array<{ key: string; count: number }>; ai: Array<{ key: string; count: number }> }; | |
| 916 | + charts: { byStatus: Array<{ key: string; count: number; mw: number | null }>; byCountry: Array<{ key: string; name: string; count: number; mw: number | null }>; byYear: Array<{ year: number; count: number; mw: number | null }> }; | |
| 917 | + map: { points: MapPoint[]; total: number; degraded: boolean }; | |
| 918 | + mwCoverage: number; | |
| 919 | +} | |
| 920 | + | |
| 921 | +export interface DownloadDataset { | |
| 922 | + key: "facilities" | "projects" | "operators" | "events" | "cloud-regions" | "ixps" | "countries" | "markets"; | |
| 923 | + label: string; | |
| 924 | + description: string; | |
| 925 | + formats: Array<"csv" | "json" | "geojson">; | |
| 926 | + filters: string[]; | |
| 927 | + rows: number; | |
| 928 | + license: string; | |
| 929 | + attribution: string[]; | |
| 930 | + /** sources excluded from redistribution (their rows are omitted or their fields blanked) */ | |
| 931 | + excludedSources: Array<{ id: string; name: string; reason: string }>; | |
| 932 | +} | |
| 933 | + | |
| 934 | +export interface WatchlistItem { id: string; entityType: "operator" | "metro" | "country" | "project" | "facility"; entityId: string; slug: string; name: string; createdAt: string } | |
| 935 | + | |
| 936 | +export interface ApiEndpointDoc { method: "GET" | "POST" | "DELETE"; path: string; summary: string; group: string; params: Array<{ name: string; in: "query" | "path"; type: string; description: string; example?: string }>; example: { curl: string; js: string; python: string }; responseSchema: string } | |
| 937 | + | |
| 938 | +export interface QualityFlagDTO { | |
| 939 | + id: string; | |
| 940 | + entityType: string; | |
| 941 | + entityId: string; | |
| 942 | + entity: { slug: string; name: string } | null; | |
| 943 | + claimId: string | null; | |
| 944 | + code: string; | |
| 945 | + severity: "info" | "warn" | "critical"; | |
| 946 | + field: string | null; | |
| 947 | + message: string; | |
| 948 | + details: Record<string, unknown> | null; | |
| 949 | + priority: number; | |
| 950 | + status: "open" | "resolved" | "dismissed"; | |
| 951 | + resolution: string | null; | |
| 952 | + resolvedBy: string | null; | |
| 953 | + resolvedAt: string | null; | |
| 954 | + runId: string | null; | |
| 955 | + createdAt: string; | |
| 956 | + updatedAt: string; | |
| 957 | +} | |
| 958 | + | |
| 959 | +export interface QualityOverview { | |
| 960 | + openFlags: number; | |
| 961 | + bySeverity: Record<string, number>; | |
| 962 | + byCode: Array<{ code: string; count: number; label: string }>; | |
| 963 | + possibleDuplicates: number; | |
| 964 | + suspiciousMw: number; | |
| 965 | + suspiciousInvestment: number; | |
| 966 | + projectFalsePositiveCandidates: number; | |
| 967 | + unscopedClaims: number; | |
| 968 | + claimsInReview: number; | |
| 969 | + unverifiedLargeProjects: number; | |
| 970 | + facilitiesMissingCountry: number; | |
| 971 | + projectsWithoutFacility: number; | |
| 972 | + orphanOperators: number; | |
| 973 | + largest: { operationalFacilityMw: Array<{ id: string; slug: string; name: string; mw: number; scope: string | null; semantics: string | null }>; constructionFacilityMw: Array<{ id: string; slug: string; name: string; mw: number }>; projectMw: Array<{ id: string; slug: string; name: string; mw: number; scope: string | null }>; investment: Array<{ id: string; slug: string; name: string; investmentUsd: number; scope: string | null }>; operatorPipelineMw: Array<{ id: string; slug: string; name: string; mw: number }>; metroPipelineMw: Array<{ id: string; slug: string; name: string; mw: number }> }; | |
| 974 | + reviewQueue: QualityFlagDTO[]; | |
| 975 | + alerts: Array<{ id: string; level: string; component: string; message: string; createdAt: string }>; | |
| 419 | 976 | } |
| 420 | 977 | |
| 421 | 978 | export interface ConnectorHealthDTO { |
@@ -425,9 +982,11 @@ export interface ConnectorHealthDTO { | ||
| 425 | 982 | kind: SourceKind; |
| 426 | 983 | mode: string; |
| 427 | 984 | enabled: boolean; |
| 428 | − health: "ok" | "degraded" | "failing" | "paused" | "never_run"; | |
| 985 | + health: "ok" | "degraded" | "failing" | "paused" | "never_run" | "blocked" | "schema_change" | "no_new_content" | "quarantine"; | |
| 429 | 986 | parserVersion: string; |
| 430 | 987 | lastRunAt: string | null; |
| 988 | + lastSuccessAt?: string | null; | |
| 989 | + lastFailureAt?: string | null; | |
| 431 | 990 | nextRunAt: string | null; |
| 432 | 991 | lastStatus: string | null; |
| 433 | 992 | discovered: number; |
@@ -439,6 +998,23 @@ export interface ConnectorHealthDTO { | ||
| 439 | 998 | avgResponseMs: number | null; |
| 440 | 999 | cost: { direct: number; firecrawl: number; scrapfly: number; credits: number }; |
| 441 | 1000 | schedule: Record<string, string>; |
| 1001 | + quarantine?: boolean; | |
| 1002 | + consecutiveFailures?: number; | |
| 1003 | + blockedSince?: string | null; | |
| 1004 | + urlsDiscovered?: number; | |
| 1005 | + urlsFetched?: number; | |
| 1006 | + newDocs?: number; | |
| 1007 | + changedDocs?: number; | |
| 1008 | + recordsCreated?: number; | |
| 1009 | + recordsModified?: number; | |
| 1010 | + rejectedClaims?: number; | |
| 1011 | + httpErrors?: number; | |
| 1012 | + antiBotEscalations?: number; | |
| 1013 | + scrapflyRequests?: number; | |
| 1014 | + firecrawlRequests?: number; | |
| 1015 | + priorityScore?: number | null; | |
| 1016 | + license?: string | null; | |
| 1017 | + redistribution?: string | null; | |
| 442 | 1018 | } |
| 443 | 1019 | |
| 444 | 1020 | export interface FacilityFilters { |
added
packages/core/src/claims.test.ts
+138 −0
@@ -0,0 +1,138 @@ | ||
| 1 | +import { describe, expect, it } from "vitest"; | |
| 2 | +import { authorityTier, capacitySanity, classifyAiEvidence, classifyCapacitySemantics, classifyInvestmentSemantics, classifyProjectEvent, classifyScope, findEvidence, investmentSanity, moneyMwCollision, projectTransition, sentenceSpans } from "./claims.js"; | |
| 3 | + | |
| 4 | +describe("classifyScope", () => { | |
| 5 | + it("detects portfolio / company / country figures", () => { | |
| 6 | + expect(classifyScope("bringing its total capacity to over 410MW in Japan").scope).toBe("portfolio"); | |
| 7 | + expect(classifyScope("STACK now operates more than 1.2 GW across its global portfolio").scope).toBe("portfolio"); | |
| 8 | + expect(classifyScope("Microsoft plans to spend $80 billion in capex in fiscal 2025").scope).toBe("company"); | |
| 9 | + expect(classifyScope("the national programme will add 5 GW across the country").scope).toBe("country"); | |
| 10 | + expect(classifyScope("Hyperscale Data Center Market to Surpass USD 624.2 Billion by 2035, CAGR of 9%").scope).toBe("company"); | |
| 11 | + }); | |
| 12 | + it("detects campus / building / facility figures", () => { | |
| 13 | + expect(classifyScope("the campus will deliver 300 MW at full build-out").scope).toBe("campus"); | |
| 14 | + expect(classifyScope("the first building offers 24 MW of critical IT load").scope).toBe("building"); | |
| 15 | + expect(classifyScope("the facility will provide 36 MW").scope).toBe("facility"); | |
| 16 | + }); | |
| 17 | + it("falls back to the subject hint only without any signal", () => { | |
| 18 | + expect(classifyScope("36 MW", "facility").scope).toBe("facility"); | |
| 19 | + expect(classifyScope("36 MW").scope).toBe("unknown"); | |
| 20 | + }); | |
| 21 | +}); | |
| 22 | + | |
| 23 | +describe("classifyCapacitySemantics", () => { | |
| 24 | + it("separates IT load, grid, utility, planned, ultimate and phase figures", () => { | |
| 25 | + expect(classifyCapacitySemantics("0.779MW Critical IT Load").predicate).toBe("critical_power_mw"); | |
| 26 | + expect(classifyCapacitySemantics("offers 24 MW of IT capacity").predicate).toBe("it_capacity_mw"); | |
| 27 | + expect(classifyCapacitySemantics("secured a 300 MW grid connection agreement with Dominion").predicate).toBe("grid_connection_mw"); | |
| 28 | + expect(classifyCapacitySemantics("dual utility feeds totaling 60 MW of utility power").predicate).toBe("utility_capacity_mw"); | |
| 29 | + expect(classifyCapacitySemantics("the campus will scale up to 1 GW at full build-out").predicate).toBe("ultimate_campus_mw"); | |
| 30 | + expect(classifyCapacitySemantics("the first phase will deliver 48 MW").predicate).toBe("phase_mw"); | |
| 31 | + expect(classifyCapacitySemantics("the planned 200 MW facility").predicate).toBe("planned_power_mw"); | |
| 32 | + expect(classifyCapacitySemantics("currently operating 12 MW").predicate).toBe("current_power_mw"); | |
| 33 | + expect(classifyCapacitySemantics("36 MW").predicate).toBeNull(); | |
| 34 | + }); | |
| 35 | +}); | |
| 36 | + | |
| 37 | +describe("classifyInvestmentSemantics", () => { | |
| 38 | + it("keeps deal values and company capex away from project investment", () => { | |
| 39 | + expect(classifyInvestmentSemantics("Microsoft will invest $80 billion through 2030 across its global fleet").predicate).toBe("multi_year_capex_usd"); | |
| 40 | + expect(classifyInvestmentSemantics("acquires the portfolio for $2.3 billion").predicate).toBe("deal_value_usd"); | |
| 41 | + expect(classifyInvestmentSemantics("the $1.2 billion campus in Abilene").predicate).toBe("campus_investment_usd"); | |
| 42 | + expect(classifyInvestmentSemantics("a $500 million investment in the new facility", "facility").predicate).toBe("project_investment_usd"); | |
| 43 | + }); | |
| 44 | +}); | |
| 45 | + | |
| 46 | +describe("sanity engines", () => { | |
| 47 | + it("blocks unscoped and market-statistic MW", () => { | |
| 48 | + const flags = capacitySanity({ value: 12_000, predicate: null, scope: "company", recordScope: "project" }); | |
| 49 | + expect(flags.some((f) => f.code === "scope_company" && f.blocks)).toBe(true); | |
| 50 | + expect(capacitySanity({ value: 25_000, predicate: "planned_power_mw", scope: "campus", recordScope: "project" }).some((f) => f.code === "mw_market_statistic")).toBe(true); | |
| 51 | + }); | |
| 52 | + it("flags single facilities above 1 GW, 5x changes and money collisions", () => { | |
| 53 | + expect(capacitySanity({ value: 1_500, predicate: "it_capacity_mw", scope: "facility", recordScope: "facility" }).map((f) => f.code)).toContain("mw_single_site_gt_1000"); | |
| 54 | + expect(capacitySanity({ value: 1_500, predicate: "ultimate_campus_mw", scope: "campus", recordScope: "campus", campusDesignation: true }).map((f) => f.code)).not.toContain("mw_single_site_gt_1000"); | |
| 55 | + expect(capacitySanity({ value: 779, predicate: "it_capacity_mw", scope: "facility", recordScope: "facility", previous: 0.779 }).map((f) => f.code)).toContain("mw_change_5x"); | |
| 56 | + expect(moneyMwCollision("a $300 million, 300 MW campus")).toBe(true); | |
| 57 | + expect(moneyMwCollision("a $300 million, 48 MW campus")).toBe(false); | |
| 58 | + }); | |
| 59 | + it("flags implausible investments", () => { | |
| 60 | + expect(investmentSanity({ value: 175e9, scope: "facility", predicate: "project_investment_usd", recordScope: "project" }).map((f) => f.code)).toContain("inv_single_site_gt_50b"); | |
| 61 | + expect(investmentSanity({ value: 624e9, scope: "company", predicate: "multi_year_capex_usd", recordScope: "project" }).some((f) => f.blocks)).toBe(true); | |
| 62 | + }); | |
| 63 | +}); | |
| 64 | + | |
| 65 | +describe("findEvidence", () => { | |
| 66 | + const text = "STACK Infrastructure today announced the appointment of Matt VanderZanden as CEO. The company operates more than 13 GW across its portfolio. Its newest campus in Lancaster will deliver 500 MW at full build-out."; | |
| 67 | + it("returns the sentence quoting the figure with offsets", () => { | |
| 68 | + const ev = findEvidence(text, 500, "mw"); | |
| 69 | + expect(ev?.text).toMatch(/Lancaster will deliver 500 MW/); | |
| 70 | + expect(text.slice(ev!.start, ev!.end)).toBe(ev!.text); | |
| 71 | + expect(findEvidence(text, 13_000, "mw")?.text).toMatch(/13 GW/); | |
| 72 | + expect(findEvidence(text, 42, "mw")).toBeNull(); | |
| 73 | + }); | |
| 74 | + it("finds money in billions and millions", () => { | |
| 75 | + expect(findEvidence("Meta will spend $10 billion on the Richland Parish site.", 10e9, "usd")?.text).toMatch(/\$10 billion/); | |
| 76 | + expect(findEvidence("A US$450 million facility.", 450e6, "usd")).not.toBeNull(); | |
| 77 | + }); | |
| 78 | + it("splits sentences with offsets", () => { | |
| 79 | + const spans = sentenceSpans("One. Two two. Three?"); | |
| 80 | + expect(spans.map((s) => s.text)).toEqual(["One.", "Two two.", "Three?"]); | |
| 81 | + }); | |
| 82 | +}); | |
| 83 | + | |
| 84 | +describe("classifyProjectEvent", () => { | |
| 85 | + it("vetoes appointments, market research, PPAs, HQs and financing", () => { | |
| 86 | + expect(classifyProjectEvent({ title: "STACK Infrastructure Appoints Matt VanderZanden as Chief Executive Officer, STACK Americas" }).class).toBe("EXECUTIVE_APPOINTMENT"); | |
| 87 | + expect(classifyProjectEvent({ title: "Hyperscale Data Center Market to Surpass USD 624.2 Billion by 2035" }).class).toBe("GENERAL_COMPANY_NEWS"); | |
| 88 | + expect(classifyProjectEvent({ title: "Ormat Technologies Signs 20-Year PPA with Switch for ~13 MW of Carbon-Free Geothermal Capacity to Power Data Centers" }).class).toBe("POWER_AGREEMENT"); | |
| 89 | + expect(classifyProjectEvent({ title: "STACK Secures $1.3B Financing for Development" }).class).toBe("FINANCING"); | |
| 90 | + expect(classifyProjectEvent({ title: "AirTrunk unveils new global headquarters in Sydney" }).mayCreateProject).toBe(false); | |
| 91 | + expect(classifyProjectEvent({ title: "Element Critical Enters Chicago Market Acquiring Two Data Centers in the Suburbs" }).class).toBe("ACQUISITION"); | |
| 92 | + expect(classifyProjectEvent({ title: "Deepening our investment in Richland Parish, Louisiana" }).mayCreateProject).toBe(false); | |
| 93 | + }); | |
| 94 | + it("recognises physical development with strong evidence", () => { | |
| 95 | + const c = classifyProjectEvent({ title: "Vantage breaks ground on 192 MW campus in Frederick, Maryland", hasOperator: true, hasLocation: true, hasExplicitName: false }); | |
| 96 | + expect(c.class).toBe("CONSTRUCTION_START"); | |
| 97 | + expect(c.mayCreateProject).toBe(true); | |
| 98 | + expect(classifyProjectEvent({ title: "Microsoft to build $3.3 billion data center campus in Mount Pleasant, Wisconsin", hasOperator: true, hasLocation: true }).class).toBe("NEW_BUILD"); | |
| 99 | + expect(classifyProjectEvent({ title: "Loudoun County approves rezoning for 300 MW data center campus", hasLocation: true, hasExplicitName: true }).class).toBe("PERMIT"); | |
| 100 | + expect(classifyProjectEvent({ title: "Compass acquires 400-acre site in Red Oak for data center campus", hasOperator: true, hasLocation: true }).class).toBe("LAND_ACQUISITION"); | |
| 101 | + expect(classifyProjectEvent({ title: "STACK Grows Northern Virginia Data Center Campus", hasOperator: true, hasLocation: true }).class).toBe("EXPANSION"); | |
| 102 | + }); | |
| 103 | + it("needs name-or-operator + location + verb to create a project", () => { | |
| 104 | + const weak = classifyProjectEvent({ title: "Massive data center investment announced", hasOperator: false, hasLocation: false }); | |
| 105 | + expect(weak.mayCreateProject).toBe(false); | |
| 106 | + const build = classifyProjectEvent({ title: "Data center campus planned", hasOperator: false, hasLocation: true, hasExplicitName: false }); | |
| 107 | + expect(build.evidence.strength).toBe("weak"); | |
| 108 | + }); | |
| 109 | + it("lets a build headline override a financing keyword", () => { | |
| 110 | + const c = classifyProjectEvent({ title: "Aligned secures $2B financing to build 400 MW campus in Phoenix", hasOperator: true, hasLocation: true }); | |
| 111 | + expect(c.physical).toBe(true); | |
| 112 | + }); | |
| 113 | +}); | |
| 114 | + | |
| 115 | +describe("authority and transitions", () => { | |
| 116 | + it("uses field-level authority tiers", () => { | |
| 117 | + expect(authorityTier({ field: "capacity", sourceKind: "utility" })).toBe("A"); | |
| 118 | + expect(authorityTier({ field: "capacity", sourceKind: "community" })).toBe("E"); | |
| 119 | + expect(authorityTier({ field: "geometry", sourceKind: "community" })).toBe("A"); | |
| 120 | + expect(authorityTier({ field: "interconnection", sourceKind: "registry" })).toBe("A"); | |
| 121 | + expect(authorityTier({ field: "capacity", sourceKind: "operator", isEstimate: true })).toBe("C"); | |
| 122 | + expect(authorityTier({ field: "capacity", sourceKind: "news", method: "llm:extract" })).toBe("E"); | |
| 123 | + }); | |
| 124 | + it("validates project lifecycle transitions", () => { | |
| 125 | + expect(projectTransition("announced", "permitting")).toBe("forward"); | |
| 126 | + expect(projectTransition("under_construction", "delayed")).toBe("side"); | |
| 127 | + expect(projectTransition("delayed", "under_construction")).toBe("resume"); | |
| 128 | + expect(projectTransition("operational", "announced")).toBe("backward"); | |
| 129 | + expect(projectTransition("operational", "cancelled")).toBe("invalid"); | |
| 130 | + expect(projectTransition(null, "announced")).toBe("forward"); | |
| 131 | + }); | |
| 132 | + it("grades AI evidence", () => { | |
| 133 | + expect(classifyAiEvidence("a purpose-built AI campus with NVIDIA GB200 systems").level).toBe("confirmed"); | |
| 134 | + expect(classifyAiEvidence("AI-ready, liquid-cooled halls at 130 kW per rack").level).toBe("likely"); | |
| 135 | + expect(classifyAiEvidence("leased to CoreWeave").level).toBe("associated"); | |
| 136 | + expect(classifyAiEvidence("a colocation facility").level).toBe("unknown"); | |
| 137 | + }); | |
| 138 | +}); | |
added
packages/core/src/claims.ts
+459 −0
@@ -0,0 +1,459 @@ | ||
| 1 | +/** | |
| 2 | + * Claim-first data architecture — shared vocabulary and deterministic classifiers. | |
| 3 | + * | |
| 4 | + * A CLAIM is one figure or value asserted by one document about one subject, with a SCOPE (what the number describes: | |
| 5 | + * a building, a facility, a campus, a metro, a country, a company portfolio…), the supporting sentence, the parser that | |
| 6 | + * produced it and the field-level authority of its source. Reconciliation (apps/worker/src/ingest/claims.ts) decides | |
| 7 | + * which claim becomes the displayed column value; nothing here touches the database. | |
| 8 | + * | |
| 9 | + * Rules of the house: | |
| 10 | + * - a figure whose scope is not the subject's own scope (portfolio / company / country / metro / unknown) is stored | |
| 11 | + * as a claim but NEVER written to the subject's capacity or investment column ("unscoped"); | |
| 12 | + * - capacity semantics are kept apart: IT load ≠ utility power ≠ grid connection ≠ planned ≠ ultimate build-out; | |
| 13 | + * - every numerical claim needs a supporting sentence — no sentence, no claim. | |
| 14 | + */ | |
| 15 | +import type { SourceKind } from "./types.js"; | |
| 16 | + | |
| 17 | +// ─── scopes ──────────────────────────────────────────────────────────────────────────────────────── | |
| 18 | +export const CLAIM_SCOPES = ["building", "facility", "campus", "metro", "country", "portfolio", "company", "unknown"] as const; | |
| 19 | +export type ClaimScope = (typeof CLAIM_SCOPES)[number]; | |
| 20 | +/** Scopes that may populate a facility / campus / project record's own figures. */ | |
| 21 | +export const SITE_SCOPES: ReadonlySet<ClaimScope> = new Set<ClaimScope>(["building", "facility", "campus"]); | |
| 22 | +export function isSiteScope(s: ClaimScope | string | null | undefined): boolean { return !!s && SITE_SCOPES.has(s as ClaimScope); } | |
| 23 | + | |
| 24 | +// ─── predicates ──────────────────────────────────────────────────────────────────────────────────── | |
| 25 | +export const CAPACITY_PREDICATES = ["it_capacity_mw", "critical_power_mw", "utility_capacity_mw", "grid_connection_mw", "current_power_mw", "planned_power_mw", "ultimate_campus_mw", "phase_mw"] as const; | |
| 26 | +export type CapacityPredicate = (typeof CAPACITY_PREDICATES)[number]; | |
| 27 | +export const INVESTMENT_PREDICATES = ["project_investment_usd", "campus_investment_usd", "company_investment_usd", "country_program_usd", "multi_year_capex_usd", "deal_value_usd"] as const; | |
| 28 | +export type InvestmentPredicate = (typeof INVESTMENT_PREDICATES)[number]; | |
| 29 | +export const CLAIM_STATUSES = ["current", "superseded", "rejected", "review", "unscoped"] as const; | |
| 30 | +export type ClaimStatus = (typeof CLAIM_STATUSES)[number]; | |
| 31 | + | |
| 32 | +export const CAPACITY_PREDICATE_LABEL: Record<CapacityPredicate, string> = { | |
| 33 | + it_capacity_mw: "IT capacity", | |
| 34 | + critical_power_mw: "Critical power", | |
| 35 | + utility_capacity_mw: "Utility capacity", | |
| 36 | + grid_connection_mw: "Grid connection", | |
| 37 | + current_power_mw: "Current power", | |
| 38 | + planned_power_mw: "Planned capacity", | |
| 39 | + ultimate_campus_mw: "Ultimate campus build-out", | |
| 40 | + phase_mw: "Phase capacity", | |
| 41 | +}; | |
| 42 | +export const INVESTMENT_PREDICATE_LABEL: Record<InvestmentPredicate, string> = { | |
| 43 | + project_investment_usd: "Project investment", | |
| 44 | + campus_investment_usd: "Campus investment", | |
| 45 | + company_investment_usd: "Company investment", | |
| 46 | + country_program_usd: "Country program", | |
| 47 | + multi_year_capex_usd: "Multi-year capex", | |
| 48 | + deal_value_usd: "Deal value", | |
| 49 | +}; | |
| 50 | + | |
| 51 | +/** Which facility column a capacity predicate feeds (null = stored as a claim only). */ | |
| 52 | +export const CAPACITY_COLUMN: Record<CapacityPredicate, "itCapacityMw" | "totalPowerMw" | "plannedPowerMw" | "utilityCapacityMw" | "gridConnectionMw" | "ultimateCampusMw" | null> = { | |
| 53 | + it_capacity_mw: "itCapacityMw", | |
| 54 | + critical_power_mw: "itCapacityMw", | |
| 55 | + utility_capacity_mw: "utilityCapacityMw", | |
| 56 | + grid_connection_mw: "gridConnectionMw", | |
| 57 | + current_power_mw: "totalPowerMw", | |
| 58 | + planned_power_mw: "plannedPowerMw", | |
| 59 | + ultimate_campus_mw: "ultimateCampusMw", | |
| 60 | + phase_mw: null, | |
| 61 | +}; | |
| 62 | + | |
| 63 | +// ─── field-level source authority ────────────────────────────────────────────────────────────────── | |
| 64 | +export type AuthorityTier = "A" | "B" | "C" | "D" | "E"; | |
| 65 | +export const TIER_RANK: Record<AuthorityTier, number> = { A: 5, B: 4, C: 3, D: 2, E: 1 }; | |
| 66 | +export const AUTHORITY_FIELDS = ["capacity", "investment", "geometry", "interconnection", "status", "ownership", "dates", "identity", "planning"] as const; | |
| 67 | +export type AuthorityField = (typeof AUTHORITY_FIELDS)[number]; | |
| 68 | + | |
| 69 | +/** | |
| 70 | + * Field-level authority hierarchy. One universal rank is wrong: OpenStreetMap is excellent for geometry and useless for | |
| 71 | + * capacity; PeeringDB (registry) is the reference for interconnection; planning filings beat press releases for approved | |
| 72 | + * capacity; utility filings beat everything for grid requirements. | |
| 73 | + */ | |
| 74 | +export const FIELD_AUTHORITY: Record<AuthorityField, Record<SourceKind, AuthorityTier>> = { | |
| 75 | + capacity: { utility: "A", government: "A", filing: "A", operator: "B", cloud_provider: "B", registry: "C", secondary: "C", news: "D", dataset: "E", community: "E" }, | |
| 76 | + investment: { filing: "A", government: "A", operator: "B", cloud_provider: "B", utility: "C", secondary: "C", news: "D", registry: "E", dataset: "E", community: "E" }, | |
| 77 | + geometry: { government: "A", community: "A", registry: "B", operator: "B", dataset: "B", cloud_provider: "C", utility: "C", filing: "C", secondary: "D", news: "E" }, | |
| 78 | + interconnection: { registry: "A", operator: "B", community: "C", dataset: "C", secondary: "D", news: "D", government: "D", filing: "D", utility: "E", cloud_provider: "E" }, | |
| 79 | + status: { operator: "A", government: "A", filing: "A", cloud_provider: "A", utility: "B", registry: "B", secondary: "C", news: "C", dataset: "D", community: "D" }, | |
| 80 | + ownership: { filing: "A", operator: "A", government: "B", registry: "B", cloud_provider: "B", secondary: "C", news: "C", dataset: "D", community: "D", utility: "D" }, | |
| 81 | + dates: { operator: "A", government: "A", filing: "A", cloud_provider: "A", secondary: "B", news: "B", registry: "C", dataset: "C", utility: "C", community: "D" }, | |
| 82 | + identity: { operator: "A", registry: "A", government: "B", filing: "B", cloud_provider: "B", dataset: "C", community: "C", secondary: "D", news: "D", utility: "D" }, | |
| 83 | + planning: { government: "A", filing: "A", utility: "B", operator: "B", secondary: "C", news: "C", cloud_provider: "C", registry: "D", dataset: "E", community: "E" }, | |
| 84 | +}; | |
| 85 | + | |
| 86 | +/** Field a predicate belongs to (for the authority lookup). */ | |
| 87 | +export function authorityFieldFor(predicate: string): AuthorityField { | |
| 88 | + if ((CAPACITY_PREDICATES as readonly string[]).includes(predicate) || /mw$/i.test(predicate)) return "capacity"; | |
| 89 | + if ((INVESTMENT_PREDICATES as readonly string[]).includes(predicate) || /usd$|investment/i.test(predicate)) return "investment"; | |
| 90 | + if (/^(geo|lat|lng|address|postal)/i.test(predicate)) return "geometry"; | |
| 91 | + if (/^(carriers|ixps|cloudProviders|networks)/i.test(predicate)) return "interconnection"; | |
| 92 | + if (/status/i.test(predicate)) return "status"; | |
| 93 | + if (/(operator|owner|tenant|developer)/i.test(predicate)) return "ownership"; | |
| 94 | + if (/(On|Opening|date)$/i.test(predicate) || /^(opened|announced|construction|expected)/i.test(predicate)) return "dates"; | |
| 95 | + if (/permit|planning|approv/i.test(predicate)) return "planning"; | |
| 96 | + return "identity"; | |
| 97 | +} | |
| 98 | + | |
| 99 | +export interface AuthorityInput { field: AuthorityField; sourceKind: SourceKind | string | null | undefined; isEstimate?: boolean | null; /** extraction method: structured / parser / regex on prose / llm */ method?: string | null; reviewed?: boolean } | |
| 100 | + | |
| 101 | +/** Authority tier of one observation for one field. Estimates and LLM-only extractions drop a tier; a human review raises one. */ | |
| 102 | +export function authorityTier(i: AuthorityInput): AuthorityTier { | |
| 103 | + const table = FIELD_AUTHORITY[i.field]; | |
| 104 | + let rank = TIER_RANK[(i.sourceKind && table[i.sourceKind as SourceKind]) || "D"]; | |
| 105 | + if (i.isEstimate) rank -= 1; | |
| 106 | + if (i.method && /^llm/i.test(i.method)) rank -= 1; | |
| 107 | + if (i.reviewed) rank += 1; | |
| 108 | + rank = Math.max(1, Math.min(5, rank)); | |
| 109 | + return (Object.keys(TIER_RANK) as AuthorityTier[]).find((t) => TIER_RANK[t] === rank) ?? "E"; | |
| 110 | +} | |
| 111 | + | |
| 112 | +// ─── scope classification ────────────────────────────────────────────────────────────────────────── | |
| 113 | +const PORTFOLIO_RE = /\b(across (?:its|our|their|the company'?s?|a) (?:global |entire |whole |growing |existing )?(?:portfolio|footprint|platform|network|fleet|estate)|portfolio|footprint|under management|under development globally|(?:global|total|combined|aggregate|cumulative) (?:capacity|footprint|pipeline|portfolio)|pipeline of|development pipeline|(?:brings?|bringing|takes?|taking|lifts?|lifting|grows?|growing) (?:its|the company'?s?|our|their|total) [^.]{0,60}(?:capacity |footprint |portfolio )?to (?:over |more than |nearly |approximately |about |roughly )?\d|now (?:operates|owns|manages|has) (?:over |more than |nearly )?\d|operates (?:over |more than |nearly )?\d+[\d.,]* ?(?:mw|gw|megawatts?|gigawatts?) (?:across|in|of)|(?:more than|over) \d+ (?:data cent(?:er|re)s|facilities|sites|campuses|locations)|globally|worldwide|company-?wide|enterprise-?wide|group-?wide)\b/i; | |
| 114 | +const COMPANY_RE = /\b(capex|capital expenditure|will invest [^.]{0,40}(?:over the next|through|by) (?:the next )?(?:\d+ years|20\d\d)|(?:through|by|over the next|over the coming) (?:20\d\d|\d+ years|the decade)|(?:total|planned|annual) investment (?:of|in) [^.]{0,30}(?:across|in) (?:the (?:us|uk|country|region)|[A-Z][a-z]+ and [A-Z][a-z]+)|(?:in|for) (?:its|our|their) (?:fiscal|financial) (?:year|quarter)|(?:fiscal|fy) ?20\d\d|guidance|earnings|revenue|quarterly results|annual report|10-k|market (?:to|will|set to|expected to|projected to) (?:surpass|reach|hit|grow|exceed|top)|market size|cagr|forecast period)\b/i; | |
| 115 | +const COUNTRY_RE = /\b(nationwide|nation-?wide|national (?:program|programme|plan|strategy|investment|pipeline)|across the country|the country'?s|country-?wide|(?:in|across) (?:the )?(?:united states|u\.s\.|us|uk|united kingdom|europe|emea|apac|apj|asia|asia-pacific|latin america|the americas|the middle east|africa|india|japan|germany|france|australia|canada|brazil|malaysia|singapore|the netherlands|ireland|spain|italy|the nordics|scandinavia|the region)\b)/i; | |
| 116 | +const METRO_RE = /\b((?:in|across) the (?:[A-Z][\w.]+ )?(?:market|metro|region|area|corridor|cluster)|(?:the )?(?:northern virginia|dfw|dallas-fort worth|silicon valley|greater [A-Z][a-z]+|the bay area) (?:market|region|area)|market'?s (?:total|inventory|capacity)|inventory|absorption|vacancy)\b/i; | |
| 117 | +const CAMPUS_RE = /\b(campus|campuses|park|complex|estate|hub|gigafactory|(?:at|when|upon|once) (?:full(?:y)? )?(?:built[- ]out|build[- ]out|buildout|complete|completed|completion)|full build-?out|ultimate(?:ly)?|(?:total|overall|eventual) (?:site|campus|planned) capacity|master-?plan(?:ned)?|(?:across|over|in) (?:\w+ )?(?:phases|buildings|data halls)|multi-?building|multi-?phase|acre (?:site|campus|development)|the (?:site|development|project) (?:will|would|could) (?:deliver|provide|offer|support|house|host|total|reach)|entire (?:site|development|project))\b/i; | |
| 118 | +const BUILDING_RE = /\b(building [A-Z0-9]+|(?:first|second|third|fourth|fifth|initial|new|final) (?:building|data hall|hall|structure|phase)|data hall|phase (?:\d+|one|two|three|i|ii|iii)\b|bldg\.?|the (?:building|hall) (?:will|would|is|has|offers|provides|delivers)|single[- ]storey|two-?stor(?:e)?y|three-?stor(?:e)?y|\d+[- ]stor(?:e)?y)\b/i; | |
| 119 | +const FACILITY_RE = /\b(facility|the data cent(?:er|re)|this data cent(?:er|re)|the site|the new data cent(?:er|re)|the (?:\w+ )?data cent(?:er|re) (?:will|would|is|has|offers|provides|delivers))\b/i; | |
| 120 | + | |
| 121 | +export interface ScopeClassification { scope: ClaimScope; reason: string } | |
| 122 | + | |
| 123 | +/** | |
| 124 | + * Deterministic scope of a figure from the sentence (and the few words) around it. Precedence: company → portfolio → | |
| 125 | + * country → metro → campus → building → facility → unknown. `subjectHint` (what the record itself is) only breaks a tie | |
| 126 | + * when the sentence says nothing. An UNKNOWN scope must never populate a facility column. | |
| 127 | + */ | |
| 128 | +export function classifyScope(context: string | null | undefined, subjectHint?: ClaimScope | null): ScopeClassification { | |
| 129 | + const s = (context ?? "").replace(/\s+/g, " "); | |
| 130 | + if (!s.trim()) return { scope: subjectHint ?? "unknown", reason: subjectHint ? "context:empty+subject" : "context:empty" }; | |
| 131 | + if (COMPANY_RE.test(s)) return { scope: "company", reason: "regex:company" }; | |
| 132 | + if (PORTFOLIO_RE.test(s)) return { scope: "portfolio", reason: "regex:portfolio" }; | |
| 133 | + if (COUNTRY_RE.test(s)) return { scope: "country", reason: "regex:country" }; | |
| 134 | + if (METRO_RE.test(s)) return { scope: "metro", reason: "regex:metro" }; | |
| 135 | + if (CAMPUS_RE.test(s)) return { scope: "campus", reason: "regex:campus" }; | |
| 136 | + if (BUILDING_RE.test(s)) return { scope: "building", reason: "regex:building" }; | |
| 137 | + if (FACILITY_RE.test(s)) return { scope: "facility", reason: "regex:facility" }; | |
| 138 | + if (subjectHint) return { scope: subjectHint, reason: "subject" }; | |
| 139 | + return { scope: "unknown", reason: "no-signal" }; | |
| 140 | +} | |
| 141 | + | |
| 142 | +// ─── capacity semantics ──────────────────────────────────────────────────────────────────────────── | |
| 143 | +const IT_RE = /\b(it (?:load|capacity|power)|critical (?:it )?(?:load|capacity)|of it\b|it-?load|white ?space power|customer (?:power|load)|usable (?:power|capacity)|for it equipment|critical power)\b/i; | |
| 144 | +const CRITICAL_RE = /\bcritical (?:it )?(?:load|power|capacity)\b/i; | |
| 145 | +const GRID_RE = /\b(grid connection|grid-?connected|interconnection (?:agreement|capacity|request|queue)|connection (?:agreement|to the grid|capacity)|from the (?:grid|utility|substation)|substation|transmission (?:line|capacity|connection)|(?:secured|reserved|contracted|allocated|committed|approved) (?:\w+ )?(?:mw|gw|megawatts?|gigawatts?) of (?:grid |utility )?(?:power|electricity|capacity|supply|load)|load (?:request|letter|study|agreement)|large[- ]load|energi[sz]ation of)\b/i; | |
| 146 | +const UTILITY_RE = /\b(utility (?:power|capacity|supply|feed|service)|power (?:supply|feed|service) (?:agreement|contract|capacity)|(?:power|electricity) (?:supply|allocation) (?:of|totaling|totalling)|(?:dual|redundant) (?:utility )?feeds?|incoming power|electrical (?:capacity|supply|service)|total (?:site )?power|power capacity of|available power)\b/i; | |
| 147 | +const ULTIMATE_RE = /\b((?:at|when|upon|once) (?:full(?:y)? )?(?:built[- ]out|build[- ]out|buildout|complete|completed|completion)|full build-?out|ultimate(?:ly)?|final build-?out|(?:total|overall|eventual|potential|maximum) (?:site|campus|planned|development) (?:capacity|power|load)|(?:scale|scalable|scaling|grow|growing|expand|expandable) (?:up )?to (?:over |more than )?\d|up to (?:over |more than |approximately |nearly )?\d|master-?plan)\b/i; | |
| 148 | +const PLANNED_RE = /\b(planned|proposed|will (?:deliver|provide|offer|have|house|support|feature|bring|add|total)|would (?:deliver|provide|offer|have|house|support)|expected to (?:deliver|provide|offer|reach|have)|is (?:set|slated|expected|designed) to|designed (?:for|to)|target(?:s|ed|ing)? (?:capacity|of)|to be (?:built|developed|delivered)|future|when complete|on completion|approved (?:for|capacity)|permitted|zoned for|application for|seeks? (?:approval|permission) for)\b/i; | |
| 149 | +const PHASE_RE = /\b((?:first|second|third|fourth|initial|next|final) phase|phase (?:\d+|one|two|three|four|i|ii|iii|iv)\b|(?:first|initial) (?:building|data hall|tranche)|initial(?:ly)? (?:deliver|provide|offer)\w*)\b/i; | |
| 150 | +const CURRENT_RE = /\b(currently|existing|in operation|operational|operating|live|online|now (?:offers|provides|delivers|has|houses)|today|present(?:ly)?|commissioned|energi[sz]ed|installed|delivered|is (?:offering|providing|delivering))\b/i; | |
| 151 | + | |
| 152 | +export interface CapacitySemantics { predicate: CapacityPredicate | null; reason: string } | |
| 153 | + | |
| 154 | +/** | |
| 155 | + * Which kind of MW a sentence describes. Precedence: grid → utility → IT → phase → ultimate → planned → current. | |
| 156 | + * Returns null when the sentence gives no semantic cue; the caller decides a default from the record status | |
| 157 | + * (pipeline record → planned, operational record → total power) and flags the figure as "semantics: default". | |
| 158 | + */ | |
| 159 | +export function classifyCapacitySemantics(context: string | null | undefined): CapacitySemantics { | |
| 160 | + const s = (context ?? "").replace(/\s+/g, " "); | |
| 161 | + if (!s.trim()) return { predicate: null, reason: "context:empty" }; | |
| 162 | + const isIt = IT_RE.test(s); | |
| 163 | + if (GRID_RE.test(s) && !isIt) return { predicate: "grid_connection_mw", reason: "regex:grid" }; | |
| 164 | + if (UTILITY_RE.test(s) && !isIt) return { predicate: "utility_capacity_mw", reason: "regex:utility" }; | |
| 165 | + if (PHASE_RE.test(s) && !ULTIMATE_RE.test(s)) return { predicate: "phase_mw", reason: "regex:phase" }; | |
| 166 | + if (ULTIMATE_RE.test(s)) return { predicate: "ultimate_campus_mw", reason: "regex:ultimate" }; | |
| 167 | + if (isIt) return { predicate: CRITICAL_RE.test(s) ? "critical_power_mw" : "it_capacity_mw", reason: "regex:it" }; | |
| 168 | + if (PLANNED_RE.test(s)) return { predicate: "planned_power_mw", reason: "regex:planned" }; | |
| 169 | + if (CURRENT_RE.test(s)) return { predicate: "current_power_mw", reason: "regex:current" }; | |
| 170 | + return { predicate: null, reason: "no-signal" }; | |
| 171 | +} | |
| 172 | + | |
| 173 | +// ─── investment semantics ────────────────────────────────────────────────────────────────────────── | |
| 174 | +const MULTI_YEAR_RE = /\b((?:through|by|over the next|over the coming|until|to) (?:20\d\d|\d+ years|the decade|the end of the decade)|multi-?year|\d+-year (?:plan|investment|programme|program|commitment)|(?:annual|yearly) (?:capex|capital expenditure|investment)|capex|capital expenditure)\b/i; | |
| 175 | +const COUNTRY_PROGRAM_RE = /\b((?:national|country|sovereign|government|federal|state) (?:program|programme|plan|initiative|strategy|fund|investment)|(?:invest|investment|commit\w*|pledge\w*) [^.]{0,60}(?:in|across) (?:the )?(?:united states|u\.s\.|us|uk|united kingdom|europe|india|japan|germany|france|australia|canada|brazil|malaysia|singapore|the country|the nation|the region|emea|apac)\b)/i; | |
| 176 | +const DEAL_RE = /\b(acqui(?:re|res|red|sition)|takeover|buyout|merger|stake|valuation|valued at|purchase price|sale of|sells?|sold|buys?|bought|financing|refinanc\w+|loan|credit facility|bond|notes|debt|equity|raises?|raised|funding round|series [a-e]|lease(?:d|s)? (?:valued|worth)|contract (?:valued|worth))\b/i; | |
| 177 | +const CAMPUS_INV_RE = /\b(campus|site|park|complex|development|project|phase|build-?out|the facility|the data cent(?:er|re))\b/i; | |
| 178 | + | |
| 179 | +export interface InvestmentSemantics { predicate: InvestmentPredicate; scope: ClaimScope; reason: string } | |
| 180 | + | |
| 181 | +/** Which kind of money a sentence describes and which scope it applies to. Deal values and company capex never become a project's investment. */ | |
| 182 | +export function classifyInvestmentSemantics(context: string | null | undefined, subjectHint?: ClaimScope | null): InvestmentSemantics { | |
| 183 | + const s = (context ?? "").replace(/\s+/g, " "); | |
| 184 | + if (!s.trim()) return { predicate: "project_investment_usd", scope: subjectHint ?? "unknown", reason: "context:empty" }; | |
| 185 | + if (DEAL_RE.test(s) && !/\b(invest(?:s|ed|ing|ment)? (?:of |about |approximately |over |more than |up to )?[$€£]|to build|construction|will invest|plans to invest)\b/i.test(s)) return { predicate: "deal_value_usd", scope: "company", reason: "regex:deal" }; | |
| 186 | + if (COUNTRY_PROGRAM_RE.test(s)) return { predicate: "country_program_usd", scope: "country", reason: "regex:country-program" }; | |
| 187 | + if (MULTI_YEAR_RE.test(s) || COMPANY_RE.test(s)) return { predicate: "multi_year_capex_usd", scope: "company", reason: "regex:multi-year" }; | |
| 188 | + if (PORTFOLIO_RE.test(s)) return { predicate: "company_investment_usd", scope: "portfolio", reason: "regex:portfolio" }; | |
| 189 | + const sc = classifyScope(s, subjectHint); | |
| 190 | + if (sc.scope === "campus") return { predicate: "campus_investment_usd", scope: "campus", reason: sc.reason }; | |
| 191 | + if (sc.scope === "building" || sc.scope === "facility") return { predicate: "project_investment_usd", scope: sc.scope, reason: sc.reason }; | |
| 192 | + if (CAMPUS_INV_RE.test(s)) return { predicate: "project_investment_usd", scope: subjectHint ?? "facility", reason: "regex:project" }; | |
| 193 | + return { predicate: "project_investment_usd", scope: sc.scope, reason: sc.reason }; | |
| 194 | +} | |
| 195 | + | |
| 196 | +// ─── sanity engines ──────────────────────────────────────────────────────────────────────────────── | |
| 197 | +export type FlagSeverity = "info" | "warn" | "critical"; | |
| 198 | +export interface SanityFlag { code: string; severity: FlagSeverity; message: string; field?: string; value?: number | null; blocks?: boolean } | |
| 199 | + | |
| 200 | +export interface CapacitySanityInput { | |
| 201 | + value: number; | |
| 202 | + predicate: CapacityPredicate | null; | |
| 203 | + scope: ClaimScope; | |
| 204 | + /** the record the figure is being written to */ | |
| 205 | + recordScope: "building" | "facility" | "campus" | "project"; | |
| 206 | + previous?: number | null; | |
| 207 | + context?: string | null; | |
| 208 | + /** true when the facility / project name says campus / park / complex */ | |
| 209 | + campusDesignation?: boolean; | |
| 210 | +} | |
| 211 | + | |
| 212 | +const MONEY_NUM_RE = /(?:US\$|USD|\$|€|£|A\$|C\$|S\$)\s?(\d{1,3}(?:[.,]\d{3})*(?:\.\d+)?)\s*(?:million|billion|bn|m\b|b\b)?/gi; | |
| 213 | +const MW_NUM_RE = /(\d{1,3}(?:[.,]\d{3})*(?:\.\d+)?)\s*(?:mw|gw|megawatts?|gigawatts?)\b/gi; | |
| 214 | + | |
| 215 | +/** "$300 million" and "300 MW" in the same sentence share a number: the unit may have been confused. */ | |
| 216 | +export function moneyMwCollision(context: string | null | undefined): boolean { | |
| 217 | + if (!context) return false; | |
| 218 | + const money = new Set<string>(); | |
| 219 | + for (const m of context.matchAll(MONEY_NUM_RE)) money.add(m[1]!.replace(/,/g, "")); | |
| 220 | + if (!money.size) return false; | |
| 221 | + for (const m of context.matchAll(MW_NUM_RE)) if (money.has(m[1]!.replace(/,/g, ""))) return true; | |
| 222 | + return false; | |
| 223 | +} | |
| 224 | + | |
| 225 | +/** Deterministic capacity checks. `blocks: true` flags must keep the figure out of the record's columns. */ | |
| 226 | +export function capacitySanity(i: CapacitySanityInput): SanityFlag[] { | |
| 227 | + const out: SanityFlag[] = []; | |
| 228 | + const v = i.value; | |
| 229 | + if (!Number.isFinite(v) || v <= 0) return [{ code: "mw_invalid", severity: "critical", message: `invalid MW ${v}`, value: v, blocks: true }]; | |
| 230 | + if (!isSiteScope(i.scope)) out.push({ code: `scope_${i.scope}`, severity: "critical", message: `figure describes a ${i.scope} (${i.scope === "unknown" ? "scope could not be determined" : "not this site"}) — kept as a claim, not assigned`, value: v, blocks: true }); | |
| 231 | + if (i.scope === "campus" && i.recordScope === "building") out.push({ code: "scope_campus_on_building", severity: "critical", message: "campus-level figure on a building record", value: v, blocks: true }); | |
| 232 | + if (v >= 20_000) out.push({ code: "mw_market_statistic", severity: "critical", message: `${v} MW is an industry / market statistic, not a site`, value: v, blocks: true }); | |
| 233 | + if (v > 1_000 && i.recordScope !== "campus" && !i.campusDesignation && i.predicate !== "ultimate_campus_mw") out.push({ code: "mw_single_site_gt_1000", severity: "critical", message: `single facility above 1 000 MW (${v} MW) without a campus designation`, value: v }); | |
| 234 | + if (v > 500 && i.recordScope === "building") out.push({ code: "mw_building_gt_500", severity: "warn", message: `one building above 500 MW (${v} MW)`, value: v }); | |
| 235 | + if (i.previous != null && i.previous > 0 && (v / i.previous >= 5 || i.previous / v >= 5)) out.push({ code: "mw_change_5x", severity: "warn", message: `figure changed more than 5× (${i.previous} → ${v} MW)`, value: v }); | |
| 236 | + if (i.context && moneyMwCollision(i.context)) out.push({ code: "mw_money_collision", severity: "critical", message: "a money figure with the same number sits in the same sentence — verify the unit", value: v }); | |
| 237 | + if (i.predicate == null) out.push({ code: "mw_semantics_default", severity: "info", message: "sentence does not say what kind of MW this is; semantics defaulted from the record status", value: v }); | |
| 238 | + if (i.predicate === "grid_connection_mw" || i.predicate === "utility_capacity_mw") out.push({ code: "mw_utility_not_it", severity: "info", message: "utility / grid figure — not IT load", value: v }); | |
| 239 | + return out; | |
| 240 | +} | |
| 241 | + | |
| 242 | +export interface InvestmentSanityInput { value: number; scope: ClaimScope; predicate: InvestmentPredicate; recordScope: "facility" | "campus" | "project"; previous?: number | null; context?: string | null } | |
| 243 | + | |
| 244 | +export function investmentSanity(i: InvestmentSanityInput): SanityFlag[] { | |
| 245 | + const out: SanityFlag[] = []; | |
| 246 | + const v = i.value; | |
| 247 | + if (!Number.isFinite(v) || v <= 0) return [{ code: "inv_invalid", severity: "critical", message: `invalid investment ${v}`, value: v, blocks: true }]; | |
| 248 | + if (!isSiteScope(i.scope)) out.push({ code: `inv_scope_${i.scope}`, severity: "critical", message: `investment describes a ${i.scope} — kept as a claim, not assigned`, value: v, blocks: true }); | |
| 249 | + if (i.predicate === "deal_value_usd" || i.predicate === "multi_year_capex_usd" || i.predicate === "company_investment_usd" || i.predicate === "country_program_usd") out.push({ code: `inv_${i.predicate}`, severity: "critical", message: `${INVESTMENT_PREDICATE_LABEL[i.predicate]} is not a site investment`, value: v, blocks: true }); | |
| 250 | + if (v > 500e9) out.push({ code: "inv_gt_500b", severity: "critical", message: "above $500B — an industry statistic", value: v, blocks: true }); | |
| 251 | + if (v > 50e9) out.push({ code: "inv_single_site_gt_50b", severity: "critical", message: `single site investment above $50B (${Math.round(v / 1e9)}B) — requires verification`, value: v }); | |
| 252 | + if (i.previous != null && i.previous > 0 && (v / i.previous >= 5 || i.previous / v >= 5)) out.push({ code: "inv_change_5x", severity: "warn", message: `investment changed more than 5× (${i.previous} → ${v})`, value: v }); | |
| 253 | + if (i.context && moneyMwCollision(i.context)) out.push({ code: "inv_money_mw_collision", severity: "warn", message: "same number appears as MW in the sentence — verify", value: v }); | |
| 254 | + return out; | |
| 255 | +} | |
| 256 | + | |
| 257 | +// ─── evidence ────────────────────────────────────────────────────────────────────────────────────── | |
| 258 | +export interface Evidence { text: string; start: number; end: number } | |
| 259 | + | |
| 260 | +/** Split into sentences with absolute offsets (abbreviation-tolerant enough for press prose). */ | |
| 261 | +export function sentenceSpans(text: string): Array<{ text: string; start: number; end: number }> { | |
| 262 | + const out: Array<{ text: string; start: number; end: number }> = []; | |
| 263 | + const re = /[^.!?\n]+(?:[.!?]+(?=\s+[A-Z0-9"'“(]|\s*$)|\n|$)/g; | |
| 264 | + for (const m of text.matchAll(re)) { | |
| 265 | + const raw = m[0]; | |
| 266 | + const lead = raw.length - raw.trimStart().length; | |
| 267 | + const t = raw.trim(); | |
| 268 | + if (!t) continue; | |
| 269 | + out.push({ text: t, start: (m.index ?? 0) + lead, end: (m.index ?? 0) + lead + t.length }); | |
| 270 | + } | |
| 271 | + return out; | |
| 272 | +} | |
| 273 | + | |
| 274 | +function numberVariants(value: number, unit: "mw" | "usd"): RegExp { | |
| 275 | + if (unit === "mw") { | |
| 276 | + const alts = new Set<string>(); | |
| 277 | + const push = (n: number, u: string) => { if (Number.isFinite(n)) alts.add(`${fmtNum(n)}\\+?\\s?(?:${u})`); }; | |
| 278 | + push(value, "MW|megawatts?"); | |
| 279 | + push(value / 1000, "GW|gigawatts?"); | |
| 280 | + push(value * 1000, "kW|kilowatts?"); | |
| 281 | + return new RegExp(`(?<![\\d.])(?:${[...alts].join("|")})\\b`, "i"); | |
| 282 | + } | |
| 283 | + const alts = new Set<string>(); | |
| 284 | + const push = (n: number, u: string) => { if (Number.isFinite(n) && n >= 1) alts.add(`(?:US\\$|USD ?|\\$|€|£|A\\$|C\\$|S\\$)\\s?${fmtNum(n)}\\s?(?:${u})`); }; | |
| 285 | + push(value / 1e9, "billion|bn|b\\b"); | |
| 286 | + push(value / 1e6, "million|mn|m\\b"); | |
| 287 | + push(value / 1e12, "trillion|tn"); | |
| 288 | + if (value < 1e6) push(value, ""); | |
| 289 | + return new RegExp(`(?:${[...alts].join("|")})`, "i"); | |
| 290 | +} | |
| 291 | + | |
| 292 | +function fmtNum(n: number): string { | |
| 293 | + const r = Math.round(n * 1000) / 1000; | |
| 294 | + const plain = String(r); | |
| 295 | + const withCommas = r >= 1000 && Number.isInteger(r) ? r.toLocaleString("en-US") : null; | |
| 296 | + const variants = [plain, withCommas].filter((x): x is string => !!x).map((x) => x.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")); | |
| 297 | + // "1.5" also written "1,5"; integers may carry a trailing ".0" | |
| 298 | + if (!Number.isInteger(r)) variants.push(plain.replace(".", ",").replace(/[.*+?^${}()|[\]\\]/g, "\\$&")); | |
| 299 | + return `(?:${variants.join("|")})`; | |
| 300 | +} | |
| 301 | + | |
| 302 | +/** The sentence that supports a figure (value in MW or USD). null when no sentence quotes it — then there is no claim. */ | |
| 303 | +export function findEvidence(text: string | null | undefined, value: number, unit: "mw" | "usd"): Evidence | null { | |
| 304 | + if (!text) return null; | |
| 305 | + const re = numberVariants(value, unit); | |
| 306 | + for (const s of sentenceSpans(text)) { | |
| 307 | + if (re.test(s.text)) return { text: s.text.length > 600 ? `${s.text.slice(0, 597)}…` : s.text, start: s.start, end: s.end }; | |
| 308 | + } | |
| 309 | + return null; | |
| 310 | +} | |
| 311 | + | |
| 312 | +// ─── project event classification ───────────────────────────────────────────────────────────────── | |
| 313 | +export const PROJECT_CLASSES = ["NEW_BUILD", "EXPANSION", "CONSTRUCTION_START", "PERMIT", "LAND_ACQUISITION", "POWER_AGREEMENT", "GRID_CONNECTION", "FINANCING", "ACQUISITION", "PARTNERSHIP", "CUSTOMER_AGREEMENT", "EXECUTIVE_APPOINTMENT", "SUSTAINABILITY", "PRODUCT_NEWS", "GENERAL_COMPANY_NEWS", "UNKNOWN"] as const; | |
| 314 | +export type ProjectClass = (typeof PROJECT_CLASSES)[number]; | |
| 315 | +/** Classes that describe physical development and may create or materially modify a PROJECT. */ | |
| 316 | +export const PHYSICAL_CLASSES: ReadonlySet<ProjectClass> = new Set<ProjectClass>(["NEW_BUILD", "EXPANSION", "CONSTRUCTION_START", "PERMIT", "LAND_ACQUISITION", "GRID_CONNECTION"]); | |
| 317 | +/** Classes that may attach an EVENT to an identifiable project / company but never create a project. */ | |
| 318 | +export const ASSOCIATED_CLASSES: ReadonlySet<ProjectClass> = new Set<ProjectClass>(["POWER_AGREEMENT", "FINANCING", "ACQUISITION", "PARTNERSHIP", "CUSTOMER_AGREEMENT"]); | |
| 319 | + | |
| 320 | +const CLASS_RULES: Array<[ProjectClass, RegExp]> = [ | |
| 321 | + ["EXECUTIVE_APPOINTMENT", /\b(appoint(?:s|ed|ment)?|names? [A-Z][\w'-]+ [A-Z][\w'-]+ (?:as|to)|joins? (?:as|the board|its board)|(?:new|hires?|promotes?|welcomes?|taps?) (?:[A-Z][\w'-]+ [A-Z][\w'-]+ as )?(?:CEO|CFO|COO|CTO|CRO|CIO|CMO|chief|president|head of|director|vice president|VP|managing director|general manager|board member|chairman|chair)|leadership (?:change|team|appointments?)|steps? down|retire(?:s|ment)|passing of|obituary|in memoriam|executive (?:team|hire|appointment))\b/i], | |
| 322 | + ["GENERAL_COMPANY_NEWS", /\b(market (?:to|will|set to|expected to|projected to) (?:surpass|reach|hit|grow|exceed|top)|market size|cagr|forecast(?:s|ed)? (?:to|that)|report(?: finds| shows| reveals|:)|survey|study (?:finds|shows|reveals)|(?:index|trends?|predictions?|outlook|white ?paper|e-?book|webinar|podcast|interview|q&a|faq|explained|101\b|guide to|tips|best practices|case study|customer spotlight|award|shortlist|finalist|recogni[sz]ed|named (?:a |one of |to )?(?:top|best|leader))\b|quarterly results|half[- ]year results|annual results|earnings|revenue|ebitda|guidance|investor (?:day|presentation)|lawsuit|sues?\b|court|settlement|opinion|op-ed|why |how |what )/i], | |
| 323 | + ["POWER_AGREEMENT", /\b(ppa\b|power purchase agreement|(?:virtual|corporate|long-term|\d+-year) (?:power|energy) (?:purchase )?agreement|(?:energy|power|electricity) (?:supply|offtake) (?:agreement|deal|contract)|offtake|(?:signs?|signed|inks?|inked|secures?|secured) (?:a )?(?:\d+[ -]?(?:mw|gw) )?(?:of )?(?:solar|wind|nuclear|geothermal|hydro|gas|renewable|clean|carbon-free) (?:power|energy|capacity|supply)|(?:solar|wind|nuclear|geothermal|hydro|gas|battery|fuel cell|smr|reactor)s? (?:to power|will power|powering|deal|plant to supply|farm to supply)|behind-the-meter|on-site (?:generation|power plant|gas plant|solar)|microgrid|small modular reactor)\b/i], | |
| 324 | + ["SUSTAINABILITY", /\b(sustainability (?:report|goals?|targets?|commitment|strategy)|net[- ]zero|carbon[- ]neutral|carbon[- ]free|esg\b|renewable (?:energy )?(?:certificates?|credits?|target|goal|commitment)|100% renewable|water (?:positive|stewardship|usage|conservation)|emissions? (?:reduction|target)|leed (?:gold|platinum|certif)|green (?:building|certification)|energy efficiency (?:program|initiative)|climate (?:pledge|commitment)|biodiversity|tree planting|heat (?:reuse|recovery) (?:program|scheme|initiative))\b/i], | |
| 325 | + ["GRID_CONNECTION", /\b(grid connection|grid-?connected|interconnection (?:agreement|request|capacity|queue|study)|connection agreement|(?:new |expanded )?substation|transmission (?:line|upgrade|project|capacity)|energi[sz]ation|energi[sz]ed|large[- ]load (?:tariff|request|agreement|customer)|load (?:request|study|letter)|(?:secured|reserved|allocated|approved) (?:\d+[ -]?(?:mw|gw) of )?(?:grid |utility )?(?:power|capacity) (?:from|with) (?:the utility|[A-Z][\w&]+ (?:energy|power|electric|utility)))\b/i], | |
| 326 | + ["FINANCING", /\b(financ(?:es|ing|ed)|refinanc\w+|raises? (?:\$|€|£|us\$)|raised (?:\$|€|£|us\$)|series [a-e]\b|funding (?:round|arranged|secured)|loan|credit facility|green bond|bonds?\b|notes? offering|debt (?:facility|package|raise)|equity (?:raise|investment|stake)|ipo\b|capital raise|construction (?:loan|financing)|securitization|abs\b|term loan|(?:secures?|secured|closes?|closed) (?:\$|€|£|us\$)[\d.,]+ ?(?:m|bn|b|million|billion)? (?:in )?(?:financing|loan|facility|debt|funding))\b/i], | |
| 327 | + ["ACQUISITION", /\b(acqui(?:res?|red|sition) (?:of )?(?:[A-Z][\w&'.-]+ )+(?:group|holdings|inc|ltd|llc|limited|corp|sa|ag|plc|pte|data ?cent(?:er|re)s?|portfolio|platform|company|business)|(?:acquires?|acquired|acquiring|buys?|bought|buying|purchases?|purchased|purchasing|takes? over|took over) (?:a |an |the |its |two |three |four |five |six |\d+ )?(?:\d+[ -]?(?:mw|gw) )?(?:(?:hyperscale|colocation|carrier-neutral|edge|operational|existing) )?(?:data ?cent(?:er|re)s?|facility|facilities|portfolio|platform|operator|business|company|stake|campus from|site from)|acquisition of|takeover|buyout|merger|merges? with|to be acquired|stake in|sells?|sold|divest\w+|sale of)\b/i], | |
| 328 | + ["CUSTOMER_AGREEMENT", /\b((?:signs?|signed|secures?|secured|inks?|inked|lands?|landed|wins?|won|announces?) (?:a |an |its |the )?(?:multi-year |long-term |major |anchor |hyperscale |\d+[ -]?(?:mw|gw) )?(?:colocation |hosting |capacity |wholesale |pre-?)?(?:lease|leases|tenant|customer|contract|agreement|deal) (?:with|for|from)|(?:lease|leases|leased) (?:\d+[ -]?(?:mw|gw)|[\d,]+ (?:sq|square))|(?:selects?|selected|chooses?|chose|picks?|picked|taps?|tapped) [A-Z][\w&'.-]+ (?:for|as|to)|(?:hosting|colocation|lease|leasing|services?|master|supply|framework|capacity) agreements?|pre-?leas\w+|fully leased|anchor tenant|moves? into|deploys? (?:in|at|with)|expands? (?:its )?(?:footprint|presence) (?:in|at|with) [A-Z][\w&'.-]+(?:'s)? (?:data ?cent(?:er|re)|facility|campus))\b/i], | |
| 329 | + ["PARTNERSHIP", /\b(partnership|partners? with|joint venture|\bjv\b|join(?:s|ed)? forces|team(?:s|ed)? up|collaborat(?:es?|ion|ing)|alliance|memorandum of understanding|\bmou\b|strategic (?:agreement|relationship|cooperation)|framework agreement)\b/i], | |
| 330 | + ["PRODUCT_NEWS", /\b(launches? (?:a )?(?:new )?(?:service|platform|product|portal|program|programme|offering|solution|marketplace|tool|api|feature)|introduces? (?:a )?(?:new )?(?:service|platform|product|offering|solution)|now available|available (?:now|in|from)|unveils? (?:a )?(?:new )?(?:service|platform|product|offering|solution|brand|logo|website)|rebrand|new (?:brand|logo|website|identity)|integrat(?:es|ion) with|certified by|(?:earns?|achieves?|receives?) (?:[A-Z]\w+ )?(?:certification|rating|accreditation|iso \d))\b/i], | |
| 331 | + ["LAND_ACQUISITION", /\b((?:acquires?|acquired|buys?|bought|purchases?|purchased|secures?|secured|closes? on|closed on|options?) (?:a |an |the |its |additional |another )?(?:\d[\d,.]*[ -]?(?:acre|hectare|ha)s?|land|site|parcel|plot|property|farmland|lot)|land (?:purchase|acquisition|deal|bank)|(?:acre|hectare) (?:site|parcel|plot|property) (?:acquired|purchased|bought)|site (?:acquisition|purchase)|zoned land|land for (?:a |its |the )?(?:new )?(?:data ?cent(?:er|re)|campus))\b/i], | |
| 332 | + ["PERMIT", /\b(planning (?:application|permission|approval|consent|committee|commission|board|inquiry)|permit(?:s|ted|ting)?\b|rezon\w+|zoning (?:approval|change|request|case|board|commission)|(?:files?|filed|submits?|submitted|lodges?|lodged) (?:plans?|an application|a planning|planning|for approval|proposal)|(?:approv(?:es|ed|al)|green-?light(?:s|ed)?|consent(?:s|ed)?|clears?|cleared|greenlit|(?:votes?|voted) to approve|unanimously approved|signs? off|sign-off) (?:for |of |on )?(?:a |the |its |\d+[ -]?(?:mw|gw) )?(?:data ?cent(?:er|re)|campus|project|plans?|development|proposal|application|rezoning|site plan|special use|conditional use)|environmental (?:impact|assessment|review|permit)|eia\b|site plan (?:approval|review)|special (?:use|exception) permit|conditional use permit|comprehensive plan amendment|public hearing|(?:county|city|town|council|board|commission|supervisors) (?:approves?|approved|rejects?|rejected|denies|denied|defers?|deferred|delays?|tables?|tabled))\b/i], | |
| 333 | + ["CONSTRUCTION_START", /\b(breaks? ground|broke ground|groundbreaking|ground-?breaking|construction (?:has |have )?(?:started|begins?|began|begun|commenced|commences|kicks? off|is under ?way|under ?way|gets under ?way)|(?:begins?|began|starts?|started|commences?|commenced|kicks? off) (?:construction|building|work|site work|vertical construction)|under construction|topped out|topping out|tops out|steel (?:erection|going up)|first concrete|shovels in the ground|site work (?:begins|has begun|under ?way)|construction (?:milestone|update|progress))\b/i], | |
| 334 | + ["EXPANSION", /\b(expan(?:sion|ds?|ded|ding)|adds? (?:\d+[ -]?(?:mw|gw)|capacity|a (?:second|third|fourth|new) (?:building|data hall|phase))|additional (?:\d+[ -]?(?:mw|gw)|capacity|building|data hall|phase)|(?:second|third|fourth|fifth|next|new) (?:phase|building|data hall|hall|facility (?:at|on|in) (?:its|the) (?:existing )?campus)|phase (?:2|3|4|ii|iii|iv|two|three|four)\b|(?:grows?|growing|scales?|scaling|extends?|extending|enlarges?|doubles?|doubling|triples?|tripling) (?:its |the )?(?:capacity|campus|footprint at|data cent(?:er|re)|site|facility)|(?:grows?|expands?|extends?) [A-Z][\w .-]{0,40}?(?:campus|data cent(?:er|re)|facility|site)\b|more capacity (?:at|in|to)|further (?:capacity|building|facility|phase)|extension (?:of|to) (?:its|the) (?:existing )?(?:data cent(?:er|re)|campus|facility|site))\b/i], | |
| 335 | + ["NEW_BUILD", /\b((?:to|will|would|could|may|might|plans? to|set to|aims? to|intends? to|proposes? to|wants? to|is|are|is expected to|expected to) (?:build|construct|develop|create|deliver|establish|open|erect|bring)|(?:plans?|planned|planning|proposes?|proposed|proposal|announces?|announced|unveils?|unveiled|reveals?|revealed|confirms?|confirmed|launches?|launched|commits? to|committed to|pledges?|earmarks?|eyes|eyeing|mulls?|weighs?|considering|exploring|in talks) (?:for |to build |to develop |to construct |to invest in |to open |a |an |its |the |new |first |second |two |three |\d+[ -]?(?:mw|gw) |\$[\d.,]+ ?(?:m|bn|b|million|billion)? |€[\d.,]+ ?(?:m|bn|b|million|billion)? |£[\d.,]+ ?(?:m|bn|b|million|billion)? |hyperscale |ai |massive |major |huge |giant |sprawling |\w+-acre )*(?:data ?cent(?:er|re)s?|campus|campuses|facility|facilities|ai factory|ai factories|hub|site|development|project|complex|cluster|supercomputer)|new (?:\d+[ -]?(?:mw|gw) |hyperscale |ai |\$[\d.,]+ ?(?:m|bn|b|million|billion) )?(?:data ?cent(?:er|re)|campus|facility|ai factory|hub) (?:in|near|at|for|to|planned|proposed|coming|slated|set)|(?:\d+[ -]?(?:mw|gw)) (?:ai )?(?:data ?cent(?:er|re)|campus|facility|site|project|development|hub) (?:in|near|at|for|planned|proposed)|(?:build|building|develop|developing|construct|constructing|open|opening) (?:a |an |its |the |new |first |\d+[ -]?(?:mw|gw) )*(?:data ?cent(?:er|re)|campus|facility|ai factory) (?:in|near|at|on|for)|first (?:data ?cent(?:er|re)|campus|facility) in|(?:enters?|entering|entry into) (?:the )?[A-Z][\w.-]+ (?:market|with)|coming to|slated for|(?:to|will|would|could) (?:house|host|feature|include) (?:a |an |up to )?(?:\d+[ -]?(?:mw|gw)|data ?cent(?:er|re)|campus))\b/i], | |
| 336 | +]; | |
| 337 | + | |
| 338 | +/** Verbs / nouns that describe physical development (build, construct, break ground, expand, permit, approve, file, acquire land, energize…). */ | |
| 339 | +export const DEVELOPMENT_VERB_RE = /\b(build|builds|building|built|construct\w*|develop\w*|break\w* ground|broke ground|groundbreaking|expan\w+|open|opens|opening|opened|permit\w*|approv\w+|file[sd]?|filing|rezon\w+|acquire[sd]? (?:\d[\d,.]*[ -]?(?:acre|hectare)s?|land|a site|the site|a parcel)|land (?:purchase|acquisition)|energi[sz]\w+|deliver\w*|commission\w*|top\w* out|phase|campus|data ?cent(?:er|re)s?|facility|site plan|planning application|zoning)\b/i; | |
| 340 | + | |
| 341 | +export interface ProjectClassInput { | |
| 342 | + title: string | null | undefined; | |
| 343 | + /** the lead of the article (first ~600 characters) */ | |
| 344 | + lead?: string | null; | |
| 345 | + /** an explicit project / facility name was found */ | |
| 346 | + hasExplicitName?: boolean; | |
| 347 | + /** a recognised operator is the headline subject or the primary operator */ | |
| 348 | + hasOperator?: boolean; | |
| 349 | + /** a city / region / country was found in the title or lead */ | |
| 350 | + hasLocation?: boolean; | |
| 351 | + /** the title-derived status (pipeline / operational / null) */ | |
| 352 | + status?: string | null; | |
| 353 | +} | |
| 354 | + | |
| 355 | +export interface ProjectClassification { | |
| 356 | + class: ProjectClass; | |
| 357 | + /** true when the class may create or materially modify a project */ | |
| 358 | + physical: boolean; | |
| 359 | + /** true when the class may attach an event to an identifiable project / company */ | |
| 360 | + associated: boolean; | |
| 361 | + /** evidence threshold for creating a project: (explicit name or operator) + location + development verb */ | |
| 362 | + evidence: { name: boolean; operator: boolean; location: boolean; verb: boolean; strength: "strong" | "weak" | "none" }; | |
| 363 | + /** true only for physical classes with strong evidence */ | |
| 364 | + mayCreateProject: boolean; | |
| 365 | + reasons: string[]; | |
| 366 | +} | |
| 367 | + | |
| 368 | +/** | |
| 369 | + * Classify what an announcement is about BEFORE anything becomes a project. The title decides; the lead breaks ties when | |
| 370 | + * the title is generic ("Company announces …"). Order of the rules = precedence: people, market research and | |
| 371 | + * sustainability news are vetoed first, then money / deals / customers, then physical development classes. | |
| 372 | + */ | |
| 373 | +export function classifyProjectEvent(i: ProjectClassInput): ProjectClassification { | |
| 374 | + const title = (i.title ?? "").replace(/\s+/g, " ").trim(); | |
| 375 | + const lead = (i.lead ?? "").replace(/\s+/g, " ").slice(0, 600); | |
| 376 | + const reasons: string[] = []; | |
| 377 | + let cls: ProjectClass = "UNKNOWN"; | |
| 378 | + const physicalTitle = ["CONSTRUCTION_START", "EXPANSION", "NEW_BUILD", "PERMIT", "LAND_ACQUISITION", "GRID_CONNECTION"].some((c) => CLASS_RULES.find(([k]) => k === c)![1].test(title)); | |
| 379 | + for (const [k, re] of CLASS_RULES) { | |
| 380 | + if (re.test(title)) { | |
| 381 | + // a money / deal / partnership headline that also says "to build 300MW campus" is a build headline | |
| 382 | + if (["FINANCING", "ACQUISITION", "CUSTOMER_AGREEMENT", "PARTNERSHIP", "POWER_AGREEMENT", "PRODUCT_NEWS", "SUSTAINABILITY"].includes(k) && physicalTitle && /\b(to build|will build|to develop|to construct|breaks? ground|broke ground|groundbreaking|new (?:\d+[ -]?(?:mw|gw) )?(?:data ?cent(?:er|re)|campus)|\d+[ -]?(?:mw|gw) (?:data ?cent(?:er|re)|campus)|expansion|expands? (?:its|the) (?:campus|data ?cent(?:er|re)))\b/i.test(title)) { | |
| 383 | + reasons.push(`title:${k}+build→physical`); | |
| 384 | + continue; | |
| 385 | + } | |
| 386 | + cls = k; | |
| 387 | + reasons.push(`title:${k}`); | |
| 388 | + break; | |
| 389 | + } | |
| 390 | + } | |
| 391 | + if (cls === "UNKNOWN" && lead) { | |
| 392 | + for (const [k, re] of CLASS_RULES) { | |
| 393 | + // the lead may only add physical / associated classes — a people or market veto must be visible in the headline | |
| 394 | + if (["EXECUTIVE_APPOINTMENT", "GENERAL_COMPANY_NEWS", "SUSTAINABILITY", "PRODUCT_NEWS"].includes(k)) continue; | |
| 395 | + if (re.test(lead)) { cls = k; reasons.push(`lead:${k}`); break; } | |
| 396 | + } | |
| 397 | + } | |
| 398 | + if (cls === "UNKNOWN" && i.status && /^(announced|proposed|rumored)$/.test(i.status) && /\b(data ?cent(?:er|re)|campus|facility)\b/i.test(`${title} ${lead}`)) { cls = "NEW_BUILD"; reasons.push("status:announced→NEW_BUILD"); } | |
| 399 | + if (cls === "UNKNOWN" && i.status === "under_construction") { cls = "CONSTRUCTION_START"; reasons.push("status:construction"); } | |
| 400 | + if (cls === "UNKNOWN" && (i.status === "permitting" || i.status === "approved")) { cls = "PERMIT"; reasons.push(`status:${i.status}`); } | |
| 401 | + if (cls === "UNKNOWN" && i.status === "expansion") { cls = "EXPANSION"; reasons.push("status:expansion"); } | |
| 402 | + | |
| 403 | + const verb = DEVELOPMENT_VERB_RE.test(title) || (!!lead && DEVELOPMENT_VERB_RE.test(lead)); | |
| 404 | + const name = !!i.hasExplicitName; | |
| 405 | + const operator = !!i.hasOperator; | |
| 406 | + const location = !!i.hasLocation; | |
| 407 | + const strength: ProjectClassification["evidence"]["strength"] = (name || operator) && location && verb ? "strong" : (name || operator || location) && verb ? "weak" : "none"; | |
| 408 | + const physical = PHYSICAL_CLASSES.has(cls); | |
| 409 | + const associated = ASSOCIATED_CLASSES.has(cls); | |
| 410 | + return { class: cls, physical, associated, evidence: { name, operator, location, verb, strength }, mayCreateProject: physical && strength === "strong", reasons }; | |
| 411 | +} | |
| 412 | + | |
| 413 | +// ─── project lifecycle state machine ─────────────────────────────────────────────────────────────── | |
| 414 | +export const PROJECT_STAGES = ["rumored", "proposed", "announced", "permitting", "approved", "under_construction", "partially_operational", "operational", "delayed", "cancelled"] as const; | |
| 415 | +export type ProjectStage = (typeof PROJECT_STAGES)[number]; | |
| 416 | +const STAGE_ORDER: Record<string, number> = { rumored: 0, proposed: 1, announced: 2, permitting: 3, approved: 4, under_construction: 5, partially_operational: 6, operational: 7, expansion: 7 }; | |
| 417 | + | |
| 418 | +export type TransitionVerdict = "forward" | "side" | "resume" | "same" | "backward" | "invalid"; | |
| 419 | + | |
| 420 | +/** | |
| 421 | + * Allowed lifecycle transitions. Forward moves along the pipeline are always fine; `delayed` / `cancelled` are side | |
| 422 | + * branches reachable from any pre-operational stage; leaving `delayed` resumes the pipeline; a backward move | |
| 423 | + * (operational → announced) is rejected unless the incoming source outranks the stored one (caller decides). | |
| 424 | + */ | |
| 425 | +export function projectTransition(from: string | null | undefined, to: string | null | undefined): TransitionVerdict { | |
| 426 | + if (!to) return "invalid"; | |
| 427 | + if (!from || from === "unknown") return "forward"; | |
| 428 | + if (from === to) return "same"; | |
| 429 | + if (to === "delayed" || to === "cancelled") return from === "operational" ? "invalid" : "side"; | |
| 430 | + if (from === "cancelled") return to === "announced" || to === "proposed" || to === "rumored" ? "resume" : "invalid"; | |
| 431 | + if (from === "delayed") return "resume"; | |
| 432 | + const a = STAGE_ORDER[from], b = STAGE_ORDER[to]; | |
| 433 | + if (a == null || b == null) return "invalid"; | |
| 434 | + return b > a ? "forward" : "backward"; | |
| 435 | +} | |
| 436 | + | |
| 437 | +// ─── AI evidence levels ──────────────────────────────────────────────────────────────────────────── | |
| 438 | +export const AI_EVIDENCE_LEVELS = ["confirmed", "likely", "associated", "unknown"] as const; | |
| 439 | +export type AiEvidence = (typeof AI_EVIDENCE_LEVELS)[number]; | |
| 440 | +const AI_CONFIRMED_RE = /\b(ai (?:data ?cent(?:er|re)|campus|factory|factories|infrastructure|cluster|supercomputer|compute (?:campus|facility|site)|training (?:facility|site|campus))|(?:purpose-?built|dedicated|designed) (?:for|to) (?:ai|artificial intelligence|accelerated computing|gpu|hpc|high-performance computing|machine learning|large language models?|llms?|training)|hpc (?:data ?cent(?:er|re)|campus|facility|cluster)|supercomput(?:er|ing) (?:cent(?:er|re)|facility|campus)|gpu (?:cluster|campus|farm|cloud|data ?cent(?:er|re)|hosting)|accelerated computing (?:facility|campus|data ?cent(?:er|re)|infrastructure)|exascale|(?:nvidia|blackwell|hopper|gb\d{3}|h\d{3}|b\d{3}) (?:gpus?|systems?|superchips?|clusters?))\b/i; | |
| 441 | +const AI_LIKELY_RE = /\b(ai-?ready|ai-?optimi[sz]ed|high[- ]density|liquid[- ]cool\w*|direct-to-chip|immersion cool\w*|rear-door heat exchangers?|(?:\d{2,3}|1\d{2}) ?kw (?:per |\/ ?)rack|gpu|accelerated computing|hpc|machine learning|inference|training clusters?|sovereign ai|ai workloads?)\b/i; | |
| 442 | +const AI_ASSOCIATED_RE = /\b(openai|anthropic|xai|coreweave|lambda labs|crusoe|nebius|nscale|fluidstack|voltage park|stargate|ai (?:tenant|customer|lease|demand|boom|workloads?)|for ai|ai and cloud|cloud and ai)\b/i; | |
| 443 | + | |
| 444 | +/** Evidence level that a facility / project is AI/HPC infrastructure. One "AI" keyword never confirms it. */ | |
| 445 | +export function classifyAiEvidence(text: string | null | undefined): { level: AiEvidence; evidence: string | null } { | |
| 446 | + if (!text) return { level: "unknown", evidence: null }; | |
| 447 | + const s = text.replace(/\s+/g, " "); | |
| 448 | + const c = AI_CONFIRMED_RE.exec(s); | |
| 449 | + if (c) return { level: "confirmed", evidence: c[0] }; | |
| 450 | + const l = AI_LIKELY_RE.exec(s); | |
| 451 | + if (l) return { level: "likely", evidence: l[0] }; | |
| 452 | + const a = AI_ASSOCIATED_RE.exec(s); | |
| 453 | + if (a) return { level: "associated", evidence: a[0] }; | |
| 454 | + return { level: "unknown", evidence: null }; | |
| 455 | +} | |
| 456 | + | |
| 457 | +// ─── event significance ──────────────────────────────────────────────────────────────────────────── | |
| 458 | +export type Significance = "major" | "medium" | "minor"; | |
| 459 | +export function significanceBand(score: number): Significance { return score >= 75 ? "major" : score >= 45 ? "medium" : "minor"; } | |
modified
packages/core/src/ids.ts
+5 −0
@@ -18,6 +18,11 @@ export const ID_PREFIXES = { | ||
| 18 | 18 | match: "mtc", |
| 19 | 19 | provenance: "prv", |
| 20 | 20 | ranking: "rnk", |
| 21 | + claim: "clm", | |
| 22 | + flag: "flg", | |
| 23 | + constraint: "grd", | |
| 24 | + watch: "wtc", | |
| 25 | + cluster: "cls", | |
| 21 | 26 | } as const; |
| 22 | 27 | export type IdKind = keyof typeof ID_PREFIXES; |
| 23 | 28 | |
modified
packages/core/src/index.browser.ts
+1 −0
@@ -5,3 +5,4 @@ export * from "./geo.js"; | ||
| 5 | 5 | export * from "./confidence.js"; |
| 6 | 6 | export * from "./diff.js"; |
| 7 | 7 | export * from "./api-types.js"; |
| 8 | +export * from "./claims.js"; | |
modified
packages/core/src/index.ts
+1 −0
@@ -7,3 +7,4 @@ export * from "./confidence.js"; | ||
| 7 | 7 | export * from "./diff.js"; |
| 8 | 8 | export * from "./ssrf.js"; |
| 9 | 9 | export * from "./api-types.js"; |
| 10 | +export * from "./claims.js"; | |
added
packages/core/src/normalize.test.ts
+68 −0
@@ -0,0 +1,68 @@ | ||
| 1 | +import { describe, expect, it } from "vitest"; | |
| 2 | +import { parseMw, parseAllMw, parseMoney, parsePartialDate, inferStatus, normalizeStatus, countryFromText, parseAreaSqm } from "./normalize.js"; | |
| 3 | + | |
| 4 | +describe("parseMw", () => { | |
| 5 | + it("parses plain figures and units", () => { | |
| 6 | + expect(parseMw("36 MW")).toBe(36); | |
| 7 | + expect(parseMw("36MW")).toBe(36); | |
| 8 | + expect(parseMw("120 megawatts")).toBe(120); | |
| 9 | + expect(parseMw("1.2 GW")).toBe(1200); | |
| 10 | + expect(parseMw("36,000 kW")).toBe(36); | |
| 11 | + expect(parseMw("40+MW")).toBe(40); | |
| 12 | + }); | |
| 13 | + it("treats a single dot followed by three digits as a decimal (DataBank 0.779MW regression)", () => { | |
| 14 | + expect(parseMw("0.779MW Critical IT Load")).toBe(0.779); | |
| 15 | + expect(parseMw("1.250 MW")).toBe(1.25); | |
| 16 | + expect(parseMw("2.500MW")).toBe(2.5); | |
| 17 | + }); | |
| 18 | + it("still reads European thousands with two or more groups", () => { | |
| 19 | + expect(parseMw("1.500.000 kW")).toBe(1500); | |
| 20 | + }); | |
| 21 | + it("reads a comma decimal", () => { | |
| 22 | + expect(parseMw("1,2 GW")).toBe(1200); | |
| 23 | + }); | |
| 24 | + it("returns null without a unit", () => { | |
| 25 | + expect(parseMw("300 million")).toBeNull(); | |
| 26 | + expect(parseMw(null)).toBeNull(); | |
| 27 | + }); | |
| 28 | +}); | |
| 29 | + | |
| 30 | +describe("parseAllMw", () => { | |
| 31 | + it("dedupes and sorts descending, drops absurd figures", () => { | |
| 32 | + expect(parseAllMw("a 36 MW hall in a 300MW campus, 36 MW again, 500,000 MW nonsense")).toEqual([300, 36]); | |
| 33 | + }); | |
| 34 | +}); | |
| 35 | + | |
| 36 | +describe("parseMoney", () => { | |
| 37 | + it("parses currencies and multipliers", () => { | |
| 38 | + expect(parseMoney("$1.2 billion")).toEqual({ amount: 1_200_000_000, currency: "USD" }); | |
| 39 | + expect(parseMoney("€800m")).toEqual({ amount: 800_000_000, currency: "EUR" }); | |
| 40 | + expect(parseMoney("£2bn")).toEqual({ amount: 2_000_000_000, currency: "GBP" }); | |
| 41 | + expect(parseMoney("US$450 million")).toEqual({ amount: 450_000_000, currency: "USD" }); | |
| 42 | + }); | |
| 43 | +}); | |
| 44 | + | |
| 45 | +describe("parsePartialDate", () => { | |
| 46 | + it("handles partial dates", () => { | |
| 47 | + expect(parsePartialDate("Q2 2027")).toBe("2027-Q2"); | |
| 48 | + expect(parsePartialDate("June 2027")).toBe("2027-06"); | |
| 49 | + expect(parsePartialDate("2027")).toBe("2027"); | |
| 50 | + expect(parsePartialDate("15 June 2027")).toBe("2027-06-15"); | |
| 51 | + }); | |
| 52 | +}); | |
| 53 | + | |
| 54 | +describe("status", () => { | |
| 55 | + it("infers and normalizes statuses", () => { | |
| 56 | + expect(inferStatus("breaks ground on new campus")).toBe("under_construction"); | |
| 57 | + expect(inferStatus("planning application submitted")).toBe("permitting"); | |
| 58 | + expect(normalizeStatus("Under Construction")).toBe("under_construction"); | |
| 59 | + expect(normalizeStatus("coming soon")).toBe("announced"); | |
| 60 | + }); | |
| 61 | +}); | |
| 62 | + | |
| 63 | +describe("misc", () => { | |
| 64 | + it("country from text and areas", () => { | |
| 65 | + expect(countryFromText("a campus in the Netherlands")).toBe("NL"); | |
| 66 | + expect(parseAreaSqm("8,170 Square Feet")).toBe(759.0); | |
| 67 | + }); | |
| 68 | +}); | |
modified
packages/core/src/normalize.ts
+4 −2
@@ -123,9 +123,11 @@ export function parseMw(text: string | null | undefined): number | null { | ||
| 123 | 123 | const m = s.match(/(\d{1,3}(?:[.,]\d{3})+|\d+(?:[.,]\d+)?)\+?\s*(gigawatts?|gw|megawatts?|mw|kilowatts?|kw)\b/i); |
| 124 | 124 | if (!m) return null; |
| 125 | 125 | let num = m[1]!; |
| 126 | − // "36,000" → 36000 ; "1,2" (fr decimal) → 1.2 ; "1.2" → 1.2 | |
| 126 | + // "36,000" → 36000 ; "1,2" (fr decimal) → 1.2 ; "1.2" → 1.2 ; "0.779" → 0.779 (a decimal, never 779) ; | |
| 127 | + // "1.500.000" (two dot groups, European thousands) → 1500000. A single dot followed by three digits is a | |
| 128 | + // decimal in every English-language capacity figure ("0.779MW", "1.250 MW") — DataBank MSP4 was stored as 779 MW. | |
| 127 | 129 | if (/^\d{1,3}(,\d{3})+$/.test(num)) num = num.replace(/,/g, ""); |
| 128 | − else if (/^\d{1,3}(\.\d{3})+$/.test(num) && !/^\d+\.\d{1,2}$/.test(num)) num = num.replace(/\./g, ""); | |
| 130 | + else if (/^\d{1,3}(\.\d{3}){2,}$/.test(num)) num = num.replace(/\./g, ""); | |
| 129 | 131 | else num = num.replace(",", "."); |
| 130 | 132 | const v = Number(num); |
| 131 | 133 | if (!Number.isFinite(v)) return null; |
modified
packages/core/src/types.ts
+42 −1
@@ -127,6 +127,17 @@ export const EVENT_TYPES = [ | ||
| 127 | 127 | "incident", |
| 128 | 128 | "page_changed", |
| 129 | 129 | "news", |
| 130 | + // infrastructure graph (2026-09) | |
| 131 | + "land_acquired", | |
| 132 | + "grid_connection", | |
| 133 | + "grid_constraint", | |
| 134 | + "utility_event", | |
| 135 | + "operator_expansion", | |
| 136 | + "customer_agreement", | |
| 137 | + "partnership", | |
| 138 | + "executive_change", | |
| 139 | + "project_delayed", | |
| 140 | + "project_cancelled", | |
| 130 | 141 | ] as const; |
| 131 | 142 | export type EventType = (typeof EVENT_TYPES)[number]; |
| 132 | 143 | |
@@ -207,6 +218,16 @@ export interface NormalizedFacility { | ||
| 207 | 218 | description?: string | null; |
| 208 | 219 | /** per-field provenance overrides; default provenance applies to all other fields */ |
| 209 | 220 | facts?: Fact[]; |
| 221 | + /** supporting sentence per numeric field (itCapacityMw, totalPowerMw, plannedPowerMw…) — required for prose-extracted figures */ | |
| 222 | + claimContext?: Record<string, string>; | |
| 223 | + /** what this record is: one building, a facility, or a campus (containment-aware aggregation) */ | |
| 224 | + recordScope?: "building" | "facility" | "campus" | null; | |
| 225 | + /** name / key of the campus-level facility record this building belongs to */ | |
| 226 | + parentFacilityKey?: string | null; | |
| 227 | + aiEvidence?: "confirmed" | "likely" | "associated" | "unknown" | null; | |
| 228 | + developerName?: string | null; | |
| 229 | + landownerName?: string | null; | |
| 230 | + tenantNames?: string[]; | |
| 210 | 231 | provenance: Provenance; |
| 211 | 232 | } |
| 212 | 233 | |
@@ -292,6 +313,24 @@ export interface NormalizedProject { | ||
| 292 | 313 | sourceUrl?: string | null; |
| 293 | 314 | timeline?: Array<{ date: string; type: EventType | string; description: string; url?: string }>; |
| 294 | 315 | externalIds?: Record<string, string | number>; |
| 316 | + /** classification of the announcement (packages/core claims.ts) — non-physical classes never create a project */ | |
| 317 | + projectClass?: string | null; | |
| 318 | + evidenceLevel?: "strong" | "weak" | "none" | null; | |
| 319 | + capacityScope?: string | null; | |
| 320 | + capacitySemantics?: string | null; | |
| 321 | + investmentScope?: string | null; | |
| 322 | + investmentSemantics?: string | null; | |
| 323 | + investmentCurrency?: string | null; | |
| 324 | + investmentOriginal?: number | null; | |
| 325 | + /** supporting sentence per numeric field (plannedMw, investmentUsd) */ | |
| 326 | + claimContext?: Record<string, string>; | |
| 327 | + aiEvidence?: "confirmed" | "likely" | "associated" | "unknown" | null; | |
| 328 | + developerName?: string | null; | |
| 329 | + tenantName?: string | null; | |
| 330 | + campusName?: string | null; | |
| 331 | + constructionStartedOn?: string | null; | |
| 332 | + approvedOn?: string | null; | |
| 333 | + permitFiledOn?: string | null; | |
| 295 | 334 | provenance: Provenance; |
| 296 | 335 | } |
| 297 | 336 | |
@@ -304,7 +343,9 @@ export interface NormalizedNewsEvent { | ||
| 304 | 343 | summary?: string | null; |
| 305 | 344 | pageType?: PageType; |
| 306 | 345 | eventType?: EventType; |
| 307 | − mentions?: { operators?: string[]; countriesIso2?: string[]; cities?: string[]; mw?: number[]; facilities?: string[] }; | |
| 346 | + mentions?: { operators?: string[]; countriesIso2?: string[]; cities?: string[]; mw?: number[]; facilities?: string[]; metro?: string | null }; | |
| 347 | + projectClass?: string | null; | |
| 348 | + isAi?: boolean | null; | |
| 308 | 349 | provenance: Provenance; |
| 309 | 350 | } |
| 310 | 351 | |
added
packages/db/migrations/0003_claims_quality_graph.sql
+202 −0
@@ -0,0 +1,202 @@ | ||
| 1 | +-- 0003 — claim-first data layer, quality flags, campus containment, capacity ontology, run tracking, snapshots. | |
| 2 | +-- Idempotent (IF NOT EXISTS everywhere) so it can be re-applied on a database created before this file existed. | |
| 3 | + | |
| 4 | +-- ─── claims ──────────────────────────────────────────────────────────────────────────────────────── | |
| 5 | +CREATE TABLE IF NOT EXISTS "claims" ( | |
| 6 | + "id" text PRIMARY KEY, | |
| 7 | + "subject_type" text NOT NULL, -- facility | campus | project | operator | market | country | |
| 8 | + "subject_id" text NOT NULL, | |
| 9 | + "predicate" text NOT NULL, -- it_capacity_mw | planned_power_mw | project_investment_usd | status | … | |
| 10 | + "value" double precision, | |
| 11 | + "value_text" text, | |
| 12 | + "unit" text, -- MW | USD | … | |
| 13 | + "scope" text NOT NULL DEFAULT 'unknown', -- building | facility | campus | metro | country | portfolio | company | unknown | |
| 14 | + "scope_reason" text, | |
| 15 | + "source_id" text NOT NULL, | |
| 16 | + "connector_id" text NOT NULL, | |
| 17 | + "document_id" text, | |
| 18 | + "url" text NOT NULL, | |
| 19 | + "published_at" text, -- partial date as published | |
| 20 | + "retrieved_at" timestamp with time zone NOT NULL DEFAULT now(), | |
| 21 | + "confidence" text NOT NULL DEFAULT 'moderate', | |
| 22 | + "is_estimate" boolean NOT NULL DEFAULT false, | |
| 23 | + "authority_tier" text NOT NULL DEFAULT 'D', | |
| 24 | + "evidence_text" text, | |
| 25 | + "evidence_start" integer, | |
| 26 | + "evidence_end" integer, | |
| 27 | + "parser_name" text, | |
| 28 | + "parser_version" text, | |
| 29 | + "run_id" text, | |
| 30 | + "status" text NOT NULL DEFAULT 'current', -- current | superseded | rejected | review | unscoped | |
| 31 | + "rejection_reason" text, | |
| 32 | + "first_observed" timestamp with time zone NOT NULL DEFAULT now(), | |
| 33 | + "last_observed" timestamp with time zone NOT NULL DEFAULT now(), | |
| 34 | + "created_at" timestamp with time zone NOT NULL DEFAULT now() | |
| 35 | +); | |
| 36 | +CREATE UNIQUE INDEX IF NOT EXISTS "claims_uq" ON "claims" ("subject_type", "subject_id", "predicate", "source_id", "url", COALESCE("value", 0), COALESCE("value_text", '')); | |
| 37 | +CREATE INDEX IF NOT EXISTS "claims_subject_idx" ON "claims" ("subject_type", "subject_id", "predicate"); | |
| 38 | +CREATE INDEX IF NOT EXISTS "claims_status_idx" ON "claims" ("status"); | |
| 39 | +CREATE INDEX IF NOT EXISTS "claims_run_idx" ON "claims" ("run_id"); | |
| 40 | +CREATE INDEX IF NOT EXISTS "claims_document_idx" ON "claims" ("document_id"); | |
| 41 | + | |
| 42 | +-- ─── quality flags ───────────────────────────────────────────────────────────────────────────────── | |
| 43 | +CREATE TABLE IF NOT EXISTS "quality_flags" ( | |
| 44 | + "id" text PRIMARY KEY, | |
| 45 | + "entity_type" text NOT NULL, | |
| 46 | + "entity_id" text NOT NULL, | |
| 47 | + "claim_id" text, | |
| 48 | + "code" text NOT NULL, -- mw_single_site_gt_1000 | scope_company | inv_single_site_gt_50b | project_false_positive | duplicate | … | |
| 49 | + "severity" text NOT NULL DEFAULT 'warn', -- info | warn | critical | |
| 50 | + "field" text, | |
| 51 | + "message" text NOT NULL, | |
| 52 | + "details" jsonb, | |
| 53 | + "priority" integer NOT NULL DEFAULT 0, -- review priority (impact-weighted) | |
| 54 | + "status" text NOT NULL DEFAULT 'open', -- open | resolved | dismissed | |
| 55 | + "resolution" text, | |
| 56 | + "resolved_by" text, | |
| 57 | + "resolved_at" timestamp with time zone, | |
| 58 | + "run_id" text, | |
| 59 | + "dedupe_key" text NOT NULL, | |
| 60 | + "created_at" timestamp with time zone NOT NULL DEFAULT now(), | |
| 61 | + "updated_at" timestamp with time zone NOT NULL DEFAULT now() | |
| 62 | +); | |
| 63 | +CREATE UNIQUE INDEX IF NOT EXISTS "quality_flags_dedupe_uq" ON "quality_flags" ("dedupe_key"); | |
| 64 | +CREATE INDEX IF NOT EXISTS "quality_flags_entity_idx" ON "quality_flags" ("entity_type", "entity_id"); | |
| 65 | +CREATE INDEX IF NOT EXISTS "quality_flags_open_idx" ON "quality_flags" ("status", "priority" DESC) WHERE "status" = 'open'; | |
| 66 | +CREATE INDEX IF NOT EXISTS "quality_flags_code_idx" ON "quality_flags" ("code"); | |
| 67 | + | |
| 68 | +-- ─── facilities: containment, capacity ontology, AI evidence ─────────────────────────────────────── | |
| 69 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "parent_facility_id" text REFERENCES "facilities"("id"); | |
| 70 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "record_scope" text NOT NULL DEFAULT 'facility'; -- building | facility | campus | |
| 71 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "ai_evidence" text NOT NULL DEFAULT 'unknown'; -- confirmed | likely | associated | unknown | |
| 72 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "utility_capacity_mw" double precision; | |
| 73 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "grid_connection_mw" double precision; | |
| 74 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "ultimate_campus_mw" double precision; | |
| 75 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "capacity_scope" text; -- scope of the displayed MW figure | |
| 76 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "capacity_semantics" text; -- predicate behind the displayed MW figure | |
| 77 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "developer_id" text REFERENCES "operators"("id"); | |
| 78 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "landowner_id" text REFERENCES "operators"("id"); | |
| 79 | +ALTER TABLE "facilities" ADD COLUMN IF NOT EXISTS "review_priority" integer NOT NULL DEFAULT 0; | |
| 80 | +CREATE INDEX IF NOT EXISTS "facilities_parent_idx" ON "facilities" ("parent_facility_id"); | |
| 81 | +CREATE INDEX IF NOT EXISTS "facilities_campus_idx" ON "facilities" ("campus_id"); | |
| 82 | +CREATE INDEX IF NOT EXISTS "facilities_merged_idx" ON "facilities" ("merged_into") WHERE "merged_into" IS NOT NULL; | |
| 83 | +CREATE INDEX IF NOT EXISTS "facilities_owner_idx" ON "facilities" ("owner_id"); | |
| 84 | +CREATE INDEX IF NOT EXISTS "facilities_ai_idx" ON "facilities" ("ai_evidence") WHERE "ai_evidence" <> 'unknown'; | |
| 85 | +CREATE INDEX IF NOT EXISTS "facilities_opened_idx" ON "facilities" ("opened_on") WHERE "opened_on" IS NOT NULL; | |
| 86 | + | |
| 87 | +-- ─── projects: classification, evidence, lifecycle, geocoding, scopes ───────────────────────────── | |
| 88 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "project_class" text; -- NEW_BUILD | EXPANSION | … | UNKNOWN | |
| 89 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "evidence_level" text; -- strong | weak | none | |
| 90 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "merged_into" text; | |
| 91 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "ai_evidence" text NOT NULL DEFAULT 'unknown'; | |
| 92 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "capacity_scope" text; | |
| 93 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "capacity_semantics" text; | |
| 94 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "investment_scope" text; | |
| 95 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "investment_semantics" text; | |
| 96 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "developer_id" text REFERENCES "operators"("id"); | |
| 97 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "tenant_id" text REFERENCES "operators"("id"); | |
| 98 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "campus_id" text REFERENCES "campuses"("id"); | |
| 99 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "construction_started_on" text; | |
| 100 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "approved_on" text; | |
| 101 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "permit_filed_on" text; | |
| 102 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "opened_on" text; | |
| 103 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "review_priority" integer NOT NULL DEFAULT 0; | |
| 104 | +ALTER TABLE "projects" ADD COLUMN IF NOT EXISTS "hidden" boolean NOT NULL DEFAULT false; -- false positives kept for audit, never listed | |
| 105 | +CREATE INDEX IF NOT EXISTS "projects_class_idx" ON "projects" ("project_class"); | |
| 106 | +CREATE INDEX IF NOT EXISTS "projects_merged_idx" ON "projects" ("merged_into") WHERE "merged_into" IS NOT NULL; | |
| 107 | +CREATE INDEX IF NOT EXISTS "projects_live_idx" ON "projects" ("status", "country_iso2") WHERE "merged_into" IS NULL AND NOT "hidden"; | |
| 108 | +CREATE INDEX IF NOT EXISTS "projects_latlng_idx" ON "projects" ("lat", "lng") WHERE "lat" IS NOT NULL; | |
| 109 | +CREATE INDEX IF NOT EXISTS "projects_metro_idx" ON "projects" ("metro_id"); | |
| 110 | + | |
| 111 | +-- ─── provenance / versions: run tracking + scope ────────────────────────────────────────────────── | |
| 112 | +ALTER TABLE "provenance" ADD COLUMN IF NOT EXISTS "run_id" text; | |
| 113 | +ALTER TABLE "provenance" ADD COLUMN IF NOT EXISTS "scope" text; | |
| 114 | +ALTER TABLE "provenance" ADD COLUMN IF NOT EXISTS "is_winner" boolean NOT NULL DEFAULT false; -- the observation backing the displayed column value | |
| 115 | +CREATE INDEX IF NOT EXISTS "provenance_current_idx" ON "provenance" ("entity_type", "entity_id", "field") WHERE "is_current"; | |
| 116 | +CREATE INDEX IF NOT EXISTS "provenance_run_idx" ON "provenance" ("run_id"); | |
| 117 | +ALTER TABLE "document_versions" ADD COLUMN IF NOT EXISTS "run_id" text; | |
| 118 | +ALTER TABLE "document_versions" ADD COLUMN IF NOT EXISTS "extractor_version" text; | |
| 119 | +CREATE INDEX IF NOT EXISTS "document_versions_run_idx" ON "document_versions" ("run_id"); | |
| 120 | +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "run_id" text; | |
| 121 | +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "cluster_id" text; -- same underlying announcement across outlets | |
| 122 | +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "evidence_count" integer NOT NULL DEFAULT 1; | |
| 123 | +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "is_ai" boolean NOT NULL DEFAULT false; | |
| 124 | +ALTER TABLE "events" ADD COLUMN IF NOT EXISTS "source_kind" text; | |
| 125 | +CREATE INDEX IF NOT EXISTS "events_significance_idx" ON "events" ("significance" DESC, "detected_at" DESC); | |
| 126 | +CREATE INDEX IF NOT EXISTS "events_project_idx" ON "events" ("project_id"); | |
| 127 | +CREATE INDEX IF NOT EXISTS "events_cluster_idx" ON "events" ("cluster_id"); | |
| 128 | +CREATE INDEX IF NOT EXISTS "events_run_idx" ON "events" ("run_id"); | |
| 129 | +CREATE INDEX IF NOT EXISTS "events_metro_idx" ON "events" ("metro_id"); | |
| 130 | + | |
| 131 | +-- ─── news items ──────────────────────────────────────────────────────────────────────────────────── | |
| 132 | +ALTER TABLE "news_items" ADD COLUMN IF NOT EXISTS "project_class" text; | |
| 133 | +ALTER TABLE "news_items" ADD COLUMN IF NOT EXISTS "cluster_id" text; | |
| 134 | +ALTER TABLE "news_items" ADD COLUMN IF NOT EXISTS "metro_id" text; | |
| 135 | +CREATE INDEX IF NOT EXISTS "news_items_operator_gin" ON "news_items" USING gin ("operator_ids"); | |
| 136 | +CREATE INDEX IF NOT EXISTS "news_items_cluster_idx" ON "news_items" ("cluster_id"); | |
| 137 | + | |
| 138 | +-- ─── connectors: quarantine + health ─────────────────────────────────────────────────────────────── | |
| 139 | +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "quarantine" boolean NOT NULL DEFAULT false; | |
| 140 | +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "consecutive_failures" integer NOT NULL DEFAULT 0; | |
| 141 | +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "blocked_since" timestamp with time zone; | |
| 142 | +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "last_discovered" integer; | |
| 143 | +ALTER TABLE "connectors" ADD COLUMN IF NOT EXISTS "priority_score" real; | |
| 144 | +ALTER TABLE "connector_runs" ADD COLUMN IF NOT EXISTS "quarantined" boolean NOT NULL DEFAULT false; | |
| 145 | + | |
| 146 | +-- ─── sources: licensing ──────────────────────────────────────────────────────────────────────────── | |
| 147 | +ALTER TABLE "sources" ADD COLUMN IF NOT EXISTS "redistribution" text; -- allowed | attribution | restricted | unknown | |
| 148 | +ALTER TABLE "sources" ADD COLUMN IF NOT EXISTS "attribution_required" boolean; | |
| 149 | + | |
| 150 | +-- ─── entity keys: scope the primary key by entity type ───────────────────────────────────────────── | |
| 151 | +DO $$ | |
| 152 | +BEGIN | |
| 153 | + IF EXISTS (SELECT 1 FROM pg_constraint WHERE conname = 'entity_keys_pkey' AND conrelid = 'entity_keys'::regclass | |
| 154 | + AND array_length(conkey, 1) = 1) THEN | |
| 155 | + ALTER TABLE "entity_keys" DROP CONSTRAINT "entity_keys_pkey"; | |
| 156 | + ALTER TABLE "entity_keys" ADD PRIMARY KEY ("key", "entity_type"); | |
| 157 | + END IF; | |
| 158 | +END $$; | |
| 159 | + | |
| 160 | +-- ─── daily snapshots for "as of" views and regression checks ─────────────────────────────────────── | |
| 161 | +CREATE TABLE IF NOT EXISTS "entity_snapshots" ( | |
| 162 | + "day" date NOT NULL, | |
| 163 | + "kind" text NOT NULL, -- global_totals | ranking | facility_status | project_stage | operator_totals | country_totals | |
| 164 | + "key" text NOT NULL, | |
| 165 | + "payload" jsonb NOT NULL, | |
| 166 | + "created_at" timestamp with time zone NOT NULL DEFAULT now(), | |
| 167 | + CONSTRAINT "entity_snapshots_pk" PRIMARY KEY ("day", "kind", "key") | |
| 168 | +); | |
| 169 | +CREATE INDEX IF NOT EXISTS "entity_snapshots_kind_idx" ON "entity_snapshots" ("kind", "key", "day"); | |
| 170 | + | |
| 171 | +-- ─── watchlists (private, cookie-scoped) ─────────────────────────────────────────────────────────── | |
| 172 | +CREATE TABLE IF NOT EXISTS "watchlists" ( | |
| 173 | + "id" text PRIMARY KEY, | |
| 174 | + "owner_token" text NOT NULL, | |
| 175 | + "entity_type" text NOT NULL, | |
| 176 | + "entity_id" text NOT NULL, | |
| 177 | + "created_at" timestamp with time zone NOT NULL DEFAULT now() | |
| 178 | +); | |
| 179 | +CREATE UNIQUE INDEX IF NOT EXISTS "watchlists_uq" ON "watchlists" ("owner_token", "entity_type", "entity_id"); | |
| 180 | + | |
| 181 | +-- ─── grid / power context (events already carry the type; this keeps market-level constraint notes) ─ | |
| 182 | +CREATE TABLE IF NOT EXISTS "grid_constraints" ( | |
| 183 | + "id" text PRIMARY KEY, | |
| 184 | + "metro_id" text REFERENCES "metros"("id"), | |
| 185 | + "country_iso2" text REFERENCES "countries"("iso2"), | |
| 186 | + "kind" text NOT NULL, -- moratorium | grid_delay | capacity_restriction | load_cap | new_transmission | new_substation | regulation | large_load_queue | |
| 187 | + "title" text NOT NULL, | |
| 188 | + "summary" text, | |
| 189 | + "effective_date" text, | |
| 190 | + "source_id" text, | |
| 191 | + "document_id" text, | |
| 192 | + "url" text NOT NULL, | |
| 193 | + "event_id" text, | |
| 194 | + "confidence" text NOT NULL DEFAULT 'moderate', | |
| 195 | + "created_at" timestamp with time zone NOT NULL DEFAULT now() | |
| 196 | +); | |
| 197 | +CREATE INDEX IF NOT EXISTS "grid_constraints_metro_idx" ON "grid_constraints" ("metro_id"); | |
| 198 | +CREATE INDEX IF NOT EXISTS "grid_constraints_country_idx" ON "grid_constraints" ("country_iso2"); | |
| 199 | + | |
| 200 | +-- ─── daily metrics time-series lookups ───────────────────────────────────────────────────────────── | |
| 201 | +CREATE INDEX IF NOT EXISTS "daily_metrics_series_idx" ON "daily_metrics" ("metric", "dim", "day"); | |
| 202 | +CREATE INDEX IF NOT EXISTS "entity_matches_created_fac_idx" ON "entity_matches" ((candidate->>'createdFacilityId')) WHERE "status" = 'pending'; | |
modified
packages/db/src/schema.ts
+174 −5
@@ -139,11 +139,24 @@ export const facilities = pgTable( | ||
| 139 | 139 | firstSeen: ts("first_seen").notNull().defaultNow(), |
| 140 | 140 | lastVerified: ts("last_verified"), |
| 141 | 141 | mergedInto: text("merged_into"), |
| 142 | + /** containment: a building row points at its campus row (aggregation never counts both) */ | |
| 143 | + parentFacilityId: text("parent_facility_id"), | |
| 144 | + recordScope: text("record_scope").notNull().default("facility"), // building | facility | campus | |
| 145 | + aiEvidence: text("ai_evidence").notNull().default("unknown"), // confirmed | likely | associated | unknown | |
| 146 | + utilityCapacityMw: doublePrecision("utility_capacity_mw"), | |
| 147 | + gridConnectionMw: doublePrecision("grid_connection_mw"), | |
| 148 | + ultimateCampusMw: doublePrecision("ultimate_campus_mw"), | |
| 149 | + capacityScope: text("capacity_scope"), | |
| 150 | + capacitySemantics: text("capacity_semantics"), | |
| 151 | + developerId: text("developer_id"), | |
| 152 | + landownerId: text("landowner_id"), | |
| 153 | + reviewPriority: integer("review_priority").notNull().default(0), | |
| 142 | 154 | createdAt: createdAt(), |
| 143 | 155 | updatedAt: updatedAt(), |
| 144 | 156 | }, |
| 145 | 157 | (t) => [ |
| 146 | 158 | index("facilities_country_idx").on(t.countryIso2), |
| 159 | + index("facilities_parent_idx").on(t.parentFacilityId), | |
| 147 | 160 | index("facilities_operator_idx").on(t.operatorId), |
| 148 | 161 | index("facilities_metro_idx").on(t.metroId), |
| 149 | 162 | index("facilities_status_idx").on(t.status), |
@@ -169,13 +182,13 @@ export const facilityAliases = pgTable( | ||
| 169 | 182 | export const entityKeys = pgTable( |
| 170 | 183 | "entity_keys", |
| 171 | 184 | { |
| 172 | − key: text("key").primaryKey(), | |
| 185 | + key: text("key").notNull(), | |
| 173 | 186 | entityType: text("entity_type").notNull(), |
| 174 | 187 | entityId: text("entity_id").notNull(), |
| 175 | 188 | connectorId: text("connector_id").notNull(), |
| 176 | 189 | createdAt: createdAt(), |
| 177 | 190 | }, |
| 178 | − (t) => [index("entity_keys_entity_idx").on(t.entityType, t.entityId)], | |
| 191 | + (t) => [primaryKey({ columns: [t.key, t.entityType] }), index("entity_keys_entity_idx").on(t.entityType, t.entityId)], | |
| 179 | 192 | ); |
| 180 | 193 | |
| 181 | 194 | export const cloudRegions = pgTable( |
@@ -275,10 +288,28 @@ export const projects = pgTable( | ||
| 275 | 288 | confidence: text("confidence").notNull().default("moderate"), |
| 276 | 289 | externalIds: jsonb("external_ids").$type<Record<string, string | number>>().notNull().default(sql`'{}'::jsonb`), |
| 277 | 290 | lastUpdate: ts("last_update").notNull().defaultNow(), |
| 291 | + projectClass: text("project_class"), // NEW_BUILD | EXPANSION | CONSTRUCTION_START | PERMIT | … | UNKNOWN | |
| 292 | + evidenceLevel: text("evidence_level"), // strong | weak | none | |
| 293 | + mergedInto: text("merged_into"), | |
| 294 | + aiEvidence: text("ai_evidence").notNull().default("unknown"), | |
| 295 | + capacityScope: text("capacity_scope"), | |
| 296 | + capacitySemantics: text("capacity_semantics"), | |
| 297 | + investmentScope: text("investment_scope"), | |
| 298 | + investmentSemantics: text("investment_semantics"), | |
| 299 | + developerId: text("developer_id"), | |
| 300 | + tenantId: text("tenant_id"), | |
| 301 | + campusId: text("campus_id"), | |
| 302 | + constructionStartedOn: text("construction_started_on"), | |
| 303 | + approvedOn: text("approved_on"), | |
| 304 | + permitFiledOn: text("permit_filed_on"), | |
| 305 | + openedOn: text("opened_on"), | |
| 306 | + reviewPriority: integer("review_priority").notNull().default(0), | |
| 307 | + /** false positives are hidden, never deleted (audit trail) */ | |
| 308 | + hidden: boolean("hidden").notNull().default(false), | |
| 278 | 309 | createdAt: createdAt(), |
| 279 | 310 | updatedAt: updatedAt(), |
| 280 | 311 | }, |
| 281 | − (t) => [index("projects_country_idx").on(t.countryIso2), index("projects_operator_idx").on(t.operatorId), index("projects_status_idx").on(t.status), index("projects_norm_idx").on(t.normalizedName)], | |
| 312 | + (t) => [index("projects_country_idx").on(t.countryIso2), index("projects_operator_idx").on(t.operatorId), index("projects_status_idx").on(t.status), index("projects_norm_idx").on(t.normalizedName), index("projects_class_idx").on(t.projectClass), index("projects_metro_idx").on(t.metroId)], | |
| 282 | 313 | ); |
| 283 | 314 | |
| 284 | 315 | export const projectTimeline = pgTable( |
@@ -310,6 +341,8 @@ export const sources = pgTable("sources", { | ||
| 310 | 341 | attribution: text("attribution"), |
| 311 | 342 | robotsAllowed: boolean("robots_allowed"), |
| 312 | 343 | notes: text("notes"), |
| 344 | + redistribution: text("redistribution"), // allowed | attribution | restricted | unknown | |
| 345 | + attributionRequired: boolean("attribution_required"), | |
| 313 | 346 | createdAt: createdAt(), |
| 314 | 347 | updatedAt: updatedAt(), |
| 315 | 348 | }); |
@@ -332,6 +365,12 @@ export const connectors = pgTable("connectors", { | ||
| 332 | 365 | lastError: text("last_error"), |
| 333 | 366 | nextRunAt: ts("next_run_at"), |
| 334 | 367 | stats: jsonb("stats").$type<Record<string, number>>().notNull().default(sql`'{}'::jsonb`), |
| 368 | + /** quarantine: runs extract and preview but publish nothing (after parser changes, HTML changes, suspicious spikes) */ | |
| 369 | + quarantine: boolean("quarantine").notNull().default(false), | |
| 370 | + consecutiveFailures: integer("consecutive_failures").notNull().default(0), | |
| 371 | + blockedSince: ts("blocked_since"), | |
| 372 | + lastDiscovered: integer("last_discovered"), | |
| 373 | + priorityScore: real("priority_score"), | |
| 335 | 374 | createdAt: createdAt(), |
| 336 | 375 | updatedAt: updatedAt(), |
| 337 | 376 | }); |
@@ -348,6 +387,7 @@ export const connectorRuns = pgTable( | ||
| 348 | 387 | stats: jsonb("stats").$type<Record<string, number>>().notNull().default(sql`'{}'::jsonb`), |
| 349 | 388 | error: text("error"), |
| 350 | 389 | log: jsonb("log").$type<Array<{ t: string; level: string; msg: string }>>().notNull().default(sql`'[]'::jsonb`), |
| 390 | + quarantined: boolean("quarantined").notNull().default(false), | |
| 351 | 391 | }, |
| 352 | 392 | (t) => [index("connector_runs_connector_idx").on(t.connectorId, t.startedAt)], |
| 353 | 393 | ); |
@@ -414,8 +454,10 @@ export const documentVersions = pgTable( | ||
| 414 | 454 | diffSummary: jsonb("diff_summary").$type<{ addedCount: number; removedCount: number; ratio: number; added: string[]; removed: string[] }>(), |
| 415 | 455 | detectedChanges: jsonb("detected_changes").$type<Array<Record<string, unknown>>>().notNull().default(sql`'[]'::jsonb`), |
| 416 | 456 | significance: integer("significance").notNull().default(0), |
| 457 | + runId: text("run_id"), | |
| 458 | + extractorVersion: text("extractor_version"), | |
| 417 | 459 | }, |
| 418 | − (t) => [index("document_versions_doc_idx").on(t.documentId, t.fetchedAt)], | |
| 460 | + (t) => [index("document_versions_doc_idx").on(t.documentId, t.fetchedAt), index("document_versions_run_idx").on(t.runId)], | |
| 419 | 461 | ); |
| 420 | 462 | |
| 421 | 463 | /** Per-field provenance for every entity value. */ |
@@ -440,9 +482,14 @@ export const provenance = pgTable( | ||
| 440 | 482 | extractorVersion: text("extractor_version"), |
| 441 | 483 | isCurrent: boolean("is_current").notNull().default(true), |
| 442 | 484 | note: text("note"), |
| 485 | + runId: text("run_id"), | |
| 486 | + scope: text("scope"), | |
| 487 | + /** the observation currently backing the displayed column value */ | |
| 488 | + isWinner: boolean("is_winner").notNull().default(false), | |
| 443 | 489 | }, |
| 444 | 490 | (t) => [ |
| 445 | 491 | index("provenance_entity_idx").on(t.entityType, t.entityId), |
| 492 | + index("provenance_run_idx").on(t.runId), | |
| 446 | 493 | index("provenance_source_idx").on(t.sourceId), |
| 447 | 494 | uniqueIndex("provenance_uq").on(t.entityType, t.entityId, t.field, t.sourceId, t.url), |
| 448 | 495 | ], |
@@ -472,9 +519,18 @@ export const events = pgTable( | ||
| 472 | 519 | metroId: text("metro_id"), |
| 473 | 520 | projectId: text("project_id"), |
| 474 | 521 | fingerprint: text("fingerprint").notNull(), |
| 522 | + runId: text("run_id"), | |
| 523 | + /** documents describing the same underlying announcement share a cluster id */ | |
| 524 | + clusterId: text("cluster_id"), | |
| 525 | + evidenceCount: integer("evidence_count").notNull().default(1), | |
| 526 | + isAi: boolean("is_ai").notNull().default(false), | |
| 527 | + sourceKind: text("source_kind"), | |
| 475 | 528 | }, |
| 476 | 529 | (t) => [ |
| 477 | 530 | uniqueIndex("events_fingerprint_uq").on(t.fingerprint), |
| 531 | + index("events_significance_idx").on(t.significance, t.detectedAt), | |
| 532 | + index("events_project_idx").on(t.projectId), | |
| 533 | + index("events_cluster_idx").on(t.clusterId), | |
| 478 | 534 | index("events_detected_idx").on(t.detectedAt), |
| 479 | 535 | index("events_entity_idx").on(t.entityType, t.entityId), |
| 480 | 536 | index("events_country_idx").on(t.countryIso2), |
@@ -502,9 +558,12 @@ export const newsItems = pgTable( | ||
| 502 | 558 | projectId: text("project_id"), |
| 503 | 559 | mw: doublePrecision("mw"), |
| 504 | 560 | significance: integer("significance").notNull().default(20), |
| 561 | + projectClass: text("project_class"), | |
| 562 | + clusterId: text("cluster_id"), | |
| 563 | + metroId: text("metro_id"), | |
| 505 | 564 | createdAt: createdAt(), |
| 506 | 565 | }, |
| 507 | − (t) => [index("news_items_published_idx").on(t.publishedAt)], | |
| 566 | + (t) => [index("news_items_published_idx").on(t.publishedAt), index("news_items_cluster_idx").on(t.clusterId)], | |
| 508 | 567 | ); |
| 509 | 568 | |
| 510 | 569 | /** Reconciliation queue: candidate records whose match to an existing facility was ambiguous. */ |
@@ -585,3 +644,113 @@ export const connectorState = pgTable( | ||
| 585 | 644 | }, |
| 586 | 645 | (t) => [primaryKey({ columns: [t.connectorId, t.key] })], |
| 587 | 646 | ); |
| 647 | + | |
| 648 | +/** Claim-first store: one figure asserted by one document about one subject, with scope, evidence and authority (packages/core claims.ts). */ | |
| 649 | +export const claims = pgTable( | |
| 650 | + "claims", | |
| 651 | + { | |
| 652 | + id: text("id").primaryKey(), | |
| 653 | + subjectType: text("subject_type").notNull(), | |
| 654 | + subjectId: text("subject_id").notNull(), | |
| 655 | + predicate: text("predicate").notNull(), | |
| 656 | + value: doublePrecision("value"), | |
| 657 | + valueText: text("value_text"), | |
| 658 | + unit: text("unit"), | |
| 659 | + scope: text("scope").notNull().default("unknown"), | |
| 660 | + scopeReason: text("scope_reason"), | |
| 661 | + sourceId: text("source_id").notNull(), | |
| 662 | + connectorId: text("connector_id").notNull(), | |
| 663 | + documentId: text("document_id"), | |
| 664 | + url: text("url").notNull(), | |
| 665 | + publishedAt: text("published_at"), | |
| 666 | + retrievedAt: ts("retrieved_at").notNull().defaultNow(), | |
| 667 | + confidence: text("confidence").notNull().default("moderate"), | |
| 668 | + isEstimate: boolean("is_estimate").notNull().default(false), | |
| 669 | + authorityTier: text("authority_tier").notNull().default("D"), | |
| 670 | + evidenceText: text("evidence_text"), | |
| 671 | + evidenceStart: integer("evidence_start"), | |
| 672 | + evidenceEnd: integer("evidence_end"), | |
| 673 | + parserName: text("parser_name"), | |
| 674 | + parserVersion: text("parser_version"), | |
| 675 | + runId: text("run_id"), | |
| 676 | + status: text("status").notNull().default("current"), // current | superseded | rejected | review | unscoped | |
| 677 | + rejectionReason: text("rejection_reason"), | |
| 678 | + firstObserved: ts("first_observed").notNull().defaultNow(), | |
| 679 | + lastObserved: ts("last_observed").notNull().defaultNow(), | |
| 680 | + createdAt: createdAt(), | |
| 681 | + }, | |
| 682 | + (t) => [index("claims_subject_idx").on(t.subjectType, t.subjectId, t.predicate), index("claims_status_idx").on(t.status), index("claims_run_idx").on(t.runId), index("claims_document_idx").on(t.documentId)], | |
| 683 | +); | |
| 684 | + | |
| 685 | +/** Deterministic data-quality flags (capacity / investment sanity, scope, duplicates, project false positives…). */ | |
| 686 | +export const qualityFlags = pgTable( | |
| 687 | + "quality_flags", | |
| 688 | + { | |
| 689 | + id: text("id").primaryKey(), | |
| 690 | + entityType: text("entity_type").notNull(), | |
| 691 | + entityId: text("entity_id").notNull(), | |
| 692 | + claimId: text("claim_id"), | |
| 693 | + code: text("code").notNull(), | |
| 694 | + severity: text("severity").notNull().default("warn"), // info | warn | critical | |
| 695 | + field: text("field"), | |
| 696 | + message: text("message").notNull(), | |
| 697 | + details: jsonb("details").$type<Record<string, unknown>>(), | |
| 698 | + priority: integer("priority").notNull().default(0), | |
| 699 | + status: text("status").notNull().default("open"), // open | resolved | dismissed | |
| 700 | + resolution: text("resolution"), | |
| 701 | + resolvedBy: text("resolved_by"), | |
| 702 | + resolvedAt: ts("resolved_at"), | |
| 703 | + runId: text("run_id"), | |
| 704 | + dedupeKey: text("dedupe_key").notNull(), | |
| 705 | + createdAt: createdAt(), | |
| 706 | + updatedAt: updatedAt(), | |
| 707 | + }, | |
| 708 | + (t) => [uniqueIndex("quality_flags_dedupe_uq").on(t.dedupeKey), index("quality_flags_entity_idx").on(t.entityType, t.entityId), index("quality_flags_code_idx").on(t.code)], | |
| 709 | +); | |
| 710 | + | |
| 711 | +/** Daily JSON snapshots (global totals, rankings, status / stage counts) for "as of" views and regression checks. */ | |
| 712 | +export const entitySnapshots = pgTable( | |
| 713 | + "entity_snapshots", | |
| 714 | + { | |
| 715 | + day: date("day").notNull(), | |
| 716 | + kind: text("kind").notNull(), | |
| 717 | + key: text("key").notNull(), | |
| 718 | + payload: jsonb("payload").$type<Record<string, unknown>>().notNull(), | |
| 719 | + createdAt: createdAt(), | |
| 720 | + }, | |
| 721 | + (t) => [primaryKey({ columns: [t.day, t.kind, t.key] }), index("entity_snapshots_kind_idx").on(t.kind, t.key, t.day)], | |
| 722 | +); | |
| 723 | + | |
| 724 | +/** Private watchlists (owner = opaque cookie token, no accounts). */ | |
| 725 | +export const watchlists = pgTable( | |
| 726 | + "watchlists", | |
| 727 | + { | |
| 728 | + id: text("id").primaryKey(), | |
| 729 | + ownerToken: text("owner_token").notNull(), | |
| 730 | + entityType: text("entity_type").notNull(), | |
| 731 | + entityId: text("entity_id").notNull(), | |
| 732 | + createdAt: createdAt(), | |
| 733 | + }, | |
| 734 | + (t) => [uniqueIndex("watchlists_uq").on(t.ownerToken, t.entityType, t.entityId)], | |
| 735 | +); | |
| 736 | + | |
| 737 | +/** Public reports of grid constraints (moratoria, delays, load caps, new transmission…) attached to a market / country. */ | |
| 738 | +export const gridConstraints = pgTable( | |
| 739 | + "grid_constraints", | |
| 740 | + { | |
| 741 | + id: text("id").primaryKey(), | |
| 742 | + metroId: text("metro_id").references(() => metros.id), | |
| 743 | + countryIso2: text("country_iso2").references(() => countries.iso2), | |
| 744 | + kind: text("kind").notNull(), | |
| 745 | + title: text("title").notNull(), | |
| 746 | + summary: text("summary"), | |
| 747 | + effectiveDate: text("effective_date"), | |
| 748 | + sourceId: text("source_id"), | |
| 749 | + documentId: text("document_id"), | |
| 750 | + url: text("url").notNull(), | |
| 751 | + eventId: text("event_id"), | |
| 752 | + confidence: text("confidence").notNull().default("moderate"), | |
| 753 | + createdAt: createdAt(), | |
| 754 | + }, | |
| 755 | + (t) => [index("grid_constraints_metro_idx").on(t.metroId), index("grid_constraints_country_idx").on(t.countryIso2)], | |
| 756 | +); | |
| 588 | 757 | |