SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
2 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%

API 2.0: claims/history/provenance/nearby/pulse/compare/explore/coverage/ai-infrastructure/power/connectivity/time-machine/download/watchlist/docs-meta endpoints, map layers + density + time filter, event dedupe by cluster, list envelopes carry sources with redistribution; admin quality/data-gaps/trace proxy/claims/project hide+merge/related-campus/quarantine/run rollback; containment-aware detail payloads; env placeholder token rejected; 59 API tests

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Simon-Pierre Boucher committed 20 days ago (Sep 12, 2026) parent 4de8d18

54 changed files +3,981 −509

modified apps/api/src/app.test.ts +552 −44
@@ -1,54 +1,91 @@
1 1 /**
2 − * Integration tests against the local Postgres (repo-root .env). A small fixture (country ZZ, operator, two
3 − * facilities, one project, one event) is inserted before and removed after; every id is prefixed `*_apitest`.
2 + * Integration tests against the local Postgres (repo-root .env). A small fixture (countries ZZ/ZY, operator, three
3 + * facilities, one project, events, a connector run with provenance / claim / event, sources with licences, one quality
4 + * flag) is inserted before and removed after; every id is prefixed `*_apitest`.
4 5 */
5 6 import { afterAll, beforeAll, describe, expect, it } from "vitest";
6 7 import type { FastifyInstance } from "fastify";
7 8 import { loadEnvFile, resetEnv } from "./env.js";
8 9
9 10 loadEnvFile();
10 −process.env.DCI_ADMIN_TOKEN = process.env.DCI_ADMIN_TOKEN || "test-admin-token";
11 +process.env.DCI_ADMIN_TOKEN = process.env.DCI_ADMIN_TOKEN && process.env.DCI_ADMIN_TOKEN !== "change-me" ? process.env.DCI_ADMIN_TOKEN : "test-admin-token";
11 12 process.env.DCI_API_CACHE = "0"; // deterministic responses while fixtures change
12 13 process.env.DCI_API_CH_LOG = "0";
14 +process.env.DCI_WORKER_URL = "http://127.0.0.1:1"; // nothing listens: proxy routes must answer 502
13 15 resetEnv();
14 16
15 17 const TOKEN = process.env.DCI_ADMIN_TOKEN!;
18 +const ADMIN = { "x-dci-admin-token": TOKEN };
16 19 const F1 = "fac_apitest000001";
17 20 const F2 = "fac_apitest000002";
21 +const F3 = "fac_apitest000003";
18 22 const OP = "op_apitest00000001";
19 23 const PRJ = "prj_apitest0000001";
20 24 const EVT = "evt_apitest0000001";
25 +const EVT_RUN = "evt_apitest0000002";
21 26 const MET = "met_apitest0000001";
27 +const RUN = "run_apitest0000001";
28 +const CLAIM = "clm_apitest0000001";
29 +const PROV_OLD = "prv_apitest0000001";
30 +const PROV_RUN = "prv_apitest0000002";
31 +const PROV_F3 = "prv_apitest0000003";
32 +const FLAG = "flg_apitest0000001";
33 +const SRC = "src_apitest";
34 +const SRC_R = "src_apitest_restricted";
22 35
23 36 let app: FastifyInstance;
24 37 let sql: Awaited<ReturnType<typeof import("./lib/sql.js").pg>>;
25 38
26 39 async function cleanup(): Promise<void> {
27 − await sql`delete from events where id = ${EVT} or entity_id in (${F1}, ${F2})`;
28 − await sql`delete from provenance where entity_id in (${F1}, ${F2}, ${OP}, ${PRJ})`;
29 − await sql`delete from facility_aliases where facility_id in (${F1}, ${F2})`;
30 − await sql`delete from entity_keys where entity_id in (${F1}, ${F2})`;
40 + await sql`delete from watchlists where entity_id in (${F1}, ${F2}, ${OP}, ${PRJ})`;
41 + await sql`delete from quality_flags where dedupe_key like 'apitest%' or entity_id in (${F1}, ${F2}, ${F3}, ${PRJ})`;
42 + await sql`delete from claims where subject_id in (${F1}, ${F2}, ${F3}, ${PRJ}) or run_id = ${RUN}`;
43 + await sql`delete from events where id in (${EVT}, ${EVT_RUN}) or entity_id in (${F1}, ${F2}, ${F3}, ${PRJ}) or project_id = ${PRJ} or run_id = ${RUN}`;
44 + await sql`delete from provenance where entity_id in (${F1}, ${F2}, ${F3}, ${OP}, ${PRJ}) or run_id = ${RUN}`;
45 + await sql`delete from project_timeline where project_id = ${PRJ}`;
46 + await sql`delete from facility_aliases where facility_id in (${F1}, ${F2}, ${F3})`;
47 + await sql`delete from entity_keys where entity_id in (${F1}, ${F2}, ${F3}, ${PRJ})`;
48 + await sql`delete from connector_runs where id = ${RUN}`;
31 49 await sql`delete from projects where id = ${PRJ}`;
32 − await sql`delete from facilities where id in (${F1}, ${F2})`;
50 + await sql`update facilities set parent_facility_id = null where parent_facility_id in (${F1}, ${F2}, ${F3})`;
51 + await sql`delete from facilities where id in (${F1}, ${F2}, ${F3})`;
33 52 await sql`delete from metros where id = ${MET}`;
34 53 await sql`delete from operators where id = ${OP}`;
35 − await sql`delete from countries where iso2 = 'ZZ'`;
54 + await sql`delete from countries where iso2 in ('ZZ', 'ZY')`;
55 + await sql`delete from sources where id in (${SRC}, ${SRC_R})`;
36 56 }
37 57
38 58 beforeAll(async () => {
39 59 const { pg } = await import("./lib/sql.js");
40 60 sql = pg();
41 61 await cleanup();
42 − await sql`insert into countries (iso2, iso3, slug, name, region, subregion, lat, lng) values ('ZZ', 'ZZZ', 'apitest-land', 'Apitest Land', 'Testing', 'Unit', 45.5, -73.6)`;
62 + await sql`insert into sources (id, connector_id, name, domain, kind, priority, url, license, attribution, redistribution) values
63 + (${SRC}, 'apitest', 'Apitest Source', 'example.test', 'operator', 2, 'https://example.test', 'CC-BY-4.0', 'Apitest Source', 'attribution'),
64 + (${SRC_R}, 'apitest-restricted', 'Apitest Restricted Source', 'restricted.test', 'dataset', 3, 'https://restricted.test', 'proprietary', 'Restricted', 'restricted')`;
65 + await sql`insert into countries (iso2, iso3, slug, name, region, subregion, lat, lng, renewable_share, electricity_twh, stats_year) values ('ZZ', 'ZZZ', 'apitest-land', 'Apitest Land', 'Testing', 'Unit', 45.5, -73.6, 0.42, 123.4, 2024), ('ZY', 'ZYY', 'apitest-land-two', 'Apitest Land Two', 'Testing', 'Unit', 10, 10, null, null, null)`;
43 66 await sql`insert into operators (id, slug, name, normalized_name, kind, hq_country_iso2, website) values (${OP}, 'apitest-operator', 'Apitest Operator', 'apitest operator', 'colocation', 'ZZ', 'https://example.test')`;
44 − await sql`insert into metros (id, slug, name, country_iso2, lat, lng) values (${MET}, 'apitest-metro', 'Apitest Metro', 'ZZ', 45.5, -73.6)`;
45 − await sql`insert into facilities (id, slug, name, normalized_name, operator_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, it_capacity_mw, total_power_mw, is_ai, confidence, completeness, opened_on, last_verified)
46 − values (${F1}, 'apitest-one', 'Apitest One Data Center', 'apitest one', ${OP}, ${MET}, 'ZZ', 'Montréal-Test', 45.5017, -73.5673, 'exact', 'operational', 'colocation', 12.5, 20, true, 'high', 80, '2019-05', now())`;
67 + await sql`insert into metros (id, slug, name, country_iso2, lat, lng, aliases) values (${MET}, 'apitest-metro', 'Apitest Metro', 'ZZ', 45.5, -73.6, '{"Apitestville"}')`;
68 + await sql`insert into facilities (id, slug, name, normalized_name, operator_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, it_capacity_mw, total_power_mw, utility_capacity_mw, is_ai, ai_evidence, confidence, completeness, opened_on, last_verified, source_count)
69 + values (${F1}, 'apitest-one', 'Apitest One Data Center', 'apitest one', ${OP}, ${MET}, 'ZZ', 'Montréal-Test', 45.5017, -73.5673, 'exact', 'operational', 'colocation', 12.5, 20, 40, true, 'confirmed', 'high', 80, '2019-05', now(), 2)`;
47 70 await sql`insert into facilities (id, slug, name, normalized_name, operator_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, planned_power_mw, confidence, completeness, opened_on)
48 71 values (${F2}, 'apitest-two', 'Apitest Two Campus', 'apitest two', ${OP}, ${MET}, 'ZZ', 'Montréal-Test', 45.52, -73.58, 'city', 'under_construction', 'hyperscale', 100, 'moderate', 40, '2027')`;
49 − await sql`insert into projects (id, slug, name, normalized_name, operator_id, facility_id, metro_id, country_iso2, status, expected_opening, planned_mw, is_ai) values (${PRJ}, 'apitest-project', 'Apitest Expansion', 'apitest expansion', ${OP}, ${F2}, ${MET}, 'ZZ', 'under_construction', '2027-Q2', 100, true)`;
50 − await sql`insert into events (id, entity_type, entity_id, event_type, old_value, new_value, source_id, url, title, significance, confidence, review_status, country_iso2, operator_id, fingerprint)
51 − values (${EVT}, 'facility', ${F1}, 'capacity_changed', '10'::jsonb, '12.5'::jsonb, 'src_apitest', 'https://example.test/one', 'Apitest One capacity 10 → 12.5 MW', 80, 'high', 'auto', 'ZZ', ${OP}, 'apitest:evt:1')`;
72 + await sql`insert into facilities (id, slug, name, normalized_name, operator_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, confidence, completeness, source_count)
73 + values (${F3}, 'apitest-three', 'Apitest Three Restricted', 'apitest three', ${OP}, 'ZY', 'Restricted City', 10.01, 10.01, 'city', 'operational', 'colocation', 'moderate', 20, 1)`;
74 + await sql`insert into projects (id, slug, name, normalized_name, operator_id, facility_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, announced_on, construction_started_on, expected_opening, planned_mw, is_ai, ai_evidence, project_class, evidence_level)
75 + values (${PRJ}, 'apitest-project', 'Apitest Expansion', 'apitest expansion', ${OP}, ${F2}, ${MET}, 'ZZ', 'Montréal-Test', 45.51, -73.57, 'city', 'under_construction', '2025-01', '2026-03', '2027-Q2', 100, true, 'likely', 'EXPANSION', 'strong')`;
76 + await sql`insert into events (id, entity_type, entity_id, event_type, old_value, new_value, source_id, url, title, significance, confidence, review_status, country_iso2, operator_id, metro_id, fingerprint, is_ai)
77 + values (${EVT}, 'facility', ${F1}, 'capacity_changed', '10'::jsonb, '12.5'::jsonb, ${SRC}, 'https://example.test/one', 'Apitest One capacity 10 → 12.5 MW', 80, 'high', 'auto', 'ZZ', ${OP}, ${MET}, 'apitest:evt:1', true)`;
78 + // a connector run with one provenance row (winner), one claim and one event — for /runs/:id/changes and rollback
79 + await sql`insert into connector_runs (id, connector_id, task, started_at, finished_at, status, stats) values (${RUN}, 'apitest', 'crawl', now() - interval '1 hour', now() - interval '50 minutes', 'ok', '{"discovered": 3, "fetched": 3, "created": 1, "changed": 1, "updated": 0, "failed": 0}'::jsonb)`;
80 + await sql`insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, url, first_observed, last_observed, retrieved_at, confidence, is_current, is_winner, run_id)
81 + values (${PROV_OLD}, 'facility', ${F1}, 'itCapacityMw', '10'::jsonb, ${SRC}, 'apitest', 'https://example.test/one-old', now() - interval '30 days', now() - interval '2 days', now() - interval '2 days', 'moderate', true, false, null),
82 + (${PROV_RUN}, 'facility', ${F1}, 'itCapacityMw', '12.5'::jsonb, ${SRC}, 'apitest', 'https://example.test/one', now() - interval '1 hour', now() - interval '1 hour', now() - interval '1 hour', 'high', true, true, ${RUN}),
83 + (${PROV_F3}, 'facility', ${F3}, 'name', '"Apitest Three Restricted"'::jsonb, ${SRC_R}, 'apitest-restricted', 'https://restricted.test/three', now(), now(), now(), 'moderate', true, true, null)`;
84 + await sql`insert into claims (id, subject_type, subject_id, predicate, value, unit, scope, scope_reason, source_id, connector_id, url, published_at, confidence, authority_tier, evidence_text, status, run_id)
85 + values (${CLAIM}, 'facility', ${F1}, 'it_capacity_mw', 12.5, 'MW', 'facility', 'regex:facility', ${SRC}, 'apitest', 'https://example.test/one', '2026-09', 'high', 'B', 'The facility offers 12.5 MW of IT capacity.', 'current', ${RUN})`;
86 + await sql`insert into events (id, entity_type, entity_id, event_type, new_value, source_id, url, title, significance, confidence, review_status, country_iso2, operator_id, fingerprint, run_id)
87 + values (${EVT_RUN}, 'facility', ${F1}, 'facility_updated', '{"itCapacityMw": 12.5}'::jsonb, ${SRC}, 'https://example.test/one', 'Apitest One updated', 30, 'moderate', 'auto', 'ZZ', ${OP}, 'apitest:evt:2', ${RUN})`;
88 + await sql`insert into quality_flags (id, entity_type, entity_id, code, severity, field, message, priority, status, dedupe_key) values (${FLAG}, 'facility', ${F1}, 'mw_change_5x', 'warn', 'itCapacityMw', 'apitest flag', 999, 'open', 'apitest:flag:1')`;
52 89 const { buildApp } = await import("./app.js");
53 90 app = await buildApp({ logger: false, underPressure: false });
54 91 await app.ready();
@@ -63,6 +100,8 @@ afterAll(async () => {
63 100 await Promise.allSettled([closeQueues(), closeRedis(), closeDb()]);
64 101 });
65 102
103 +const slugs = (arr: Array<{ slug: string }>) => arr.map((f) => f.slug).filter((s) => s.startsWith("apitest"));
104 +
66 105 describe("system", () => {
67 106 it("health and ready", async () => {
68 107 const h = await app.inject({ url: "/api/health" });
@@ -77,15 +116,21 @@ describe("system", () => {
77 116 expect(m.statusCode).toBe(200);
78 117 expect(m.body).toContain("dci_api_requests_total");
79 118 });
80 − it("openapi document is served", async () => {
119 + it("openapi document is served with response schemas", async () => {
81 120 const o = await app.inject({ url: "/api/v1/openapi.json" });
82 121 expect(o.statusCode).toBe(200);
83 − expect(o.json().paths["/api/v1/datacenters"]).toBeDefined();
122 + const paths = o.json().paths;
123 + expect(paths["/api/v1/datacenters"]).toBeDefined();
124 + expect(paths["/api/v1/nearby"]).toBeDefined();
125 + const ok = paths["/api/v1/datacenters"].get.responses["200"];
126 + expect(ok).toBeDefined();
127 + const schema = ok.content?.["application/json"]?.schema ?? ok;
128 + expect(schema.properties?.data ?? schema["x-example"] ?? schema.description).toBeDefined();
84 129 });
85 130 });
86 131
87 132 describe("envelope, ETag and 404", () => {
88 − it("wraps list responses and sets caching headers", async () => {
133 + it("wraps list responses and sets caching headers + sources", async () => {
89 134 const res = await app.inject({ url: "/api/v1/datacenters?country=ZZ&sort=mw" });
90 135 expect(res.statusCode).toBe(200);
91 136 const body = res.json();
@@ -94,6 +139,9 @@ describe("envelope, ETag and 404", () => {
94 139 expect(body.meta.generatedAt).toBeTruthy();
95 140 expect(body.data[0].slug).toBe("apitest-two"); // planned 100 MW sorts first (mw = COALESCE(it,total,planned))
96 141 expect(body.data[1].operator).toEqual({ id: OP, slug: "apitest-operator", name: "Apitest Operator" });
142 + expect(body.data[1]).toMatchObject({ recordScope: "facility", aiEvidence: "confirmed", utilityCapacityMw: 40, sourceCount: 2 });
143 + expect(Array.isArray(body.sources)).toBe(true);
144 + expect(body.sources.some((s: { id: string; redistribution: string }) => s.id === SRC && s.redistribution === "attribution")).toBe(true);
97 145 expect(res.headers.etag).toMatch(/^W\//);
98 146 expect(res.headers["cache-control"]).toContain("s-maxage=120");
99 147 const again = await app.inject({ url: "/api/v1/datacenters?country=ZZ&sort=mw", headers: { "if-none-match": String(res.headers.etag) } });
@@ -110,7 +158,7 @@ describe("envelope, ETag and 404", () => {
110 158 expect(d.json().data.map((f: { slug: string }) => f.slug)).toEqual(["apitest-two"]);
111 159 });
112 160 it("returns a consistent 404 JSON", async () => {
113 − for (const url of ["/api/v1/datacenters/nope", "/api/v1/operators/nope", "/api/v1/countries/nope", "/api/v1/metros/nope", "/api/v1/projects/nope", "/api/v1/events/nope", "/api/v1/rankings/nope", "/api/v1/sitemap/nope", "/api/nothing-here"]) {
161 + for (const url of ["/api/v1/datacenters/nope", "/api/v1/operators/nope", "/api/v1/countries/nope", "/api/v1/metros/nope", "/api/v1/projects/nope", "/api/v1/events/nope", "/api/v1/rankings/nope", "/api/v1/sitemap/nope", "/api/v1/download/nope", "/api/nothing-here"]) {
114 162 const r = await app.inject({ url });
115 163 expect(r.statusCode, url).toBe(404);
116 164 expect(typeof r.json().error, url).toBe("string");
@@ -130,39 +178,124 @@ describe("details", () => {
130 178 const d = r.json().data;
131 179 expect(d.id).toBe(F1);
132 180 expect(d.metro.slug).toBe("apitest-metro");
133 − expect(d.nearby.map((f: { slug: string }) => f.slug)).toEqual(["apitest-two"]);
181 + // legacy `nearby` is capped at the 8 closest rows: real Montréal facilities may outrank the fixture — shape only
182 + expect(Array.isArray(d.nearby)).toBe(true);
183 + expect(d.nearby.length).toBeLessThanOrEqual(8);
134 184 expect(d.projects.map((p: { slug: string }) => p.slug)).toEqual(["apitest-project"]); // same operator + metro
135 − expect(d.events.length).toBe(1);
136 − expect(d.sourceHistory[0].kind).toBe("changed");
185 + expect(d.events.length).toBe(2);
186 + expect(d.sourceHistory[0].kind).toBeTruthy();
137 187 expect(Array.isArray(d.provenance)).toBe(true);
188 + expect(d.provenance.find((p: { isWinner: boolean; field: string }) => p.field === "itCapacityMw" && p.isWinner).value).toBe(12.5);
138 189 expect(d.certifications).toEqual([]);
190 + // API 2.0 fields
191 + expect(d.buildings).toEqual([]);
192 + expect(d.tenants).toEqual([]);
193 + expect(d.claims.length).toBe(1);
194 + expect(d.claims[0]).toMatchObject({ id: CLAIM, predicate: "it_capacity_mw", label: "IT capacity", scope: "facility", isWinner: true, authorityTier: "B" });
195 + expect(d.capacityHistory.length).toBeGreaterThanOrEqual(3); // 2 observations + 1 claim + capacity_changed event
196 + expect(d.capacityHistory.some((h: { kind: string }) => h.kind === "changed")).toBe(true);
197 + expect(d.nearbyInfrastructure.radiusKm).toBe(25);
198 + expect(slugs(d.nearbyInfrastructure.facilities)).toEqual(["apitest-two"]);
199 + expect(slugs(d.nearbyInfrastructure.projects)).toEqual(["apitest-project"]);
200 + expect(d.nearbyInfrastructure.landingStations).toEqual([]);
201 + expect(typeof d.nearbyInfrastructure.note).toBe("string");
202 + expect(d.powerContext.utilityCapacityMw).toBe(40);
203 + expect(d.powerContext.countryEnergy).toMatchObject({ renewableShare: 0.42, electricityTwh: 123.4, statsYear: 2024 });
204 + expect(typeof d.powerContext.note).toBe("string");
205 + expect(d.dataQuality).toMatchObject({ sourceCount: 1, primarySourceCount: 1, claimsTotal: 1, claimsCurrent: 1, pendingDuplicate: false });
206 + expect(d.dataQuality.openFlags.some((f: { code: string }) => f.code === "mw_change_5x")).toBe(true);
139 207 const byId = await app.inject({ url: `/api/v1/datacenters/${F1}` });
140 208 expect(byId.json().data.slug).toBe("apitest-one");
209 + const small = await app.inject({ url: "/api/v1/datacenters/apitest-one?radius_km=1" });
210 + expect(small.json().data.nearbyInfrastructure.radiusKm).toBe(1);
211 + });
212 + it("facility history, claims and provenance endpoints", async () => {
213 + const h = (await app.inject({ url: "/api/v1/datacenters/apitest-one/history" })).json().data;
214 + expect(h.entityType).toBe("facility");
215 + expect(h.entityId).toBe(F1);
216 + expect(h.fields.itCapacityMw.length).toBeGreaterThanOrEqual(2);
217 + expect(h.changes.some((c: { eventId: string }) => c.eventId === EVT)).toBe(true);
218 + const c = (await app.inject({ url: "/api/v1/datacenters/apitest-one/claims" })).json().data;
219 + expect(c.map((x: { id: string }) => x.id)).toEqual([CLAIM]);
220 + const p = (await app.inject({ url: `/api/v1/datacenters/${F1}/provenance` })).json().data;
221 + expect(p.filter((x: { field: string }) => x.field === "itCapacityMw").length).toBe(2);
222 + expect((await app.inject({ url: "/api/v1/datacenters/nope/history" })).statusCode).toBe(404);
141 223 });
142 224 it("operator, country, metro and project details aggregate the fixture", async () => {
143 225 const op = (await app.inject({ url: "/api/v1/operators/apitest-operator" })).json().data;
144 − expect(op.facilityCount).toBe(2);
145 − expect(op.countries[0]).toMatchObject({ iso2: "ZZ", facilityCount: 2, knownMw: 12.5 });
146 − expect(op.statusBreakdown).toEqual({ operational: 1, under_construction: 1 });
147 − expect(op.recentEvents.length).toBe(1);
226 + expect(op.facilityCount).toBe(3);
227 + expect(op.countries.find((c: { iso2: string }) => c.iso2 === "ZZ")).toMatchObject({ iso2: "ZZ", facilityCount: 2, knownMw: 12.5 });
228 + expect(op.statusBreakdown).toEqual({ operational: 2, under_construction: 1 });
229 + expect(op.recentEvents.length).toBe(2);
230 + expect(op.pipeline.operational).toMatchObject({ count: 2, mw: 12.5 });
231 + expect(op.pipeline.construction).toMatchObject({ count: 1, mw: 100 });
232 + expect(op.pipeline.projects).toEqual([{ status: "under_construction", count: 1, mw: 100 }]);
233 + expect(op.velocity.windows.map((w: { label: string }) => w.label)).toEqual(["12m", "3y", "5y"]);
234 + expect(op.velocity.countriesOverTime.length).toBeGreaterThan(0);
235 + expect(op.topCountries[0]).toMatchObject({ iso2: "ZZ", facilityCount: 2 });
236 + expect(op.topCountries[0].share).toBeCloseTo(2 / 3, 2);
237 + expect(slugs(op.aiFacilities)).toEqual(["apitest-one"]);
238 + expect(Array.isArray(op.corporateEvents)).toBe(true);
239 + expect(op.dataQuality.sourceCount).toBe(0);
148 240 const co = (await app.inject({ url: "/api/v1/countries/zz" })).json().data;
149 − expect(co).toMatchObject({ iso2: "ZZ", facilityCount: 2, operationalCount: 1, constructionCount: 1, knownMw: 12.5, constructionMw: 100, mwCoverage: 1 });
150 − expect(co.growth).toEqual([{ year: 2019, facilities: 1, knownMw: 12.5 }, { year: 2027, facilities: 2, knownMw: 112.5 }]);
241 + expect(co).toMatchObject({ iso2: "ZZ", facilityCount: 2, operationalCount: 1, constructionCount: 1, knownMw: 12.5, constructionMw: 100, mwCoverage: 1, projectCount: 1 });
242 + // known MW = operational figures only: the 100 MW planned for the 2027 campus is pipeline, not known capacity
243 + expect(co.growth).toEqual([{ year: 2019, facilities: 1, knownMw: 12.5 }, { year: 2027, facilities: 2, knownMw: 12.5 }]);
151 244 expect(co.topOperators[0].slug).toBe("apitest-operator");
245 + expect(co.energy).toMatchObject({ renewableShare: 0.42, electricityTwh: 123.4, statsYear: 2024, gridCarbonIntensity: null });
246 + expect(co.energy.note).toMatch(/national/i);
247 + expect(co.pipeline.construction.count).toBe(1);
248 + expect(co.coverage).toMatchObject({ key: "ZZ", facilities: 2, capacityCoverage: 1, anyLocationCoverage: 1, openingDateCoverage: 1 });
249 + expect(slugs(co.aiFacilities)).toEqual(["apitest-one"]);
250 + expect(co.aiProjects.map((p: { slug: string }) => p.slug)).toEqual(["apitest-project"]);
251 + expect(co.gridConstraints).toEqual([]);
252 + expect(co.ixps).toEqual([]);
152 253 const me = (await app.inject({ url: "/api/v1/metros/apitest-metro" })).json().data;
153 254 expect(me.facilityCount).toBe(2);
154 255 expect(me.projectCount).toBe(1);
155 256 expect(me.constraints).toEqual([]);
257 + expect(me.concentration.operatorCount).toBe(1);
258 + expect(me.concentration.facilities.hhi).toBe(10000);
259 + expect(me.concentration.knownMw.hhi).toBe(10000);
260 + expect(me.momentum.window).toBe("12m");
261 + expect(me.momentum.projectsEnteredConstruction).toBe(1);
262 + expect(me.pipeline.operational.count).toBe(1);
263 + expect(me.openingTimeline).toEqual([{ year: 2019, opened: 1, openedMw: 12.5 }, { year: 2027, opened: 1, openedMw: null }]);
264 + expect(me.coverage.facilities).toBe(2);
265 + expect(me.gridConstraints).toEqual([]);
156 266 const pr = (await app.inject({ url: "/api/v1/projects/apitest-project" })).json().data;
157 267 expect(pr.facility.slug).toBe("apitest-two");
158 268 expect(pr.timeline).toEqual([]);
269 + expect(pr).toMatchObject({ projectClass: "EXPANSION", evidenceLevel: "strong", aiEvidence: "likely", isAi: true, geoPrecision: "city", constructionStartedOn: "2026-03" });
270 + const stages = Object.fromEntries(pr.stages.map((s: { stage: string; date: string | null; reached: boolean; current: boolean }) => [s.stage, s]));
271 + expect(stages.announced).toMatchObject({ date: "2025-01", reached: true, current: false });
272 + expect(stages.under_construction).toMatchObject({ date: "2026-03", reached: true, current: true });
273 + expect(stages.operational).toMatchObject({ reached: false, current: false });
274 + expect(pr.velocityDays.announcedToConstruction).toBe(Math.round((Date.UTC(2026, 2, 1) - Date.UTC(2025, 0, 1)) / 86_400_000));
275 + expect(pr.velocityDays.constructionToOpening).toBeNull();
276 + expect(pr.claims).toEqual([]);
277 + expect(pr.nearbyInfrastructure).not.toBeNull();
278 + expect(slugs(pr.nearbyInfrastructure.facilities)).toEqual(["apitest-one", "apitest-two"]);
279 + expect(Array.isArray(pr.relatedEvents)).toBe(true);
280 + expect(pr.dataQuality.claimsTotal).toBe(0);
281 + expect(pr.campus).toBeNull();
282 + const ph = (await app.inject({ url: "/api/v1/projects/apitest-project/history" })).json().data;
283 + expect(ph.entityType).toBe("project");
284 + expect((await app.inject({ url: "/api/v1/projects/apitest-project/claims" })).json().data).toEqual([]);
159 285 });
160 − it("list endpoints include the fixture", async () => {
286 + it("list endpoints include the fixture and carry sources", async () => {
161 287 expect((await app.inject({ url: "/api/v1/countries" })).json().data.some((c: { iso2: string }) => c.iso2 === "ZZ")).toBe(true);
162 − expect((await app.inject({ url: "/api/v1/operators?q=apitest" })).json().data[0].facilityCount).toBe(2);
163 − expect((await app.inject({ url: "/api/v1/projects?country=ZZ&ai=1" })).json().meta.total).toBe(1);
288 + expect((await app.inject({ url: "/api/v1/operators?q=apitest" })).json().data[0].facilityCount).toBe(3);
289 + const pj = (await app.inject({ url: "/api/v1/projects?country=ZZ&ai=1" })).json();
290 + expect(pj.meta.total).toBe(1);
291 + expect(Array.isArray(pj.sources)).toBe(true);
164 292 const ev = (await app.inject({ url: "/api/v1/events?country=ZZ&type=capacity_changed" })).json();
165 − expect(ev.data[0]).toMatchObject({ id: EVT, entity: { slug: "apitest-one", name: "Apitest One Data Center" }, operator: { slug: "apitest-operator" } });
293 + expect(ev.data[0]).toMatchObject({ id: EVT, entity: { slug: "apitest-one", name: "Apitest One Data Center" }, operator: { slug: "apitest-operator" }, metro: { slug: "apitest-metro" }, isAi: true, significanceBand: "major", evidenceCount: 1, sourceName: "Apitest Source", sourceKind: "operator" });
294 + expect(ev.sources.map((s: { id: string }) => s.id)).toContain(SRC);
295 + const major = (await app.inject({ url: "/api/v1/events?country=ZZ&significance=major&ai=1&source_kind=operator&q=capacity" })).json();
296 + expect(major.data.map((e: { id: string }) => e.id)).toEqual([EVT]);
297 + const none = (await app.inject({ url: "/api/v1/events?country=ZZ&significance=minor" })).json();
298 + expect(none.data.map((e: { id: string }) => e.id)).toEqual([EVT_RUN]);
166 299 expect((await app.inject({ url: "/api/v1/sitemap/facilities" })).json().data.some((x: { slug: string }) => x.slug === "apitest-one")).toBe(true);
167 300 });
168 301 });
@@ -191,10 +324,29 @@ describe("map", () => {
191 324 const p = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=-74,45,-73,46&country=ZZ" })).json().data;
192 325 expect(p.mode).toBe("points");
193 326 expect(p.points.map((x: { slug: string }) => x.slug)).toEqual(["apitest-two", "apitest-one"]);
194 − expect(p.points[1]).toMatchObject({ n: "Apitest One Data Center", o: "Apitest Operator", s: "operational", t: "colocation", mw: 12.5, p: "exact", ai: 1, c: "ZZ" });
327 + expect(p.points[1]).toMatchObject({ n: "Apitest One Data Center", o: "Apitest Operator", s: "operational", t: "colocation", mw: 12.5, p: "exact", ai: 1, c: "ZZ", y: 2019 });
195 328 const outside = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=10,10,11,11&country=ZZ" })).json().data;
196 329 expect(outside.points).toEqual([]);
197 330 });
331 + it("layers, density and time machine", async () => {
332 + const pr = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=-74,45,-73,46&country=ZZ&layer=projects" })).json();
333 + expect(pr.statusCode ?? 200).toBe(200);
334 + expect(pr.data.layer).toBe("projects");
335 + const overlay = pr.data.overlay ?? [];
336 + expect(overlay.find((x: { slug: string }) => x.slug === "apitest-project")).toMatchObject({ k: "project", s: "under_construction", p: "city" });
337 + const dens = (await app.inject({ url: "/api/v1/map?zoom=6&country=ZZ&density=facilities" })).json().data;
338 + expect(dens.mode).toBe("density");
339 + expect(dens.density.view).toBe("facilities");
340 + expect(dens.density.cells.length).toBeGreaterThanOrEqual(1);
341 + expect(dens.density.max).toBeGreaterThanOrEqual(1);
342 + expect(dens.density.cells.reduce((a: number, c: { n: number }) => a + c.n, 0)).toBe(2);
343 + const yr = (await app.inject({ url: "/api/v1/map?zoom=3&country=ZZ&year=2020" })).json().data;
344 + expect(yr.year).toBe(2020);
345 + expect(yr.yearCoverage).toBe(1);
346 + expect(yr.clusters[0].count).toBe(1); // apitest-one opened 2019; apitest-two opens 2027
347 + const bad = await app.inject({ url: "/api/v1/map?zoom=3&layer=nope" });
348 + expect(bad.statusCode).toBe(400);
349 + });
198 350 });
199 351
200 352 describe("search", () => {
@@ -205,26 +357,266 @@ describe("search", () => {
205 357 expect(d.interpreted.status).toBe("operational");
206 358 expect(d.hits.some((h: { type: string; slug: string }) => h.type === "facility" && h.slug === "apitest-one")).toBe(true);
207 359 expect(d.hits.some((h: { type: string }) => h.type === "operator")).toBe(true);
208 − expect(d.facilities.items.map((f: { slug: string }) => f.slug)).toEqual(["apitest-one"]);
360 + expect(d.facilities.items.map((f: { slug: string }) => f.slug).filter((s: string) => s.startsWith("apitest"))).toEqual(["apitest-one", "apitest-three"]);
209 361 const city = d.hits.find((h: { type: string }) => h.type === "city");
210 362 expect(city).toBeUndefined(); // "apitest" is not a city
363 + expect(Array.isArray(r.json().sources)).toBe(true);
211 364 });
212 365 it("city hits link to the facility list", async () => {
213 366 const d = (await app.inject({ url: "/api/v1/search?q=Montr%C3%A9al-Test" })).json().data;
214 367 const city = d.hits.find((h: { type: string }) => h.type === "city");
215 368 expect(city.href).toContain("/datacenters?q=");
216 369 });
370 + it("interprets project, max MW and year phrases", async () => {
371 + const d = (await app.inject({ url: "/api/v1/search?q=projects%20under%20500%20MW%20opening%202027" })).json().data;
372 + expect(d.interpreted).toMatchObject({ entity: "project", maxMw: 500, year: 2027 });
373 + });
374 +});
375 +
376 +describe("intelligence endpoints", () => {
377 + it("/nearby returns layers around a point (empty layers are honest)", async () => {
378 + const r = await app.inject({ url: "/api/v1/nearby?lat=45.5017&lng=-73.5673&radius_km=10" });
379 + expect(r.statusCode).toBe(200);
380 + const d = r.json().data;
381 + expect(d.center).toEqual({ lat: 45.5017, lng: -73.5673 });
382 + expect(d.radiusKm).toBe(10);
383 + expect(slugs(d.facilities)).toEqual(["apitest-one", "apitest-two"]);
384 + expect(d.facilities[0].distanceKm).toBe(0);
385 + expect(slugs(d.projects)).toEqual(["apitest-project"]);
386 + expect(d.metros.some((m: { slug: string }) => m.slug === "apitest-metro")).toBe(true);
387 + expect(d.landingStations).toEqual([]);
388 + expect(d.substations).toEqual([]);
389 + expect(d.powerPlants).toEqual([]);
390 + expect(d.note).toMatch(/never/i);
391 + expect((await app.inject({ url: "/api/v1/nearby?lat=95&lng=0" })).statusCode).toBe(400);
392 + expect((await app.inject({ url: "/api/v1/nearby?lng=0" })).statusCode).toBe(400);
393 + expect((await app.inject({ url: "/api/v1/nearby?lat=45.5&lng=-73.6&radius_km=500" })).statusCode).toBe(400); // > 200 km is rejected, not clamped
394 + const typed = (await app.inject({ url: "/api/v1/nearby?lat=45.5&lng=-73.6&radius_km=200&types=metros" })).json().data;
395 + expect(typed.radiusKm).toBe(200);
396 + expect(typed.facilities).toEqual([]);
397 + expect(typed.metros.some((m: { slug: string }) => m.slug === "apitest-metro")).toBe(true);
398 + });
399 + it("/pulse is deterministic and validates the window", async () => {
400 + const r = await app.inject({ url: "/api/v1/pulse?window=7d" });
401 + expect(r.statusCode).toBe(200);
402 + const d = r.json().data;
403 + expect(d.window).toBe("7d");
404 + expect(typeof d.since).toBe("string");
405 + expect(d.newProjects).toBeGreaterThanOrEqual(1); // the fixture project was created now
406 + expect(d.eventsTotal).toBeGreaterThanOrEqual(2);
407 + expect(d.capacityChanges).toBeGreaterThanOrEqual(1);
408 + for (const k of ["projectsEnteredConstruction", "facilitiesOpened", "cloudRegionsAnnounced", "powerAgreements", "gridConstraintEvents", "acquisitions", "financingEvents", "newFacilitiesIndexed"]) expect(typeof d[k], k).toBe("number");
409 + expect(Array.isArray(d.majorEvents)).toBe(true);
410 + // top-15 lists: the fixture cannot be expected to rank in an 11 000-event database — check shape only
411 + expect(Array.isArray(d.byCountry)).toBe(true);
412 + if (d.byCountry.length) expect(d.byCountry[0]).toMatchObject({ iso2: expect.any(String), slug: expect.any(String), events: expect.any(Number), newProjects: expect.any(Number) });
413 + expect(Array.isArray(d.byOperator)).toBe(true);
414 + if (d.byOperator.length) expect(d.byOperator[0]).toMatchObject({ slug: expect.any(String), events: expect.any(Number) });
415 + expect((await app.inject({ url: "/api/v1/pulse?window=1y" })).statusCode).toBe(400);
416 + expect((await app.inject({ url: "/api/v1/pulse" })).json().data.window).toBe("24h");
417 + });
418 + it("/explore filters facilities and projects, rejects unknown keys", async () => {
419 + const f = (await app.inject({ url: "/api/v1/explore?entity=facilities&country=ZZ&per_page=10" })).json().data;
420 + expect(f.total).toBe(2);
421 + expect(f.items.map((x: { slug: string }) => x.slug).sort()).toEqual(["apitest-one", "apitest-two"]);
422 + expect(Object.fromEntries(f.facets.status.map((s: { key: string; count: number }) => [s.key, s.count]))).toEqual({ operational: 1, under_construction: 1 });
423 + expect(f.facets.country[0]).toMatchObject({ key: "ZZ", name: "Apitest Land", count: 2 });
424 + expect(f.facets.operator[0]).toMatchObject({ key: "apitest-operator", count: 2 });
425 + expect(f.facets.ai.some((a: { key: string; count: number }) => a.key === "confirmed" && a.count === 1)).toBe(true);
426 + expect(f.charts.byStatus.find((s: { key: string }) => s.key === "operational")).toMatchObject({ count: 1, mw: 12.5 });
427 + expect(f.charts.byYear.map((y: { year: number }) => y.year)).toEqual([2019, 2027]);
428 + expect(f.map.total).toBe(2);
429 + expect(f.map.degraded).toBe(false);
430 + expect(f.mwCoverage).toBe(1);
431 + const ai = (await app.inject({ url: "/api/v1/explore?country=ZZ&ai=confirmed" })).json().data;
432 + expect(ai.items.map((x: { slug: string }) => x.slug)).toEqual(["apitest-one"]);
433 + const p = (await app.inject({ url: "/api/v1/explore?entity=projects&country=ZZ&min_mw=50" })).json().data;
434 + expect(p.total).toBe(1);
435 + expect(p.items[0].slug).toBe("apitest-project");
436 + expect(p.map.points[0]).toMatchObject({ k: "project" });
437 + const bad = await app.inject({ url: "/api/v1/explore?bogus=1" });
438 + expect(bad.statusCode).toBe(400);
439 + expect((await app.inject({ url: "/api/v1/explore?entity=nope" })).statusCode).toBe(400);
440 + });
441 + it("/coverage reports per scope and per source", async () => {
442 + const r = await app.inject({ url: "/api/v1/coverage" });
443 + expect(r.statusCode).toBe(200);
444 + const d = r.json().data;
445 + expect(d.global.facilities).toBeGreaterThan(2);
446 + expect(d.global.capacityCoverage).toBeGreaterThan(0);
447 + expect(d.global.capacityCoverage).toBeLessThanOrEqual(1);
448 + const zz = d.countries.find((c: { key: string }) => c.key === "ZZ");
449 + expect(zz).toMatchObject({ facilities: 2, capacityCoverage: 1, operatorCoverage: 1, preciseLocationCoverage: 0.5, statusCoverage: 1, openingDateCoverage: 1, projects: 1, projectsWithLocation: 1 });
450 + expect(d.fields.some((f: { field: string }) => /it.?capacity|itCapacityMw/i.test(f.field))).toBe(true);
451 + expect(d.bySourceKind.some((k: { kind: string }) => k.kind === "operator")).toBe(true);
452 + const s = (await app.inject({ url: "/api/v1/coverage/sources" })).json().data;
453 + const src = s.find((x: { id: string }) => x.id === SRC);
454 + expect(src).toMatchObject({ kind: "operator", recordsContributed: 1, fieldsContributed: 2, uniqueRecords: 1, redistribution: "attribution", license: "CC-BY-4.0" });
455 + expect(src.lastSuccessfulCrawl).toBeTruthy();
456 + });
457 + it("/ai-infrastructure, /power, /connectivity and /time-machine", async () => {
458 + const ai = (await app.inject({ url: "/api/v1/ai-infrastructure" })).json().data;
459 + expect(ai.stats.confirmed).toBeGreaterThanOrEqual(1);
460 + expect(ai.stats.facilities).toBeGreaterThanOrEqual(ai.stats.confirmed);
461 + expect(ai.stats.projects).toBeGreaterThanOrEqual(1);
462 + expect(ai.evidenceNote).toMatch(/confirmed/);
463 + expect(ai.topOperators.length).toBeGreaterThan(0);
464 + expect(ai.topOperators[0]).toHaveProperty("plannedMw");
465 + expect(Array.isArray(ai.pipelineByStage)).toBe(true);
466 + const pw = (await app.inject({ url: "/api/v1/power" })).json().data;
467 + // top-50 by MW: a 40 MW fixture is outranked by the ≥ 200 MW projects in the database — check shape + ordering
468 + expect(pw.largeLoads.length).toBeGreaterThan(0);
469 + expect(pw.largeLoads.every((l: { kind: string; mw: number }) => ["utility_capacity", "grid_connection", "planned_load"].includes(l.kind) && l.mw > 0)).toBe(true);
470 + expect(pw.largeLoads.every((l: { mw: number }, i: number, a: Array<{ mw: number }>) => i === 0 || a[i - 1]!.mw >= l.mw)).toBe(true);
471 + expect(Array.isArray(pw.gridConstraints)).toBe(true);
472 + // countryEnergy lists the 60 largest markets: honest energy context, never an estimated carbon intensity
473 + expect(pw.countryEnergy.length).toBeGreaterThan(0);
474 + expect(pw.countryEnergy.every((c: { energy: { gridCarbonIntensity: unknown; note: string }; facilities: number }) => c.energy.gridCarbonIntensity === null && typeof c.energy.note === "string" && c.facilities > 0)).toBe(true);
475 + expect(typeof pw.note).toBe("string");
476 + expect(Array.isArray(pw.utilities)).toBe(true);
477 + const cn = (await app.inject({ url: "/api/v1/connectivity" })).json().data;
478 + expect(cn.landingStations).toEqual([]);
479 + expect(typeof cn.note).toBe("string");
480 + expect(Array.isArray(cn.ixps)).toBe(true);
481 + expect(Array.isArray(cn.byMetro)).toBe(true);
482 + const tm = (await app.inject({ url: "/api/v1/time-machine" })).json().data;
483 + expect(tm.frames.length).toBeGreaterThan(0);
484 + const y2019 = tm.frames.find((f: { year: number }) => f.year === 2019);
485 + expect(y2019.opened).toBeGreaterThanOrEqual(1);
486 + expect(typeof tm.earliestReliableYear).toBe("number");
487 + expect(tm.openingDateCoverage).toBeGreaterThan(0);
488 + expect(tm.openingDateCoverage).toBeLessThanOrEqual(1);
489 + expect(tm.note).toMatch(/opening date/i);
490 + });
491 + it("/compare/operators validates and compares", async () => {
492 + expect((await app.inject({ url: "/api/v1/compare/operators?slugs=apitest-operator" })).statusCode).toBe(400);
493 + expect((await app.inject({ url: "/api/v1/compare/operators?slugs=apitest-operator,does-not-exist" })).statusCode).toBe(404);
494 + const other = (await app.inject({ url: "/api/v1/operators?per_page=1&sort=facilities" })).json().data[0];
495 + const r = await app.inject({ url: `/api/v1/compare/operators?slugs=apitest-operator,${other.slug}` });
496 + expect(r.statusCode).toBe(200);
497 + const d = r.json().data;
498 + expect(d.operators.length).toBe(2);
499 + const me = d.operators.find((o: { slug: string }) => o.slug === "apitest-operator");
500 + expect(me.pipeline.construction.count).toBe(1);
501 + expect(me.countriesList.sort()).toEqual(["ZY", "ZZ"]);
502 + expect(me.aiCount).toBe(1);
503 + expect(me.velocity.windows.length).toBe(3);
504 + });
505 + it("/docs-meta lists endpoints with examples", async () => {
506 + const d = (await app.inject({ url: "/api/v1/docs-meta" })).json().data;
507 + const nearby = d.find((e: { path: string }) => e.path === "/api/v1/nearby");
508 + expect(nearby).toMatchObject({ method: "GET", group: "nearby" });
509 + expect(nearby.params.some((p: { name: string }) => p.name === "lat")).toBe(true);
510 + expect(nearby.example.curl).toContain("/api/v1/nearby");
511 + expect(typeof nearby.example.python).toBe("string");
512 + expect(d.some((e: { path: string }) => e.path.startsWith("/api/v1/download/"))).toBe(true);
513 + expect(d.some((e: { path: string; method: string }) => e.path === "/api/v1/watchlist" && e.method === "POST")).toBe(true);
514 + });
515 +});
516 +
517 +describe("dashboard", () => {
518 + it("has the API 2.0 sections", async () => {
519 + const r = await app.inject({ url: "/api/v1/dashboard" });
520 + expect(r.statusCode).toBe(200);
521 + const d = r.json().data;
522 + expect(d.stats.facilities).toBeGreaterThan(0);
523 + expect(d.pulse.window).toBe("24h");
524 + expect(Array.isArray(d.fastestGrowingMetros)).toBe(true);
525 + expect(d.majorProjects.length).toBeGreaterThan(0);
526 + expect(d.majorProjects.every((p: { plannedMw: number }) => p.plannedMw >= 100)).toBe(true);
527 + expect(Array.isArray(d.operatorExpansion)).toBe(true);
528 + expect(Array.isArray(d.powerEvents)).toBe(true);
529 + expect(Array.isArray(d.gridConstraints)).toBe(true);
530 + expect(d.coverage.key).toBe("global");
531 + expect(typeof d.ingestion.runs24h).toBe("number");
532 + expect(d.ingestion.connectorsTotal).toBeGreaterThanOrEqual(0);
533 + });
534 +});
535 +
536 +describe("download", () => {
537 + it("lists datasets", async () => {
538 + const r = await app.inject({ url: "/api/v1/download/datasets" });
539 + expect(r.statusCode).toBe(200);
540 + const d = r.json().data;
541 + const fac = d.find((x: { key: string }) => x.key === "facilities");
542 + expect(fac.formats).toEqual(expect.arrayContaining(["csv", "json", "geojson"]));
543 + expect(fac.rows).toBeGreaterThan(0);
544 + expect(fac.excludedSources.some((s: { id: string }) => s.id === SRC_R)).toBe(true);
545 + expect(d.map((x: { key: string }) => x.key)).toEqual(expect.arrayContaining(["projects", "operators", "events", "cloud-regions", "ixps", "countries", "markets"]));
546 + });
547 + it("streams CSV with a header row, license gating and content-disposition", async () => {
548 + const r = await app.inject({ url: "/api/v1/download/facilities?format=csv&country=ZZ" });
549 + expect(r.statusCode).toBe(200);
550 + expect(String(r.headers["content-type"])).toContain("text/csv");
551 + expect(String(r.headers["content-disposition"])).toMatch(/attachment; filename="?dci-facilities-\d{4}-\d{2}-\d{2}\.csv"?/);
552 + const lines = r.body.trim().split(/\r?\n/);
553 + expect(lines[0]!.startsWith("id,slug,name")).toBe(true);
554 + expect(lines.length).toBe(3);
555 + expect(r.body).toContain("apitest-one");
556 + // the only source of apitest-three forbids redistribution → the row is excluded and the source listed
557 + const zy = await app.inject({ url: "/api/v1/download/facilities?format=csv&country=ZY" });
558 + expect(zy.statusCode).toBe(200);
559 + expect(zy.body.trim().split(/\r?\n/).length).toBe(1);
560 + expect(String(zy.headers["x-dci-excluded-sources"] ?? "")).toContain(SRC_R);
561 + const js = await app.inject({ url: "/api/v1/download/facilities?format=json&country=ZY" });
562 + expect(js.json().meta.excludedSources.some((s: { id: string }) => s.id === SRC_R)).toBe(true);
563 + const gj = await app.inject({ url: "/api/v1/download/projects?format=geojson&country=ZZ" });
564 + expect(gj.statusCode).toBe(200);
565 + expect(gj.json().type).toBe("FeatureCollection");
566 + expect(gj.json().features[0].geometry).toEqual({ type: "Point", coordinates: [-73.57, 45.51] });
567 + expect((await app.inject({ url: "/api/v1/download/facilities?format=xml" })).statusCode).toBe(400);
568 + });
569 +});
570 +
571 +describe("watchlist", () => {
572 + it("is private to an httpOnly cookie token and feeds events", async () => {
573 + const first = await app.inject({ url: "/api/v1/watchlist" });
574 + expect(first.statusCode).toBe(200);
575 + expect(first.json().data).toEqual([]);
576 + const setCookie = String(first.headers["set-cookie"] ?? "");
577 + expect(setCookie).toMatch(/dci_watch=/);
578 + expect(setCookie).toMatch(/HttpOnly/i);
579 + expect(first.headers["cache-control"]).toBe("no-store");
580 + const cookie = setCookie.split(";")[0]!;
581 + const add = await app.inject({ method: "POST", url: "/api/v1/watchlist", headers: { cookie }, payload: { entityType: "facility", slug: "apitest-one" } });
582 + expect([200, 201]).toContain(add.statusCode);
583 + expect(add.json().data).toMatchObject({ entityType: "facility", entityId: F1, slug: "apitest-one", name: "Apitest One Data Center" });
584 + const id = add.json().data.id as string;
585 + const list = (await app.inject({ url: "/api/v1/watchlist", headers: { cookie } })).json().data;
586 + expect(list.map((w: { entityId: string }) => w.entityId)).toEqual([F1]);
587 + const other = (await app.inject({ url: "/api/v1/watchlist" })).json().data;
588 + expect(other).toEqual([]); // a different visitor sees nothing
589 + const feed = (await app.inject({ url: "/api/v1/watchlist/feed", headers: { cookie } })).json().data;
590 + expect(feed.map((e: { id: string }) => e.id).sort()).toEqual([EVT, EVT_RUN].sort());
591 + expect((await app.inject({ method: "POST", url: "/api/v1/watchlist", headers: { cookie }, payload: { entityType: "nope", slug: "x" } })).statusCode).toBe(400);
592 + expect((await app.inject({ method: "POST", url: "/api/v1/watchlist", headers: { cookie }, payload: { entityType: "facility", slug: "does-not-exist" } })).statusCode).toBe(404);
593 + const del = await app.inject({ method: "DELETE", url: `/api/v1/watchlist/${id}`, headers: { cookie } });
594 + expect(del.json().data.deleted).toBe(true);
595 + expect((await app.inject({ url: "/api/v1/watchlist", headers: { cookie } })).json().data).toEqual([]);
596 + });
217 597 });
218 598
219 599 describe("admin", () => {
220 600 it("requires the token", async () => {
221 601 expect((await app.inject({ url: "/api/admin/connectors" })).statusCode).toBe(401);
222 602 expect((await app.inject({ url: "/api/admin/connectors", headers: { "x-dci-admin-token": "wrong" } })).statusCode).toBe(401);
223 − const ok = await app.inject({ url: "/api/admin/connectors", headers: { "x-dci-admin-token": TOKEN } });
603 + const ok = await app.inject({ url: "/api/admin/connectors", headers: ADMIN });
224 604 expect(ok.statusCode).toBe(200);
225 605 expect(Array.isArray(ok.json().data)).toBe(true);
226 606 expect(ok.headers["cache-control"]).toBe("no-store");
227 607 });
608 + it("rejects the placeholder token with 503", async () => {
609 + const { getEnv } = await import("./env.js");
610 + const saved = process.env.DCI_ADMIN_TOKEN;
611 + process.env.DCI_ADMIN_TOKEN = "change-me";
612 + resetEnv();
613 + expect(getEnv().adminToken).toBeNull();
614 + const r = await app.inject({ url: "/api/admin/connectors", headers: { "x-dci-admin-token": "change-me" } });
615 + expect(r.statusCode).toBe(503);
616 + process.env.DCI_ADMIN_TOKEN = saved;
617 + resetEnv();
618 + expect(getEnv().adminToken).toBe(TOKEN);
619 + });
228 620 it("constant-time token compare helper", async () => {
229 621 const { tokenMatches } = await import("./routes/admin/index.js");
230 622 expect(tokenMatches("abc", "abc")).toBe(true);
@@ -232,21 +624,133 @@ describe("admin", () => {
232 624 expect(tokenMatches(undefined, "abc")).toBe(false);
233 625 expect(tokenMatches("abc", null)).toBe(false);
234 626 });
627 + it("connector health carries the full DTO", async () => {
628 + const rows = (await app.inject({ url: "/api/admin/connectors", headers: ADMIN })).json().data;
629 + if (rows.length) {
630 + const c = rows[0];
631 + for (const k of ["quarantine", "consecutiveFailures", "urlsDiscovered", "urlsFetched", "newDocs", "changedDocs", "recordsCreated", "recordsModified", "rejectedClaims", "httpErrors", "antiBotEscalations", "scrapflyRequests", "firecrawlRequests"]) expect(c, k).toHaveProperty(k);
632 + expect(["ok", "degraded", "failing", "paused", "never_run", "blocked", "schema_change", "no_new_content", "quarantine"]).toContain(c.health);
633 + }
634 + });
635 + it("quality overview, flags and review queue", async () => {
636 + const r = await app.inject({ url: "/api/admin/quality", headers: ADMIN });
637 + expect(r.statusCode).toBe(200);
638 + const d = r.json().data;
639 + expect(d.openFlags).toBeGreaterThanOrEqual(1);
640 + expect(d.bySeverity.warn).toBeGreaterThanOrEqual(1);
641 + expect(d.byCode.some((c: { code: string }) => c.code === "mw_change_5x")).toBe(true);
642 + expect(d.reviewQueue[0]).toMatchObject({ id: FLAG, entity: { slug: "apitest-one", name: "Apitest One Data Center" }, priority: 999 });
643 + expect(d.largest.operationalFacilityMw.length).toBeGreaterThan(0);
644 + expect(d.largest.operationalFacilityMw[0]).toHaveProperty("scope");
645 + expect(d.largest.projectMw.length).toBeGreaterThan(0);
646 + for (const k of ["suspiciousMw", "suspiciousInvestment", "unscopedClaims", "claimsInReview", "possibleDuplicates", "projectFalsePositiveCandidates", "unverifiedLargeProjects", "facilitiesMissingCountry", "projectsWithoutFacility", "orphanOperators"]) expect(typeof d[k], k).toBe("number");
647 + const flags = (await app.inject({ url: "/api/admin/quality/flags?status=open&code=mw_change_5x&min_priority=900", headers: ADMIN })).json();
648 + expect(flags.data.map((f: { id: string }) => f.id)).toContain(FLAG);
649 + const res = await app.inject({ method: "POST", url: `/api/admin/quality/flags/${FLAG}/resolve`, headers: ADMIN, payload: { resolution: "checked against the operator page" } });
650 + expect(res.statusCode).toBe(200);
651 + expect(res.json().data).toMatchObject({ id: FLAG, status: "resolved", resolution: "checked against the operator page" });
652 + const dq = (await app.inject({ url: "/api/v1/datacenters/apitest-one" })).json().data.dataQuality;
653 + expect(dq.openFlags.some((f: { code: string }) => f.code === "mw_change_5x")).toBe(false);
654 + expect((await app.inject({ method: "POST", url: "/api/admin/quality/flags/flg_nope/dismiss", headers: ADMIN, payload: {} })).statusCode).toBe(404);
655 + });
656 + it("claims listing and status change", async () => {
657 + const list = (await app.inject({ url: `/api/admin/claims?subject_type=facility&subject_id=${F1}`, headers: ADMIN })).json();
658 + expect(list.data.map((c: { id: string }) => c.id)).toEqual([CLAIM]);
659 + const r = await app.inject({ method: "POST", url: `/api/admin/claims/${CLAIM}/status`, headers: ADMIN, payload: { status: "review", reason: "double-check scope" } });
660 + expect(r.statusCode).toBe(200);
661 + expect(r.json().data.claim).toMatchObject({ id: CLAIM, status: "review" });
662 + expect(r.json().data.flagged).toBe(false);
663 + await app.inject({ method: "POST", url: `/api/admin/claims/${CLAIM}/status`, headers: ADMIN, payload: { status: "current" } });
664 + expect((await app.inject({ method: "POST", url: `/api/admin/claims/${CLAIM}/status`, headers: ADMIN, payload: { status: "bogus" } })).statusCode).toBe(400);
665 + });
666 + it("worker proxies answer 502 when the worker is unreachable", async () => {
667 + const g = await app.inject({ url: "/api/admin/data-gaps", headers: ADMIN });
668 + expect(g.statusCode).toBe(502);
669 + expect(typeof g.json().error).toBe("string");
670 + const t = await app.inject({ url: "/api/admin/documents/doc_nope/trace", headers: ADMIN });
671 + expect(t.statusCode).toBe(502);
672 + });
673 + it("hides a project and excludes it everywhere, then unhides", async () => {
674 + const hide = await app.inject({ method: "POST", url: `/api/admin/projects/${PRJ}/hide`, headers: ADMIN, payload: { reason: "executive appointment, not a project" } });
675 + expect(hide.statusCode).toBe(200);
676 + expect((await app.inject({ url: "/api/v1/projects?country=ZZ" })).json().meta.total).toBe(0);
677 + expect((await app.inject({ url: "/api/v1/projects/apitest-project" })).statusCode).toBe(404);
678 + expect((await app.inject({ url: "/api/v1/countries/zz" })).json().data.projectCount).toBe(0);
679 + expect((await app.inject({ url: "/api/v1/metros/apitest-metro" })).json().data.projectCount).toBe(0);
680 + expect((await app.inject({ url: "/api/v1/explore?entity=projects&country=ZZ" })).json().data.total).toBe(0);
681 + const overlay = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=-74,45,-73,46&country=ZZ&layer=projects" })).json().data.overlay ?? [];
682 + expect(overlay.some((x: { slug: string }) => x.slug === "apitest-project")).toBe(false);
683 + expect((await app.inject({ url: "/api/v1/pulse?window=7d" })).json().data.newProjects).toBe(0 + (await sql`select count(*)::int as n from projects where hidden = false and merged_into is null and created_at >= now() - interval '7 days'`)[0]!.n);
684 + const prov = await sql`select value from provenance where entity_type = 'project' and entity_id = ${PRJ} and field = 'hidden' and source_id = 'src_manual'`;
685 + expect(prov.length).toBe(1);
686 + const unhide = await app.inject({ method: "POST", url: `/api/admin/projects/${PRJ}/unhide`, headers: ADMIN, payload: {} });
687 + expect(unhide.statusCode).toBe(200);
688 + expect((await app.inject({ url: "/api/v1/projects?country=ZZ" })).json().meta.total).toBe(1);
689 + expect((await app.inject({ method: "POST", url: `/api/admin/projects/${PRJ}/hide`, headers: ADMIN, payload: {} })).statusCode).toBe(400);
690 + });
691 + it("patches a project with provenance", async () => {
692 + const r = await app.inject({ method: "PATCH", url: `/api/admin/projects/${PRJ}`, headers: ADMIN, payload: { planned_mw: 120, expected_opening: "2027-Q3", note: "operator update" } });
693 + expect(r.statusCode).toBe(200);
694 + const d = (await app.inject({ url: "/api/v1/projects/apitest-project" })).json().data;
695 + expect(d.plannedMw).toBe(120);
696 + expect(d.expectedOpening).toBe("2027-Q3");
697 + const prov = await sql`select field from provenance where entity_type = 'project' and entity_id = ${PRJ} and source_id = 'src_manual' and is_current order by field`;
698 + expect(prov.map((p) => p.field)).toEqual(expect.arrayContaining(["expectedOpening", "plannedMw"]));
699 + expect((await app.inject({ method: "PATCH", url: `/api/admin/projects/${PRJ}`, headers: ADMIN, payload: { expected_opening: "someday" } })).statusCode).toBe(400);
700 + expect((await app.inject({ method: "PATCH", url: `/api/admin/projects/${PRJ}`, headers: ADMIN, payload: { bogus: 1 } })).statusCode).toBe(400);
701 + });
702 + it("links a building to its campus (containment)", async () => {
703 + const r = await app.inject({ method: "POST", url: `/api/admin/facilities/${F1}/parent`, headers: ADMIN, payload: { parentId: F2 } });
704 + expect(r.statusCode).toBe(200);
705 + const child = (await app.inject({ url: "/api/v1/datacenters/apitest-one" })).json().data;
706 + expect(child.recordScope).toBe("building");
707 + expect(child.parentFacility).toEqual({ id: F2, slug: "apitest-two", name: "Apitest Two Campus" });
708 + const parent = (await app.inject({ url: "/api/v1/datacenters/apitest-two" })).json().data;
709 + expect(parent.recordScope).toBe("campus");
710 + expect(parent.buildings.map((b: { slug: string }) => b.slug)).toEqual(["apitest-one"]);
711 + // containment-aware aggregate: the campus row with a building is not counted; the building keeps its own MW
712 + const co = (await app.inject({ url: "/api/v1/countries/zz" })).json().data;
713 + expect(co.pipeline.operational).toMatchObject({ count: 1, mw: 12.5 });
714 + expect(co.pipeline.construction.count).toBe(0);
715 + expect((await app.inject({ method: "POST", url: `/api/admin/facilities/${F2}/parent`, headers: ADMIN, payload: { parentId: F1 } })).statusCode).toBe(400); // cycle
716 + const undo = await app.inject({ method: "POST", url: `/api/admin/facilities/${F1}/parent`, headers: ADMIN, payload: { parentId: null } });
717 + expect(undo.statusCode).toBe(200);
718 + expect((await app.inject({ url: "/api/v1/datacenters/apitest-one" })).json().data.recordScope).toBe("facility");
719 + });
720 + it("run changes and rollback", async () => {
721 + const ch = (await app.inject({ url: `/api/admin/runs/${RUN}/changes`, headers: ADMIN })).json().data;
722 + expect(ch.provenance.length).toBe(1);
723 + expect(ch.claims.map((c: { id: string }) => c.id)).toEqual([CLAIM]);
724 + expect(ch.events.map((e: { id: string }) => e.id)).toEqual([EVT_RUN]);
725 + expect((await app.inject({ url: "/api/admin/runs/run_nope/changes", headers: ADMIN })).statusCode).toBe(404);
726 + const rb = await app.inject({ method: "POST", url: `/api/admin/runs/${RUN}/rollback`, headers: ADMIN, payload: {} });
727 + expect(rb.statusCode).toBe(200);
728 + const claim = await sql`select status, rejection_reason from claims where id = ${CLAIM}`;
729 + expect(claim[0]).toMatchObject({ status: "rejected", rejection_reason: "rollback" });
730 + const prov = await sql`select id, is_current, is_winner from provenance where entity_id = ${F1} and field = 'itCapacityMw' order by id`;
731 + expect(prov.find((p) => p.id === PROV_RUN)).toMatchObject({ is_current: false, is_winner: false });
732 + expect(prov.find((p) => p.id === PROV_OLD)).toMatchObject({ is_current: true, is_winner: true });
733 + const ev = await sql`select review_status from events where id = ${EVT_RUN}`;
734 + expect(ev[0]!.review_status).toBe("rejected");
735 + const ids = (await app.inject({ url: "/api/v1/events?country=ZZ&per_page=100" })).json().data.map((e: { id: string }) => e.id);
736 + expect(ids).toContain(EVT);
737 + expect(ids).not.toContain(EVT_RUN);
738 + });
235 739 it("curates a facility with provenance + event", async () => {
236 − const r = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: { "x-dci-admin-token": TOKEN }, payload: { total_power_mw: 25, status: "expansion", note: "test" } });
740 + const r = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: ADMIN, payload: { total_power_mw: 25, status: "expansion", note: "test" } });
237 741 expect(r.statusCode).toBe(200);
238 742 const body = r.json().data;
239 743 expect(body.changed.map((c: { field: string }) => c.field).sort()).toEqual(["status", "totalPowerMw"]);
240 744 expect(body.facility.totalPowerMw).toBe(25);
241 745 const prov = await sql`select field, value, source_id from provenance where entity_id = ${F1} and source_id = 'src_manual' order by field`;
242 − expect(prov.map((p) => p.field)).toEqual(["status", "totalPowerMw"]);
243 − const ev = await sql`select event_type, significance from events where entity_id = ${F1} and source_id = 'src_manual'`;
746 + expect(prov.map((p) => p.field)).toEqual(expect.arrayContaining(["status", "totalPowerMw"]));
747 + const ev = await sql`select event_type, significance from events where entity_id = ${F1} and source_id = 'src_manual' and event_type = 'status_changed'`;
244 748 expect(ev[0]).toMatchObject({ event_type: "status_changed", significance: 90 });
245 − const bad = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: { "x-dci-admin-token": TOKEN }, payload: { status: "not-a-status" } });
749 + const bad = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: ADMIN, payload: { status: "not-a-status" } });
246 750 expect(bad.statusCode).toBe(400);
247 751 });
248 752 it("merges a duplicate facility", async () => {
249 − const r = await app.inject({ method: "POST", url: `/api/admin/facilities/${F2}/merge`, headers: { "x-dci-admin-token": TOKEN }, payload: { into: "apitest-one" } });
753 + const r = await app.inject({ method: "POST", url: `/api/admin/facilities/${F2}/merge`, headers: ADMIN, payload: { into: "apitest-one" } });
250 754 expect(r.statusCode).toBe(200);
251 755 expect(r.json().data).toMatchObject({ into: F1, merged: F2 });
252 756 const list = (await app.inject({ url: "/api/v1/datacenters?country=ZZ" })).json();
@@ -258,15 +762,15 @@ describe("admin", () => {
258 762 expect(prj.facility.id).toBe(F1);
259 763 });
260 764 it("validates connector yaml", async () => {
261 − const good = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: { "x-dci-admin-token": TOKEN }, payload: { yaml: "id: apitest\nname: Apitest\ndomain: example.test\nkind: operator\n" } });
765 + const good = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: ADMIN, payload: { yaml: "id: apitest\nname: Apitest\ndomain: example.test\nkind: operator\n" } });
262 766 expect(good.json().data.ok).toBe(true);
263 − const bad = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: { "x-dci-admin-token": TOKEN }, payload: { yaml: "id: Bad Id\nname: x\n" } });
767 + const bad = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: ADMIN, payload: { yaml: "id: Bad Id\nname: x\n" } });
264 768 expect(bad.json().data.ok).toBe(false);
265 769 expect(bad.json().data.errors.length).toBeGreaterThan(0);
266 770 });
267 771 it("previews an extractor against inline html", async () => {
268 772 const html = `<html><head><title>Apitest DC1</title><script type="application/ld+json">{"@type":"Place","name":"Apitest DC1","address":{"streetAddress":"1 Rue Test","addressLocality":"Montréal","addressCountry":"CA"},"geo":{"latitude":45.5,"longitude":-73.6}}</script></head><body><h1>Apitest DC1</h1><div class="specs">Total power 36 MW</div></body></html>`;
269 − const r = await app.inject({ method: "POST", url: "/api/admin/devtool/preview", headers: { "x-dci-admin-token": TOKEN }, payload: { html, url: "https://example.test/data-centers/dc1", extractor: { kind: "facility", key: "apitest:{code}", fields: { name: "h1", code: { selector: "h1", regex: "\\b(DC\\d+)\\b" }, city: { jsonld: "Place", jsonPath: "address.addressLocality" }, countryIso2: { jsonld: "Place", jsonPath: "address.addressCountry", transform: "country" }, lat: { jsonld: "Place", jsonPath: "geo.latitude", transform: "float" }, lng: { jsonld: "Place", jsonPath: "geo.longitude", transform: "float" }, totalPowerMw: { selector: ".specs", regex: "(\\d+(?:\\.\\d+)?\\s*MW)", transform: "mw" } } } } });
773 + const r = await app.inject({ method: "POST", url: "/api/admin/devtool/preview", headers: ADMIN, payload: { html, url: "https://example.test/data-centers/dc1", extractor: { kind: "facility", key: "apitest:{code}", fields: { name: "h1", code: { selector: "h1", regex: "\\b(DC\\d+)\\b" }, city: { jsonld: "Place", jsonPath: "address.addressLocality" }, countryIso2: { jsonld: "Place", jsonPath: "address.addressCountry", transform: "country" }, lat: { jsonld: "Place", jsonPath: "geo.latitude", transform: "float" }, lng: { jsonld: "Place", jsonPath: "geo.longitude", transform: "float" }, totalPowerMw: { selector: ".specs", regex: "(\\d+(?:\\.\\d+)?\\s*MW)", transform: "mw" } } } } });
270 774 expect(r.statusCode).toBe(200);
271 775 const d = r.json().data;
272 776 expect(d.records.length).toBe(1);
@@ -275,4 +779,8 @@ describe("admin", () => {
275 779 expect(d.entities[0].geo).toMatchObject({ lat: 45.5, lng: -73.6 });
276 780 expect(d.validation).toMatchObject({ total: 1, valid: 1, rejected: 0 });
277 781 });
782 + it("maintenance body is strict", async () => {
783 + const bad = await app.inject({ method: "POST", url: "/api/admin/maintenance/rankings", headers: ADMIN, payload: { evil: "x" } });
784 + expect(bad.statusCode).toBe(400);
785 + });
278 786 });
modified apps/api/src/app.ts +2 −2
@@ -60,9 +60,9 @@ export async function buildApp(opts: BuildOptions = {}): Promise<FastifyInstance
60 60 await app.register(swagger, {
61 61 openapi: {
62 62 openapi: "3.1.0",
63 − info: { title: "DataCenterIndex API", version: "1.0.0", description: "Public read API (`/api/v1`, envelope `{ data, meta, sources }`) and token-protected admin API (`/api/admin`, header `x-dci-admin-token`). See packages/core/src/api-types.ts for response types." },
63 + info: { title: "DataCenterIndex API", version: "2.0.0", description: "Public read API (`/api/v1`, envelope `{ data, meta, sources }`, weak ETags, `s-maxage` caching) and token-protected admin API (`/api/admin`, header `x-dci-admin-token`). Response `data` types are named after packages/core/src/api-types.ts (the contract); every 200 response documents the envelope and an `x-example`. Figures are published values only — coverage and methodology are stated in `meta.methodology`." },
64 64 servers: [{ url: env.siteUrl }, { url: `http://127.0.0.1:${env.port}` }],
65 − tags: [{ name: "facilities" }, { name: "operators" }, { name: "countries" }, { name: "metros" }, { name: "cloud-regions" }, { name: "ixps" }, { name: "projects" }, { name: "events" }, { name: "map" }, { name: "search" }, { name: "rankings" }, { name: "dashboard" }, { name: "stats" }, { name: "sources" }, { name: "sitemap" }, { name: "system" }, { name: "admin" }],
65 + tags: [{ name: "facilities" }, { name: "operators" }, { name: "countries" }, { name: "metros" }, { name: "cloud-regions" }, { name: "ixps" }, { name: "projects" }, { name: "events" }, { name: "map" }, { name: "search" }, { name: "rankings" }, { name: "dashboard" }, { name: "stats" }, { name: "sources" }, { name: "sitemap" }, { name: "nearby" }, { name: "pulse" }, { name: "compare" }, { name: "explore" }, { name: "coverage" }, { name: "ai" }, { name: "power" }, { name: "connectivity" }, { name: "time-machine" }, { name: "download" }, { name: "watchlist" }, { name: "docs" }, { name: "system" }, { name: "admin" }],
66 66 components: { securitySchemes: { adminToken: { type: "apiKey", in: "header", name: "x-dci-admin-token" } } },
67 67 },
68 68 });
modified apps/api/src/env.ts +5 −1
@@ -19,7 +19,10 @@ export interface ApiEnv {
19 19 s3AccessKey: string;
20 20 s3SecretKey: string;
21 21 s3Region: string;
22 + /** null when DCI_ADMIN_TOKEN is missing, empty or the "change-me" placeholder → admin API answers 503 */
22 23 adminToken: string | null;
24 + /** internal HTTP server of the worker (GET /trace/<documentId>, GET /data-gaps) */
25 + workerUrl: string;
23 26 /** directory holding connector YAML files */
24 27 configDir: string;
25 28 /** optional path to the worker connectors registry (parsers) for the dev tool; null = generic only */
@@ -79,7 +82,8 @@ export function getEnv(): ApiEnv {
79 82 s3AccessKey: e.S3_ACCESS_KEY ?? "dci",
80 83 s3SecretKey: e.S3_SECRET_KEY ?? "",
81 84 s3Region: e.S3_REGION ?? "us-east-1",
82 − adminToken: e.DCI_ADMIN_TOKEN && e.DCI_ADMIN_TOKEN !== "change-me" ? e.DCI_ADMIN_TOKEN : e.DCI_ADMIN_TOKEN ?? null,
85 + adminToken: e.DCI_ADMIN_TOKEN && e.DCI_ADMIN_TOKEN.trim() !== "" && e.DCI_ADMIN_TOKEN !== "change-me" ? e.DCI_ADMIN_TOKEN : null,
86 + workerUrl: (e.DCI_WORKER_URL && e.DCI_WORKER_URL.trim()) || "http://127.0.0.1:8320",
83 87 configDir: e.DCI_CONFIG_DIR ? resolve(e.DCI_CONFIG_DIR) : join(root, "config", "connectors"),
84 88 workerConnectorsDir: existsSync(workerDir) ? workerDir : null,
85 89 logLevel: (e.DCI_LOG_LEVEL ?? "info").toLowerCase(),
modified apps/api/src/queues.ts +1 −1
@@ -24,7 +24,7 @@ export interface CrawlJobData {
24 24 requestedBy?: string;
25 25 }
26 26
27 −export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup";
27 +export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup" | "quality" | "snapshot";
28 28 export interface MaintenanceJobData { kind: MaintenanceKind; task: MaintenanceKind; requestedBy?: string; [k: string]: unknown }
29 29
30 30 const queues = new Map<string, Queue>();
added apps/api/src/repositories/admin/claims.ts +47 −0
@@ -0,0 +1,47 @@
1 +/** Admin claim reads / status changes. Changing a claim's status never rewrites an entity column: the worker's next reconciliation does. */
2 +import type { ClaimDTO } from "@dci/core";
3 +import { CAPACITY_COLUMN } from "@dci/core";
4 +import { pg, claimCols, andAll, page, type Fragment } from "../../lib/sql.js";
5 +import { int, num, reqStr, str, type Row } from "../../lib/rows.js";
6 +import { claimDto } from "../../lib/dto.js";
7 +
8 +export interface ClaimFilters { subjectType?: string; subjectId?: string; status?: string; predicate?: string; page?: number; perPage?: number }
9 +
10 +export async function listClaims(f: ClaimFilters): Promise<{ items: ClaimDTO[]; total: number; page: number; perPage: number }> {
11 + const sql = pg();
12 + const pg_ = page(f.page, f.perPage, 500, 50);
13 + const c: Fragment[] = [];
14 + if (f.subjectType) c.push(sql`k.subject_type = ${f.subjectType}`);
15 + if (f.subjectId) c.push(sql`k.subject_id = ${f.subjectId}`);
16 + if (f.status && f.status !== "all") c.push(sql`k.status = ${f.status}`);
17 + if (f.predicate) c.push(sql`k.predicate = ${f.predicate}`);
18 + const rows = await sql<Row[]>`select ${claimCols(sql)}, count(*) over() as total from claims k left join sources s on s.id = k.source_id where ${andAll(sql, c)} order by k.last_observed desc, k.id limit ${pg_.perPage} offset ${pg_.offset}`;
19 + return { items: rows.map((r) => claimDto(r, false)), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage };
20 +}
21 +
22 +const FACILITY_COL: Record<string, string> = { itCapacityMw: "it_capacity_mw", totalPowerMw: "total_power_mw", plannedPowerMw: "planned_power_mw", utilityCapacityMw: "utility_capacity_mw", gridConnectionMw: "grid_connection_mw", ultimateCampusMw: "ultimate_campus_mw" };
23 +
24 +/** Set a claim status; when a rejected claim was backing a displayed facility MW column, a quality flag is raised (column untouched). */
25 +export async function setClaimStatus(id: string, status: "current" | "rejected" | "review", reason: string | null): Promise<{ claim: ClaimDTO; flagged: boolean } | null> {
26 + const sql = pg();
27 + const rows = await sql<Row[]>`update claims set status = ${status}, rejection_reason = ${status === "rejected" ? reason : null} where id = ${id} returning id`;
28 + if (!rows[0]) return null;
29 + const full = (await sql<Row[]>`select ${claimCols(sql)} from claims k left join sources s on s.id = k.source_id where k.id = ${id}`)[0]!;
30 + let flagged = false;
31 + if (status === "rejected" && (str(full.subject_type) === "facility" || str(full.subject_type) === "campus")) {
32 + const camel = (CAPACITY_COLUMN as Record<string, string | null>)[reqStr(full.predicate)] ?? null;
33 + const col = camel ? FACILITY_COL[camel] : null;
34 + const v = num(full.value);
35 + if (col && v != null) {
36 + const fac = (await sql<Row[]>`select ${sql(col)} as v, name from facilities where id = ${reqStr(full.subject_id)}`)[0];
37 + const cur = num(fac?.v);
38 + if (fac && cur != null && Math.abs(cur - v) < 1e-9) {
39 + await sql`insert into quality_flags (id, entity_type, entity_id, claim_id, code, severity, field, message, details, priority, status, dedupe_key)
40 + values (${"flg_" + id.replace(/^clm_/, "").slice(0, 24)}, ${reqStr(full.subject_type)}, ${reqStr(full.subject_id)}, ${id}, 'claim_rejected_backing_value', 'warn', ${camel}, ${`rejected claim (${reqStr(full.predicate)} = ${v}) is still the displayed ${camel} value — re-reconcile`}, ${JSON.stringify({ claimId: id, value: v, reason })}::jsonb, 60, 'open', ${"claim_rejected:" + id})
41 + on conflict (dedupe_key) do update set status = 'open', message = excluded.message, details = excluded.details, updated_at = now()`;
42 + flagged = true;
43 + }
44 + }
45 + }
46 + return { claim: claimDto(full, false), flagged };
47 +}
modified apps/api/src/repositories/admin/connectors.ts +82 −15
@@ -1,26 +1,32 @@
1 +/**
2 + * Connector health (ConnectorHealthDTO) from the connectors row, its latest run, the last-24 h run stats, document
3 + * counters, the source licence and (best effort) ClickHouse crawl_log latency / premium request counts.
4 + */
1 5 import type { ConnectorHealthDTO } from "@dci/core";
2 6 import { chQuery } from "@dci/db/clickhouse";
3 7 import { pg } from "../../lib/sql.js";
4 8 import { bool, int, iso, json, num, reqStr, str, type Row } from "../../lib/rows.js";
5 9 import { asSourceKind } from "../../lib/dto.js";
6 10
7 −interface ChCost { connector_id: string; fetcher: string; n: string | number; avg_ms: string | number | null; credits: string | number | null }
11 +interface ChCost { connector_id: string; fetcher: string; n: string | number; avg_ms: string | number | null; credits: string | number | null; n24: string | number | null }
12 +interface ChStats { avgMs: number | null; cost: ConnectorHealthDTO["cost"]; requests24h: { scrapfly: number; firecrawl: number } }
8 13
9 −/** Per-connector cost / latency from ClickHouse crawl_log over the last 7 days (null when CH unavailable). */
10 −async function crawlLogStats(): Promise<Map<string, { avgMs: number | null; cost: ConnectorHealthDTO["cost"] }> | null> {
14 +/** Per-connector cost / latency from ClickHouse crawl_log over the last 7 days + premium request counts over 24 h (null when CH unavailable). */
15 +async function crawlLogStats(): Promise<Map<string, ChStats> | null> {
11 16 try {
12 17 const rows = await Promise.race([
13 − chQuery<ChCost>("select connector_id, fetcher, count() as n, avg(duration_ms) as avg_ms, sum(credits) as credits from crawl_log where ts >= now() - interval 7 day group by connector_id, fetcher"),
18 + chQuery<ChCost>("select connector_id, fetcher, count() as n, avg(duration_ms) as avg_ms, sum(credits) as credits, countIf(ts >= now() - interval 24 hour) as n24 from crawl_log where ts >= now() - interval 7 day group by connector_id, fetcher"),
14 19 new Promise<never>((_, rej) => setTimeout(() => rej(new Error("clickhouse timeout")), 3_000)),
15 20 ]);
16 − const out = new Map<string, { avgMs: number | null; cost: ConnectorHealthDTO["cost"]; _n: number; _sum: number }>();
21 + const out = new Map<string, ChStats & { _n: number; _sum: number }>();
17 22 for (const r of rows) {
18 23 let e = out.get(r.connector_id);
19 − if (!e) { e = { avgMs: null, cost: { direct: 0, firecrawl: 0, scrapfly: 0, credits: 0 }, _n: 0, _sum: 0 }; out.set(r.connector_id, e); }
24 + if (!e) { e = { avgMs: null, cost: { direct: 0, firecrawl: 0, scrapfly: 0, credits: 0 }, requests24h: { scrapfly: 0, firecrawl: 0 }, _n: 0, _sum: 0 }; out.set(r.connector_id, e); }
20 25 const n = int(r.n);
21 26 const f = r.fetcher === "firecrawl" ? "firecrawl" : r.fetcher === "scrapfly" ? "scrapfly" : "direct";
22 27 e.cost[f] += n;
23 28 e.cost.credits += num(r.credits) ?? 0;
29 + if (f !== "direct") e.requests24h[f] += int(r.n24);
24 30 e._n += n;
25 31 e._sum += (num(r.avg_ms) ?? 0) * n;
26 32 e.avgMs = e._n ? Math.round(e._sum / e._n) : null;
@@ -36,17 +42,36 @@ function runStat(stats: Record<string, unknown>, ...keys: string[]): number | nu
36 42 return null;
37 43 }
38 44
39 −export function connectorHealth(r: Row, ch: Map<string, { avgMs: number | null; cost: ConnectorHealthDTO["cost"] }> | null): ConnectorHealthDTO {
45 +export function connectorHealth(r: Row, ch: Map<string, ChStats> | null): ConnectorHealthDTO {
40 46 const stats = json<Record<string, unknown>>(r.stats, {});
41 47 const runStats = json<Record<string, unknown>>(r.run_stats, {});
48 + const day = json<Record<string, unknown>>(r.day_stats, {});
42 49 const merged = { ...stats, ...runStats };
43 50 const paused = bool(r.paused);
51 + const quarantine = bool(r.quarantine);
52 + const consecutiveFailures = int(r.consecutive_failures);
53 + const blockedSince = iso(r.blocked_since);
44 54 const healthCol = str(r.health) ?? "never_run";
45 − const health: ConnectorHealthDTO["health"] = paused ? "paused" : !r.last_run_at && !r.run_started ? "never_run" : healthCol === "ok" || healthCol === "degraded" || healthCol === "failing" ? healthCol : healthCol === "never_run" ? "never_run" : "ok";
55 + const lastError = str(r.last_error) ?? "";
46 56 const docsFetched = int(r.doc_fetched);
47 57 const extractTried = int(r.extract_tried);
58 + const extractionSuccess = extractTried ? Math.round((int(r.extract_ok) / extractTried) * 1000) / 1000 : runStat(runStats, "extractionSuccess");
59 + const neverRun = !r.last_run_at && !r.run_started;
60 + const weekRuns = int(r.week_runs);
61 + const weekNew = int(r.week_created) + int(r.week_changed);
62 + let health: ConnectorHealthDTO["health"];
63 + if (paused) health = "paused";
64 + else if (quarantine) health = "quarantine";
65 + else if (neverRun) health = "never_run";
66 + else if (blockedSince) health = "blocked";
67 + else if ((extractionSuccess != null && extractionSuccess < 0.2 && extractTried >= 10) || /schema|selector|parser/i.test(lastError)) health = "schema_change";
68 + else if (consecutiveFailures >= 3) health = "failing";
69 + else if (weekRuns >= 3 && weekNew === 0 && (str(r.run_status) === "ok" || str(r.last_status) === "ok")) health = "no_new_content";
70 + else if (healthCol === "ok" || healthCol === "degraded" || healthCol === "failing") health = healthCol;
71 + else health = str(r.run_status) === "failed" ? "failing" : "ok"; // stored health stale (e.g. never_run) but a run exists
48 72 const chStats = ch?.get(reqStr(r.id));
49 73 const cost: ConnectorHealthDTO["cost"] = chStats?.cost ?? { direct: runStat(merged, "direct", "fetch_direct") ?? 0, firecrawl: runStat(merged, "firecrawl", "fetch_firecrawl") ?? 0, scrapfly: runStat(merged, "scrapfly", "fetch_scrapfly") ?? 0, credits: runStat(merged, "credits", "creditsSpent") ?? 0 };
74 + const d = (k: string, ...alts: string[]) => runStat(day, k, ...alts) ?? 0;
50 75 return {
51 76 id: reqStr(r.id),
52 77 sourceName: reqStr(r.source_name),
@@ -57,6 +82,8 @@ export function connectorHealth(r: Row, ch: Map<string, { avgMs: number | null;
57 82 health,
58 83 parserVersion: reqStr(r.parser_version, "v1"),
59 84 lastRunAt: iso(r.last_run_at) ?? iso(r.run_started),
85 + lastSuccessAt: iso(r.last_success_at),
86 + lastFailureAt: iso(r.last_failure_at),
60 87 nextRunAt: iso(r.next_run_at),
61 88 lastStatus: str(r.last_status) ?? str(r.run_status),
62 89 discovered: runStat(runStats, "discovered", "urls") ?? int(r.doc_total),
@@ -64,19 +91,59 @@ export function connectorHealth(r: Row, ch: Map<string, { avgMs: number | null;
64 91 changed: runStat(runStats, "changed") ?? int(r.doc_changed),
65 92 failed: runStat(runStats, "failed", "errors") ?? int(r.doc_failed),
66 93 extracted: runStat(runStats, "extracted", "entities", "received") ?? int(r.doc_extracted),
67 − extractionSuccess: extractTried ? Math.round((int(r.extract_ok) / extractTried) * 1000) / 1000 : runStat(runStats, "extractionSuccess"),
94 + extractionSuccess,
68 95 avgResponseMs: chStats?.avgMs ?? runStat(merged, "avgMs", "avg_ms", "avgResponseMs"),
69 96 cost,
70 97 schedule: json<Record<string, string>>(r.schedule, {}),
98 + quarantine,
99 + consecutiveFailures,
100 + blockedSince,
101 + urlsDiscovered: d("discovered"),
102 + urlsFetched: d("fetched"),
103 + newDocs: d("created"),
104 + changedDocs: d("changed"),
105 + recordsCreated: d("created"),
106 + recordsModified: d("updated"),
107 + rejectedClaims: d("unscopedClaims"),
108 + httpErrors: d("failed"),
109 + antiBotEscalations: d("blockedFetches", "robotsBlocked"),
110 + scrapflyRequests: chStats ? chStats.requests24h.scrapfly : d("scrapfly", "fetch_scrapfly"),
111 + firecrawlRequests: chStats ? chStats.requests24h.firecrawl : d("firecrawl", "fetch_firecrawl"),
112 + priorityScore: num(r.priority_score),
113 + license: str(r.src_license),
114 + redistribution: str(r.src_redistribution),
71 115 };
72 116 }
73 117
118 +const DAY_KEYS = ["discovered", "fetched", "created", "changed", "updated", "unscopedClaims", "failed", "blockedFetches", "robotsBlocked", "scrapfly", "firecrawl", "fetch_scrapfly", "fetch_firecrawl"];
119 +
74 120 const CONNECTOR_SELECT = (sql: ReturnType<typeof pg>) => sql`
75 121 select c.*, r.id as run_id, r.status as run_status, r.stats as run_stats, r.started_at as run_started, r.finished_at as run_finished,
122 + ls.finished_at as last_success_at, lf.finished_at as last_failure_at,
76 123 coalesce(d.total, 0) as doc_total, coalesce(d.fetched, 0) as doc_fetched, coalesce(d.changed, 0) as doc_changed, coalesce(d.failed, 0) as doc_failed, coalesce(d.extracted, 0) as doc_extracted,
77 − coalesce(d.extract_ok, 0) as extract_ok, coalesce(d.extract_tried, 0) as extract_tried, coalesce(d.quarantined, 0) as doc_quarantined
124 + coalesce(d.extract_ok, 0) as extract_ok, coalesce(d.extract_tried, 0) as extract_tried, coalesce(d.quarantined, 0) as doc_quarantined,
125 + coalesce(s24.day_stats, '{}'::jsonb) as day_stats,
126 + coalesce(w.runs, 0) as week_runs, coalesce(w.created, 0) as week_created, coalesce(w.changed, 0) as week_changed,
127 + src.license as src_license, src.redistribution as src_redistribution
78 128 from connectors c
79 129 left join lateral (select id, status, stats, started_at, finished_at from connector_runs where connector_id = c.id order by started_at desc limit 1) r on true
130 + left join lateral (select finished_at from connector_runs where connector_id = c.id and status = 'ok' order by started_at desc limit 1) ls on true
131 + left join lateral (select finished_at from connector_runs where connector_id = c.id and status = 'failed' order by started_at desc limit 1) lf on true
132 + left join lateral (select license, redistribution from sources where connector_id = c.id order by priority, id limit 1) src on true
133 + left join (
134 + select connector_id, count(*)::int as runs, coalesce(sum((stats->>'created')::numeric), 0) as created, coalesce(sum((stats->>'changed')::numeric), 0) as changed
135 + from connector_runs where started_at >= now() - interval '7 days' and status <> 'running' group by connector_id
136 + ) w on w.connector_id = c.id
137 + left join (
138 + select connector_id, jsonb_object_agg(key, total) as day_stats
139 + from (
140 + select cr.connector_id, kv.key, sum(kv.value) as total
141 + from connector_runs cr
142 + cross join lateral (select key, (value #>> '{}')::numeric as value from jsonb_each(cr.stats) where key = any(${DAY_KEYS}) and jsonb_typeof(value) = 'number') kv
143 + where cr.started_at >= now() - interval '24 hours'
144 + group by cr.connector_id, kv.key
145 + ) x group by connector_id
146 + ) s24 on s24.connector_id = c.id
80 147 left join (
81 148 select connector_id, count(*)::int as total, count(*) filter (where last_fetched is not null)::int as fetched, count(*) filter (where change_count > 0)::int as changed,
82 149 count(*) filter (where error_count > 0)::int as failed, coalesce(sum(extract_count), 0)::int as extracted, count(*) filter (where extract_ok)::int as extract_ok,
@@ -98,7 +165,7 @@ export async function getConnectorAdmin(id: string): Promise<ConnectorAdminDetai
98 165 const r = rows[0];
99 166 if (!r) return null;
100 167 const [runs, byType, errors] = await Promise.all([
101 − sql<Row[]>`select id, task, started_at, finished_at, status, stats, error from connector_runs where connector_id = ${id} order by started_at desc limit 30`,
168 + sql<Row[]>`select id, task, started_at, finished_at, status, stats, error, quarantined from connector_runs where connector_id = ${id} order by started_at desc limit 30`,
102 169 sql<Row[]>`select page_type, count(*)::int as n, count(*) filter (where change_count > 0)::int as changed, count(*) filter (where error_count > 0)::int as failed, count(*) filter (where quarantined)::int as quarantined from documents where connector_id = ${id} group by page_type order by n desc`,
103 170 sql<Row[]>`select id, url, error, status_code, error_count, last_checked from documents where connector_id = ${id} and error is not null order by last_checked desc nulls last limit 10`,
104 171 ]);
@@ -107,7 +174,7 @@ export async function getConnectorAdmin(id: string): Promise<ConnectorAdminDetai
107 174 config: json<Record<string, unknown>>(r.config, {}),
108 175 paused: bool(r.paused),
109 176 lastError: str(r.last_error),
110 − runs: runs.map((x) => ({ id: reqStr(x.id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error) })),
177 + runs: runs.map((x) => ({ id: reqStr(x.id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), quarantined: bool(x.quarantined) })),
111 178 documentsByPageType: byType.map((x) => ({ pageType: reqStr(x.page_type), count: int(x.n), changed: int(x.changed), failed: int(x.failed), quarantined: int(x.quarantined) })),
112 179 errorSamples: errors.map((x) => ({ id: reqStr(x.id), url: reqStr(x.url), error: str(x.error), statusCode: num(x.status_code), errorCount: int(x.error_count), lastChecked: iso(x.last_checked) })),
113 180 createdAt: iso(r.created_at),
@@ -117,8 +184,8 @@ export async function getConnectorAdmin(id: string): Promise<ConnectorAdminDetai
117 184
118 185 export async function listRuns(connectorId: string | undefined, limit = 100): Promise<Array<Record<string, unknown>>> {
119 186 const sql = pg();
120 − const rows = await sql<Row[]>`select id, connector_id, task, started_at, finished_at, status, stats, error from connector_runs where ${connectorId ? sql`connector_id = ${connectorId}` : sql`true`} order by started_at desc limit ${limit}`;
121 − return rows.map((x) => ({ id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error) }));
187 + const rows = await sql<Row[]>`select id, connector_id, task, started_at, finished_at, status, stats, error, quarantined from connector_runs where ${connectorId ? sql`connector_id = ${connectorId}` : sql`true`} order by started_at desc limit ${limit}`;
188 + return rows.map((x) => ({ id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), quarantined: bool(x.quarantined) }));
122 189 }
123 190
124 191 export async function getRun(id: string): Promise<Record<string, unknown> | null> {
@@ -126,5 +193,5 @@ export async function getRun(id: string): Promise<Record<string, unknown> | null
126 193 const rows = await sql<Row[]>`select * from connector_runs where id = ${id}`;
127 194 const x = rows[0];
128 195 if (!x) return null;
129 − return { id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), log: json(x.log, []) };
196 + return { id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), log: json(x.log, []), quarantined: bool(x.quarantined) };
130 197 }
modified apps/api/src/repositories/admin/merge.ts +55 −0
@@ -4,6 +4,7 @@
4 4 * detail lookups follow the pointer).
5 5 */
6 6 import { pg } from "../../lib/sql.js";
7 +import { HttpError } from "../../lib/http.js";
7 8 import type { Row } from "../../lib/rows.js";
8 9
9 10 export interface MergeResult { into: string; merged: string; moved: Record<string, number> }
@@ -48,6 +49,60 @@ export async function mergeFacility(dupId: string, intoId: string, decidedBy = "
48 49 });
49 50 }
50 51
52 +export interface ParentResult { id: string; parentId: string | null; recordScope: string; parentRecordScope: string | null; changed: boolean }
53 +
54 +/**
55 + * Containment link: `childId` becomes a building of `parentId` (null detaches). Guards: parent exists, is not merged,
56 + * is not the child, and the parent's own chain never leads back to the child (no cycles, checked 5 levels up).
57 + * record_scope: child → building (or facility when detached), parent → campus (stays campus while it has children).
58 + */
59 +export async function setFacilityParent(childIdOrSlug: string, parentIdOrSlug: string | null, decidedBy = "admin"): Promise<ParentResult> {
60 + const sql = pg();
61 + const child = (await sql<Row[]>`select id, slug, name, parent_facility_id, record_scope, merged_into from facilities where id = ${childIdOrSlug} or slug = ${childIdOrSlug} order by (id = ${childIdOrSlug}) desc limit 1`)[0];
62 + if (!child) throw new HttpError(404, "facility not found");
63 + const childId = String(child.id);
64 + let parentId: string | null = null;
65 + if (parentIdOrSlug != null) {
66 + const parent = (await sql<Row[]>`select id, merged_into from facilities where id = ${parentIdOrSlug} or slug = ${parentIdOrSlug} order by (id = ${parentIdOrSlug}) desc limit 1`)[0];
67 + if (!parent) throw new HttpError(400, "parent facility not found");
68 + if (parent.merged_into) throw new HttpError(400, `parent is merged into ${String(parent.merged_into)}`);
69 + parentId = String(parent.id);
70 + if (parentId === childId) throw new HttpError(400, "a facility cannot be its own parent");
71 + // cycle check: walk up from the parent
72 + let cur: string | null = parentId;
73 + for (let i = 0; i < 5 && cur; i++) {
74 + const up: Row | undefined = (await sql<Row[]>`select parent_facility_id from facilities where id = ${cur}`)[0];
75 + cur = up?.parent_facility_id == null ? null : String(up.parent_facility_id);
76 + if (cur === childId) throw new HttpError(400, "containment cycle: the parent is already contained by this facility");
77 + }
78 + }
79 + const previous = child.parent_facility_id == null ? null : String(child.parent_facility_id);
80 + const scope = parentId ? "building" : "facility";
81 + if (previous === parentId && String(child.record_scope) === scope) return { id: childId, parentId, recordScope: scope, parentRecordScope: null, changed: false };
82 + await ensureManualSource();
83 + const now = new Date().toISOString();
84 + const url = `https://www.datacenterindex.io/admin/facilities/${childId}`;
85 + let parentScope: string | null = null;
86 + await sql.begin(async (tx) => {
87 + await tx`update facilities set parent_facility_id = ${parentId}, record_scope = ${scope}, updated_at = now() where id = ${childId}`;
88 + for (const [field, value] of [["parentFacilityId", parentId], ["recordScope", scope]] as Array<[string, unknown]>) {
89 + await tx`insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, document_id, url, first_observed, last_observed, retrieved_at, confidence, is_estimate, method, extractor_version, is_current, is_winner, note)
90 + values (${"prov_" + Math.random().toString(36).slice(2, 14)}, 'facility', ${childId}, ${field}, ${JSON.stringify(value)}::jsonb, 'src_manual', 'manual', null, ${url}, ${now}, ${now}, ${now}, 'high', false, 'manual', 'admin', true, true, ${"containment set by " + decidedBy})
91 + on conflict (entity_type, entity_id, field, source_id, url) do update set value = excluded.value, last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, is_current = true, is_winner = true, note = excluded.note`;
92 + }
93 + if (parentId) {
94 + await tx`update facilities set record_scope = 'campus', updated_at = now() where id = ${parentId} and record_scope <> 'campus'`;
95 + parentScope = "campus";
96 + }
97 + // a former parent stays a campus only while it still has building rows
98 + if (previous && previous !== parentId) {
99 + const r = await tx`update facilities set record_scope = 'facility', updated_at = now() where id = ${previous} and record_scope = 'campus' and not exists (select 1 from facilities c where c.parent_facility_id = ${previous} and c.merged_into is null)`;
100 + if (r.count && !parentId) parentScope = "facility";
101 + }
102 + });
103 + return { id: childId, parentId, recordScope: scope, parentRecordScope: parentScope, changed: true };
104 +}
105 +
51 106 /** Ensure the manual-curation source exists (kind registry, "Manual curation"). */
52 107 export async function ensureManualSource(): Promise<void> {
53 108 const sql = pg();
added apps/api/src/repositories/admin/projects.ts +126 −0
@@ -0,0 +1,126 @@
1 +/**
2 + * Project curation: hide / unhide (false positives are never deleted), merge (duplicate folded into a survivor),
3 + * PATCH curated fields. Every change writes src_manual provenance; only lifecycle-relevant changes emit an event
4 + * (status → project_status_changed, planned MW → planned_capacity_changed, hide/unhide → project_status_changed with
5 + * {hidden} values). Everything else is provenance-only.
6 + */
7 +import { createHash } from "node:crypto";
8 +import type postgres from "postgres";
9 +import { newId, normalizeName } from "@dci/core";
10 +import { pg } from "../../lib/sql.js";
11 +import { str, type Row } from "../../lib/rows.js";
12 +import { ensureManualSource } from "./merge.js";
13 +
14 +const ADMIN_URL = (id: string) => `https://www.datacenterindex.io/admin/projects/${id}`;
15 +
16 +export async function findProject(idOrSlug: string): Promise<Row | null> {
17 + const sql = pg();
18 + return (await sql<Row[]>`select * from projects where id = ${idOrSlug} or slug = ${idOrSlug} order by (id = ${idOrSlug}) desc limit 1`)[0] ?? null;
19 +}
20 +
21 +type Tx = postgres.TransactionSql;
22 +
23 +async function manualProvenance(tx: Tx, projectId: string, field: string, value: unknown, now: string, note: string | null, scope: string | null = null): Promise<void> {
24 + await tx`insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, document_id, url, first_observed, last_observed, retrieved_at, confidence, is_estimate, method, extractor_version, is_current, is_winner, scope, note)
25 + values (${newId("provenance")}, 'project', ${projectId}, ${field}, ${JSON.stringify(value)}::jsonb, 'src_manual', 'manual', null, ${ADMIN_URL(projectId)}, ${now}, ${now}, ${now}, 'high', false, 'manual', 'admin', true, true, ${scope}, ${note})
26 + on conflict (entity_type, entity_id, field, source_id, url) do update set value = excluded.value, last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, is_current = true, is_winner = true, scope = excluded.scope, note = excluded.note`;
27 + // the manual value is the displayed one: other current rows for the field are no longer winners
28 + await tx`update provenance set is_winner = false where entity_type = 'project' and entity_id = ${projectId} and field = ${field} and source_id <> 'src_manual' and is_winner`;
29 +}
30 +
31 +export async function setProjectHidden(idOrSlug: string, hidden: boolean, reason: string | null): Promise<{ id: string; hidden: boolean; changed: boolean } | null> {
32 + const sql = pg();
33 + const cur = await findProject(idOrSlug);
34 + if (!cur) return null;
35 + const id = String(cur.id);
36 + if (Boolean(cur.hidden) === hidden) return { id, hidden, changed: false };
37 + await ensureManualSource();
38 + const now = new Date().toISOString();
39 + await sql.begin(async (tx) => {
40 + await tx`update projects set hidden = ${hidden}, updated_at = now() where id = ${id}`;
41 + await manualProvenance(tx, id, "hidden", hidden, now, reason);
42 + await tx`insert into events (id, entity_type, entity_id, event_type, detected_at, effective_date, old_value, new_value, source_id, document_id, url, title, summary, significance, confidence, review_status, country_iso2, operator_id, project_id, fingerprint)
43 + values (${newId("event")}, 'project', ${id}, 'project_status_changed', ${now}, ${now.slice(0, 10)}, ${JSON.stringify({ hidden: !hidden })}::jsonb, ${JSON.stringify({ hidden })}::jsonb, 'src_manual', null, ${ADMIN_URL(id)}, ${`${String(cur.name)}: ${hidden ? "hidden by review" : "restored by review"}${reason ? ` (${reason})` : ""}`}, ${reason}, 10, 'high', 'approved', ${str(cur.country_iso2)}, ${str(cur.operator_id)}, ${id}, ${`manual:${hidden ? "hide" : "unhide"}:${id}:${now.slice(0, 10)}`})
44 + on conflict (fingerprint) do nothing`;
45 + });
46 + return { id, hidden, changed: true };
47 +}
48 +
49 +export interface ProjectMergeResult { into: string; merged: string; moved: Record<string, number> }
50 +
51 +export async function mergeProject(dupIdOrSlug: string, intoIdOrSlug: string): Promise<ProjectMergeResult> {
52 + const sql = pg();
53 + const [dup, into] = await Promise.all([findProject(dupIdOrSlug), findProject(intoIdOrSlug)]);
54 + if (!dup) throw new Error("project not found");
55 + if (!into) throw new Error("target project not found");
56 + const dupId = String(dup.id), intoId = String(into.id);
57 + if (dupId === intoId) throw new Error("cannot merge a project into itself");
58 + if (into.hidden === true) throw new Error("target project is hidden");
59 + if (into.merged_into) throw new Error(`target is itself merged into ${String(into.merged_into)}`);
60 + if (dup.merged_into) throw new Error(`project is already merged into ${String(dup.merged_into)}`);
61 + await ensureManualSource();
62 + return sql.begin(async (tx) => {
63 + const moved: Record<string, number> = {};
64 + const count = (r: { count: number }) => r.count;
65 + moved.events = count(await tx`update events set project_id = ${intoId} where project_id = ${dupId}`);
66 + moved.eventsEntity = count(await tx`update events set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${dupId}`);
67 + moved.timeline = count(await tx`update project_timeline set project_id = ${intoId} where project_id = ${dupId}`);
68 + moved.claims = count(await tx`update claims k set subject_id = ${intoId} where k.subject_type = 'project' and k.subject_id = ${dupId} and not exists (select 1 from claims q where q.subject_type = 'project' and q.subject_id = ${intoId} and q.predicate = k.predicate and q.source_id = k.source_id and q.url = k.url and coalesce(q.value, 0) = coalesce(k.value, 0) and coalesce(q.value_text, '') = coalesce(k.value_text, ''))`);
69 + await tx`delete from claims where subject_type = 'project' and subject_id = ${dupId}`;
70 + moved.provenance = count(await tx`update provenance p set entity_id = ${intoId} where p.entity_type = 'project' and p.entity_id = ${dupId} and not exists (select 1 from provenance q where q.entity_type = 'project' and q.entity_id = ${intoId} and q.field = p.field and q.source_id = p.source_id and q.url = p.url)`);
71 + await tx`delete from provenance where entity_type = 'project' and entity_id = ${dupId}`;
72 + moved.entityKeys = count(await tx`update entity_keys set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${dupId}`);
73 + moved.qualityFlags = count(await tx`update quality_flags set status = 'resolved', resolution = ${"merged into " + intoId}, resolved_by = 'admin', resolved_at = now(), updated_at = now() where entity_type = 'project' and entity_id = ${dupId} and status = 'open'`);
74 + await tx`update projects set merged_into = ${intoId}, updated_at = now() where id = ${dupId}`;
75 + await tx`update projects set last_update = now(), updated_at = now() where id = ${intoId}`;
76 + await tx`insert into events (id, entity_type, entity_id, event_type, detected_at, old_value, new_value, source_id, url, title, summary, significance, confidence, review_status, country_iso2, operator_id, project_id, fingerprint)
77 + values (${newId("event")}, 'project', ${intoId}, 'facility_updated', now(), ${JSON.stringify({ mergedProject: dupId, name: dup.name })}::jsonb, ${JSON.stringify({ into: intoId })}::jsonb, 'src_manual', ${ADMIN_URL(intoId)}, ${"Duplicate project merged: " + String(dup.name)}, 'Merged by admin', 20, 'high', 'approved', ${str(into.country_iso2)}, ${str(into.operator_id)}, ${intoId}, ${"merge:project:" + dupId + ":" + intoId})
78 + on conflict (fingerprint) do nothing`;
79 + return { into: intoId, merged: dupId, moved };
80 + });
81 +}
82 +
83 +/** snake_case column → provenance field name */
84 +const COLUMN_TO_FIELD: Record<string, string> = { planned_mw: "plannedMw", investment_usd: "investmentUsd", expected_opening: "expectedOpening", announced_on: "announcedOn", construction_started_on: "constructionStartedOn", approved_on: "approvedOn", permit_filed_on: "permitFiledOn", opened_on: "openedOn", operator_id: "operatorId", country_iso2: "countryIso2", metro_id: "metroId", geo_precision: "geoPrecision", project_class: "projectClass", ai_evidence: "aiEvidence", is_ai: "isAi" };
85 +
86 +export interface PatchChange { field: string; column: string; oldValue: unknown; newValue: unknown }
87 +
88 +export async function patchProject(idOrSlug: string, fields: Record<string, unknown>, note: string | null): Promise<{ id: string; changed: PatchChange[] } | null> {
89 + const sql = pg();
90 + const cur = await findProject(idOrSlug);
91 + if (!cur) return null;
92 + const id = String(cur.id);
93 + const changed: PatchChange[] = [];
94 + const set: Record<string, unknown> = {};
95 + for (const [col, val] of Object.entries(fields)) {
96 + if (val === undefined) continue;
97 + const before = cur[col] ?? null;
98 + if (JSON.stringify(before) === JSON.stringify(val)) continue;
99 + set[col] = val;
100 + changed.push({ field: COLUMN_TO_FIELD[col] ?? col, column: col, oldValue: before, newValue: val });
101 + }
102 + if (!changed.length) return { id, changed };
103 + await ensureManualSource();
104 + const now = new Date().toISOString();
105 + if (set.name) set.normalized_name = normalizeName(String(set.name));
106 + if (set.planned_mw !== undefined) { set.capacity_scope = "facility"; set.capacity_semantics = "planned_power_mw"; }
107 + if (set.investment_usd !== undefined) { set.investment_scope = "facility"; set.investment_semantics = "project_investment_usd"; }
108 + if (set.lat !== undefined && set.lat !== null && set.geo_precision === undefined) set.geo_precision = "approximate";
109 + set.updated_at = now;
110 + set.last_update = now;
111 + set.confidence = "high";
112 + await sql.begin(async (tx) => {
113 + await tx`update projects set ${tx(set)} where id = ${id}`;
114 + for (const c of changed) await manualProvenance(tx, id, c.field, c.newValue, now, note, /mw$|investment/i.test(c.column) ? "facility" : null);
115 + const statusChange = changed.find((c) => c.column === "status");
116 + const mwChange = changed.find((c) => c.column === "planned_mw");
117 + const evt = statusChange ? { type: "project_status_changed", old: statusChange.oldValue, nw: statusChange.newValue, sig: 85 } : mwChange ? { type: "planned_capacity_changed", old: mwChange.oldValue, nw: mwChange.newValue, sig: 75 } : null;
118 + if (evt) {
119 + const digest = createHash("sha256").update(JSON.stringify([evt.type, evt.nw])).digest("hex").slice(0, 16);
120 + await tx`insert into events (id, entity_type, entity_id, event_type, detected_at, effective_date, old_value, new_value, source_id, document_id, url, title, summary, significance, confidence, review_status, country_iso2, operator_id, project_id, fingerprint)
121 + values (${newId("event")}, 'project', ${id}, ${evt.type}, ${now}, ${now.slice(0, 10)}, ${JSON.stringify(evt.old)}::jsonb, ${JSON.stringify(evt.nw)}::jsonb, 'src_manual', null, ${"https://www.datacenterindex.io/projects/" + String(cur.slug)}, ${`${String(set.name ?? cur.name)}: ${statusChange ? "status" : "planned capacity"} updated by curation`}, ${note}, ${evt.sig}, 'high', 'approved', ${(set.country_iso2 as string | undefined) ?? str(cur.country_iso2)}, ${(set.operator_id as string | undefined) ?? str(cur.operator_id)}, ${id}, ${"manual:project:" + id + ":" + now.slice(0, 10) + ":" + digest})
122 + on conflict (fingerprint) do nothing`;
123 + }
124 + });
125 + return { id, changed };
126 +}
added apps/api/src/repositories/admin/quality.ts +140 −0
@@ -0,0 +1,140 @@
1 +/**
2 + * Admin data-quality reads: QualityOverview (largest values, review queue, counters) and the quality_flags list.
3 + * Every "largest" query is a plain ORDER BY on published figures — the point is to surface outliers for review.
4 + */
5 +import type { QualityFlagDTO, QualityOverview } from "@dci/core";
6 +import { pg, facilityView, pipelineMwAgg, projectLive, andAll, page, PLANNED_SET, CONSTRUCTION_SET, type Fragment } from "../../lib/sql.js";
7 +import { int, iso, num, reqStr, str, type Row } from "../../lib/rows.js";
8 +import { qualityFlagDto } from "../../lib/dto.js";
9 +import { entityKey, resolveEntityRefs } from "../../lib/resolve.js";
10 +
11 +const CODE_LABELS: Record<string, string> = {
12 + mw_single_site_gt_1000: "Single site > 1 000 MW",
13 + mw_building_gt_500: "One building > 500 MW",
14 + mw_market_statistic: "MW figure is a market statistic",
15 + mw_change_5x: "MW figure changed > 5×",
16 + mw_money_collision: "MW / money figure collision",
17 + mw_semantics_default: "MW semantics defaulted",
18 + mw_utility_not_it: "Utility / grid MW (not IT load)",
19 + mw_project_gt_2000: "Project > 2 000 MW",
20 + mw_density_implausible: "Implausible power density",
21 + mw_invalid: "Invalid MW figure",
22 + inv_single_site_gt_50b: "Single site investment > $50B",
23 + inv_gt_500b: "Investment > $500B (industry statistic)",
24 + inv_change_5x: "Investment changed > 5×",
25 + inv_money_mw_collision: "Investment / MW figure collision",
26 + inv_invalid: "Invalid investment figure",
27 + project_false_positive_candidate: "Project false-positive candidate",
28 + project_title_like_name: "Project named after a headline",
29 + project_no_location: "Project without a location",
30 + project_unknown_scope: "Project figure with unknown scope",
31 + facility_no_country: "Facility without a country",
32 + operator_bad_slug: "Operator with a broken slug",
33 + claim_rejected_backing_value: "Rejected claim was backing the displayed value",
34 +};
35 +
36 +export function flagLabel(code: string): string {
37 + if (CODE_LABELS[code]) return CODE_LABELS[code]!;
38 + if (code.startsWith("scope_")) return `Figure scope: ${code.slice(6).replace(/_/g, " ")}`;
39 + if (code.startsWith("inv_scope_")) return `Investment scope: ${code.slice(10).replace(/_/g, " ")}`;
40 + if (code.startsWith("inv_")) return `Investment: ${code.slice(4).replace(/_/g, " ")}`;
41 + if (code.startsWith("duplicate")) return "Possible duplicate";
42 + return code.replace(/_/g, " ").replace(/^\w/, (c) => c.toUpperCase());
43 +}
44 +
45 +function named(r: Row): { id: string; slug: string; name: string } {
46 + return { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) };
47 +}
48 +
49 +/** Attach {slug,name} refs to quality flag rows (one lookup per entity type). */
50 +export async function flagsWithRefs(rows: Row[]): Promise<QualityFlagDTO[]> {
51 + const refs = await resolveEntityRefs(rows.map((r) => ({ type: reqStr(r.entity_type), id: str(r.entity_id) })));
52 + return rows.map((r) => qualityFlagDto(r, refs.get(entityKey(r.entity_type, r.entity_id)) ?? null));
53 +}
54 +
55 +export async function qualityOverview(): Promise<QualityOverview> {
56 + const sql = pg();
57 + const live = projectLive(sql);
58 + const [counts, sev, codes, largeOps, largeCon, largePrj, largeInv, opPipe, metPipe, queue, alerts] = await Promise.all([
59 + sql<Row[]>`select
60 + (select count(*) from quality_flags where status = 'open')::int as open_flags,
61 + (select count(*) from entity_matches where status = 'pending')::int + (select count(*) from quality_flags where status = 'open' and code like 'duplicate%')::int as possible_duplicates,
62 + (select count(*) from quality_flags where status = 'open' and code like 'mw\\_%')::int as suspicious_mw,
63 + (select count(*) from quality_flags where status = 'open' and code like 'inv\\_%')::int as suspicious_inv,
64 + (select count(*) from quality_flags where status = 'open' and code = 'project_false_positive_candidate')::int as project_fp,
65 + (select count(*) from claims where status = 'unscoped')::int as unscoped_claims,
66 + (select count(*) from claims where status = 'review')::int as claims_review,
67 + (select count(*) from projects p where ${live} and p.planned_mw >= 500 and (coalesce(p.evidence_level, 'none') <> 'strong' or p.confidence in ('moderate', 'estimated', 'unverified')))::int as unverified_large,
68 + (select count(*) from facilities where merged_into is null and country_iso2 is null)::int as fac_no_country,
69 + (select count(*) from projects p where ${live} and p.facility_id is null)::int as prj_no_facility,
70 + (select count(*) from operators o where not exists (select 1 from facilities f where (f.operator_id = o.id or f.owner_id = o.id) and f.merged_into is null)
71 + and not exists (select 1 from projects p where p.operator_id = o.id and ${live})
72 + and not exists (select 1 from cloud_regions r where r.provider_id = o.id)
73 + and not exists (select 1 from facility_tenants t where t.operator_id = o.id))::int as orphan_operators`,
74 + sql<Row[]>`select severity, count(*)::int as n from quality_flags where status = 'open' group by 1`,
75 + sql<Row[]>`select code, count(*)::int as n from quality_flags where status = 'open' group by 1 order by n desc`,
76 + sql<Row[]>`select id, slug, name, coalesce(it_capacity_mw, total_power_mw) as mw, capacity_scope as scope, capacity_semantics as semantics from facilities where merged_into is null and status in ('operational', 'partially_operational', 'expansion') and coalesce(it_capacity_mw, total_power_mw) is not null order by 4 desc limit 15`,
77 + sql<Row[]>`select id, slug, name, coalesce(planned_power_mw, it_capacity_mw, total_power_mw) as mw from facilities where merged_into is null and status = 'under_construction' and coalesce(planned_power_mw, it_capacity_mw, total_power_mw) is not null order by 4 desc limit 15`,
78 + sql<Row[]>`select p.id, p.slug, p.name, p.planned_mw as mw, p.capacity_scope as scope from projects p where ${live} and p.planned_mw is not null order by p.planned_mw desc limit 15`,
79 + sql<Row[]>`select p.id, p.slug, p.name, p.investment_usd as inv, p.investment_scope as scope from projects p where ${live} and p.investment_usd is not null order by p.investment_usd desc limit 15`,
80 + sql<Row[]>`select o.id, o.slug, o.name,
81 + coalesce((select sum(${pipelineMwAgg(sql)}) from ${facilityView(sql)} f where f.operator_id = o.id and f.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0)
82 + + coalesce((select sum(p.planned_mw) from projects p where p.operator_id = o.id and ${live} and p.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0) as mw
83 + from operators o order by mw desc nulls last limit 15`,
84 + sql<Row[]>`select m.id, m.slug, m.name,
85 + coalesce((select sum(${pipelineMwAgg(sql)}) from ${facilityView(sql)} f where f.metro_id = m.id and f.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0)
86 + + coalesce((select sum(p.planned_mw) from projects p where p.metro_id = m.id and ${live} and p.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0) as mw
87 + from metros m order by mw desc nulls last limit 15`,
88 + sql<Row[]>`select * from quality_flags where status = 'open' order by priority desc, created_at desc limit 50`,
89 + sql<Row[]>`select id, level, component, message, created_at from system_alerts where resolved_at is null order by created_at desc limit 20`,
90 + ]);
91 + const c = counts[0] ?? {};
92 + const bySeverity: Record<string, number> = {};
93 + for (const r of sev) bySeverity[reqStr(r.severity)] = int(r.n);
94 + return {
95 + openFlags: int(c.open_flags),
96 + bySeverity,
97 + byCode: codes.map((r) => ({ code: reqStr(r.code), count: int(r.n), label: flagLabel(reqStr(r.code)) })),
98 + possibleDuplicates: int(c.possible_duplicates),
99 + suspiciousMw: int(c.suspicious_mw),
100 + suspiciousInvestment: int(c.suspicious_inv),
101 + projectFalsePositiveCandidates: int(c.project_fp),
102 + unscopedClaims: int(c.unscoped_claims),
103 + claimsInReview: int(c.claims_review),
104 + unverifiedLargeProjects: int(c.unverified_large),
105 + facilitiesMissingCountry: int(c.fac_no_country),
106 + projectsWithoutFacility: int(c.prj_no_facility),
107 + orphanOperators: int(c.orphan_operators),
108 + largest: {
109 + operationalFacilityMw: largeOps.map((r) => ({ ...named(r), mw: num(r.mw) ?? 0, scope: str(r.scope), semantics: str(r.semantics) })),
110 + constructionFacilityMw: largeCon.map((r) => ({ ...named(r), mw: num(r.mw) ?? 0 })),
111 + projectMw: largePrj.map((r) => ({ ...named(r), mw: num(r.mw) ?? 0, scope: str(r.scope) })),
112 + investment: largeInv.map((r) => ({ ...named(r), investmentUsd: num(r.inv) ?? 0, scope: str(r.scope) })),
113 + operatorPipelineMw: opPipe.filter((r) => (num(r.mw) ?? 0) > 0).map((r) => ({ ...named(r), mw: Math.round((num(r.mw) ?? 0) * 100) / 100 })),
114 + metroPipelineMw: metPipe.filter((r) => (num(r.mw) ?? 0) > 0).map((r) => ({ ...named(r), mw: Math.round((num(r.mw) ?? 0) * 100) / 100 })),
115 + },
116 + reviewQueue: await flagsWithRefs(queue),
117 + alerts: alerts.map((a) => ({ id: reqStr(a.id), level: reqStr(a.level), component: reqStr(a.component), message: reqStr(a.message), createdAt: iso(a.created_at) ?? "" })),
118 + };
119 +}
120 +
121 +export interface FlagFilters { status?: string; code?: string; entityType?: string; minPriority?: number; page?: number; perPage?: number }
122 +
123 +export async function listFlags(f: FlagFilters): Promise<{ items: QualityFlagDTO[]; total: number; page: number; perPage: number }> {
124 + const sql = pg();
125 + const pg_ = page(f.page, f.perPage, 200, 50);
126 + const c: Fragment[] = [];
127 + if (f.status && f.status !== "all") c.push(sql`q.status = ${f.status}`);
128 + if (f.code) c.push(sql`q.code = ${f.code}`);
129 + if (f.entityType) c.push(sql`q.entity_type = ${f.entityType}`);
130 + if (f.minPriority != null) c.push(sql`q.priority >= ${f.minPriority}`);
131 + const rows = await sql<Row[]>`select q.*, count(*) over() as total from quality_flags q where ${andAll(sql, c)} order by q.priority desc, q.created_at desc limit ${pg_.perPage} offset ${pg_.offset}`;
132 + return { items: await flagsWithRefs(rows), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage };
133 +}
134 +
135 +export async function setFlagStatus(id: string, status: "resolved" | "dismissed", resolution: string | null): Promise<QualityFlagDTO | null> {
136 + const sql = pg();
137 + const rows = await sql<Row[]>`update quality_flags set status = ${status}, resolution = ${resolution}, resolved_by = 'admin', resolved_at = now(), updated_at = now() where id = ${id} returning *`;
138 + if (!rows[0]) return null;
139 + return (await flagsWithRefs(rows))[0] ?? null;
140 +}
added apps/api/src/repositories/admin/runs.ts +70 −0
@@ -0,0 +1,70 @@
1 +/**
2 + * Per-run change inspection and rollback. A rollback never deletes: claims → rejected (reason rollback), provenance rows
3 + * → is_current = false (winner restored to the latest remaining current row per field), events → review_status rejected.
4 + * Entity columns are left as they are; the worker's reconciliation re-derives them from the remaining claims.
5 + */
6 +import type { ClaimDTO, EventDTO, ProvenanceDTO } from "@dci/core";
7 +import { pg, claimCols, eventCols, eventJoins } from "../../lib/sql.js";
8 +import { bool, int, iso, reqStr, str, type Row } from "../../lib/rows.js";
9 +import { claimDto, eventDto, provenanceDto } from "../../lib/dto.js";
10 +import { entityKey, resolveEntityRefs } from "../../lib/resolve.js";
11 +
12 +const CAP = 2000;
13 +
14 +export interface RunChanges {
15 + runId: string;
16 + provenance: Array<ProvenanceDTO & { entityType: string; entityId: string; isCurrent: boolean }>;
17 + claims: ClaimDTO[];
18 + events: EventDTO[];
19 + documentVersions: Array<{ id: string; documentId: string; url: string | null; fetchedAt: string | null; significance: number; changes: number }>;
20 + counts: { provenance: number; claims: number; events: number; documentVersions: number };
21 +}
22 +
23 +export async function runExists(id: string): Promise<boolean> {
24 + const sql = pg();
25 + return (await sql`select 1 from connector_runs where id = ${id}`).length > 0;
26 +}
27 +
28 +export async function runChanges(id: string): Promise<RunChanges | null> {
29 + const sql = pg();
30 + if (!(await runExists(id))) return null;
31 + const [prov, claims, events, versions, counts] = await Promise.all([
32 + sql<Row[]>`select p.*, s.name as source_name, s.kind as source_kind from provenance p left join sources s on s.id = p.source_id where p.run_id = ${id} order by p.entity_type, p.entity_id, p.field limit ${CAP}`,
33 + sql<Row[]>`select ${claimCols(sql)} from claims k left join sources s on s.id = k.source_id where k.run_id = ${id} order by k.subject_type, k.subject_id, k.predicate limit ${CAP}`,
34 + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.run_id = ${id} order by e.detected_at desc limit ${CAP}`,
35 + sql<Row[]>`select v.id, v.document_id, d.url, v.fetched_at, v.significance, jsonb_array_length(coalesce(v.detected_changes, '[]'::jsonb)) as changes from document_versions v left join documents d on d.id = v.document_id where v.run_id = ${id} order by v.fetched_at desc limit ${CAP}`,
36 + sql<Row[]>`select (select count(*) from provenance where run_id = ${id})::int as p, (select count(*) from claims where run_id = ${id})::int as c, (select count(*) from events where run_id = ${id})::int as e, (select count(*) from document_versions where run_id = ${id})::int as v`,
37 + ]);
38 + const refs = await resolveEntityRefs(events.map((r) => ({ type: reqStr(r.entity_type), id: str(r.entity_id) })));
39 + const c = counts[0] ?? {};
40 + return {
41 + runId: id,
42 + provenance: prov.map((p) => ({ ...provenanceDto(p), entityType: reqStr(p.entity_type), entityId: reqStr(p.entity_id), isCurrent: bool(p.is_current) })),
43 + claims: claims.map((k) => claimDto(k, false)),
44 + events: events.map((e) => eventDto(e, refs.get(entityKey(e.entity_type, e.entity_id)) ?? null)),
45 + documentVersions: versions.map((v) => ({ id: reqStr(v.id), documentId: reqStr(v.document_id), url: str(v.url), fetchedAt: iso(v.fetched_at), significance: int(v.significance), changes: int(v.changes) })),
46 + counts: { provenance: int(c.p), claims: int(c.c), events: int(c.e), documentVersions: int(c.v) },
47 + };
48 +}
49 +
50 +export interface RollbackResult { runId: string; claimsRejected: number; provenanceRetired: number; winnersRestored: number; eventsRejected: number }
51 +
52 +export async function rollbackRun(id: string): Promise<RollbackResult | null> {
53 + const sql = pg();
54 + if (!(await runExists(id))) return null;
55 + return sql.begin(async (tx) => {
56 + const claims = await tx`update claims set status = 'rejected', rejection_reason = 'rollback' where run_id = ${id} and status <> 'rejected'`;
57 + const touched = await tx<Row[]>`select distinct entity_type, entity_id, field from provenance where run_id = ${id} and is_current`;
58 + const prov = await tx`update provenance set is_current = false, is_winner = false where run_id = ${id} and is_current`;
59 + let restored = 0;
60 + for (const t of touched) {
61 + const r = await tx`update provenance set is_winner = true where id = (
62 + select id from provenance where entity_type = ${reqStr(t.entity_type)} and entity_id = ${reqStr(t.entity_id)} and field = ${reqStr(t.field)} and is_current order by last_observed desc, retrieved_at desc limit 1)
63 + and not exists (select 1 from provenance w where w.entity_type = ${reqStr(t.entity_type)} and w.entity_id = ${reqStr(t.entity_id)} and w.field = ${reqStr(t.field)} and w.is_current and w.is_winner)`;
64 + restored += r.count;
65 + }
66 + const events = await tx`update events set review_status = 'rejected' where run_id = ${id} and review_status <> 'rejected'`;
67 + await tx`update connector_runs set log = coalesce(log, '[]'::jsonb) || ${JSON.stringify([{ t: new Date().toISOString(), level: "warn", msg: `rolled back by admin: ${claims.count} claims rejected, ${prov.count} provenance rows retired, ${events.count} events rejected` }])}::jsonb where id = ${id}`;
68 + return { runId: id, claimsRejected: claims.count, provenanceRetired: prov.count, winnersRestored: restored, eventsRejected: events.count };
69 + });
70 +}
added apps/api/src/repositories/ai.ts +76 −0
@@ -0,0 +1,76 @@
1 +/**
2 + * /ai-infrastructure — AI / HPC index. Counts use ai_evidence confirmed | likely (or the legacy is_ai flag);
3 + * "associated" facilities are reported separately and never counted as AI infrastructure.
4 + */
5 +import type { AiIndex } from "@dci/core";
6 +import { pg, facilityJoins, facilitySummaryCols, facilityView, knownMwAgg, pipelineMwAgg, countedAgg, hasMwAgg, mwExpr, projectJoins, projectSummaryCols, projectLive, CONSTRUCTION_SET, type Sql } from "../lib/sql.js";
7 +import { int, num, reqStr, type Row } from "../lib/rows.js";
8 +import { asStatus, facilitySummary, projectSummary, round2, share } from "../lib/dto.js";
9 +import { eventsWhere } from "./events.js";
10 +
11 +export const AI_EVIDENCE_NOTE = "AI evidence levels: confirmed = a source explicitly describes the site as AI / HPC / GPU / accelerated-computing infrastructure; likely = AI-ready, high-density, liquid- or direct-to-chip cooling or GPU signals in the source; associated = only an AI tenant, customer or company mention (reported separately, NOT counted as AI infrastructure); unknown = no signal. A single keyword never confirms a site. Counts and MW cover confirmed + likely only; MW figures are published site-scoped figures (containment-aware), never extrapolated.";
12 +
13 +const AI_F = (sql: Sql) => sql`(f.ai_evidence in ('confirmed', 'likely') or f.is_ai or f.facility_type = 'ai')`;
14 +const AI_P = (sql: Sql) => sql`(p.ai_evidence in ('confirmed', 'likely') or p.is_ai)`;
15 +
16 +export async function aiIndex(): Promise<AiIndex> {
17 + const sql = pg();
18 + const known = knownMwAgg(sql), pipe = pipelineMwAgg(sql), counted = countedAgg(sql), hasMw = hasMwAgg(sql);
19 + const [fs, ps, topOps, topMetros, topCountries, stages, recentProjects, facilities, recentAnnouncements] = await Promise.all([
20 + sql<Row[]>`select count(*) filter (where ${counted} and ${AI_F(sql)})::int as facilities,
21 + count(*) filter (where ${counted} and f.ai_evidence = 'confirmed')::int as confirmed,
22 + count(*) filter (where ${counted} and f.ai_evidence = 'likely')::int as likely,
23 + count(*) filter (where ${counted} and f.ai_evidence = 'associated')::int as associated,
24 + sum(${pipe}) filter (where ${AI_F(sql)} and f.status = any(${CONSTRUCTION_SET}))::float as construction_mw,
25 + sum(${known}) filter (where ${AI_F(sql)} and f.status in ('operational', 'partially_operational', 'expansion'))::float as known_mw,
26 + count(*) filter (where ${counted} and ${AI_F(sql)} and ${hasMw})::int as with_mw,
27 + count(distinct f.country_iso2) filter (where ${AI_F(sql)})::int as f_countries,
28 + count(distinct f.operator_id) filter (where ${AI_F(sql)})::int as f_operators
29 + from ${facilityView(sql)} f`,
30 + sql<Row[]>`select count(*)::int as projects, sum(p.planned_mw)::float as planned_mw, count(distinct p.country_iso2)::int as countries, count(distinct p.operator_id)::int as operators from projects p where ${projectLive(sql)} and ${AI_P(sql)}`,
31 + sql<Row[]>`select * from (select o.id, o.slug, o.name,
32 + (select count(*) from ${facilityView(sql)} f where f.operator_id = o.id and ${counted} and ${AI_F(sql)})::int as facilities,
33 + (select count(*) from projects p where p.operator_id = o.id and ${projectLive(sql)} and ${AI_P(sql)})::int as projects,
34 + (select sum(p.planned_mw) from projects p where p.operator_id = o.id and ${projectLive(sql)} and ${AI_P(sql)})::float as planned_mw
35 + from operators o where exists (select 1 from facilities f where f.operator_id = o.id and f.merged_into is null and ${AI_F(sql)}) or exists (select 1 from projects p where p.operator_id = o.id and ${projectLive(sql)} and ${AI_P(sql)})) t order by t.facilities + t.projects desc, t.planned_mw desc nulls last, t.name limit 10`,
36 + sql<Row[]>`select * from (select m.id, m.slug, m.name, m.country_iso2,
37 + (select count(*) from ${facilityView(sql)} f where f.metro_id = m.id and ${counted} and ${AI_F(sql)})::int as facilities,
38 + (select count(*) from projects p where p.metro_id = m.id and ${projectLive(sql)} and ${AI_P(sql)})::int as projects,
39 + (select sum(p.planned_mw) from projects p where p.metro_id = m.id and ${projectLive(sql)} and ${AI_P(sql)})::float as planned_mw
40 + from metros m where exists (select 1 from facilities f where f.metro_id = m.id and f.merged_into is null and ${AI_F(sql)}) or exists (select 1 from projects p where p.metro_id = m.id and ${projectLive(sql)} and ${AI_P(sql)})) t order by t.facilities + t.projects desc, t.planned_mw desc nulls last, t.name limit 10`,
41 + sql<Row[]>`select * from (select c.iso2, c.slug, c.name,
42 + (select count(*) from ${facilityView(sql)} f where f.country_iso2 = c.iso2 and ${counted} and ${AI_F(sql)})::int as facilities,
43 + (select count(*) from projects p where p.country_iso2 = c.iso2 and ${projectLive(sql)} and ${AI_P(sql)})::int as projects,
44 + (select sum(p.planned_mw) from projects p where p.country_iso2 = c.iso2 and ${projectLive(sql)} and ${AI_P(sql)})::float as planned_mw
45 + from countries c where exists (select 1 from facilities f where f.country_iso2 = c.iso2 and f.merged_into is null and ${AI_F(sql)}) or exists (select 1 from projects p where p.country_iso2 = c.iso2 and ${projectLive(sql)} and ${AI_P(sql)})) t order by t.facilities + t.projects desc, t.planned_mw desc nulls last, t.name limit 10`,
46 + sql<Row[]>`select p.status, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${projectLive(sql)} and ${AI_P(sql)} group by 1 order by n desc`,
47 + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and ${AI_P(sql)} order by p.last_update desc limit 10`,
48 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and ${AI_F(sql)} order by ${mwExpr(sql)} desc nulls last, f.completeness desc limit 30`,
49 + eventsWhere(sql`e.is_ai = true`, 15),
50 + ]);
51 + const f = fs[0] ?? {};
52 + const p = ps[0] ?? {};
53 + const facilitiesN = int(f.facilities);
54 + return {
55 + stats: {
56 + facilities: facilitiesN,
57 + confirmed: int(f.confirmed),
58 + likely: int(f.likely),
59 + associated: int(f.associated),
60 + projects: int(p.projects),
61 + plannedMw: round2(num(p.planned_mw)),
62 + constructionMw: round2(num(f.construction_mw)),
63 + countries: Math.max(int(f.f_countries), int(p.countries)),
64 + operators: Math.max(int(f.f_operators), int(p.operators)),
65 + mwCoverage: share(int(f.with_mw), facilitiesN),
66 + },
67 + topOperators: topOps.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), facilities: int(r.facilities), projects: int(r.projects), plannedMw: round2(num(r.planned_mw)) })),
68 + topMetros: topMetros.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), countryIso2: reqStr(r.country_iso2), facilities: int(r.facilities), projects: int(r.projects), plannedMw: round2(num(r.planned_mw)) })),
69 + topCountries: topCountries.map((r) => ({ iso2: reqStr(r.iso2), slug: reqStr(r.slug), name: reqStr(r.name), facilities: int(r.facilities), projects: int(r.projects), plannedMw: round2(num(r.planned_mw)) })),
70 + pipelineByStage: stages.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: round2(num(r.mw)) })),
71 + recentAnnouncements,
72 + recentProjects: recentProjects.map(projectSummary),
73 + facilities: facilities.map(facilitySummary),
74 + evidenceNote: AI_EVIDENCE_NOTE,
75 + };
76 +}
modified apps/api/src/repositories/cloud-regions.ts +33 −15
@@ -1,8 +1,10 @@
1 −import type { CloudRegionSummary, FacilitySummary } from "@dci/core";
2 −import { pg, andAll, cloudRegionCols, facilityJoins, facilitySummaryCols, type Fragment } from "../lib/sql.js";
3 −import { str, type Row } from "../lib/rows.js";
1 +import type { CloudRegionDetail as CloudRegionDetailContract, CloudRegionSummary, FacilitySummary } from "@dci/core";
2 +import { pg, andAll, cloudRegionCols, facilityJoins, facilitySummaryCols, mwExpr, type Fragment } from "../lib/sql.js";
3 +import { record, reqIso, str, type Row } from "../lib/rows.js";
4 4 import { cloudRegionSummary, facilitySummary } from "../lib/dto.js";
5 5 import { findBySlugOrId } from "../lib/resolve.js";
6 +import { eventsForEntity } from "./events.js";
7 +import { provenanceFor } from "./facilities.js";
6 8
7 9 export async function listCloudRegions(f: { provider?: string; country?: string; metroId?: string; status?: string }): Promise<CloudRegionSummary[]> {
8 10 const sql = pg();
@@ -15,40 +17,56 @@ export async function listCloudRegions(f: { provider?: string; country?: string;
15 17 return rows.map(cloudRegionSummary);
16 18 }
17 19
18 −export interface CloudRegionDetail extends CloudRegionSummary {
20 +/** Contract CloudRegionDetail + the extra fields the web already reads. */
21 +export interface CloudRegionDetail extends CloudRegionDetailContract {
19 22 regionName: string | null;
20 − metro: { id: string; slug: string; name: string } | null;
21 23 countryName: string | null;
22 − siblingRegions: CloudRegionSummary[];
23 24 facilitiesInMetro: FacilitySummary[];
24 − externalIds: Record<string, string | number>;
25 25 updatedAt: string;
26 26 }
27 27
28 +export const MARKET_ASSOCIATION_NOTE = "Region associated with market, not hosted at a specific facility";
29 +
28 30 export async function getCloudRegion(idOrSlug: string): Promise<CloudRegionDetail | null> {
29 31 const sql = pg();
30 32 const row = await findBySlugOrId("cloud_regions", idOrSlug);
31 33 if (!row) return null;
32 34 const id = String(row.id);
33 35 const metroId = str(row.metro_id);
34 − const [main, siblings, facs, metroRows, countryRows] = await Promise.all([
36 + const providerId = String(row.provider_id);
37 + const [main, siblings, facs, hosts, metroRows, countryRows, events, provenance] = await Promise.all([
35 38 sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.id = ${id}`,
36 − sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.provider_id = ${String(row.provider_id)} and r.id <> ${id} and r.country_iso2 is not distinct from ${str(row.country_iso2)} order by r.code limit 12`,
39 + sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.provider_id = ${providerId} and r.id <> ${id} and r.country_iso2 is not distinct from ${str(row.country_iso2)} order by r.code limit 12`,
37 40 metroId
38 − ? sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.metro_id = ${metroId} order by coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) desc nulls last, f.name limit 12`
41 + ? sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.metro_id = ${metroId} order by ${mwExpr(sql)} desc nulls last, f.name limit 12`
39 42 : Promise.resolve([] as Row[]),
43 + // explicit hosting link only: a cloud tenancy of this provider at a facility in the region's metro / country
44 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} join facility_tenants t on t.facility_id = f.id and t.role = 'cloud' and t.operator_id = ${providerId}
45 + where f.merged_into is null and ${metroId ? sql`f.metro_id = ${metroId}` : row.country_iso2 ? sql`f.country_iso2 = ${String(row.country_iso2)}` : sql`false`} order by f.name limit 50`,
40 46 metroId ? sql<Row[]>`select id, slug, name from metros where id = ${metroId}` : Promise.resolve([] as Row[]),
41 47 row.country_iso2 ? sql<Row[]>`select name from countries where iso2 = ${String(row.country_iso2)}` : Promise.resolve([] as Row[]),
48 + eventsForEntity("cloud_region", id, 50),
49 + provenanceFor("cloud_region", id),
42 50 ]);
43 51 const base = cloudRegionSummary(main[0] ?? row);
52 + const hostFacilities = hosts.map(facilitySummary);
53 + const marketFacilities = facs.map(facilitySummary);
54 + const announcedProv = provenance.find((p) => p.field === "announcedOn" && typeof p.value === "string");
55 + const announcedOn = announcedProv ? String(announcedProv.value) : base.status === "announced" && base.launchedOn ? base.launchedOn : null;
44 56 return {
45 57 ...base,
46 − regionName: str(row.region_name),
47 58 metro: metroRows[0] ? { id: String(metroRows[0].id), slug: String(metroRows[0].slug), name: String(metroRows[0].name) } : null,
48 − countryName: countryRows[0] ? String(countryRows[0].name) : null,
59 + hostFacilities,
60 + marketFacilities,
49 61 siblingRegions: siblings.map(cloudRegionSummary),
50 − facilitiesInMetro: facs.map(facilitySummary),
51 − externalIds: (row.external_ids as Record<string, string | number> | null) ?? {},
52 − updatedAt: String(row.updated_at ?? ""),
62 + announcedOn,
63 + events,
64 + provenance,
65 + externalIds: record(row.external_ids),
66 + note: hostFacilities.length ? `Hosting publicly verified for ${hostFacilities.length} facilit${hostFacilities.length > 1 ? "ies" : "y"} (cloud tenancy recorded at the facility); marketFacilities lists the other facilities of the market.` : MARKET_ASSOCIATION_NOTE,
67 + regionName: str(row.region_name),
68 + countryName: countryRows[0] ? String(countryRows[0].name) : null,
69 + facilitiesInMetro: marketFacilities,
70 + updatedAt: reqIso(row.updated_at),
53 71 };
54 72 }
added apps/api/src/repositories/compare.ts +34 −0
@@ -0,0 +1,34 @@
1 +/** Side-by-side operator comparison (2–5 operators): summary, pipeline, velocity, new markets, recent projects. */
2 +import type { OperatorComparison } from "@dci/core";
3 +import { pg, projectLive } from "../lib/sql.js";
4 +import { reqStr, type Row } from "../lib/rows.js";
5 +import { notFound } from "../lib/http.js";
6 +import { expansionVelocity, pipelineBreakdown, scopeFor } from "../lib/pipeline.js";
7 +import { newMarkets, operatorSummaryById } from "./operators.js";
8 +import { projectsWhere } from "./projects.js";
9 +
10 +export const COMPARE_METHODOLOGY = "Each column is computed the same way as the operator page: containment-aware facility counts, known MW = published operational figures only (coverage given per operator), pipeline from facility statuses and live project records (hidden / merged projects excluded), velocity windows from opened_on (else first_seen) and announced_on (else first indexed).";
11 +
12 +export async function compareOperators(slugs: string[]): Promise<OperatorComparison> {
13 + const sql = pg();
14 + const ids: string[] = [];
15 + for (const slug of slugs) {
16 + const r = (await sql<Row[]>`select id from operators where slug = ${slug} or id = ${slug} limit 1`)[0];
17 + if (!r) throw notFound(`operator ${slug}`);
18 + ids.push(reqStr(r.id));
19 + }
20 + const operators = await Promise.all(ids.map(async (id) => {
21 + const scope = scopeFor(sql, { operatorId: id });
22 + const [summary, pipeline, velocity, markets, recentProjects, countries] = await Promise.all([
23 + operatorSummaryById(id),
24 + pipelineBreakdown(scope),
25 + expansionVelocity(scope),
26 + newMarkets(id, 12),
27 + projectsWhere(sql`${projectLive(sql)} and p.operator_id = ${id}`, 5),
28 + sql<Row[]>`select distinct country_iso2 from facilities where operator_id = ${id} and merged_into is null and country_iso2 is not null order by 1`,
29 + ]);
30 + if (!summary) throw notFound(`operator ${id}`);
31 + return { ...summary, pipeline, velocity, aiCount: summary.aiCount ?? 0, newMarkets12m: markets, recentProjects, countriesList: countries.map((c) => reqStr(c.country_iso2)), coverage: summary.mwCoverage ?? 0 };
32 + }));
33 + return { operators, generatedAt: new Date().toISOString() };
34 +}
added apps/api/src/repositories/connectivity.ts +39 −0
@@ -0,0 +1,39 @@
1 +/**
2 + * /connectivity — IXPs, cloud regions, carrier hotels and per-metro interconnection density. Cable landing stations
3 + * have no connector yet and are returned empty with a note.
4 + */
5 +import type { ConnectivityOverview } from "@dci/core";
6 +import { pg, cloudRegionCols, facilityJoins, facilitySummaryCols } from "../lib/sql.js";
7 +import { int, reqStr, type Row } from "../lib/rows.js";
8 +import { cloudRegionSummary, facilitySummary, ixpSummary } from "../lib/dto.js";
9 +
10 +export const CONNECTIVITY_NOTE = "IXP facility / operator counts come from published IXP↔facility links (facility_ixps); IXPs have no coordinates of their own and are placed at their metro reference point on maps. Carrier hotels = facility_type carrier_hotel or ≥ 20 carriers published on the facility. Submarine cable landing stations are empty until a licensed connector exists — never inferred. Cloud regions are associated with a market, not hosted at a specific facility unless a source says so.";
11 +
12 +export async function connectivityOverview(): Promise<ConnectivityOverview> {
13 + const sql = pg();
14 + const [ixps, cloudRegions, carrierHotels, byMetro] = await Promise.all([
15 + sql<Row[]>`select x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count,
16 + (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count,
17 + (select count(distinct f.operator_id)::int from facility_ixps fx join facilities f on f.id = fx.facility_id where fx.ixp_id = x.id and f.operator_id is not null) as operator_count,
18 + m.id as met_id, m.slug as met_slug, m.name as met_name, m.lat, m.lng
19 + from ixps x left join metros m on m.id = x.metro_id
20 + order by x.network_count desc nulls last, facility_count desc, x.name limit 100`,
21 + sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.status <> 'retired' order by pr.name, r.country_iso2, r.code limit 2000`,
22 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and (f.facility_type = 'carrier_hotel' or coalesce(f.carriers_count, 0) >= 20) order by f.carriers_count desc nulls last, f.completeness desc, f.name limit 50`,
23 + sql<Row[]>`select * from (select m.id, m.slug, m.name, m.country_iso2,
24 + (select count(*)::int from ixps x where x.metro_id = m.id) as ixps,
25 + (select count(*)::int from cloud_regions r where r.metro_id = m.id and r.status <> 'retired') as cloud_regions,
26 + (select count(*)::int from facilities f where f.metro_id = m.id and f.merged_into is null and (coalesce(f.carriers_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id and t.role in ('carrier', 'network')))) as facilities_with_carriers,
27 + (select count(*)::int from facilities f where f.metro_id = m.id and f.merged_into is null) as facilities
28 + from metros m
29 + where exists (select 1 from ixps x where x.metro_id = m.id) or exists (select 1 from cloud_regions r where r.metro_id = m.id) or exists (select 1 from facilities f where f.metro_id = m.id and f.merged_into is null and (coalesce(f.carriers_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id)))) t order by t.ixps + t.cloud_regions + t.facilities_with_carriers desc, t.facilities desc limit 50`,
30 + ]);
31 + return {
32 + ixps: ixps.map((r) => ({ ...ixpSummary(r), operatorCount: int(r.operator_count) })),
33 + cloudRegions: cloudRegions.map(cloudRegionSummary),
34 + carrierHotels: carrierHotels.map(facilitySummary),
35 + landingStations: [],
36 + byMetro: byMetro.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), countryIso2: reqStr(r.country_iso2), ixps: int(r.ixps), cloudRegions: int(r.cloud_regions), facilitiesWithCarriers: int(r.facilities_with_carriers), facilities: int(r.facilities) })),
37 + note: CONNECTIVITY_NOTE,
38 + };
39 +}
modified apps/api/src/repositories/countries.ts +65 −40
@@ -1,15 +1,18 @@
1 −import type { CountryDetail, CountrySummary } from "@dci/core";
2 −import { pg, mwExpr, plannedMwExpr, OPERATIONAL_SET, CONSTRUCTION_SET, PLANNED_SET, yearExpr, type Fragment, type Sql } from "../lib/sql.js";
1 +import type { CountryDetail, CountrySummary, EnergyContext } from "@dci/core";
2 +import { pg, mwExpr, yearExpr, facilityView, knownMwAgg, countedAgg, projectLive, facilityJoins, facilitySummaryCols, projectJoins, projectSummaryCols, cloudRegionCols, AI_LEVELS, type Sql } from "../lib/sql.js";
3 3 import { int, num, reqStr, str, type Row } from "../lib/rows.js";
4 −import { growthSeries, statusBreakdown, typeBreakdown, cloudRegionSummary } from "../lib/dto.js";
4 +import { growthSeries, statusBreakdown, typeBreakdown, cloudRegionSummary, facilitySummary, projectSummary, ixpSummary, round2, share } from "../lib/dto.js";
5 5 import { findCountry } from "../lib/resolve.js";
6 6 import { listFacilities } from "./facilities.js";
7 7 import { operatorsForScope } from "./operators.js";
8 −import { listMetros } from "./metros.js";
8 +import { gridConstraintsFor, listMetros } from "./metros.js";
9 9 import { projectsWhere } from "./projects.js";
10 10 import { eventsForCountry } from "./events.js";
11 11 import { rankingPositions } from "./rankings.js";
12 −import { cloudRegionCols } from "../lib/sql.js";
12 +import { coverageRow, dimAggCols, pipelineBreakdown, scopeFor } from "../lib/pipeline.js";
13 +import { claimsFor } from "../lib/quality.js";
14 +
15 +export const ENERGY_NOTE = "National grid averages (renewable share, generation) describe the country's electricity mix, not the electricity a facility contracts; utility/grid MW on facilities are supply figures, not IT load.";
13 16
14 17 function countrySummary(r: Row): CountrySummary {
15 18 const n = int(r.facility_count);
@@ -27,79 +30,101 @@ function countrySummary(r: Row): CountrySummary {
27 30 operationalCount: int(r.operational),
28 31 constructionCount: int(r.construction),
29 32 plannedCount: int(r.planned),
30 − knownMw: num(r.known_mw),
31 − constructionMw: num(r.construction_mw),
32 − plannedMw: num(r.planned_mw),
33 + knownMw: round2(num(r.known_mw)),
34 + constructionMw: round2(num(r.construction_mw)),
35 + plannedMw: round2(num(r.planned_mw)),
33 36 operatorCount: int(r.operator_count),
34 37 cloudRegionCount: int(r.cloud_region_count),
35 38 hyperscaleCount: int(r.hyperscale),
36 39 aiCount: int(r.ai),
37 40 projectCount: int(r.project_count),
38 − mwCoverage: n ? Math.round((withMw / n) * 1000) / 1000 : 0,
41 + projectPlannedMw: round2(num(r.project_planned_mw)),
42 + projectConstructionMw: round2(num(r.project_construction_mw)),
43 + ixpCount: int(r.ixp_count),
44 + mwCoverage: n ? share(withMw, n) : 0,
39 45 lat: num(r.lat),
40 46 lng: num(r.lng),
41 47 };
42 48 }
43 49
44 −function facilityAgg(sql: Sql): Fragment {
45 − const mw = mwExpr(sql);
46 − return sql`(
47 − select f.country_iso2, count(*)::int as facility_count,
48 − count(*) filter (where f.status = any(${OPERATIONAL_SET}))::int as operational,
49 − count(*) filter (where f.status = any(${CONSTRUCTION_SET}))::int as construction,
50 − count(*) filter (where f.status = any(${PLANNED_SET}))::int as planned,
51 − sum(${mw}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw,
52 − sum(${mw}) filter (where f.status = any(${CONSTRUCTION_SET}))::float as construction_mw,
53 − sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PLANNED_SET}))::float as planned_mw,
54 − count(distinct f.operator_id)::int as operator_count,
55 − count(*) filter (where f.is_hyperscale or f.facility_type = 'hyperscale')::int as hyperscale,
56 − count(*) filter (where f.is_ai or f.facility_type = 'ai')::int as ai,
57 − count(*) filter (where ${mw} is not null)::int as with_mw
58 − from facilities f where f.merged_into is null and f.country_iso2 is not null group by f.country_iso2
59 − )`;
60 −}
61 −
62 50 const SELECT = (sql: Sql) => sql`
63 − select c.iso2, c.iso3, c.slug, c.name, c.region, c.subregion, c.population, c.gdp_usd, c.lat, c.lng, c.electricity_twh, c.renewable_share,
64 − coalesce(fa.facility_count, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned,
65 − fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operator_count, 0) as operator_count, coalesce(fa.hyperscale, 0) as hyperscale, coalesce(fa.ai, 0) as ai, coalesce(fa.with_mw, 0) as with_mw,
66 − coalesce(cr.n, 0) as cloud_region_count, coalesce(pj.n, 0) as project_count
51 + select c.iso2, c.iso3, c.slug, c.name, c.region, c.subregion, c.population, c.gdp_usd, c.lat, c.lng, c.electricity_twh, c.renewable_share, c.stats_year, c.stats,
52 + coalesce(fa.facilities, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned,
53 + fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operators, 0) as operator_count, coalesce(fa.hyperscale, 0) as hyperscale, coalesce(fa.ai, 0) as ai, coalesce(fa.with_mw, 0) as with_mw,
54 + coalesce(cr.n, 0) as cloud_region_count, coalesce(pj.n, 0) as project_count, pj.planned_mw as project_planned_mw, pj.construction_mw as project_construction_mw, coalesce(ix.n, 0) as ixp_count
67 55 from countries c
68 − left join ${facilityAgg(sql)} fa on fa.country_iso2 = c.iso2
69 − left join (select country_iso2, count(*)::int as n from cloud_regions group by 1) cr on cr.country_iso2 = c.iso2
70 − left join (select country_iso2, count(*)::int as n from projects group by 1) pj on pj.country_iso2 = c.iso2`;
56 + left join (select f.country_iso2, ${dimAggCols(sql)} from ${facilityView(sql)} f where f.country_iso2 is not null group by f.country_iso2) fa on fa.country_iso2 = c.iso2
57 + left join (select country_iso2, count(*)::int as n from cloud_regions where status <> 'retired' group by 1) cr on cr.country_iso2 = c.iso2
58 + left join (select country_iso2, count(*)::int as n from ixps group by 1) ix on ix.country_iso2 = c.iso2
59 + left join (select p.country_iso2, count(*)::int as n,
60 + sum(p.planned_mw) filter (where p.status in ('rumored','proposed','announced','permitting','approved','delayed'))::float as planned_mw,
61 + sum(p.planned_mw) filter (where p.status = 'under_construction')::float as construction_mw
62 + from projects p where ${projectLive(sql)} group by 1) pj on pj.country_iso2 = c.iso2`;
71 63
72 64 export async function listCountries(opts: { all?: boolean; region?: string } = {}): Promise<CountrySummary[]> {
73 65 const sql = pg();
74 66 const rows = await sql<Row[]>`${SELECT(sql)}
75 − where ${opts.all ? sql`true` : sql`(coalesce(fa.facility_count, 0) > 0 or coalesce(cr.n, 0) > 0)`}
67 + where ${opts.all ? sql`true` : sql`(coalesce(fa.facilities, 0) > 0 or coalesce(cr.n, 0) > 0)`}
76 68 and ${opts.region ? sql`(c.region ilike ${opts.region} or c.subregion ilike ${opts.region})` : sql`true`}
77 − order by coalesce(fa.facility_count, 0) desc, coalesce(cr.n, 0) desc, c.name`;
69 + order by coalesce(fa.facilities, 0) desc, coalesce(cr.n, 0) desc, c.name`;
78 70 return rows.map(countrySummary);
79 71 }
80 72
73 +async function energyContext(base: Row): Promise<EnergyContext> {
74 + const sql = pg();
75 + const stats = (base.stats as Record<string, unknown> | null) ?? {};
76 + let sourceName: string | null = null;
77 + let sourceUrl: string | null = null;
78 + const es = stats.energySource;
79 + if (es && typeof es === "object") { sourceName = str((es as Record<string, unknown>).name); sourceUrl = str((es as Record<string, unknown>).url); }
80 + else if (typeof es === "string") sourceName = es;
81 + if (!sourceName) {
82 + const rows = await sql<Row[]>`select s.name, p.url from provenance p left join sources s on s.id = p.source_id where p.entity_type = 'country' and p.entity_id = ${String(base.iso2)} and p.is_current and p.field in ('renewableShare', 'electricityTwh', 'renewable_share', 'electricity_twh') order by p.last_observed desc limit 1`;
83 + if (rows[0]) { sourceName = str(rows[0].name); sourceUrl = str(rows[0].url); }
84 + }
85 + return { renewableShare: num(base.renewable_share), electricityTwh: num(base.electricity_twh), statsYear: num(base.stats_year), gridCarbonIntensity: null, sourceName, sourceUrl, note: ENERGY_NOTE };
86 +}
87 +
81 88 export async function getCountryDetail(slugOrIso2: string, fPage = 1): Promise<CountryDetail | null> {
82 89 const sql = pg();
83 90 const base = await findCountry(slugOrIso2);
84 91 if (!base) return null;
85 92 const iso2 = String(base.iso2);
86 − const [sumRows, topOperators, metros, cloudRegionRows, recentProjects, recentEvents, growthRows, statusRows, typeRows, invRows, rankings, facilities] = await Promise.all([
93 + const scope = scopeFor(sql, { countryIso2: iso2 });
94 + const [sumRows, topOperators, metros, cloudRegionRows, recentProjects, recentEvents, growthRows, statusRows, typeRows, invRows, rankings, facilities, ixpRows, gridConstraints, energy, aiFacRows, aiPrjRows, pipeline, coverage, claims] = await Promise.all([
87 95 sql<Row[]>`${SELECT(sql)} where c.iso2 = ${iso2}`,
88 96 operatorsForScope({ countryIso2: iso2 }, 10),
89 97 listMetros({ country: iso2 }),
90 98 sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.country_iso2 = ${iso2} order by pr.name, r.code`,
91 − projectsWhere(sql`p.country_iso2 = ${iso2}`, 10),
99 + projectsWhere(sql`${projectLive(sql)} and p.country_iso2 = ${iso2}`, 10),
92 100 eventsForCountry(iso2, 20),
93 − sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*)::int as n, sum(${mwExpr(sql)})::float as mw from facilities f where f.country_iso2 = ${iso2} and f.merged_into is null and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
101 + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${knownMwAgg(sql)})::float as mw from ${facilityView(sql)} f where f.country_iso2 = ${iso2} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
94 102 sql<Row[]>`select status, count(*)::int as n from facilities where country_iso2 = ${iso2} and merged_into is null group by status`,
95 103 sql<Row[]>`select facility_type, count(*)::int as n from facilities where country_iso2 = ${iso2} and merged_into is null group by facility_type`,
96 − sql<Row[]>`select sum(investment_usd)::float as inv from projects where country_iso2 = ${iso2} and status not in ('cancelled')`,
104 + sql<Row[]>`select sum(p.investment_usd)::float as inv from projects p where ${projectLive(sql)} and p.country_iso2 = ${iso2} and p.status not in ('cancelled') and (p.investment_scope is null or p.investment_scope in ('facility', 'campus', 'building'))`,
97 105 rankingPositions("countries", [iso2, String(base.slug)]),
98 106 listFacilities({ countryIso2: iso2, page: fPage, per_page: 50, sort: "mw", order: "desc" }),
107 + sql<Row[]>`select x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count, m.id as met_id, m.slug as met_slug, m.name as met_name
108 + from ixps x left join metros m on m.id = x.metro_id where x.country_iso2 = ${iso2} order by x.network_count desc nulls last, x.name limit 500`,
109 + gridConstraintsFor({ countryIso2: iso2 }),
110 + energyContext(base),
111 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.country_iso2 = ${iso2} and f.ai_evidence = any(${AI_LEVELS}) order by ${mwExpr(sql)} desc nulls last, f.name limit 20`,
112 + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and p.country_iso2 = ${iso2} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai) order by p.planned_mw desc nulls last, p.last_update desc limit 20`,
113 + pipelineBreakdown(scope),
114 + coverageRow(iso2, String(base.name), String(base.slug), scope),
115 + claimsFor("country", iso2),
99 116 ]);
100 117 const summary = countrySummary(sumRows[0] ?? { ...base, facility_count: 0 });
101 118 return {
102 119 ...summary,
120 + ixps: ixpRows.map(ixpSummary),
121 + gridConstraints,
122 + energy,
123 + aiFacilities: aiFacRows.map(facilitySummary),
124 + aiProjects: aiPrjRows.map(projectSummary),
125 + pipeline,
126 + coverage,
127 + claims,
103 128 topOperators,
104 129 metros,
105 130 cloudRegions: cloudRegionRows.map(cloudRegionSummary),
added apps/api/src/repositories/coverage.ts +124 −0
@@ -0,0 +1,124 @@
1 +/**
2 + * Coverage report: how much of the index is actually known, per country / metro / operator / field / source kind.
3 + * Containment-aware (campus rows with buildings are not counted). Shares are 0..1.
4 + */
5 +import type { CoverageReport, CoverageRow, SourceCoverage } from "@dci/core";
6 +import { authorityTier } from "@dci/core";
7 +import { pg, facilityView, countedAgg, hasMwAgg, projectLive, type Fragment, type Sql } from "../lib/sql.js";
8 +import { int, iso, reqStr, str, type Row } from "../lib/rows.js";
9 +import { asSourceKind, share } from "../lib/dto.js";
10 +import { coverageRow, scopeFor } from "../lib/pipeline.js";
11 +
12 +const PRIMARY_KINDS = ["operator", "government", "filing", "utility", "cloud_provider", "registry"];
13 +
14 +/** Grouped version of lib/pipeline coverageRow: one row per dimension value. */
15 +function coverageCols(sql: Sql): Fragment {
16 + const counted = countedAgg(sql), hasMw = hasMwAgg(sql);
17 + return sql`
18 + count(*) filter (where ${counted})::int as n,
19 + count(*) filter (where ${counted} and ${hasMw})::int as with_mw,
20 + count(*) filter (where ${counted} and f.operator_id is not null)::int as with_op,
21 + count(*) filter (where ${counted} and f.lat is not null and f.geo_precision in ('exact', 'parcel', 'street'))::int as precise,
22 + count(*) filter (where ${counted} and f.lat is not null)::int as any_loc,
23 + count(*) filter (where ${counted} and f.status <> 'unknown')::int as with_status,
24 + count(*) filter (where ${counted} and f.opened_on ~ '^\\d{4}')::int as with_open,
25 + count(*) filter (where ${counted} and f.source_count >= 2)::int as multi,
26 + count(*) filter (where ${counted} and (coalesce(f.carriers_count, 0) > 0 or coalesce(f.ixp_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id) or exists (select 1 from facility_ixps x where x.facility_id = f.id)))::int as conn,
27 + count(*) filter (where ${counted} and exists (select 1 from provenance p join sources s on s.id = p.source_id where p.entity_type = 'facility' and p.entity_id = f.id and p.is_current and s.kind = any(${PRIMARY_KINDS})))::int as primary_src`;
28 +}
29 +
30 +function toRow(r: Row, projects: { n: number; loc: number }): CoverageRow {
31 + const n = int(r.n);
32 + return {
33 + key: reqStr(r.key), name: reqStr(r.name), slug: reqStr(r.slug),
34 + facilities: n,
35 + capacityCoverage: share(int(r.with_mw), n),
36 + operatorCoverage: share(int(r.with_op), n),
37 + preciseLocationCoverage: share(int(r.precise), n),
38 + anyLocationCoverage: share(int(r.any_loc), n),
39 + statusCoverage: share(int(r.with_status), n),
40 + openingDateCoverage: share(int(r.with_open), n),
41 + multiSourceCoverage: share(int(r.multi), n),
42 + projects: projects.n,
43 + projectsWithLocation: projects.loc,
44 + connectivityCoverage: share(int(r.conn), n),
45 + primarySourceShare: share(int(r.primary_src), n),
46 + };
47 +}
48 +
49 +async function projectCounts(sql: Sql, col: "country_iso2" | "metro_id" | "operator_id"): Promise<Map<string, { n: number; loc: number }>> {
50 + const rows = await sql<Row[]>`select p.${sql(col)} as key, count(*)::int as n, count(*) filter (where p.lat is not null)::int as loc from projects p where ${projectLive(sql)} and p.${sql(col)} is not null group by 1`;
51 + return new Map(rows.map((r) => [reqStr(r.key), { n: int(r.n), loc: int(r.loc) }]));
52 +}
53 +
54 +export const COVERAGE_METHODOLOGY = "Shares of counted facilities (campus rows with building rows excluded; merged duplicates excluded) with each field known. capacityCoverage counts a building covered by its campus figure as covered. Precise location = exact / parcel / street. Primary sources = operator, government, filing, utility, cloud provider, registry. Nothing is extrapolated: a low share means the index does not know, not that the value is zero.";
55 +
56 +export async function coverageReport(): Promise<CoverageReport> {
57 + const sql = pg();
58 + const [global, countries, metros, operators, pc, pm, po, fields, bySourceKind] = await Promise.all([
59 + coverageRow("global", "Global", "global", scopeFor(sql, {})),
60 + sql<Row[]>`select c.iso2 as key, c.name, c.slug, ${coverageCols(sql)} from ${facilityView(sql)} f join countries c on c.iso2 = f.country_iso2 group by c.iso2, c.name, c.slug having count(*) filter (where ${countedAgg(sql)}) >= 1 order by n desc, c.name`,
61 + sql<Row[]>`select m.id as key, m.name, m.slug, ${coverageCols(sql)} from ${facilityView(sql)} f join metros m on m.id = f.metro_id group by m.id, m.name, m.slug having count(*) filter (where ${countedAgg(sql)}) >= 1 order by n desc, m.name`,
62 + sql<Row[]>`select o.id as key, o.name, o.slug, ${coverageCols(sql)} from ${facilityView(sql)} f join operators o on o.id = f.operator_id group by o.id, o.name, o.slug having count(*) filter (where ${countedAgg(sql)}) >= 5 order by n desc, o.name`,
63 + projectCounts(sql, "country_iso2"),
64 + projectCounts(sql, "metro_id"),
65 + projectCounts(sql, "operator_id"),
66 + sql<Row[]>`select count(*)::int as n,
67 + count(*) filter (where f.operator_id is not null)::int as operator,
68 + count(*) filter (where f.lat is not null)::int as coordinates,
69 + count(*) filter (where f.lat is not null and f.geo_precision in ('exact', 'parcel', 'street'))::int as precise_coordinates,
70 + count(*) filter (where f.status <> 'unknown')::int as status,
71 + count(*) filter (where f.facility_type <> 'unknown')::int as facility_type,
72 + count(*) filter (where f.it_capacity_mw is not null)::int as it_capacity_mw,
73 + count(*) filter (where f.total_power_mw is not null)::int as total_power_mw,
74 + count(*) filter (where f.planned_power_mw is not null)::int as planned_power_mw,
75 + count(*) filter (where f.opened_on ~ '^\\d{4}')::int as opened_on,
76 + count(*) filter (where f.address is not null)::int as address,
77 + count(*) filter (where coalesce(f.carriers_count, 0) > 0 or coalesce(f.ixp_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id) or exists (select 1 from facility_ixps x where x.facility_id = f.id))::int as connectivity,
78 + count(*) filter (where f.website is not null)::int as website,
79 + count(*) filter (where f.description is not null)::int as description,
80 + count(*) filter (where f.tier is not null)::int as tier
81 + from ${facilityView(sql)} f where ${countedAgg(sql)}`,
82 + sql<Row[]>`select s.kind, count(distinct p.entity_id) filter (where p.entity_type = 'facility')::int as facilities, count(*)::int as fields from provenance p join sources s on s.id = p.source_id where p.is_current group by s.kind order by facilities desc`,
83 + ]);
84 + const fr = fields[0] ?? {};
85 + const total = int(fr.n);
86 + const FIELDS: Array<[string, string]> = [["operator", "Operator"], ["coordinates", "Coordinates (any precision)"], ["precise_coordinates", "Precise coordinates (exact / parcel / street)"], ["status", "Lifecycle status"], ["facility_type", "Facility type"], ["it_capacity_mw", "IT capacity (MW)"], ["total_power_mw", "Total power (MW)"], ["planned_power_mw", "Planned power (MW)"], ["opened_on", "Opening date"], ["address", "Street address"], ["connectivity", "Carriers / IXPs"], ["website", "Website"], ["description", "Description"], ["tier", "Tier / certification"]];
87 + return {
88 + global,
89 + countries: countries.map((r) => toRow(r, pc.get(reqStr(r.key)) ?? { n: 0, loc: 0 })),
90 + metros: metros.map((r) => toRow(r, pm.get(reqStr(r.key)) ?? { n: 0, loc: 0 })),
91 + operators: operators.map((r) => toRow(r, po.get(reqStr(r.key)) ?? { n: 0, loc: 0 })),
92 + fields: FIELDS.map(([field, label]) => ({ field, label, coverage: share(int(fr[field]), total), count: int(fr[field]) })),
93 + bySourceKind: bySourceKind.map((r) => ({ kind: asSourceKind(r.kind), facilities: int(r.facilities), fields: int(r.fields) })),
94 + generatedAt: new Date().toISOString(),
95 + };
96 +}
97 +
98 +export async function sourceCoverage(): Promise<SourceCoverage[]> {
99 + const sql = pg();
100 + const rows = await sql<Row[]>`
101 + with cur as (select source_id, entity_type, entity_id from provenance where is_current),
102 + per_entity as (select entity_type, entity_id, count(distinct source_id) as sources from cur group by 1, 2)
103 + select s.id, s.name, s.kind, s.license, s.redistribution, s.connector_id,
104 + (select count(distinct (c.entity_type, c.entity_id)) from cur c where c.source_id = s.id)::int as records,
105 + (select count(*) from cur c where c.source_id = s.id)::int as fields,
106 + (select count(distinct (c.entity_type, c.entity_id)) from cur c join per_entity pe on pe.entity_type = c.entity_type and pe.entity_id = c.entity_id where c.source_id = s.id and pe.sources = 1)::int as unique_records,
107 + (select max(r.finished_at) from connector_runs r where r.connector_id = s.connector_id and r.status in ('ok', 'partial')) as last_ok,
108 + (select count(*) from connector_runs r where r.connector_id = s.connector_id and r.started_at >= now() - interval '30 days' and r.status <> 'running')::int as runs30,
109 + (select count(*) from connector_runs r where r.connector_id = s.connector_id and r.started_at >= now() - interval '30 days' and r.status in ('failed', 'aborted'))::int as failed30
110 + from sources s order by records desc, s.name`;
111 + return rows.map((r) => {
112 + const lastOk = iso(r.last_ok);
113 + const runs = int(r.runs30);
114 + return {
115 + id: reqStr(r.id), name: reqStr(r.name), kind: asSourceKind(r.kind),
116 + recordsContributed: int(r.records), fieldsContributed: int(r.fields), uniqueRecords: int(r.unique_records),
117 + lastSuccessfulCrawl: lastOk,
118 + freshnessDays: lastOk ? Math.max(0, Math.round((Date.now() - new Date(lastOk).getTime()) / 86_400_000)) : null,
119 + authority: authorityTier({ field: "identity", sourceKind: str(r.kind) }),
120 + failureRate: runs ? Math.round((int(r.failed30) / runs) * 1000) / 1000 : null,
121 + license: str(r.license), redistribution: str(r.redistribution),
122 + };
123 + });
124 +}
modified apps/api/src/repositories/dashboard.ts +98 −55
@@ -1,74 +1,108 @@
1 −import type { Dashboard, DashboardStats, ProjectSummary, RankingRow } from "@dci/core";
2 −import { pg, mwExpr, plannedMwExpr, OPERATIONAL_SET, CONSTRUCTION_SET, PLANNED_SET, yearExpr, projectJoins, projectSummaryCols } from "../lib/sql.js";
1 +import type { Dashboard, DashboardStats, GridConstraintDTO, MarketMomentum, ProjectSummary, RankingRow } from "@dci/core";
2 +import { pg, yearExpr, projectJoins, projectSummaryCols, projectLive, facilityView, knownMwAgg, countedAgg, eventCols, eventJoins, gridConstraintCols, gridConstraintJoins, AI_LEVELS, POWER_EVENT_TYPES, GRID_EVENT_TYPES } from "../lib/sql.js";
3 3 import { int, iso, num, reqStr, str, type Row } from "../lib/rows.js";
4 −import { asStatus, asType, projectSummary } from "../lib/dto.js";
4 +import { asStatus, asType, gridConstraintDto, gridConstraintFromEvent, projectSummary, round2, share } from "../lib/dto.js";
5 5 import { hrefFor } from "../lib/resolve.js";
6 −import { latestEvents } from "./events.js";
6 +import { latestEvents, toEventDtos } from "./events.js";
7 7 import { recentlyVerifiedFacilities } from "./facilities.js";
8 8 import { topRows } from "./rankings.js";
9 +import { pulse } from "./pulse.js";
10 +import { coverageRow, dimAggCols, momentum, scopeFor } from "../lib/pipeline.js";
9 11
10 −async function liveTopCountries(limit: number): Promise<RankingRow[]> {
12 +async function liveTop(dim: "country" | "metro" | "operator", limit: number): Promise<RankingRow[]> {
11 13 const sql = pg();
12 − const rows = await sql<Row[]>`
13 − select c.iso2 as id, c.slug, c.name, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw,
14 − (count(*) filter (where ${mwExpr(sql)} is not null))::float / count(*) as coverage
15 − from facilities f join countries c on c.iso2 = f.country_iso2 where f.merged_into is null group by 1, 2, 3 order by n desc limit ${limit}`;
16 − return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor("country", reqStr(r.slug)), value: int(r.n), secondary: num(r.mw), countryIso2: reqStr(r.id), coverage: num(r.coverage) }));
14 + const known = knownMwAgg(sql), counted = countedAgg(sql), hasMw = sql`(coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is not null or f.covered_by_parent)`;
15 + const rows = dim === "country"
16 + ? await sql<Row[]>`select c.iso2 as id, c.slug, c.name, c.iso2 as country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational','partially_operational','expansion'))::float as mw, count(*) filter (where ${counted} and ${hasMw})::int as with_mw from ${facilityView(sql)} f join countries c on c.iso2 = f.country_iso2 group by 1, 2, 3, 4 order by n desc limit ${limit}`
17 + : dim === "metro"
18 + ? await sql<Row[]>`select m.id, m.slug, m.name, m.country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational','partially_operational','expansion'))::float as mw, count(*) filter (where ${counted} and ${hasMw})::int as with_mw from ${facilityView(sql)} f join metros m on m.id = f.metro_id group by 1, 2, 3, 4 order by n desc limit ${limit}`
19 + : await sql<Row[]>`select o.id, o.slug, o.name, o.hq_country_iso2 as country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational','partially_operational','expansion'))::float as mw, count(*) filter (where ${counted} and ${hasMw})::int as with_mw from ${facilityView(sql)} f join operators o on o.id = f.operator_id group by 1, 2, 3, 4 order by n desc limit ${limit}`;
20 + return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor(dim, reqStr(r.slug)), value: int(r.n), secondary: round2(num(r.mw)), countryIso2: str(r.country_iso2), coverage: share(int(r.with_mw), int(r.n)) }));
17 21 }
18 22
19 −async function liveTopMetros(limit: number): Promise<RankingRow[]> {
23 +async function fastestGrowingMetros(limit = 8): Promise<Dashboard["fastestGrowingMetros"]> {
20 24 const sql = pg();
25 + const since = sql`(current_date - interval '12 months')`;
26 + const pd = (col: string) => sql`(case when p.${sql(col)} ~ '^\\d{4}-\\d{2}-\\d{2}' then p.${sql(col)}::date when p.${sql(col)} ~ '^\\d{4}-\\d{2}$' then (p.${sql(col)} || '-01')::date when p.${sql(col)} ~ '^\\d{4}$' then (p.${sql(col)} || '-01-01')::date else null end)`;
21 27 const rows = await sql<Row[]>`
22 − select m.id, m.slug, m.name, m.country_iso2, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw,
23 − (count(*) filter (where ${mwExpr(sql)} is not null))::float / count(*) as coverage
24 − from facilities f join metros m on m.id = f.metro_id where f.merged_into is null group by 1, 2, 3, 4 order by n desc limit ${limit}`;
25 − return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor("metro", reqStr(r.slug)), value: int(r.n), secondary: num(r.mw), countryIso2: str(r.country_iso2), coverage: num(r.coverage) }));
28 + with pa as (select p.metro_id, count(*) filter (where coalesce(${pd("announced_on")}, p.created_at::date) >= ${since})::int as announced, count(*) filter (where ${pd("construction_started_on")} >= ${since})::int as construction from projects p where ${projectLive(sql)} and p.metro_id is not null group by 1),
29 + fo as (select f.metro_id, count(*)::int as opened from facilities f where f.merged_into is null and f.metro_id is not null and f.status in ('operational','partially_operational','expansion') and (case when f.opened_on ~ '^\\d{4}-\\d{2}-\\d{2}' then f.opened_on::date when f.opened_on ~ '^\\d{4}-\\d{2}$' then (f.opened_on || '-01')::date when f.opened_on ~ '^\\d{4}$' then (f.opened_on || '-01-01')::date else null end) >= ${since} group by 1)
30 + select m.id, m.slug, m.name, m.country_iso2, coalesce(pa.announced, 0) + coalesce(pa.construction, 0) + coalesce(fo.opened, 0) as score
31 + from metros m left join pa on pa.metro_id = m.id left join fo on fo.metro_id = m.id
32 + where coalesce(pa.announced, 0) + coalesce(pa.construction, 0) + coalesce(fo.opened, 0) > 0
33 + order by score desc, m.name limit ${limit}`;
34 + return Promise.all(rows.map(async (r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), countryIso2: reqStr(r.country_iso2), momentum: (await momentum(scopeFor(sql, { metroId: reqStr(r.id) }))) as MarketMomentum })));
26 35 }
27 36
28 −async function liveTopOperators(limit: number): Promise<RankingRow[]> {
37 +async function operatorExpansion(limit = 10): Promise<Dashboard["operatorExpansion"]> {
29 38 const sql = pg();
39 + const since = sql`(current_date - interval '12 months')`;
30 40 const rows = await sql<Row[]>`
31 − select o.id, o.slug, o.name, o.hq_country_iso2, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw,
32 − (count(*) filter (where ${mwExpr(sql)} is not null))::float / count(*) as coverage
33 − from facilities f join operators o on o.id = f.operator_id where f.merged_into is null group by 1, 2, 3, 4 order by n desc limit ${limit}`;
34 − return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor("operator", reqStr(r.slug)), value: int(r.n), secondary: num(r.mw), countryIso2: str(r.hq_country_iso2), coverage: num(r.coverage) }));
41 + with entries as (
42 + select f.operator_id, f.metro_id, f.country_iso2, coalesce(case when f.opened_on ~ '^\\d{4}-\\d{2}-\\d{2}' then f.opened_on::date when f.opened_on ~ '^\\d{4}-\\d{2}$' then (f.opened_on || '-01')::date when f.opened_on ~ '^\\d{4}$' then (f.opened_on || '-01-01')::date else null end, f.first_seen::date) as d
43 + from facilities f where f.merged_into is null and f.operator_id is not null
44 + union all
45 + select p.operator_id, p.metro_id, p.country_iso2, case when p.announced_on ~ '^\\d{4}-\\d{2}-\\d{2}' then p.announced_on::date when p.announced_on ~ '^\\d{4}-\\d{2}$' then (p.announced_on || '-01')::date when p.announced_on ~ '^\\d{4}$' then (p.announced_on || '-01-01')::date else p.created_at::date end
46 + from projects p where ${projectLive(sql)} and p.operator_id is not null
47 + ),
48 + fc as (select operator_id, country_iso2, min(d) as first_d from entries where country_iso2 is not null group by 1, 2),
49 + fm as (select operator_id, metro_id, min(d) as first_d from entries where metro_id is not null group by 1, 2),
50 + agg as (
51 + select o.id, o.slug, o.name,
52 + (select array_agg(fc.country_iso2 order by fc.first_d desc) from fc where fc.operator_id = o.id and fc.first_d >= ${since}) as new_countries,
53 + (select array_agg(m.name order by fm.first_d desc) from fm join metros m on m.id = fm.metro_id where fm.operator_id = o.id and fm.first_d >= ${since}) as new_metros,
54 + (select count(*)::int from projects p where ${projectLive(sql)} and p.operator_id = o.id and coalesce(case when p.announced_on ~ '^\\d{4}' then (left(p.announced_on, 4) || '-01-01')::date else null end, p.created_at::date) >= ${since}) as projects12m,
55 + (select count(*) from fc where fc.operator_id = o.id and fc.first_d < ${since}) as had_before
56 + from operators o
57 + )
58 + select * from agg where (coalesce(array_length(new_countries, 1), 0) > 0 or coalesce(array_length(new_metros, 1), 0) > 0) and had_before > 0
59 + order by coalesce(array_length(new_countries, 1), 0) + coalesce(array_length(new_metros, 1), 0) desc, projects12m desc, name limit ${limit}`;
60 + return rows.map((r) => ({ operator: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, newCountries: ((r.new_countries as string[] | null) ?? []).map(String), newMetros: ((r.new_metros as string[] | null) ?? []).map(String), projects12m: int(r.projects12m) }));
61 +}
62 +
63 +async function globalGridConstraints(limit = 10): Promise<GridConstraintDTO[]> {
64 + const sql = pg();
65 + const [rows, ev] = await Promise.all([
66 + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} order by g.created_at desc limit ${limit}`,
67 + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.event_type = any(${GRID_EVENT_TYPES}) and e.review_status <> 'rejected' order by e.detected_at desc limit ${limit}`,
68 + ]);
69 + const out = rows.map(gridConstraintDto);
70 + const seen = new Set(out.map((g) => g.eventId).filter(Boolean));
71 + for (const e of await toEventDtos(ev)) { if (seen.has(e.id)) continue; seen.add(e.id); out.push(gridConstraintFromEvent(e)); }
72 + return out.slice(0, limit);
35 73 }
36 74
37 75 export async function dashboard(): Promise<Dashboard> {
38 76 const sql = pg();
39 − const mw = mwExpr(sql);
40 − const [facStats, countsRow, evRow, crawlRow, capSeries, capFallback, projStatus, typeRows, pipeline, aiRow, aiRecent, latest, newProjects, recentlyVerified, rkCountries, rkMetros, rkOperators] = await Promise.all([
41 − sql<Row[]>`
42 − select count(*)::int as facilities,
43 − count(*) filter (where f.status = any(${OPERATIONAL_SET}))::int as operational,
44 − count(*) filter (where f.status = any(${CONSTRUCTION_SET}))::int as under_construction,
45 − count(*) filter (where f.status = any(${PLANNED_SET}))::int as planned,
46 − coalesce(sum(${mw}) filter (where f.status = any(${OPERATIONAL_SET})), 0)::float as known_operational_mw,
47 − coalesce(sum(${mw}) filter (where f.status = any(${CONSTRUCTION_SET})), 0)::float as construction_mw,
48 − coalesce(sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PLANNED_SET})), 0)::float as planned_mw,
49 − count(*) filter (where ${mw} is not null)::int as with_mw,
50 − count(distinct f.country_iso2)::int as countries,
51 − count(distinct f.metro_id)::int as metros,
52 − count(distinct f.operator_id)::int as operators
53 − from facilities f where f.merged_into is null`,
77 + const known = knownMwAgg(sql);
78 + const [facStats, countsRow, evRow, crawlRow, capSeries, capFallback, projStatus, typeRows, pipeline, aiRow, aiRecent, latest, newProjects, recentlyVerified, rkCountries, rkMetros, rkOperators, pulseData, growing, major, expansion, powerRows, gridConstraints, coverage, ingRow] = await Promise.all([
79 + sql<Row[]>`select ${dimAggCols(sql)}, count(distinct f.country_iso2)::int as countries, count(distinct f.metro_id)::int as metros from ${facilityView(sql)} f`,
54 80 sql<Row[]>`
55 − select (select count(*)::int from cloud_regions) as cloud_regions, (select count(*)::int from ixps) as ixps, (select count(*)::int from projects) as projects,
81 + select (select count(*)::int from cloud_regions where status <> 'retired') as cloud_regions, (select count(*)::int from ixps) as ixps, (select count(*)::int from projects p where ${projectLive(sql)}) as projects,
56 82 (select count(*)::int from sources) as sources, (select count(*)::int from documents) as documents, (select count(*)::int from operators) as all_operators`,
57 83 sql<Row[]>`select count(*) filter (where detected_at >= now() - interval '24 hours')::int as e24, count(*) filter (where detected_at >= now() - interval '7 days')::int as e7 from events where review_status <> 'rejected'`,
58 − sql<Row[]>`select max(finished_at) as last_finished, max(started_at) as last_started from connector_runs`,
84 + sql<Row[]>`select max(finished_at) as last_finished, max(started_at) as last_started, count(*) filter (where started_at >= now() - interval '24 hours')::int as runs24 from connector_runs`,
59 85 sql<Row[]>`select day, metric, value from daily_metrics where dim = 'global' and metric in ('facilities_total', 'known_mw') order by day`,
60 − sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*)::int as n, sum(${mw})::float as mw from facilities f where f.merged_into is null and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
61 − sql<Row[]>`select status, count(*)::int as n, sum(planned_mw)::float as mw from projects group by status order by n desc`,
86 + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${known})::float as mw from ${facilityView(sql)} f where f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
87 + sql<Row[]>`select p.status, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${projectLive(sql)} group by p.status order by n desc`,
62 88 sql<Row[]>`select facility_type, count(*)::int as n from facilities where merged_into is null group by facility_type order by n desc`,
63 − sql<Row[]>`select left(expected_opening, 4) as year, count(*)::int as n, sum(planned_mw)::float as mw from projects where expected_opening ~ '^\\d{4}' and status not in ('cancelled', 'closed') group by 1 order by 1`,
64 − sql<Row[]>`select (select count(*)::int from facilities where merged_into is null and (is_ai or facility_type = 'ai')) as facilities, (select count(*)::int from projects where is_ai) as projects, (select sum(planned_mw)::float from projects where is_ai and status not in ('cancelled')) as planned_mw`,
65 − sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where p.is_ai order by p.last_update desc limit 6`,
89 + sql<Row[]>`select left(p.expected_opening, 4) as year, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${projectLive(sql)} and p.expected_opening ~ '^\\d{4}' and p.status not in ('cancelled', 'closed') group by 1 order by 1`,
90 + sql<Row[]>`select (select count(*)::int from ${facilityView(sql)} f where ${countedAgg(sql)} and (f.ai_evidence = any(${AI_LEVELS}) or f.is_ai or f.facility_type = 'ai')) as facilities, (select count(*)::int from projects p where ${projectLive(sql)} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai)) as projects, (select sum(p.planned_mw)::float from projects p where ${projectLive(sql)} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai) and p.status not in ('cancelled')) as planned_mw`,
91 + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai) order by p.last_update desc limit 6`,
66 92 latestEvents(20),
67 − sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} order by p.created_at desc limit 10`,
93 + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} order by p.created_at desc limit 10`,
68 94 recentlyVerifiedFacilities(10),
69 95 topRows(["countries_by_facilities", "countries_by_known_mw", "countries_facilities", "top_countries"], "countries", 10),
70 96 topRows(["metros_by_facilities", "metros_by_known_mw", "metros_facilities", "top_metros"], "metros", 10),
71 97 topRows(["operators_by_facilities", "operators_by_known_mw", "operators_facilities", "top_operators"], "operators", 10),
98 + pulse("24h"),
99 + fastestGrowingMetros(8),
100 + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and p.planned_mw >= 100 and p.status not in ('cancelled', 'closed') order by p.planned_mw desc, p.last_update desc limit 12`,
101 + operatorExpansion(10),
102 + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.event_type = any(${POWER_EVENT_TYPES}) and e.review_status <> 'rejected' order by e.detected_at desc limit 10`,
103 + globalGridConstraints(10),
104 + coverageRow("global", "Global", "global", scopeFor(sql, {})),
105 + sql<Row[]>`select (select count(*)::int from documents where last_fetched >= now() - interval '24 hours') as docs24, (select count(*)::int from connectors where enabled and not paused) as total, (select count(*)::int from connectors where enabled and not paused and health = 'ok') as healthy`,
72 106 ]);
73 107 const fs = facStats[0] ?? {};
74 108 const cn = countsRow[0] ?? {};
@@ -76,12 +110,12 @@ export async function dashboard(): Promise<Dashboard> {
76 110 const stats: DashboardStats = {
77 111 facilities,
78 112 operational: int(fs.operational),
79 − underConstruction: int(fs.under_construction),
113 + underConstruction: int(fs.construction),
80 114 planned: int(fs.planned),
81 − knownOperationalMw: Math.round((num(fs.known_operational_mw) ?? 0) * 100) / 100,
82 − constructionMw: Math.round((num(fs.construction_mw) ?? 0) * 100) / 100,
83 − plannedMw: Math.round((num(fs.planned_mw) ?? 0) * 100) / 100,
84 − mwCoverage: facilities ? Math.round((int(fs.with_mw) / facilities) * 1000) / 1000 : 0,
115 + knownOperationalMw: round2(num(fs.known_mw)) ?? 0,
116 + constructionMw: round2(num(fs.construction_mw)) ?? 0,
117 + plannedMw: round2(num(fs.planned_mw)) ?? 0,
118 + mwCoverage: share(int(fs.with_mw), facilities),
85 119 countries: int(fs.countries),
86 120 metros: int(fs.metros),
87 121 operators: int(fs.operators) || int(cn.all_operators),
@@ -112,22 +146,31 @@ export async function dashboard(): Promise<Dashboard> {
112 146 }
113 147 if (!capacityOverTime.length) {
114 148 let cumF = 0, cumMw = 0, saw = false;
115 − capacityOverTime = capFallback.map((r) => { cumF += int(r.n); const m = num(r.mw); if (m != null) { cumMw += m; saw = true; } return { year: int(r.year), facilities: cumF, knownMw: m, cumulativeMw: saw ? Math.round(cumMw * 100) / 100 : null }; }).filter((x) => x.year > 0);
149 + capacityOverTime = capFallback.map((r) => { cumF += int(r.n); const m = num(r.mw); if (m != null) { cumMw += m; saw = true; } return { year: int(r.year), facilities: cumF, knownMw: round2(m), cumulativeMw: saw ? Math.round(cumMw * 100) / 100 : null }; }).filter((x) => x.year > 0);
116 150 }
117 151
118 152 const ai = aiRow[0] ?? {};
153 + const ing = ingRow[0] ?? {};
119 154 return {
120 155 stats,
121 156 capacityOverTime,
122 − projectsByStatus: projStatus.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: num(r.mw) })),
123 − topCountries: rkCountries ?? (await liveTopCountries(10)),
124 − topMetros: rkMetros ?? (await liveTopMetros(10)),
125 − topOperators: rkOperators ?? (await liveTopOperators(10)),
157 + projectsByStatus: projStatus.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: round2(num(r.mw)) })),
158 + topCountries: rkCountries ?? (await liveTop("country", 10)),
159 + topMetros: rkMetros ?? (await liveTop("metro", 10)),
160 + topOperators: rkOperators ?? (await liveTop("operator", 10)),
126 161 typeBreakdown: typeRows.map((r) => ({ type: asType(r.facility_type), count: int(r.n) })),
127 − pipelineByYear: pipeline.map((r) => ({ year: reqStr(r.year), count: int(r.n), mw: num(r.mw) })),
128 − aiExpansion: { facilities: int(ai.facilities), projects: int(ai.projects), plannedMw: num(ai.planned_mw), recent: aiRecent.map(projectSummary) as ProjectSummary[] },
162 + pipelineByYear: pipeline.map((r) => ({ year: reqStr(r.year), count: int(r.n), mw: round2(num(r.mw)) })),
163 + aiExpansion: { facilities: int(ai.facilities), projects: int(ai.projects), plannedMw: round2(num(ai.planned_mw)), recent: aiRecent.map(projectSummary) as ProjectSummary[] },
129 164 latestEvents: latest,
130 165 newProjects: newProjects.map(projectSummary),
131 166 recentlyVerified,
167 + pulse: pulseData,
168 + fastestGrowingMetros: growing,
169 + majorProjects: major.map(projectSummary),
170 + operatorExpansion: expansion,
171 + powerEvents: await toEventDtos(powerRows),
172 + gridConstraints,
173 + coverage,
174 + ingestion: { lastRunAt: iso(crawlRow[0]?.last_started), runs24h: int(crawlRow[0]?.runs24), documents24h: int(ing.docs24), events24h: int(evRow[0]?.e24), connectorsHealthy: int(ing.healthy), connectorsTotal: int(ing.total) },
132 175 };
133 176 }
added apps/api/src/repositories/download.ts +196 −0
@@ -0,0 +1,196 @@
1 +/**
2 + * Dataset downloads (CSV / JSON / GeoJSON), streamed in 1 000-row batches, ≤ 50 000 rows, with license gating:
3 + * a row whose ONLY current provenance sources forbid redistribution (sources.redistribution = 'restricted') is
4 + * excluded and the excluded sources are listed. Nothing is ever blanked silently.
5 + */
6 +import type { DownloadDataset } from "@dci/core";
7 +import { pg, andAll, facilityJoins, projectLive, yearExpr, type Fragment, type Sql } from "../lib/sql.js";
8 +import { int, num, reqStr, str, type Row } from "../lib/rows.js";
9 +import { csvLine } from "../lib/csv.js";
10 +import { csv } from "../lib/http.js";
11 +import { facilityConds, toFacilityQuery } from "./facilities.js";
12 +
13 +export const DOWNLOAD_KEYS = ["facilities", "projects", "operators", "events", "cloud-regions", "ixps", "countries", "markets"] as const;
14 +export type DownloadKey = (typeof DownloadKeys)[number];
15 +const DownloadKeys = DOWNLOAD_KEYS;
16 +export type DownloadFormat = "csv" | "json" | "geojson";
17 +export const MAX_ROWS = 50_000;
18 +export const DOWNLOAD_LICENSE = "Each row carries the names of the sources it was built from; re-users must keep that attribution and respect each source's licence (see /api/v1/sources). Rows whose only sources forbid redistribution are excluded from every download and listed in excludedSources. Figures are published values only (no estimates unless flagged), with the coverage caveats of the API.";
19 +
20 +export interface DownloadFilters { country?: string; status?: string; type?: string; operator?: string; metro?: string; min_mw?: number; max_mw?: number; ai?: boolean; hyperscale?: boolean; has_mw?: boolean; project_status?: string; expected_from?: number; expected_to?: number; event_type?: string; since?: string; until?: string; min_significance?: number }
21 +
22 +interface Spec { key: DownloadKey; label: string; description: string; formats: DownloadFormat[]; filters: string[]; columns: string[]; entityType: string | null; gated: boolean }
23 +
24 +export const SPECS: Spec[] = [
25 + { key: "facilities", label: "Facilities", description: "Every indexed data center / campus / building record (merged duplicates excluded) with location, status, capacity ontology, AI evidence and sources.", formats: ["csv", "json", "geojson"], filters: ["country", "status", "type", "operator", "metro", "min_mw", "max_mw", "ai", "hyperscale", "has_mw"], columns: ["id", "slug", "name", "operator", "country_iso2", "city", "metro", "lat", "lng", "geo_precision", "status", "facility_type", "it_capacity_mw", "total_power_mw", "planned_power_mw", "utility_capacity_mw", "grid_connection_mw", "mw_is_estimate", "record_scope", "parent_facility_id", "ai_evidence", "opened_on", "confidence", "completeness", "source_count", "sources", "updated_at"], entityType: "facility", gated: true },
26 + { key: "projects", label: "Projects", description: "Live infrastructure projects (false positives hidden by review and merged duplicates excluded) with lifecycle dates, planned MW and investment scopes.", formats: ["csv", "json", "geojson"], filters: ["country", "project_status", "operator", "metro", "min_mw", "max_mw", "ai", "expected_from", "expected_to"], columns: ["id", "slug", "name", "operator", "country_iso2", "city", "metro", "lat", "lng", "geo_precision", "status", "project_class", "evidence_level", "announced_on", "permit_filed_on", "approved_on", "construction_started_on", "expected_opening", "opened_on", "planned_mw", "capacity_scope", "investment_usd", "investment_scope", "ai_evidence", "confidence", "sources", "last_update"], entityType: "project", gated: true },
27 + { key: "operators", label: "Operators", description: "Operators with containment-aware facility counts and known MW (from the worker-refreshed stats).", formats: ["csv", "json"], filters: ["country"], columns: ["id", "slug", "name", "kind", "hq_country_iso2", "website", "is_cloud_provider", "is_carrier", "facility_count", "country_count", "known_mw", "planned_mw", "project_count", "mw_coverage", "sources", "updated_at"], entityType: "operator", gated: true },
28 + { key: "events", label: "Events", description: "Change feed: detected facility / project / market events with source and significance (rejected events excluded).", formats: ["csv", "json"], filters: ["event_type", "country", "operator", "since", "until", "min_significance"], columns: ["id", "entity_type", "entity_id", "event_type", "detected_at", "effective_date", "title", "url", "source", "source_kind", "significance", "significance_band", "confidence", "country_iso2", "operator", "old_value", "new_value"], entityType: null, gated: true },
29 + { key: "cloud-regions", label: "Cloud regions", description: "Public cloud regions by provider with city-level coordinates.", formats: ["csv", "json", "geojson"], filters: ["country", "operator"], columns: ["id", "slug", "provider", "code", "name", "city", "country_iso2", "lat", "lng", "geo_precision", "availability_zones", "launched_on", "status", "is_sovereign", "source_url", "sources", "updated_at"], entityType: "cloud_region", gated: true },
30 + { key: "ixps", label: "Internet exchanges", description: "IXPs with network counts and facility links; coordinates are the metro reference point when known.", formats: ["csv", "json", "geojson"], filters: ["country"], columns: ["id", "slug", "name", "name_long", "city", "country_iso2", "website", "network_count", "facility_count", "metro", "lat", "lng", "sources", "updated_at"], entityType: "ixp", gated: true },
31 + { key: "countries", label: "Countries", description: "Countries with containment-aware facility totals (worker-refreshed stats) and public energy indicators.", formats: ["csv", "json", "geojson"], filters: [], columns: ["iso2", "iso3", "slug", "name", "region", "subregion", "population", "gdp_usd", "electricity_twh", "renewable_share", "stats_year", "facility_count", "known_mw", "mw_coverage", "lat", "lng", "updated_at"], entityType: null, gated: false },
32 + { key: "markets", label: "Markets (metros)", description: "Data center markets / metros with reference point, radius and containment-aware totals.", formats: ["csv", "json", "geojson"], filters: ["country"], columns: ["id", "slug", "name", "country_iso2", "region_name", "lat", "lng", "radius_km", "facility_count", "known_mw", "mw_coverage", "operator_count", "updated_at"], entityType: null, gated: false },
33 +];
34 +
35 +export function specFor(key: string): Spec | null {
36 + return SPECS.find((s) => s.key === key) ?? null;
37 +}
38 +
39 +/** Sources that forbid redistribution (their rows are gated out). */
40 +export async function restrictedSources(): Promise<Array<{ id: string; name: string; reason: string }>> {
41 + const sql = pg();
42 + const rows = await sql<Row[]>`select id, name from sources where redistribution = 'restricted' order by name`;
43 + return rows.map((r) => ({ id: reqStr(r.id), name: reqStr(r.name), reason: "redistribution = restricted" }));
44 +}
45 +
46 +/** A row is gated out when it has provenance and every current source is restricted. */
47 +function gate(sql: Sql, entityType: string, idCol: Fragment): Fragment {
48 + return sql`not (exists (select 1 from provenance gp where gp.entity_type = ${entityType} and gp.entity_id = ${idCol} and gp.is_current)
49 + and not exists (select 1 from provenance gp join sources gs on gs.id = gp.source_id where gp.entity_type = ${entityType} and gp.entity_id = ${idCol} and gp.is_current and coalesce(gs.redistribution, 'unknown') <> 'restricted'))`;
50 +}
51 +
52 +function sourcesCol(sql: Sql, entityType: string, idCol: Fragment): Fragment {
53 + return sql`(select string_agg(distinct s.name, ';' order by s.name) from provenance sp join sources s on s.id = sp.source_id where sp.entity_type = ${entityType} and sp.entity_id = ${idCol} and sp.is_current)`;
54 +}
55 +
56 +/** SELECT for a dataset, columns aliased exactly as in the spec (extra `lat`/`lng` are used for GeoJSON). */
57 +function query(sql: Sql, key: DownloadKey, f: DownloadFilters): Fragment {
58 + switch (key) {
59 + case "facilities": {
60 + const conds = facilityConds(sql, toFacilityQuery({ country: f.country, status: f.status, type: f.type, operator: f.operator, metro: f.metro, min_mw: f.min_mw, max_mw: f.max_mw, ai: f.ai, hyperscale: f.hyperscale, has_mw: f.has_mw }));
61 + conds.push(gate(sql, "facility", sql`f.id`));
62 + return sql`select f.id, f.slug, f.name, o.name as operator, f.country_iso2, f.city, m.name as metro, f.lat, f.lng, f.geo_precision, f.status, f.facility_type, f.it_capacity_mw, f.total_power_mw, f.planned_power_mw, f.utility_capacity_mw, f.grid_connection_mw, f.mw_is_estimate, f.record_scope, f.parent_facility_id, f.ai_evidence, f.opened_on, f.confidence, f.completeness, f.source_count, ${sourcesCol(sql, "facility", sql`f.id`)} as sources, f.updated_at
63 + from facilities f ${facilityJoins(sql)} where ${andAll(sql, conds)} order by f.country_iso2, f.name limit ${MAX_ROWS}`;
64 + }
65 + case "projects": {
66 + const c: Fragment[] = [projectLive(sql), gate(sql, "project", sql`p.id`)];
67 + const st = csv(f.project_status ?? f.status);
68 + if (st.length) c.push(sql`p.status = any(${st})`);
69 + const co = csv(f.country).map((x) => x.toUpperCase());
70 + if (co.length) c.push(sql`p.country_iso2 = any(${co})`);
71 + if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`);
72 + if (f.metro) c.push(sql`(m.slug = ${f.metro} or m.id = ${f.metro})`);
73 + if (f.min_mw != null) c.push(sql`p.planned_mw >= ${f.min_mw}`);
74 + if (f.max_mw != null) c.push(sql`p.planned_mw <= ${f.max_mw}`);
75 + if (f.ai === true) c.push(sql`(p.is_ai or p.ai_evidence in ('confirmed', 'likely'))`);
76 + if (f.expected_from != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} >= ${f.expected_from}`);
77 + if (f.expected_to != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} <= ${f.expected_to}`);
78 + return sql`select p.id, p.slug, p.name, o.name as operator, p.country_iso2, p.city, m.name as metro, p.lat, p.lng, p.geo_precision, p.status, p.project_class, p.evidence_level, p.announced_on, p.permit_filed_on, p.approved_on, p.construction_started_on, p.expected_opening, p.opened_on, p.planned_mw, p.capacity_scope, p.investment_usd, p.investment_scope, p.ai_evidence, p.confidence, ${sourcesCol(sql, "project", sql`p.id`)} as sources, p.last_update
79 + from projects p left join operators o on o.id = p.operator_id left join metros m on m.id = p.metro_id where ${andAll(sql, c)} order by p.planned_mw desc nulls last, p.name limit ${MAX_ROWS}`;
80 + }
81 + case "operators": {
82 + const c: Fragment[] = [gate(sql, "operator", sql`o.id`)];
83 + if (f.country) c.push(sql`(o.hq_country_iso2 = ${f.country.toUpperCase()} or exists (select 1 from facilities x where x.operator_id = o.id and x.country_iso2 = ${f.country.toUpperCase()} and x.merged_into is null))`);
84 + return sql`select o.id, o.slug, o.name, o.kind, o.hq_country_iso2, o.website, o.is_cloud_provider, o.is_carrier, (o.stats->>'facilityCount')::int as facility_count, (o.stats->>'countryCount')::int as country_count, (o.stats->>'knownMw')::float as known_mw, (o.stats->>'plannedMw')::float as planned_mw, (o.stats->>'projectCount')::int as project_count, (o.stats->>'mwCoverage')::float as mw_coverage, ${sourcesCol(sql, "operator", sql`o.id`)} as sources, o.updated_at
85 + from operators o where ${andAll(sql, c)} order by (o.stats->>'facilityCount')::int desc nulls last, o.name limit ${MAX_ROWS}`;
86 + }
87 + case "events": {
88 + const c: Fragment[] = [sql`e.review_status <> 'rejected'`, sql`coalesce(s.redistribution, 'unknown') <> 'restricted'`];
89 + const types = csv(f.event_type);
90 + if (types.length) c.push(sql`e.event_type = any(${types})`);
91 + if (f.country) c.push(sql`e.country_iso2 = ${f.country.toUpperCase()}`);
92 + if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`);
93 + if (f.since) c.push(sql`e.detected_at >= ${f.since}::timestamptz`);
94 + if (f.until) c.push(sql`e.detected_at <= ${f.until}::timestamptz`);
95 + if (f.min_significance != null) c.push(sql`e.significance >= ${f.min_significance}`);
96 + return sql`select e.id, e.entity_type, e.entity_id, e.event_type, e.detected_at, e.effective_date, e.title, e.url, s.name as source, coalesce(e.source_kind, s.kind) as source_kind, e.significance, case when e.significance >= 75 then 'major' when e.significance >= 45 then 'medium' else 'minor' end as significance_band, e.confidence, e.country_iso2, o.name as operator, e.old_value, e.new_value
97 + from events e left join sources s on s.id = e.source_id left join operators o on o.id = e.operator_id where ${andAll(sql, c)} order by e.detected_at desc limit ${MAX_ROWS}`;
98 + }
99 + case "cloud-regions": {
100 + const c: Fragment[] = [gate(sql, "cloud_region", sql`r.id`)];
101 + if (f.country) c.push(sql`r.country_iso2 = ${f.country.toUpperCase()}`);
102 + if (f.operator) c.push(sql`(pr.slug = ${f.operator} or pr.id = ${f.operator})`);
103 + return sql`select r.id, r.slug, pr.name as provider, r.code, r.name, r.city, r.country_iso2, r.lat, r.lng, r.geo_precision, r.availability_zones, r.launched_on, r.status, r.is_sovereign, r.source_url, ${sourcesCol(sql, "cloud_region", sql`r.id`)} as sources, r.updated_at
104 + from cloud_regions r join operators pr on pr.id = r.provider_id where ${andAll(sql, c)} order by pr.name, r.code limit ${MAX_ROWS}`;
105 + }
106 + case "ixps": {
107 + const c: Fragment[] = [gate(sql, "ixp", sql`x.id`)];
108 + if (f.country) c.push(sql`x.country_iso2 = ${f.country.toUpperCase()}`);
109 + return sql`select x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count, m.name as metro, m.lat, m.lng, ${sourcesCol(sql, "ixp", sql`x.id`)} as sources, x.updated_at
110 + from ixps x left join metros m on m.id = x.metro_id where ${andAll(sql, c)} order by x.network_count desc nulls last, x.name limit ${MAX_ROWS}`;
111 + }
112 + case "countries":
113 + return sql`select c.iso2, c.iso3, c.slug, c.name, c.region, c.subregion, c.population, c.gdp_usd, c.electricity_twh, c.renewable_share, c.stats_year, (c.stats->>'facilityCount')::int as facility_count, (c.stats->>'knownMw')::float as known_mw, (c.stats->>'mwCoverage')::float as mw_coverage, c.lat, c.lng, c.updated_at
114 + from countries c where coalesce((c.stats->>'facilityCount')::int, 0) > 0 or exists (select 1 from cloud_regions r where r.country_iso2 = c.iso2) order by (c.stats->>'facilityCount')::int desc nulls last, c.name limit ${MAX_ROWS}`;
115 + case "markets": {
116 + const c: Fragment[] = [sql`true`];
117 + if (f.country) c.push(sql`m.country_iso2 = ${f.country.toUpperCase()}`);
118 + return sql`select m.id, m.slug, m.name, m.country_iso2, m.region_name, m.lat, m.lng, m.radius_km, (m.stats->>'facilityCount')::int as facility_count, (m.stats->>'knownMw')::float as known_mw, (m.stats->>'mwCoverage')::float as mw_coverage, (m.stats->>'operatorCount')::int as operator_count, m.updated_at
119 + from metros m where ${andAll(sql, c)} order by (m.stats->>'facilityCount')::int desc nulls last, m.name limit ${MAX_ROWS}`;
120 + }
121 + }
122 +}
123 +
124 +/** Dataset catalogue with live row counts. */
125 +export async function listDatasets(): Promise<DownloadDataset[]> {
126 + const sql = pg();
127 + const [counts, excluded, attributions] = await Promise.all([
128 + sql<Row[]>`select
129 + (select count(*) from facilities where merged_into is null)::int as facilities,
130 + (select count(*) from projects p where ${projectLive(sql)})::int as projects,
131 + (select count(*) from operators)::int as operators,
132 + (select count(*) from events where review_status <> 'rejected')::int as events,
133 + (select count(*) from cloud_regions)::int as "cloud-regions",
134 + (select count(*) from ixps)::int as ixps,
135 + (select count(*) from countries c where coalesce((c.stats->>'facilityCount')::int, 0) > 0 or exists (select 1 from cloud_regions r where r.country_iso2 = c.iso2))::int as countries,
136 + (select count(*) from metros)::int as markets`,
137 + restrictedSources(),
138 + sql<Row[]>`select p.entity_type, array_agg(distinct s.attribution) filter (where s.attribution is not null) as attributions from provenance p join sources s on s.id = p.source_id where p.is_current group by p.entity_type`,
139 + ]);
140 + const attr = new Map<string, string[]>(attributions.map((r) => [reqStr(r.entity_type), ((r.attributions as string[] | null) ?? []).slice(0, 30)]));
141 + const eventAttr = (await sql<Row[]>`select distinct s.attribution from events e join sources s on s.id = e.source_id where s.attribution is not null limit 30`).map((r) => reqStr(r.attribution));
142 + const c = counts[0] ?? {};
143 + return SPECS.map((s) => ({
144 + key: s.key,
145 + label: s.label,
146 + description: s.description,
147 + formats: s.formats,
148 + filters: s.filters,
149 + rows: Math.min(MAX_ROWS, int(c[s.key])),
150 + license: DOWNLOAD_LICENSE,
151 + attribution: s.entityType ? attr.get(s.entityType) ?? [] : s.key === "events" ? eventAttr : [],
152 + excludedSources: s.gated ? excluded : [],
153 + }));
154 +}
155 +
156 +function cell(v: unknown): unknown {
157 + if (v == null) return null;
158 + if (typeof v === "object" && !(v instanceof Date)) return JSON.stringify(v);
159 + return v;
160 +}
161 +
162 +/** Async generator of chunks for a dataset in the requested format (header first). */
163 +export async function* streamDataset(key: DownloadKey, format: DownloadFormat, f: DownloadFilters, excluded: Array<{ id: string; name: string; reason: string }>): AsyncGenerator<string> {
164 + const sql = pg();
165 + const spec = specFor(key)!;
166 + const cols = spec.columns;
167 + const q = query(sql, key, f);
168 + let total = 0, skipped = 0, first = true;
169 + if (format === "csv") yield csvLine(cols);
170 + else if (format === "json") yield `{"data":[`;
171 + else yield `{"type":"FeatureCollection","features":[`;
172 + const cursor = q.cursor(1000);
173 + for await (const rows of cursor) {
174 + let chunk = "";
175 + for (const r of rows as Row[]) {
176 + total++;
177 + if (format === "csv") { chunk += csvLine(cols.map((c) => cell(r[c]))); continue; }
178 + const obj: Record<string, unknown> = {};
179 + for (const c of cols) obj[c] = r[c] ?? null;
180 + if (format === "json") { chunk += `${first ? "" : ","}${JSON.stringify(obj)}`; first = false; continue; }
181 + const lat = num(r.lat), lng = num(r.lng);
182 + if (lat == null || lng == null) { skipped++; continue; }
183 + const { lat: _a, lng: _b, ...props } = obj;
184 + chunk += `${first ? "" : ","}${JSON.stringify({ type: "Feature", id: str(r.id ?? r.iso2), geometry: { type: "Point", coordinates: [lng, lat] }, properties: props })}`;
185 + first = false;
186 + }
187 + if (chunk) yield chunk;
188 + }
189 + const meta = { total: format === "geojson" ? total - skipped : total, rowsWithoutCoordinates: format === "geojson" ? skipped : undefined, excludedSources: excluded, generatedAt: new Date().toISOString(), license: DOWNLOAD_LICENSE, maxRows: MAX_ROWS };
190 + if (format === "json") yield `],"meta":${JSON.stringify(meta)}}`;
191 + else if (format === "geojson") yield `],"meta":${JSON.stringify(meta)}}`;
192 +}
193 +
194 +export function contentType(format: DownloadFormat): string {
195 + return format === "csv" ? "text/csv; charset=utf-8" : format === "json" ? "application/json; charset=utf-8" : "application/geo+json; charset=utf-8";
196 +}
modified apps/api/src/repositories/events.ts +121 −52
@@ -1,7 +1,7 @@
1 −import type { EventDTO } from "@dci/core";
2 −import { pg, andAll, eventCols, eventJoins, page, type Fragment } from "../lib/sql.js";
3 −import { int, type Row } from "../lib/rows.js";
4 −import { eventDto } from "../lib/dto.js";
1 +import type { EventDTO, SourceKind } from "@dci/core";
2 +import { pg, andAll, eventCols, eventJoins, page, likePattern, type Fragment } from "../lib/sql.js";
3 +import { int, reqStr, type Row } from "../lib/rows.js";
4 +import { asSourceKind, eventDto } from "../lib/dto.js";
5 5 import { entityKey, resolveEntityRefs } from "../lib/resolve.js";
6 6
7 7 export interface EventFilters {
@@ -13,102 +13,171 @@ export interface EventFilters {
13 13 entityType?: string;
14 14 entityId?: string;
15 15 minSignificance?: number;
16 + significance?: "major" | "medium" | "minor";
17 + confidence?: string[];
18 + sourceKind?: string[];
19 + ai?: boolean;
16 20 since?: string;
21 + until?: string;
22 + q?: string;
23 + /** collapse events sharing a cluster id into one row (default true) */
24 + dedupe?: boolean;
17 25 reviewStatus?: string;
18 26 page?: number;
19 27 perPage?: number;
20 28 }
21 29
30 +/** Primary-source order used to pick the representative member of a cluster. */
31 +const SOURCE_RANK = ["operator", "government", "filing", "utility", "cloud_provider", "registry", "secondary", "news", "dataset", "community"];
32 +
33 +export const EVENTS_DEDUPE_METHODOLOGY = "dedupe=true (default) collapses events that share a cluster_id (documents describing the same announcement) into one row: the member from the most authoritative source kind (operator > government > filing > utility > cloud provider > registry > secondary > news > dataset > community), then the highest significance, then the earliest detection. evidenceCount = number of documents in the cluster; otherSources lists the other members (or same-day events of the same operator and type when no cluster id exists). Significance bands: major ≥ 75, medium 45–74, minor < 45.";
34 +
35 +function sourceRankExpr(sql: ReturnType<typeof pg>): Fragment {
36 + return sql`(case coalesce(e.source_kind, s.kind) ${sql.unsafe(SOURCE_RANK.map((k, i) => `when '${k}' then ${i}`).join(" "))} else 99 end)`;
37 +}
38 +
22 39 /** Map rows to EventDTOs with {slug,name} resolved per entity type. */
23 40 export async function toEventDtos(rows: Row[]): Promise<EventDTO[]> {
24 41 const refs = await resolveEntityRefs(rows.map((r) => ({ type: String(r.entity_type ?? ""), id: r.entity_id == null ? null : String(r.entity_id) })));
25 42 return rows.map((r) => eventDto(r, refs.get(entityKey(r.entity_type, r.entity_id)) ?? null));
26 43 }
27 44
45 +/**
46 + * Fill evidenceCount / otherSources for a page of events in one query: other members of the same cluster, or — when
47 + * the event has no cluster id — same-day events with the same operator and event type.
48 + */
49 +export async function attachOtherSources(dtos: EventDTO[]): Promise<EventDTO[]> {
50 + if (!dtos.length) return dtos;
51 + const sql = pg();
52 + const ids = dtos.map((d) => d.id);
53 + const rows = await sql<Row[]>`
54 + select x.id as for_id, e.id, s.name as source_name, coalesce(e.source_kind, s.kind) as source_kind, e.url
55 + from events x
56 + join events e on e.id <> x.id and e.review_status <> 'rejected' and (
57 + (x.cluster_id is not null and e.cluster_id = x.cluster_id)
58 + or (x.cluster_id is null and x.operator_id is not null and e.operator_id = x.operator_id and e.event_type = x.event_type and e.detected_at::date = x.detected_at::date)
59 + )
60 + left join sources s on s.id = e.source_id
61 + where x.id = any(${ids})
62 + order by x.id, e.detected_at asc`;
63 + const by = new Map<string, Array<{ sourceName: string; sourceKind: SourceKind; url: string }>>();
64 + const seen = new Set<string>();
65 + for (const r of rows) {
66 + const forId = reqStr(r.for_id);
67 + const url = reqStr(r.url);
68 + const k = `${forId}|${url}`;
69 + if (seen.has(k)) continue;
70 + seen.add(k);
71 + const list = by.get(forId) ?? [];
72 + if (list.length < 10) list.push({ sourceName: reqStr(r.source_name, "unknown source"), sourceKind: asSourceKind(r.source_kind), url });
73 + by.set(forId, list);
74 + }
75 + for (const d of dtos) {
76 + const others = by.get(d.id) ?? [];
77 + d.otherSources = others;
78 + d.evidenceCount = Math.max(d.evidenceCount ?? 1, others.length + 1);
79 + }
80 + return dtos;
81 +}
82 +
83 +function conds(f: EventFilters): Fragment[] {
84 + const sql = pg();
85 + const c: Fragment[] = [];
86 + if (f.type?.length) c.push(sql`e.event_type = any(${f.type})`);
87 + if (f.country) c.push(sql`e.country_iso2 = ${f.country.toUpperCase()}`);
88 + if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`);
89 + if (f.metro) c.push(sql`e.metro_id in (select id from metros where slug = ${f.metro} or id = ${f.metro})`);
90 + if (f.project) c.push(sql`(e.project_id in (select id from projects where slug = ${f.project} or id = ${f.project}) or (e.entity_type = 'project' and e.entity_id in (select id from projects where slug = ${f.project} or id = ${f.project})))`);
91 + if (f.entityType) c.push(sql`e.entity_type = ${f.entityType}`);
92 + if (f.entityId) c.push(sql`e.entity_id = ${f.entityId}`);
93 + if (f.minSignificance != null) c.push(sql`e.significance >= ${f.minSignificance}`);
94 + if (f.significance === "major") c.push(sql`e.significance >= 75`);
95 + if (f.significance === "medium") c.push(sql`e.significance >= 45 and e.significance < 75`);
96 + if (f.significance === "minor") c.push(sql`e.significance < 45`);
97 + if (f.confidence?.length) c.push(sql`e.confidence = any(${f.confidence})`);
98 + if (f.sourceKind?.length) c.push(sql`coalesce(e.source_kind, s.kind) = any(${f.sourceKind})`);
99 + if (f.ai === true) c.push(sql`e.is_ai`);
100 + if (f.ai === false) c.push(sql`not e.is_ai`);
101 + if (f.since) c.push(sql`e.detected_at >= ${f.since}::timestamptz`);
102 + if (f.until) c.push(sql`e.detected_at <= ${f.until}::timestamptz`);
103 + if (f.q) { const t = f.q.trim(); if (t) c.push(sql`(e.title ilike ${likePattern(t)} or e.summary ilike ${likePattern(t)})`); }
104 + if (f.reviewStatus) c.push(sql`e.review_status = ${f.reviewStatus}`);
105 + else c.push(sql`e.review_status <> 'rejected'`);
106 + return c;
107 +}
108 +
28 109 export async function listEvents(f: EventFilters): Promise<{ items: EventDTO[]; total: number; page: number; perPage: number }> {
29 110 const sql = pg();
30 111 const pg_ = page(f.page, f.perPage, 100, 50);
31 − const conds: Fragment[] = [];
32 − if (f.type?.length) conds.push(sql`e.event_type = any(${f.type})`);
33 − if (f.country) conds.push(sql`e.country_iso2 = ${f.country.toUpperCase()}`);
34 − if (f.operator) conds.push(sql`o.slug = ${f.operator}`);
35 − if (f.metro) conds.push(sql`e.metro_id in (select id from metros where slug = ${f.metro} or id = ${f.metro})`);
36 − if (f.project) conds.push(sql`(e.project_id in (select id from projects where slug = ${f.project} or id = ${f.project}) or (e.entity_type = 'project' and e.entity_id in (select id from projects where slug = ${f.project} or id = ${f.project})))`);
37 − if (f.entityType) conds.push(sql`e.entity_type = ${f.entityType}`);
38 − if (f.entityId) conds.push(sql`e.entity_id = ${f.entityId}`);
39 − if (f.minSignificance != null) conds.push(sql`e.significance >= ${f.minSignificance}`);
40 − if (f.since) conds.push(sql`e.detected_at >= ${f.since}::timestamptz`);
41 − if (f.reviewStatus) conds.push(sql`e.review_status = ${f.reviewStatus}`);
42 − else conds.push(sql`e.review_status <> 'rejected'`);
43 − const rows = await sql<Row[]>`
44 − select ${eventCols(sql)}, count(*) over() as total
45 − from events e ${eventJoins(sql)}
46 − where ${andAll(sql, conds)}
47 − order by e.detected_at desc, e.id desc
48 − limit ${pg_.perPage} offset ${pg_.offset}`;
112 + const dedupe = f.dedupe !== false;
113 + const where = andAll(sql, conds(f));
114 + const rows = dedupe
115 + ? await sql<Row[]>`
116 + with ranked as (
117 + select ${eventCols(sql)}, row_number() over (partition by coalesce(e.cluster_id, e.id) order by ${sourceRankExpr(sql)} asc, e.significance desc, e.detected_at asc, e.id asc) as rn
118 + from events e ${eventJoins(sql)}
119 + where ${where}
120 + )
121 + select *, count(*) over() as total from ranked where rn = 1
122 + order by detected_at desc, id desc
123 + limit ${pg_.perPage} offset ${pg_.offset}`
124 + : await sql<Row[]>`
125 + select ${eventCols(sql)}, count(*) over() as total
126 + from events e ${eventJoins(sql)}
127 + where ${where}
128 + order by e.detected_at desc, e.id desc
129 + limit ${pg_.perPage} offset ${pg_.offset}`;
49 130 const total = rows.length ? int(rows[0]!.total) : 0;
50 − return { items: await toEventDtos(rows), total, page: pg_.page, perPage: pg_.perPage };
131 + const items = await attachOtherSources(await toEventDtos(rows));
132 + return { items, total, page: pg_.page, perPage: pg_.perPage };
51 133 }
52 134
53 135 export async function getEvent(id: string): Promise<EventDTO | null> {
54 136 const sql = pg();
55 137 const rows = await sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.id = ${id} limit 1`;
56 138 if (!rows.length) return null;
57 − return (await toEventDtos(rows))[0] ?? null;
139 + const dtos = await attachOtherSources(await toEventDtos(rows));
140 + return dtos[0] ?? null;
58 141 }
59 142
60 −/** Events attached to an entity (entity_type + entity_id), newest first. */
61 −export async function eventsForEntity(entityType: string, entityId: string, limit = 50): Promise<EventDTO[]> {
143 +/** Arbitrary condition over `events e` (joined with sources s, operators o, metros em, projects ep), newest first. */
144 +export async function eventsWhere(cond: Fragment, limit = 20): Promise<EventDTO[]> {
62 145 const sql = pg();
63 146 const rows = await sql<Row[]>`
64 147 select ${eventCols(sql)} from events e ${eventJoins(sql)}
65 − where e.entity_type = ${entityType} and e.entity_id = ${entityId} and e.review_status <> 'rejected'
148 + where (${cond}) and e.review_status <> 'rejected'
66 149 order by e.detected_at desc limit ${limit}`;
67 150 return toEventDtos(rows);
68 151 }
69 152
153 +/** Events attached to an entity (entity_type + entity_id), newest first. */
154 +export async function eventsForEntity(entityType: string, entityId: string, limit = 50): Promise<EventDTO[]> {
155 + const sql = pg();
156 + return eventsWhere(sql`e.entity_type = ${entityType} and e.entity_id = ${entityId}`, limit);
157 +}
158 +
70 159 /** Recent events for an operator (by operator_id or entity = operator). */
71 160 export async function eventsForOperator(operatorId: string, limit = 20): Promise<EventDTO[]> {
72 161 const sql = pg();
73 − const rows = await sql<Row[]>`
74 − select ${eventCols(sql)} from events e ${eventJoins(sql)}
75 − where (e.operator_id = ${operatorId} or (e.entity_type = 'operator' and e.entity_id = ${operatorId})) and e.review_status <> 'rejected'
76 − order by e.detected_at desc limit ${limit}`;
77 − return toEventDtos(rows);
162 + return eventsWhere(sql`e.operator_id = ${operatorId} or (e.entity_type = 'operator' and e.entity_id = ${operatorId})`, limit);
78 163 }
79 164
80 165 export async function eventsForCountry(iso2: string, limit = 20): Promise<EventDTO[]> {
81 166 const sql = pg();
82 − const rows = await sql<Row[]>`
83 − select ${eventCols(sql)} from events e ${eventJoins(sql)}
84 − where e.country_iso2 = ${iso2} and e.review_status <> 'rejected'
85 − order by e.detected_at desc limit ${limit}`;
86 − return toEventDtos(rows);
167 + return eventsWhere(sql`e.country_iso2 = ${iso2}`, limit);
87 168 }
88 169
89 170 export async function eventsForMetro(metroId: string, limit = 20): Promise<EventDTO[]> {
90 171 const sql = pg();
91 − const rows = await sql<Row[]>`
92 − select ${eventCols(sql)} from events e ${eventJoins(sql)}
93 − where (e.metro_id = ${metroId} or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = ${metroId}))) and e.review_status <> 'rejected'
94 − order by e.detected_at desc limit ${limit}`;
95 − return toEventDtos(rows);
172 + return eventsWhere(sql`e.metro_id = ${metroId} or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = ${metroId}))`, limit);
96 173 }
97 174
98 175 export async function eventsForProject(projectId: string, limit = 50): Promise<EventDTO[]> {
99 176 const sql = pg();
100 − const rows = await sql<Row[]>`
101 − select ${eventCols(sql)} from events e ${eventJoins(sql)}
102 − where (e.project_id = ${projectId} or (e.entity_type = 'project' and e.entity_id = ${projectId})) and e.review_status <> 'rejected'
103 − order by e.detected_at desc limit ${limit}`;
104 − return toEventDtos(rows);
177 + return eventsWhere(sql`e.project_id = ${projectId} or (e.entity_type = 'project' and e.entity_id = ${projectId})`, limit);
105 178 }
106 179
107 180 export async function latestEvents(limit = 20, minSignificance = 0): Promise<EventDTO[]> {
108 181 const sql = pg();
109 − const rows = await sql<Row[]>`
110 − select ${eventCols(sql)} from events e ${eventJoins(sql)}
111 − where e.review_status <> 'rejected' and e.significance >= ${minSignificance}
112 − order by e.detected_at desc limit ${limit}`;
113 − return toEventDtos(rows);
182 + return eventsWhere(sql`e.significance >= ${minSignificance}`, limit);
114 183 }
added apps/api/src/repositories/explore.ts +198 −0
@@ -0,0 +1,198 @@
1 +/**
2 + * /explore — structured query over facilities or live projects with facets, charts (containment-aware MW) and
3 + * map points. Facets and charts are computed on the FULL filtered set; `items` is one page.
4 + */
5 +import type { ExploreQuery, ExploreResponse, MapPoint } from "@dci/core";
6 +import { pg, andAll, facilityJoins, facilitySummaryCols, facilityView, knownMwAgg, pipelineMwAgg, countedAgg, hasMwAgg, mwExpr, projectJoins, projectSummaryCols, projectLive, yearExpr, likePattern, page, PIPELINE_SET, OPERATIONAL_SET, type Fragment, type Sql } from "../lib/sql.js";
7 +import { int, num, reqStr, str, type Row } from "../lib/rows.js";
8 +import { asPrecision, asStatus, asType, facilitySummary, projectSummary, round2, share } from "../lib/dto.js";
9 +import { csv } from "../lib/http.js";
10 +import { facilityConds, toFacilityQuery } from "./facilities.js";
11 +import { MAX_POINTS } from "./map.js";
12 +
13 +export const EXPLORE_METHODOLOGY = "Facets and charts cover the whole filtered set (items are one page). Facility MW in charts is containment-aware (a campus and its buildings are never both summed) and uses published figures only: IT capacity, else total power, else planned power. Project MW is the planned MW published on live project records (false positives hidden by review and merged duplicates excluded). mwCoverage is the share of rows with any MW figure. Map points are capped at 5 000 (degraded = true when the cap was hit).";
14 +
15 +const AI_LEVELS_FOR: Record<string, string[]> = { confirmed: ["confirmed"], likely: ["confirmed", "likely"], associated: ["confirmed", "likely", "associated"] };
16 +
17 +function aiCond(sql: Sql, alias: Fragment, ai: ExploreQuery["ai"], legacyFlag: Fragment): Fragment | null {
18 + if (!ai) return null;
19 + if (ai === "any") return sql`(${alias}.ai_evidence <> 'unknown' or ${legacyFlag})`;
20 + const levels = AI_LEVELS_FOR[ai] ?? ["confirmed", "likely"];
21 + return ai === "confirmed" ? sql`${alias}.ai_evidence = any(${levels})` : sql`(${alias}.ai_evidence = any(${levels}) or ${legacyFlag})`;
22 +}
23 +
24 +function facilityExploreConds(sql: Sql, q: ExploreQuery): Fragment[] {
25 + const conds = facilityConds(sql, toFacilityQuery({ q: q.q, country: q.country, metro: q.metro, operator: q.operator, status: q.status, type: q.type, min_mw: q.min_mw, max_mw: q.max_mw, hyperscale: q.hyperscale, has_mw: q.has_mw, confidence: q.confidence, opened_from: q.opened_from, opened_to: q.opened_to }));
26 + const ai = aiCond(sql, sql`f`, q.ai, sql`f.is_ai`);
27 + if (ai) conds.push(ai);
28 + const prec = csv(q.location_precision);
29 + if (prec.length) conds.push(sql`f.geo_precision = any(${prec})`);
30 + const expected = yearExpr(sql, sql`coalesce(f.opened_on, f.construction_started_on, f.announced_on)`);
31 + if (q.expected_before != null) conds.push(sql`f.status = any(${PIPELINE_SET}) and ${expected} <= ${q.expected_before}`);
32 + if (q.expected_after != null) conds.push(sql`f.status = any(${PIPELINE_SET}) and ${expected} >= ${q.expected_after}`);
33 + if (q.announced_since) conds.push(sql`f.announced_on >= ${q.announced_since}`);
34 + return conds;
35 +}
36 +
37 +function projectExploreConds(sql: Sql, q: ExploreQuery): Fragment[] {
38 + const conds: Fragment[] = [projectLive(sql)];
39 + const statuses = csv(q.project_status ?? q.status);
40 + if (statuses.length) conds.push(sql`p.status = any(${statuses})`);
41 + const countries = csv(q.country).map((c) => c.toUpperCase());
42 + if (countries.length) conds.push(sql`p.country_iso2 = any(${countries})`);
43 + if (q.metro) conds.push(sql`(m.slug = ${q.metro} or m.id = ${q.metro})`);
44 + if (q.operator) conds.push(sql`(o.slug = ${q.operator} or o.id = ${q.operator})`);
45 + if (q.min_mw != null) conds.push(sql`p.planned_mw >= ${q.min_mw}`);
46 + if (q.max_mw != null) conds.push(sql`p.planned_mw <= ${q.max_mw}`);
47 + const ai = aiCond(sql, sql`p`, q.ai, sql`p.is_ai`);
48 + if (ai) conds.push(ai);
49 + if (q.hyperscale === true) conds.push(sql`exists (select 1 from operators ho where ho.id = p.operator_id and ho.kind = 'hyperscaler')`);
50 + const expected = yearExpr(sql, sql`p.expected_opening`);
51 + if (q.expected_before != null) conds.push(sql`${expected} <= ${q.expected_before}`);
52 + if (q.expected_after != null) conds.push(sql`${expected} >= ${q.expected_after}`);
53 + if (q.announced_since) conds.push(sql`p.announced_on >= ${q.announced_since}`);
54 + const conf = csv(q.confidence);
55 + if (conf.length) conds.push(sql`p.confidence = any(${conf})`);
56 + const cls = csv(q.project_class);
57 + if (cls.length) conds.push(sql`p.project_class = any(${cls})`);
58 + const prec = csv(q.location_precision);
59 + if (prec.length) conds.push(sql`p.geo_precision = any(${prec})`);
60 + if (q.has_mw === true) conds.push(sql`p.planned_mw is not null`);
61 + if (q.has_mw === false) conds.push(sql`p.planned_mw is null`);
62 + if (q.q) conds.push(sql`(p.name ilike ${likePattern(q.q)} or o.name ilike ${likePattern(q.q)} or p.city ilike ${likePattern(q.q)})`);
63 + return conds;
64 +}
65 +
66 +function facilityOrder(sql: Sql, sort: string | undefined, order: "asc" | "desc" | undefined): Fragment {
67 + const asc = (order ?? (sort === "name" ? "asc" : "desc")) === "asc";
68 + const dir = asc ? sql`asc` : sql`desc`;
69 + switch (sort) {
70 + case "name": return sql`f.name ${dir}, f.id`;
71 + case "mw": return sql`${mwExpr(sql)} ${dir} nulls last, f.name`;
72 + case "opened": case "opening": return sql`f.opened_on ${dir} nulls last, f.name`;
73 + case "announced": return sql`f.announced_on ${dir} nulls last, f.name`;
74 + case "completeness": return sql`f.completeness ${dir}, f.name`;
75 + default: return sql`f.updated_at ${dir}, f.id`;
76 + }
77 +}
78 +
79 +function projectOrder(sql: Sql, sort: string | undefined, order: "asc" | "desc" | undefined): Fragment {
80 + const asc = (order ?? (sort === "name" ? "asc" : "desc")) === "asc";
81 + const dir = asc ? sql`asc` : sql`desc`;
82 + switch (sort) {
83 + case "name": return sql`p.name ${dir}, p.id`;
84 + case "mw": return sql`p.planned_mw ${dir} nulls last, p.name`;
85 + case "announced": return sql`p.announced_on ${dir} nulls last, p.name`;
86 + case "opening": case "opened": return sql`p.expected_opening ${dir} nulls last, p.name`;
87 + default: return sql`p.last_update ${dir}, p.id`;
88 + }
89 +}
90 +
91 +export async function explore(q: ExploreQuery): Promise<ExploreResponse> {
92 + const sql = pg();
93 + const pg_ = page(q.page, q.per_page, 100, 24);
94 + const entity = q.entity ?? "facilities";
95 + if (entity === "projects") {
96 + const where = andAll(sql, projectExploreConds(sql, q));
97 + const [items, facets, charts, pts, cov] = await Promise.all([
98 + sql<Row[]>`select ${projectSummaryCols(sql)}, count(*) over() as total from projects p ${projectJoins(sql)} where ${where} order by ${projectOrder(sql, q.sort, q.order)} limit ${pg_.perPage} offset ${pg_.offset}`,
99 + Promise.all([
100 + sql<Row[]>`select p.status as key, count(*)::int as n from projects p ${projectJoins(sql)} where ${where} group by 1 order by n desc`,
101 + sql<Row[]>`select p.country_iso2 as key, c.name, count(*)::int as n from projects p ${projectJoins(sql)} left join countries c on c.iso2 = p.country_iso2 where ${where} and p.country_iso2 is not null group by 1, 2 order by n desc limit 30`,
102 + sql<Row[]>`select o.slug as key, o.name, count(*)::int as n from projects p ${projectJoins(sql)} where ${where} and o.id is not null group by 1, 2 order by n desc limit 15`,
103 + sql<Row[]>`select p.ai_evidence as key, count(*)::int as n from projects p ${projectJoins(sql)} where ${where} group by 1 order by n desc`,
104 + ]),
105 + Promise.all([
106 + sql<Row[]>`select p.status as key, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p ${projectJoins(sql)} where ${where} group by 1 order by n desc`,
107 + sql<Row[]>`select p.country_iso2 as key, c.name, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p ${projectJoins(sql)} left join countries c on c.iso2 = p.country_iso2 where ${where} and p.country_iso2 is not null group by 1, 2 order by n desc limit 15`,
108 + sql<Row[]>`select ${yearExpr(sql, sql`p.expected_opening`)} as year, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p ${projectJoins(sql)} where ${where} and p.expected_opening ~ '^\\d{4}' group by 1 order by 1`,
109 + ]),
110 + sql<Row[]>`select p.id, p.slug, p.name, o.name as op_name, p.lat, p.lng, p.status, p.planned_mw, p.geo_precision, p.is_ai, p.ai_evidence, p.country_iso2, ${yearExpr(sql, sql`p.expected_opening`)} as y from projects p ${projectJoins(sql)} where ${where} and p.lat is not null and p.lng is not null order by p.planned_mw desc nulls last limit ${MAX_POINTS + 1}`,
111 + sql<Row[]>`select count(*)::int as n, count(*) filter (where p.planned_mw is not null)::int as with_mw from projects p ${projectJoins(sql)} where ${where}`,
112 + ]);
113 + const [fStatus, fCountry, fOperator, fAi] = facets;
114 + const [cStatus, cCountry, cYear] = charts;
115 + const degraded = pts.length > MAX_POINTS;
116 + const points: MapPoint[] = (degraded ? [] : pts).map((r) => {
117 + const p: MapPoint = { id: reqStr(r.id), slug: reqStr(r.slug), n: reqStr(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: "unknown", mw: num(r.planned_mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2), k: "project" };
118 + const ai = str(r.ai_evidence);
119 + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1;
120 + const y = num(r.y);
121 + if (y != null) p.y = y;
122 + return p;
123 + });
124 + return {
125 + query: q,
126 + total: items.length ? int(items[0]!.total) : int(cov[0]?.n),
127 + items: items.map(projectSummary),
128 + facets: {
129 + status: fStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n) })),
130 + country: fCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n) })),
131 + operator: fOperator.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name), count: int(r.n) })),
132 + ai: fAi.map((r) => ({ key: reqStr(r.key, "unknown"), count: int(r.n) })),
133 + },
134 + charts: {
135 + byStatus: cStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n), mw: round2(num(r.mw)) })),
136 + byCountry: cCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n), mw: round2(num(r.mw)) })),
137 + byYear: cYear.map((r) => ({ year: int(r.year), count: int(r.n), mw: round2(num(r.mw)) })).filter((x) => x.year > 0),
138 + },
139 + map: { points, total: degraded ? int(cov[0]?.n) : points.length, degraded },
140 + mwCoverage: share(int(cov[0]?.with_mw), int(cov[0]?.n)),
141 + };
142 + }
143 +
144 + const conds = facilityExploreConds(sql, q);
145 + const where = andAll(sql, conds);
146 + const known = knownMwAgg(sql), pipe = pipelineMwAgg(sql), counted = countedAgg(sql), hasMw = hasMwAgg(sql);
147 + const chartMw = sql`sum(case when f.status = any(${OPERATIONAL_SET}) then ${known} else ${pipe} end)::float`;
148 + const [items, facets, charts, pts, cov] = await Promise.all([
149 + sql<Row[]>`select ${facilitySummaryCols(sql)}, count(*) over() as total from facilities f ${facilityJoins(sql)} where ${where} order by ${facilityOrder(sql, q.sort, q.order)} limit ${pg_.perPage} offset ${pg_.offset}`,
150 + Promise.all([
151 + sql<Row[]>`select f.status as key, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`,
152 + sql<Row[]>`select f.country_iso2 as key, c.name, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} and f.country_iso2 is not null group by 1, 2 order by n desc limit 30`,
153 + sql<Row[]>`select o.slug as key, o.name, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} and o.id is not null group by 1, 2 order by n desc limit 15`,
154 + sql<Row[]>`select f.facility_type as key, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`,
155 + sql<Row[]>`select f.ai_evidence as key, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`,
156 + ]),
157 + Promise.all([
158 + sql<Row[]>`select f.status as key, count(*) filter (where ${counted})::int as n, ${chartMw} as mw from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`,
159 + sql<Row[]>`select f.country_iso2 as key, c.name, count(*) filter (where ${counted})::int as n, ${chartMw} as mw from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where} and f.country_iso2 is not null group by 1, 2 order by n desc limit 15`,
160 + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${counted})::int as n, sum(${known})::float as mw from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
161 + ]),
162 + sql<Row[]>`select f.id, f.slug, f.name, o.name as op_name, f.lat, f.lng, f.status, f.facility_type, ${mwExpr(sql)} as mw, f.geo_precision, f.is_ai, f.ai_evidence, f.is_hyperscale, f.country_iso2, f.record_scope, ${yearExpr(sql, sql`f.opened_on`)} as y from facilities f ${facilityJoins(sql)} where ${where} and f.lat is not null and f.lng is not null order by ${mwExpr(sql)} desc nulls last, f.id limit ${MAX_POINTS + 1}`,
163 + sql<Row[]>`select count(*) filter (where ${counted})::int as n, count(*) filter (where ${counted} and ${hasMw})::int as with_mw, count(*)::int as rows from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where}`,
164 + ]);
165 + const [fStatus, fCountry, fOperator, fType, fAi] = facets;
166 + const [cStatus, cCountry, cYear] = charts;
167 + const degraded = pts.length > MAX_POINTS;
168 + const points: MapPoint[] = (degraded ? [] : pts).map((r) => {
169 + const p: MapPoint = { id: reqStr(r.id), slug: reqStr(r.slug), n: reqStr(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: asType(r.facility_type), mw: num(r.mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2) };
170 + const ai = str(r.ai_evidence);
171 + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1;
172 + if (r.is_hyperscale === true) p.hs = 1;
173 + const y = num(r.y);
174 + if (y != null) p.y = y;
175 + const rs = str(r.record_scope);
176 + if (rs === "building" || rs === "campus") p.rs = rs;
177 + return p;
178 + });
179 + return {
180 + query: q,
181 + total: items.length ? int(items[0]!.total) : int(cov[0]?.rows),
182 + items: items.map(facilitySummary),
183 + facets: {
184 + status: fStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n) })),
185 + country: fCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n) })),
186 + operator: fOperator.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name), count: int(r.n) })),
187 + type: fType.map((r) => ({ key: reqStr(r.key), count: int(r.n) })),
188 + ai: fAi.map((r) => ({ key: reqStr(r.key, "unknown"), count: int(r.n) })),
189 + },
190 + charts: {
191 + byStatus: cStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n), mw: round2(num(r.mw)) })),
192 + byCountry: cCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n), mw: round2(num(r.mw)) })),
193 + byYear: cYear.map((r) => ({ year: int(r.year), count: int(r.n), mw: round2(num(r.mw)) })).filter((x) => x.year > 0),
194 + },
195 + map: { points, total: degraded ? int(cov[0]?.rows) : points.length, degraded },
196 + mwCoverage: share(int(cov[0]?.with_mw), int(cov[0]?.n)),
197 + };
198 +}
modified apps/api/src/repositories/facilities.ts +122 −19
@@ -1,10 +1,12 @@
1 −import type { CloudRegionSummary, FacilityDetail, FacilityFilters, FacilitySummary, ProvenanceDTO, SourceRef } from "@dci/core";
2 −import { pg, andAll, facilityJoins, facilitySummaryCols, mwExpr, page, bboxAround, haversineExpr, yearExpr, likePattern, PIPELINE_SET, cloudRegionCols, type Fragment, type Sql } from "../lib/sql.js";
3 −import { int, num, record, reqIso, str, strArray, type Row } from "../lib/rows.js";
4 −import { cloudRegionSummary, facilitySummary, provenanceDto } from "../lib/dto.js";
1 +import type { ClaimDTO, CloudRegionSummary, EntityHistory, EventDTO, FacilityDetail, FacilityFilters, FacilitySummary, GridConstraintDTO, ProvenanceDTO, SourceRef } from "@dci/core";
2 +import { pg, andAll, facilityJoins, facilitySummaryCols, mwExpr, page, bboxAround, haversineExpr, yearExpr, likePattern, PIPELINE_SET, cloudRegionCols, gridConstraintCols, gridConstraintJoins, POWER_EVENT_TYPES, GRID_EVENT_TYPES, AI_LEVELS, type Fragment, type Sql } from "../lib/sql.js";
3 +import { int, num, record, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js";
4 +import { cloudRegionSummary, facilitySummary, gridConstraintDto, gridConstraintFromEvent, provenanceDto } from "../lib/dto.js";
5 5 import { csv } from "../lib/http.js";
6 6 import { findBySlugOrId } from "../lib/resolve.js";
7 −import { eventsForEntity } from "./events.js";
7 +import { nearby } from "../lib/nearby.js";
8 +import { capacityHistory, claimsFor, dataQualityFor, entityHistory, eventsForSubject, provenanceAll } from "../lib/quality.js";
9 +import { eventsForEntity, eventsWhere } from "./events.js";
8 10 import { projectsForFacility } from "./projects.js";
9 11 import { buildSourceHistory, documentVersionsFor, sourceIdsOf, sourceRefsFor } from "../lib/source-history.js";
10 12
@@ -21,6 +23,12 @@ export interface FacilityQuery extends Omit<FacilityFilters, "country" | "status
21 23 metroId?: string;
22 24 countryIso2?: string;
23 25 operatorId?: string;
26 + /** extra: AI evidence levels (confirmed, likely, associated, unknown) */
27 + aiEvidence?: string[];
28 + /** extra: geo precision values */
29 + locationPrecision?: string[];
30 + /** extra: record scope (building | facility | campus) */
31 + recordScope?: string[];
24 32 }
25 33
26 34 /** Translate raw FacilityFilters (csv strings) into a FacilityQuery. */
@@ -63,8 +71,11 @@ export function facilityConds(sql: Sql, q: FacilityQuery): Fragment[] {
63 71 if (q.hyperscale === true) conds.push(sql`(f.is_hyperscale or f.facility_type = 'hyperscale')`);
64 72 if (q.hyperscale === false) conds.push(sql`(not f.is_hyperscale and f.facility_type <> 'hyperscale')`);
65 73 if (q.colocation === true) conds.push(sql`f.facility_type in ('colocation', 'carrier_hotel', 'wholesale')`);
66 − if (q.ai === true) conds.push(sql`(f.is_ai or f.facility_type = 'ai')`);
67 − if (q.ai === false) conds.push(sql`(not f.is_ai and f.facility_type <> 'ai')`);
74 + if (q.ai === true) conds.push(sql`(f.is_ai or f.facility_type = 'ai' or f.ai_evidence = any(${AI_LEVELS}))`);
75 + if (q.ai === false) conds.push(sql`(not f.is_ai and f.facility_type <> 'ai' and f.ai_evidence <> all(${AI_LEVELS}))`);
76 + if (q.aiEvidence?.length) conds.push(sql`f.ai_evidence = any(${q.aiEvidence})`);
77 + if (q.locationPrecision?.length) conds.push(sql`f.geo_precision = any(${q.locationPrecision})`);
78 + if (q.recordScope?.length) conds.push(sql`f.record_scope = any(${q.recordScope})`);
68 79 if (q.renewable === true) conds.push(sql`f.renewable_claim is not null`);
69 80 if (q.has_mw === true) conds.push(sql`${mw} is not null`);
70 81 if (q.has_mw === false) conds.push(sql`${mw} is null`);
@@ -113,6 +124,13 @@ export async function facilitiesByIds(ids: string[]): Promise<FacilitySummary[]>
113 124 return ids.map((id) => by.get(id)).filter((x): x is FacilitySummary => Boolean(x));
114 125 }
115 126
127 +/** Facilities matching an arbitrary condition over `facilities f` (+ facilityJoins aliases), ordered by MW. */
128 +export async function facilitiesWhere(cond: Fragment, limit = 20, orderBy?: Fragment): Promise<FacilitySummary[]> {
129 + const sql = pg();
130 + const rows = await sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and (${cond}) order by ${orderBy ?? sql`${mwExpr(sql)} desc nulls last, f.name`} limit ${limit}`;
131 + return rows.map(facilitySummary);
132 +}
133 +
116 134 export async function recentlyVerifiedFacilities(limit = 10): Promise<FacilitySummary[]> {
117 135 const sql = pg();
118 136 const rows = await sql<Row[]>`
@@ -144,42 +162,99 @@ async function cloudRegionsInMetro(metroId: string | null, limit = 12): Promise<
144 162 export async function provenanceFor(entityType: string, entityId: string, limit = 400): Promise<ProvenanceDTO[]> {
145 163 const sql = pg();
146 164 const rows = await sql<Row[]>`
147 − select p.field, p.value, p.source_id, s.name as source_name, s.kind as source_kind, p.url, p.first_observed, p.last_observed, p.retrieved_at, p.confidence, p.is_estimate, p.method
165 + select p.field, p.value, p.source_id, s.name as source_name, s.kind as source_kind, p.url, p.first_observed, p.last_observed, p.retrieved_at, p.confidence, p.is_estimate, p.method, p.is_winner, p.scope, p.run_id, p.document_id
148 166 from provenance p left join sources s on s.id = p.source_id
149 167 where p.entity_type = ${entityType} and p.entity_id = ${entityId} and p.is_current
150 − order by p.field, p.last_observed desc limit ${limit}`;
168 + order by p.field, p.is_winner desc, p.last_observed desc limit ${limit}`;
151 169 return rows.map(provenanceDto);
152 170 }
153 171
154 −export async function getFacilityDetail(idOrSlug: string): Promise<{ detail: FacilityDetail; sources: SourceRef[] } | null> {
172 +/** Grid constraints (grid_constraints rows + grid / utility / power events) for a metro and/or country. */
173 +export async function gridConstraintsFor(scope: { metroId?: string | null; countryIso2?: string | null }, limit = 20): Promise<GridConstraintDTO[]> {
174 + const sql = pg();
175 + const conds: Fragment[] = [];
176 + if (scope.metroId) conds.push(sql`g.metro_id = ${scope.metroId}`);
177 + if (scope.countryIso2 && !scope.metroId) conds.push(sql`g.country_iso2 = ${scope.countryIso2}`);
178 + if (!conds.length) return [];
179 + const [rows, events] = await Promise.all([
180 + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} where ${conds.reduce<Fragment>((a, c) => sql`${a} or ${c}`, sql`false`)} order by g.effective_date desc nulls last, g.created_at desc limit ${limit}`,
181 + scope.metroId ? eventsWhere(sql`e.metro_id = ${scope.metroId} and e.event_type = any(${GRID_EVENT_TYPES})`, limit) : Promise.resolve([] as EventDTO[]),
182 + ]);
183 + const out = rows.map(gridConstraintDto);
184 + const seen = new Set(out.map((g) => g.eventId).filter(Boolean));
185 + for (const e of events) if (!seen.has(e.id)) out.push(gridConstraintFromEvent(e));
186 + return out.slice(0, limit);
187 +}
188 +
189 +export const POWER_CONTEXT_NOTE = "utilityCapacityMw / gridConnectionMw are utility-side figures (power available or contracted from the grid), not IT load — never compare them with itCapacityMw. countryEnergy shows national grid averages (renewable share, generation), which do not describe this facility's contracted electricity or power purchase agreements. Grid constraints are public reports attached to the facility's market or country, not to the site itself.";
190 +
191 +/** Resolve a facility by slug or id, following merged_into to the survivor. */
192 +export async function resolveFacilityRow(idOrSlug: string): Promise<Row | null> {
155 193 const sql = pg();
156 194 const base = await findBySlugOrId("facilities", idOrSlug);
157 195 if (!base) return null;
158 − // follow merges: a merged facility redirects to its survivor
159 196 const mergedInto = str(base.merged_into);
160 − const row = mergedInto ? ((await sql<Row[]>`select * from facilities where id = ${mergedInto} limit 1`)[0] ?? base) : base;
197 + return mergedInto ? ((await sql<Row[]>`select * from facilities where id = ${mergedInto} limit 1`)[0] ?? base) : base;
198 +}
199 +
200 +export async function getFacilityDetail(idOrSlug: string, opts: { radiusKm?: number } = {}): Promise<{ detail: FacilityDetail; sources: SourceRef[] } | null> {
201 + const sql = pg();
202 + const row = await resolveFacilityRow(idOrSlug);
203 + if (!row) return null;
161 204 const id = String(row.id);
162 − const [sumRows, aliasRows, tenantRows, ixpRows, cloudRegions, nearby, projects, provenance, events, versions, ownerRows, campusRows] = await Promise.all([
205 + const lat = num(row.lat), lng = num(row.lng);
206 + const metroId = str(row.metro_id);
207 + const countryIso2 = str(row.country_iso2);
208 + const operatorId = str(row.operator_id);
209 + const [sumRows, aliasRows, tenantRows, ixpRows, cloudRegions, nearbyRows, projects, provenance, events, versions, ownerRows, campusRows, buildingRows, roleRows, claims, nearbyInfra, powerEvents, gridConstraints, energyRows, dataQuality] = await Promise.all([
163 210 sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.id = ${id}`,
164 211 sql<Row[]>`select alias from facility_aliases where facility_id = ${id} order by alias`,
165 212 sql<Row[]>`select t.role, t.asn, o.id, o.slug, o.name from facility_tenants t join operators o on o.id = t.operator_id where t.facility_id = ${id} order by o.name`,
166 213 sql<Row[]>`select x.id, x.slug, x.name from facility_ixps fx join ixps x on x.id = fx.ixp_id where fx.facility_id = ${id} order by x.name`,
167 − cloudRegionsInMetro(str(row.metro_id)),
168 − num(row.lat) != null && num(row.lng) != null ? nearbyFacilities(id, num(row.lat)!, num(row.lng)!) : Promise.resolve([] as FacilitySummary[]),
169 − projectsForFacility(id, str(row.operator_id), str(row.metro_id)),
214 + cloudRegionsInMetro(metroId),
215 + lat != null && lng != null ? nearbyFacilities(id, lat, lng) : Promise.resolve([] as FacilitySummary[]),
216 + projectsForFacility(id, operatorId, metroId),
170 217 provenanceFor("facility", id),
171 218 eventsForEntity("facility", id, 50),
172 219 documentVersionsFor("facility", id),
173 220 row.owner_id ? sql<Row[]>`select id, slug, name from operators where id = ${String(row.owner_id)}` : Promise.resolve([] as Row[]),
174 221 row.campus_id ? sql<Row[]>`select id, slug, name from campuses where id = ${String(row.campus_id)}` : Promise.resolve([] as Row[]),
222 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.parent_facility_id = ${id} and f.merged_into is null order by f.name`,
223 + sql<Row[]>`select o.id, o.slug, o.name, (o.id = ${str(row.developer_id)}) as is_developer, (o.id = ${str(row.landowner_id)}) as is_landowner from operators o where o.id in (${str(row.developer_id) ?? ""}, ${str(row.landowner_id) ?? ""})`,
224 + claimsFor("facility", id),
225 + lat != null && lng != null ? nearby({ lat, lng, radiusKm: opts.radiusKm ?? 25, excludeFacilityId: id, limitPerType: 50 }) : Promise.resolve(null),
226 + eventsWhere(sql`e.event_type = any(${POWER_EVENT_TYPES}) and ((e.entity_type = 'facility' and e.entity_id = ${id}) ${operatorId && metroId ? sql`or (e.operator_id = ${operatorId} and e.metro_id = ${metroId})` : sql``})`, 20),
227 + gridConstraintsFor({ metroId, countryIso2 }),
228 + countryIso2 ? sql<Row[]>`select renewable_share, electricity_twh, stats_year from countries where iso2 = ${countryIso2}` : Promise.resolve([] as Row[]),
229 + dataQualityFor("facility", id, int(row.completeness), str(row.last_verified)),
175 230 ]);
176 231 const summary = facilitySummary(sumRows[0] ?? row);
177 232 const carriers = tenantRows.filter((t) => t.role !== "cloud").map((t) => ({ id: String(t.id), slug: String(t.slug), name: String(t.name), asn: num(t.asn) }));
178 233 const cloudProviders = tenantRows.filter((t) => t.role === "cloud").map((t) => ({ id: String(t.id), slug: String(t.slug), name: String(t.name) }));
179 234 const owner = ownerRows[0] ? { id: String(ownerRows[0].id), slug: String(ownerRows[0].slug), name: String(ownerRows[0].name) } : null;
180 235 const campus = campusRows[0] ? { id: String(campusRows[0].id), slug: String(campusRows[0].slug), name: String(campusRows[0].name) } : null;
236 + const refOf = (r: Row | undefined) => (r ? { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) } : null);
237 + const developer = refOf(roleRows.find((r) => r.is_developer === true));
238 + const landowner = refOf(roleRows.find((r) => r.is_landowner === true));
239 + const energy = energyRows[0];
181 240 const detail: FacilityDetail = {
182 241 ...summary,
242 + buildings: buildingRows.map(facilitySummary),
243 + developer,
244 + landowner,
245 + tenants: tenantRows.map((t) => ({ id: String(t.id), slug: String(t.slug), name: String(t.name), role: reqStr(t.role, "tenant") })),
246 + claims,
247 + capacityHistory: capacityHistory(provenance, claims, events),
248 + nearbyInfrastructure: nearbyInfra,
249 + powerContext: {
250 + utilityCapacityMw: num(row.utility_capacity_mw),
251 + gridConnectionMw: num(row.grid_connection_mw),
252 + powerEvents,
253 + gridConstraints,
254 + countryEnergy: energy ? { renewableShare: num(energy.renewable_share), electricityTwh: num(energy.electricity_twh), statsYear: num(energy.stats_year) } : null,
255 + note: POWER_CONTEXT_NOTE,
256 + },
257 + dataQuality,
183 258 aliases: aliasRows.map((a) => String(a.alias)),
184 259 owner,
185 260 campus,
@@ -202,7 +277,7 @@ export async function getFacilityDetail(idOrSlug: string): Promise<{ detail: Fac
202 277 cloudProviders,
203 278 ixps: ixpRows.map((x) => ({ id: String(x.id), slug: String(x.slug), name: String(x.name) })),
204 279 cloudRegions,
205 − nearby,
280 + nearby: nearbyRows,
206 281 projects,
207 282 provenance,
208 283 events,
@@ -211,10 +286,39 @@ export async function getFacilityDetail(idOrSlug: string): Promise<{ detail: Fac
211 286 firstSeen: reqIso(row.first_seen),
212 287 updatedAt: reqIso(row.updated_at),
213 288 };
214 − const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions));
289 + const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions, claims));
215 290 return { detail, sources };
216 291 }
217 292
293 +/** /datacenters/:id/history */
294 +export async function getFacilityHistory(idOrSlug: string): Promise<{ history: EntityHistory; sources: SourceRef[] } | null> {
295 + const row = await resolveFacilityRow(idOrSlug);
296 + if (!row) return null;
297 + const id = String(row.id);
298 + const [prov, claims, events] = await Promise.all([provenanceAll("facility", id), claimsFor("facility", id), eventsForSubject("facility", id)]);
299 + return { history: entityHistory("facility", id, prov, claims, events), sources: await sourceRefsFor(sourceIdsOf(prov, claims, events)) };
300 +}
301 +
302 +/** /datacenters/:id/claims */
303 +export async function getFacilityClaims(idOrSlug: string, opts: { status?: string[]; predicate?: string } = {}): Promise<{ id: string; claims: ClaimDTO[]; sources: SourceRef[] } | null> {
304 + const row = await resolveFacilityRow(idOrSlug);
305 + if (!row) return null;
306 + const id = String(row.id);
307 + const claims = await claimsFor("facility", id, opts);
308 + return { id, claims, sources: await sourceRefsFor(sourceIdsOf(claims)) };
309 +}
310 +
311 +/** /datacenters/:id/provenance — every observation, current and superseded. */
312 +export async function getFacilityProvenance(idOrSlug: string): Promise<{ id: string; provenance: ProvenanceDTO[]; current: number; sources: SourceRef[] } | null> {
313 + const row = await resolveFacilityRow(idOrSlug);
314 + if (!row) return null;
315 + const id = String(row.id);
316 + const provenance = await provenanceAll("facility", id);
317 + const sql = pg();
318 + const cur = await sql<Row[]>`select count(*)::int as n from provenance where entity_type = 'facility' and entity_id = ${id} and is_current`;
319 + return { id, provenance, current: int(cur[0]?.n), sources: await sourceRefsFor(sourceIdsOf(provenance)) };
320 +}
321 +
218 322 /** Distinct facility cities matching a text (for search). */
219 323 export async function searchCities(text: string, limit = 5): Promise<Array<{ city: string; countryIso2: string | null; count: number }>> {
220 324 const sql = pg();
@@ -226,4 +330,3 @@ export async function searchCities(text: string, limit = 5): Promise<Array<{ cit
226 330 order by (lower(f.city) = lower(${text})) desc, n desc limit ${limit}`;
227 331 return rows.map((r) => ({ city: String(r.city), countryIso2: str(r.country_iso2), count: int(r.n) }));
228 332 }
229 −
modified apps/api/src/repositories/ixps.ts +43 −16
@@ -1,51 +1,78 @@
1 −import type { FacilitySummary, IxpSummary } from "@dci/core";
2 −import { pg, facilityJoins, facilitySummaryCols } from "../lib/sql.js";
3 −import { record, reqIso, str, type Row } from "../lib/rows.js";
1 +import type { FacilitySummary, IxpDetail as IxpDetailContract, IxpSummary } from "@dci/core";
2 +import { pg, facilityJoins, facilitySummaryCols, haversineExpr, withinBbox, likePattern } from "../lib/sql.js";
3 +import { num, record, reqIso, reqStr, round, str, type Row } from "../lib/rows.js";
4 4 import { facilitySummary, ixpSummary } from "../lib/dto.js";
5 5 import { findBySlugOrId } from "../lib/resolve.js";
6 +import { eventsForEntity } from "./events.js";
7 +import { provenanceFor } from "./facilities.js";
8 +import { buildSourceHistory, documentVersionsFor } from "../lib/source-history.js";
6 9
7 −const COLS = "x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count";
10 +/** IXP columns joined with the metro (ixps have no coordinates of their own — the metro reference point stands in). */
11 +const COLS = "x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, xm.id as met_id, xm.slug as met_slug, xm.name as met_name, xm.lat as lat, xm.lng as lng";
8 12
9 13 export async function listIxps(f: { country?: string; metroId?: string; q?: string }): Promise<IxpSummary[]> {
10 14 const sql = pg();
11 15 const rows = await sql<Row[]>`
12 16 select ${sql.unsafe(COLS)}, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count
13 − from ixps x
17 + from ixps x left join metros xm on xm.id = x.metro_id
14 18 where ${f.country ? sql`x.country_iso2 = ${f.country.toUpperCase()}` : sql`true`}
15 19 and ${f.metroId ? sql`x.metro_id = ${f.metroId}` : sql`true`}
16 − and ${f.q ? sql`(x.name ilike ${"%" + f.q + "%"} or x.name_long ilike ${"%" + f.q + "%"})` : sql`true`}
20 + and ${f.q ? sql`(x.name ilike ${likePattern(f.q)} or x.name_long ilike ${likePattern(f.q)})` : sql`true`}
17 21 order by x.network_count desc nulls last, x.name limit 2000`;
18 22 return rows.map(ixpSummary);
19 23 }
20 24
21 −export interface IxpDetail extends IxpSummary {
25 +/** Contract IxpDetail + the extra fields the web already reads. */
26 +export interface IxpDetail extends IxpDetailContract {
22 27 countryName: string | null;
23 − metro: { id: string; slug: string; name: string } | null;
24 28 regionContinent: string | null;
25 − facilities: FacilitySummary[];
26 − externalIds: Record<string, string | number>;
27 29 updatedAt: string;
28 30 }
29 31
32 +export const IXP_COORDS_NOTE = "This IXP has no published coordinates in the index; lat/lng and the nearby facilities use its metro reference point. Traffic figures are not indexed (no licensed source).";
33 +
30 34 export async function getIxp(idOrSlug: string): Promise<IxpDetail | null> {
31 35 const sql = pg();
32 36 const row = await findBySlugOrId("ixps", idOrSlug);
33 37 if (!row) return null;
34 38 const id = String(row.id);
35 39 const metroId = str(row.metro_id);
36 − const [main, facs, metroRows, countryRows] = await Promise.all([
37 − sql<Row[]>`select ${sql.unsafe(COLS)}, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count from ixps x where x.id = ${id}`,
40 + const [main, facs, metroRows, countryRows, provenance, events, versions] = await Promise.all([
41 + sql<Row[]>`select ${sql.unsafe(COLS)}, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count from ixps x left join metros xm on xm.id = x.metro_id where x.id = ${id}`,
38 42 sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} join facility_ixps fx on fx.facility_id = f.id where fx.ixp_id = ${id} and f.merged_into is null order by f.name limit 200`,
39 − metroId ? sql<Row[]>`select id, slug, name from metros where id = ${metroId}` : Promise.resolve([] as Row[]),
43 + metroId ? sql<Row[]>`select id, slug, name, lat, lng from metros where id = ${metroId}` : Promise.resolve([] as Row[]),
40 44 row.country_iso2 ? sql<Row[]>`select name from countries where iso2 = ${String(row.country_iso2)}` : Promise.resolve([] as Row[]),
45 + provenanceFor("ixp", id),
46 + eventsForEntity("ixp", id, 50),
47 + documentVersionsFor("ixp", id),
41 48 ]);
49 + const metro = metroRows[0] ?? null;
50 + const lat = metro ? num(metro.lat) : null, lng = metro ? num(metro.lng) : null;
51 + const facilities = facs.map(facilitySummary);
52 + const linkedIds = facilities.map((f) => f.id);
53 + let nearbyFacilities: Array<FacilitySummary & { distanceKm: number }> = [];
54 + if (lat != null && lng != null) {
55 + const dist = haversineExpr(sql, lat, lng);
56 + const rows = await sql<Row[]>`select ${facilitySummaryCols(sql)}, ${dist} as distance_km from facilities f ${facilityJoins(sql)}
57 + where f.merged_into is null and f.lat is not null and ${withinBbox(sql, lat, lng, 25, sql`f.lat`, sql`f.lng`)} and ${dist} <= 25 ${linkedIds.length ? sql`and f.id <> all(${linkedIds})` : sql``}
58 + order by ${dist} asc limit 20`;
59 + nearbyFacilities = rows.map((r) => ({ ...facilitySummary(r), distanceKm: round(num(r.distance_km), 2) ?? 0 }));
60 + }
61 + const opMap = new Map<string, { id: string; slug: string; name: string; facilityCount: number }>();
62 + for (const f of facilities) if (f.operator) { const cur = opMap.get(f.operator.id) ?? { ...f.operator, facilityCount: 0 }; cur.facilityCount++; opMap.set(f.operator.id, cur); }
42 63 return {
43 64 ...ixpSummary(main[0] ?? { ...row, facility_count: facs.length }),
65 + metro: metro ? { id: reqStr(metro.id), slug: reqStr(metro.slug), name: reqStr(metro.name) } : null,
66 + lat,
67 + lng,
68 + facilities,
69 + operators: [...opMap.values()].sort((a, b) => b.facilityCount - a.facilityCount || a.name.localeCompare(b.name)),
70 + nearbyFacilities,
71 + trafficNote: metro ? IXP_COORDS_NOTE : null,
72 + externalIds: record(row.external_ids),
73 + sourceHistory: buildSourceHistory(provenance, events, versions),
44 74 countryName: countryRows[0] ? String(countryRows[0].name) : null,
45 − metro: metroRows[0] ? { id: String(metroRows[0].id), slug: String(metroRows[0].slug), name: String(metroRows[0].name) } : null,
46 75 regionContinent: str(row.region_continent),
47 − facilities: facs.map(facilitySummary),
48 − externalIds: record(row.external_ids),
49 76 updatedAt: reqIso(row.updated_at),
50 77 };
51 78 }
modified apps/api/src/repositories/map.ts +241 −34
@@ -1,5 +1,12 @@
1 −import type { FacilityStatus, MapCluster, MapPoint, MapResponse } from "@dci/core";
2 −import { pg, andAll, facilityJoins, mwExpr, type Fragment, type Sql } from "../lib/sql.js";
1 +/**
2 + * /map data: zoom-tiered clusters / points for facilities, thematic layers (capacity, projects, AI, cloud, IXPs,
3 + * connectivity, power, pipeline), density grids and a "time machine" year filter. Only entities with coordinates are
4 + * drawn; IXPs without a published location fall back to their metro reference point (p = "metro"). Layers without a
5 + * connector (landing stations, substations, power plants) are never drawn — power events are exposed as metro-level
6 + * clusters, not as invented plant locations.
7 + */
8 +import type { DensityCell, DensityView, FacilityStatus, MapCluster, MapLayer, MapPoint, MapResponse } from "@dci/core";
9 +import { pg, andAll, facilityJoins, facilityView, knownMwAgg, pipelineMwAgg, countedAgg, mwExpr, projectLive, yearExpr, PIPELINE_SET, POWER_EVENT_TYPES, GRID_EVENT_TYPES, type Fragment, type Sql } from "../lib/sql.js";
3 10 import { facilityConds, type FacilityQuery } from "./facilities.js";
4 11 import { int, num, str, type Row } from "../lib/rows.js";
5 12 import { asPrecision, asStatus, asType } from "../lib/dto.js";
@@ -12,15 +19,30 @@ export interface MapQuery {
12 19 status?: string[];
13 20 type?: string[];
14 21 operator?: string;
22 + metro?: string;
15 23 country?: string[];
24 + q?: string;
16 25 min_mw?: number;
17 26 max_mw?: number;
18 27 ai?: boolean;
19 28 hyperscale?: boolean;
29 + has_mw?: boolean;
30 + confidence?: string[];
31 + opened_from?: number;
32 + opened_to?: number;
20 33 cloud_regions?: boolean;
34 + layer?: MapLayer;
35 + density?: DensityView;
36 + year?: number;
37 + project_status?: string[];
38 + expected_from?: number;
39 + expected_to?: number;
40 + location_precision?: string[];
41 + ai_evidence?: string[];
21 42 }
22 43
23 44 export const MAX_POINTS = 5000;
45 +export const MAP_METHODOLOGY = "Only entities with coordinates are drawn; `p` is the geo precision (city/metro-level points are approximate). Facility `mw` = best known figure (IT, else total, else planned) — on the power layer it is utility / grid supply capacity, NOT IT load. `k` marks overlay kinds (project, ixp, cloud_region); IXPs without a published location sit at their metro reference point. Density weights are containment-aware (a campus and its buildings are never both summed). `year` keeps facilities whose opening date is ≤ year; `yearCoverage` is the share of matching facilities that have an opening date at all — undated facilities are absent from every year frame.";
24 46
25 47 /** Map mode from zoom: <5 countries, 5–8 grid, ≥9 points. Exported for tests. */
26 48 export function mapMode(zoom: number): "country" | "grid" | "points" {
@@ -39,9 +61,46 @@ function bboxCond(sql: Sql, b: Bbox | undefined, latCol: Fragment, lngCol: Fragm
39 61 return sql`(${latCol} between ${b.s} and ${b.n} and ${lng})`;
40 62 }
41 63
64 +const AI_COND = (sql: Sql) => sql`(f.ai_evidence in ('confirmed', 'likely') or f.is_ai)`;
65 +const PIPELINE_PROJECT_STATUSES = ["rumored", "proposed", "announced", "permitting", "approved", "under_construction", "delayed"];
66 +
67 +/** Layer-specific facility conditions (on top of the generic filters). */
68 +function layerConds(sql: Sql, q: MapQuery): Fragment[] {
69 + const out: Fragment[] = [];
70 + switch (q.layer) {
71 + case "capacity": out.push(sql`${mwExpr(sql)} is not null`); break;
72 + case "ai": out.push(AI_COND(sql)); break;
73 + case "power": out.push(sql`(f.utility_capacity_mw is not null or f.grid_connection_mw is not null)`); break;
74 + case "pipeline": out.push(sql`f.status = any(${PIPELINE_SET})`); break;
75 + case "connectivity": out.push(sql`(f.facility_type = 'carrier_hotel' or coalesce(f.carriers_count, 0) >= 20)`); break;
76 + default: break;
77 + }
78 + if (q.location_precision?.length) out.push(sql`f.geo_precision = any(${q.location_precision})`);
79 + if (q.ai_evidence?.length) out.push(sql`f.ai_evidence = any(${q.ai_evidence})`);
80 + return out;
81 +}
82 +
83 +/** Time-machine condition: opened_on year ≤ year; pipeline rows only when the caller asked for pipeline statuses. */
84 +function yearCond(sql: Sql, q: MapQuery): Fragment {
85 + if (q.year == null) return sql`true`;
86 + const opened = sql`${yearExpr(sql, sql`f.opened_on`)} <= ${q.year}`;
87 + const pipelineAsked = (q.status ?? []).some((s) => (PIPELINE_SET as string[]).includes(s));
88 + if (!pipelineAsked) return opened;
89 + return sql`(${opened} or (f.status = any(${PIPELINE_SET}) and ${yearExpr(sql, sql`coalesce(f.announced_on, f.construction_started_on, f.opened_on)`)} <= ${q.year}))`;
90 +}
91 +
92 +function baseConds(sql: Sql, q: MapQuery): Fragment[] {
93 + const fq: FacilityQuery = { status: q.status, type: q.type, operator: q.operator, metro: q.metro, country: q.country, q: q.q, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, confidence: q.confidence, opened_from: q.opened_from, opened_to: q.opened_to };
94 + return [...facilityConds(sql, fq), ...layerConds(sql, q), sql`f.lat is not null and f.lng is not null`, bboxCond(sql, q.bbox, sql`f.lat`, sql`f.lng`)];
95 +}
96 +
42 97 function conds(sql: Sql, q: MapQuery): Fragment {
43 − const fq: FacilityQuery = { status: q.status, type: q.type, operator: q.operator, country: q.country, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale };
44 − return andAll(sql, [...facilityConds(sql, fq), sql`f.lat is not null and f.lng is not null`, bboxCond(sql, q.bbox, sql`f.lat`, sql`f.lng`)]);
98 + return andAll(sql, [...baseConds(sql, q), yearCond(sql, q)]);
99 +}
100 +
101 +/** Layers whose facility points are hidden (overlay-only layers). */
102 +function facilitiesDrawn(q: MapQuery): boolean {
103 + return q.layer !== "cloud" && q.layer !== "ixps" && q.layer !== "projects";
45 104 }
46 105
47 106 function addStatus(target: Partial<Record<FacilityStatus, number>>, status: unknown, n: number): void {
@@ -49,9 +108,15 @@ function addStatus(target: Partial<Record<FacilityStatus, number>>, status: unkn
49 108 target[s] = (target[s] ?? 0) + n;
50 109 }
51 110
111 +/** MW column drawn for the layer: power → utility/grid supply, else the best known figure. */
112 +function pointMw(sql: Sql, q: MapQuery): Fragment {
113 + return q.layer === "power" ? sql`coalesce(f.utility_capacity_mw, f.grid_connection_mw)` : mwExpr(sql);
114 +}
115 +
52 116 async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> {
117 + const mw = pointMw(sql, q);
53 118 const rows = await sql<Row[]>`
54 − select f.country_iso2, f.status, count(*)::int as n, sum(${mwExpr(sql)})::float as mw, sum(f.lat)::float as slat, sum(f.lng)::float as slng, c.lat as clat, c.lng as clng, c.name as cname
119 + select f.country_iso2, f.status, count(*)::int as n, sum(${mw})::float as mw, sum(f.lat)::float as slat, sum(f.lng)::float as slng, c.lat as clat, c.lng as clng, c.name as cname
55 120 from facilities f ${facilityJoins(sql)}
56 121 where ${conds(sql, q)}
57 122 group by f.country_iso2, f.status, c.lat, c.lng, c.name`;
@@ -64,8 +129,8 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> {
64 129 c.count += n;
65 130 c._slat += num(r.slat) ?? 0;
66 131 c._slng += num(r.slng) ?? 0;
67 − const mw = num(r.mw);
68 − if (mw != null) { c.mw = (c.mw ?? 0) + mw; c._hasMw = true; }
132 + const m = num(r.mw);
133 + if (m != null) { c.mw = (c.mw ?? 0) + m; c._hasMw = true; }
69 134 addStatus(c.statuses, r.status, n);
70 135 const clat = num(r.clat), clng = num(r.clng);
71 136 if (clat != null && clng != null) { c.lat = clat; c.lng = clng; }
@@ -77,7 +142,7 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> {
77 142 by.delete("unknown");
78 143 const size = cellSize(4);
79 144 const cells = await sql<Row[]>`
80 − select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mwExpr(sql)})::float as mw, avg(f.lat)::float as alat, avg(f.lng)::float as alng
145 + select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mw})::float as mw, avg(f.lat)::float as alat, avg(f.lng)::float as alng
81 146 from facilities f ${facilityJoins(sql)}
82 147 where ${conds(sql, q)} and f.country_iso2 is null
83 148 group by 1, 2, 3`;
@@ -87,8 +152,8 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> {
87 152 if (!c) { c = { key, lat: num(r.alat) ?? 0, lng: num(r.alng) ?? 0, count: 0, mw: null, statuses: {}, label: null, _slat: 0, _slng: 0, _hasMw: false }; by.set(key, c); }
88 153 const n = int(r.n);
89 154 c.count += n;
90 − const mw = num(r.mw);
91 − if (mw != null) { c.mw = (c.mw ?? 0) + mw; c._hasMw = true; }
155 + const m = num(r.mw);
156 + if (m != null) { c.mw = (c.mw ?? 0) + m; c._hasMw = true; }
92 157 addStatus(c.statuses, r.status, n);
93 158 }
94 159 }
@@ -103,8 +168,9 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> {
103 168
104 169 async function gridClusters(sql: Sql, q: MapQuery, zoom: number): Promise<MapCluster[]> {
105 170 const size = cellSize(zoom);
171 + const mw = pointMw(sql, q);
106 172 const rows = await sql<Row[]>`
107 − select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mwExpr(sql)})::float as mw
173 + select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mw})::float as mw
108 174 from facilities f ${facilityJoins(sql)}
109 175 where ${conds(sql, q)}
110 176 group by 1, 2, 3`;
@@ -116,25 +182,31 @@ async function gridClusters(sql: Sql, q: MapQuery, zoom: number): Promise<MapClu
116 182 if (!c) { c = { key, lat: Math.round((j * size - 90 + size / 2) * 1e4) / 1e4, lng: Math.round((i * size - 180 + size / 2) * 1e4) / 1e4, count: 0, mw: null, statuses: {}, label: null }; by.set(key, c); }
117 183 const n = int(r.n);
118 184 c.count += n;
119 − const mw = num(r.mw);
120 − if (mw != null) c.mw = Math.round(((c.mw ?? 0) + mw) * 100) / 100;
185 + const m = num(r.mw);
186 + if (m != null) c.mw = Math.round(((c.mw ?? 0) + m) * 100) / 100;
121 187 addStatus(c.statuses, r.status, n);
122 188 }
123 189 return [...by.values()].sort((a, b) => b.count - a.count);
124 190 }
125 191
126 192 async function points(sql: Sql, q: MapQuery): Promise<MapPoint[] | null> {
193 + const mw = pointMw(sql, q);
127 194 const rows = await sql<Row[]>`
128 − select f.id, f.slug, f.name, o.name as op_name, f.lat, f.lng, f.status, f.facility_type, ${mwExpr(sql)} as mw, f.geo_precision, f.is_ai, f.is_hyperscale, f.country_iso2
195 + select f.id, f.slug, f.name, o.name as op_name, f.lat, f.lng, f.status, f.facility_type, ${mw} as mw, f.geo_precision, f.is_ai, f.ai_evidence, f.is_hyperscale, f.country_iso2, f.record_scope, ${yearExpr(sql, sql`f.opened_on`)} as y
129 196 from facilities f ${facilityJoins(sql)}
130 197 where ${conds(sql, q)}
131 − order by ${mwExpr(sql)} desc nulls last, f.id
198 + order by ${mw} desc nulls last, f.id
132 199 limit ${MAX_POINTS + 1}`;
133 200 if (rows.length > MAX_POINTS) return null;
134 201 return rows.map((r) => {
135 202 const p: MapPoint = { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: asType(r.facility_type), mw: num(r.mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2) };
136 − if (r.is_ai === true) p.ai = 1;
203 + const ai = str(r.ai_evidence);
204 + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1;
137 205 if (r.is_hyperscale === true) p.hs = 1;
206 + const y = num(r.y);
207 + if (y != null) p.y = y;
208 + const rs = str(r.record_scope);
209 + if (rs === "building" || rs === "campus") p.rs = rs;
138 210 return p;
139 211 });
140 212 }
@@ -143,36 +215,171 @@ async function cloudRegionPoints(sql: Sql, q: MapQuery): Promise<MapPoint[]> {
143 215 const c: Fragment[] = [sql`r.lat is not null and r.lng is not null`, bboxCond(sql, q.bbox, sql`r.lat`, sql`r.lng`)];
144 216 if (q.country?.length) c.push(sql`r.country_iso2 = any(${q.country})`);
145 217 if (q.operator) c.push(sql`(pr.slug = ${q.operator} or pr.id = ${q.operator})`);
146 − const rows = await sql<Row[]>`select r.id, r.slug, r.name, pr.name as pr_name, r.lat, r.lng, r.status, r.geo_precision, r.country_iso2 from cloud_regions r join operators pr on pr.id = r.provider_id where ${andAll(sql, c)} limit 2000`;
218 + if (q.metro) c.push(sql`r.metro_id in (select id from metros where slug = ${q.metro} or id = ${q.metro})`);
219 + const rows = await sql<Row[]>`select r.id, r.slug, r.name, pr.name as pr_name, r.lat, r.lng, r.status, r.geo_precision, r.country_iso2, ${yearExpr(sql, sql`r.launched_on`)} as y from cloud_regions r join operators pr on pr.id = r.provider_id where ${andAll(sql, c)} limit 2000`;
147 220 return rows.map((r) => {
148 221 const st = str(r.status);
149 − return { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.pr_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: st === "announced" ? "announced" : st === "retired" ? "closed" : "operational", t: "cloud_region", p: asPrecision(r.geo_precision ?? "city"), c: str(r.country_iso2) };
222 + const p: MapPoint = { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.pr_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: st === "announced" ? "announced" : st === "retired" ? "closed" : "operational", t: "cloud_region", p: asPrecision(r.geo_precision ?? "city"), c: str(r.country_iso2), k: "cloud_region" };
223 + const y = num(r.y);
224 + if (y != null) p.y = y;
225 + return p;
150 226 });
151 227 }
152 228
229 +/** Live projects with coordinates (city-level geocodes → p = city). */
230 +async function projectPoints(sql: Sql, q: MapQuery, onlyAi = false): Promise<MapPoint[]> {
231 + const c: Fragment[] = [projectLive(sql), sql`p.lat is not null and p.lng is not null`, bboxCond(sql, q.bbox, sql`p.lat`, sql`p.lng`)];
232 + if (q.country?.length) c.push(sql`p.country_iso2 = any(${q.country})`);
233 + if (q.operator) c.push(sql`p.operator_id in (select id from operators where slug = ${q.operator} or id = ${q.operator})`);
234 + if (q.metro) c.push(sql`p.metro_id in (select id from metros where slug = ${q.metro} or id = ${q.metro})`);
235 + if (q.project_status?.length) c.push(sql`p.status = any(${q.project_status})`);
236 + else if (q.layer === "pipeline") c.push(sql`p.status = any(${PIPELINE_PROJECT_STATUSES})`);
237 + if (q.expected_from != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} >= ${q.expected_from}`);
238 + if (q.expected_to != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} <= ${q.expected_to}`);
239 + if (q.min_mw != null) c.push(sql`p.planned_mw >= ${q.min_mw}`);
240 + if (q.max_mw != null) c.push(sql`p.planned_mw <= ${q.max_mw}`);
241 + if (onlyAi || q.ai === true) c.push(sql`(p.is_ai or p.ai_evidence in ('confirmed', 'likely'))`);
242 + if (q.ai_evidence?.length) c.push(sql`p.ai_evidence = any(${q.ai_evidence})`);
243 + if (q.year != null) c.push(sql`${yearExpr(sql, sql`coalesce(p.announced_on, p.expected_opening)`)} <= ${q.year}`);
244 + const rows = await sql<Row[]>`
245 + select p.id, p.slug, p.name, o.name as op_name, p.lat, p.lng, p.status, p.planned_mw, p.geo_precision, p.is_ai, p.ai_evidence, p.country_iso2, pf.facility_type, ${yearExpr(sql, sql`p.expected_opening`)} as y
246 + from projects p left join operators o on o.id = p.operator_id left join facilities pf on pf.id = p.facility_id
247 + where ${andAll(sql, c)} order by p.planned_mw desc nulls last, p.id limit ${MAX_POINTS}`;
248 + return rows.map((r) => {
249 + const p: MapPoint = { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: r.facility_type ? asType(r.facility_type) : "unknown", mw: num(r.planned_mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2), k: "project" };
250 + const ai = str(r.ai_evidence);
251 + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1;
252 + const y = num(r.y);
253 + if (y != null) p.y = y;
254 + return p;
255 + });
256 +}
257 +
258 +/** IXPs at their metro's reference point (ixps carry no coordinates); metro resolved by id, else by city name. */
259 +function ixpMetroJoin(sql: Sql): Fragment {
260 + return sql`
261 + left join metros xm on xm.id = x.metro_id
262 + left join lateral (select m2.id, m2.slug, m2.name, m2.lat, m2.lng from metros m2 where x.metro_id is null and x.country_iso2 = m2.country_iso2 and x.city is not null and (lower(m2.name) = lower(x.city) or exists (select 1 from unnest(m2.aliases) a where lower(a) = lower(x.city))) limit 1) xm2 on true`;
263 +}
264 +
265 +async function ixpPoints(sql: Sql, q: MapQuery): Promise<MapPoint[]> {
266 + const lat = sql`coalesce(xm.lat, xm2.lat)`, lng = sql`coalesce(xm.lng, xm2.lng)`;
267 + const c: Fragment[] = [sql`${lat} is not null`, bboxCond(sql, q.bbox, lat, lng)];
268 + if (q.country?.length) c.push(sql`x.country_iso2 = any(${q.country})`);
269 + if (q.metro) c.push(sql`coalesce(xm.id, xm2.id) in (select id from metros where slug = ${q.metro} or id = ${q.metro})`);
270 + const rows = await sql<Row[]>`select x.id, x.slug, x.name, x.country_iso2, x.network_count, ${lat} as lat, ${lng} as lng from ixps x ${ixpMetroJoin(sql)} where ${andAll(sql, c)} order by x.network_count desc nulls last limit 2000`;
271 + return rows.map((r) => ({ id: String(r.id), slug: String(r.slug), n: String(r.name), o: null, lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: "operational", t: "internet_exchange", mw: null, p: "metro", c: str(r.country_iso2), k: "ixp" }));
272 +}
273 +
274 +/** Power / grid events by metro, as clusters at the metro reference point (no invented plant locations). */
275 +async function powerEventClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> {
276 + const types = [...new Set([...POWER_EVENT_TYPES, ...GRID_EVENT_TYPES])];
277 + const c: Fragment[] = [sql`e.event_type = any(${types})`, sql`e.review_status <> 'rejected'`, bboxCond(sql, q.bbox, sql`m.lat`, sql`m.lng`)];
278 + if (q.country?.length) c.push(sql`m.country_iso2 = any(${q.country})`);
279 + if (q.metro) c.push(sql`(m.slug = ${q.metro} or m.id = ${q.metro})`);
280 + const rows = await sql<Row[]>`select m.id, m.name, m.lat, m.lng, count(*)::int as n from events e join metros m on m.id = e.metro_id where ${andAll(sql, c)} group by m.id, m.name, m.lat, m.lng order by n desc limit 500`;
281 + return rows.map((r) => ({ key: `power:${String(r.id)}`, lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, count: int(r.n), mw: null, statuses: {}, label: `${String(r.name)} — ${int(r.n)} power / grid event${int(r.n) > 1 ? "s" : ""}` }));
282 +}
283 +
284 +/** Share of filter-matching facilities that carry an opening year (time machine honesty figure). */
285 +async function yearCoverage(sql: Sql, q: MapQuery): Promise<number> {
286 + const rows = await sql<Row[]>`select count(*)::int as n, count(*) filter (where f.opened_on ~ '^\\d{4}')::int as dated from facilities f ${facilityJoins(sql)} where ${andAll(sql, baseConds(sql, q))}`;
287 + const n = int(rows[0]?.n);
288 + return n ? Math.round((int(rows[0]?.dated) / n) * 1000) / 1000 : 0;
289 +}
290 +
291 +async function densityCells(sql: Sql, q: MapQuery, zoom: number): Promise<{ view: DensityView; cells: DensityCell[]; max: number; total: number }> {
292 + const view = q.density ?? "facilities";
293 + const size = cellSize(zoom);
294 + const cell = (latCol: Fragment, lngCol: Fragment) => sql`floor((${lngCol} + 180) / ${size})::int as i, floor((${latCol} + 90) / ${size})::int as j`;
295 + let rows: Row[];
296 + if (view === "cloud") {
297 + const c: Fragment[] = [sql`r.lat is not null and r.lng is not null and r.status <> 'retired'`, bboxCond(sql, q.bbox, sql`r.lat`, sql`r.lng`)];
298 + if (q.country?.length) c.push(sql`r.country_iso2 = any(${q.country})`);
299 + if (q.operator) c.push(sql`r.provider_id in (select id from operators where slug = ${q.operator} or id = ${q.operator})`);
300 + rows = await sql<Row[]>`select ${cell(sql`r.lat`, sql`r.lng`)}, count(*)::int as n, count(*)::float as w from cloud_regions r where ${andAll(sql, c)} group by 1, 2`;
301 + } else if (view === "ixps") {
302 + const lat = sql`coalesce(xm.lat, xm2.lat)`, lng = sql`coalesce(xm.lng, xm2.lng)`;
303 + const c: Fragment[] = [sql`${lat} is not null`, bboxCond(sql, q.bbox, lat, lng)];
304 + if (q.country?.length) c.push(sql`x.country_iso2 = any(${q.country})`);
305 + rows = await sql<Row[]>`select ${cell(lat, lng)}, count(*)::int as n, count(*)::float as w from ixps x ${ixpMetroJoin(sql)} where ${andAll(sql, c)} group by 1, 2`;
306 + } else {
307 + const weight: Fragment =
308 + view === "known_mw" ? sql`coalesce(sum(${knownMwAgg(sql)}) filter (where f.status in ('operational', 'partially_operational', 'expansion')), 0)::float`
309 + : view === "pipeline_mw" ? sql`coalesce(sum(${pipelineMwAgg(sql)}) filter (where f.status = any(${PIPELINE_SET})), 0)::float`
310 + : view === "ai" ? sql`count(*) filter (where ${countedAgg(sql)} and ${AI_COND(sql)})::float`
311 + : view === "operators" ? sql`count(distinct f.operator_id)::float`
312 + : sql`count(*) filter (where ${countedAgg(sql)})::float`;
313 + rows = await sql<Row[]>`select ${cell(sql`f.lat`, sql`f.lng`)}, count(*)::int as n, ${weight} as w
314 + from ${facilityView(sql)} f ${facilityJoins(sql)} where ${conds(sql, q)} group by 1, 2`;
315 + }
316 + const cells: DensityCell[] = [];
317 + let max = 0, total = 0;
318 + for (const r of rows) {
319 + const w = Math.round((num(r.w) ?? 0) * 100) / 100;
320 + const n = int(r.n);
321 + if (w <= 0 && n <= 0) continue;
322 + total += n;
323 + if (w > max) max = w;
324 + cells.push({ lat: Math.round((int(r.j) * size - 90 + size / 2) * 1e4) / 1e4, lng: Math.round((int(r.i) * size - 180 + size / 2) * 1e4) / 1e4, w, n });
325 + }
326 + cells.sort((a, b) => b.w - a.w);
327 + return { view, cells, max, total };
328 +}
329 +
330 +/** Overlay points for the requested layer (non-facility entities). */
331 +async function overlayFor(sql: Sql, q: MapQuery): Promise<{ overlay: MapPoint[]; clusters: MapCluster[] }> {
332 + const layer = q.layer ?? "facilities";
333 + const tasks: Array<Promise<MapPoint[]>> = [];
334 + let clusters: Promise<MapCluster[]> = Promise.resolve([]);
335 + if (layer === "projects" || layer === "pipeline") tasks.push(projectPoints(sql, q));
336 + if (layer === "ai") tasks.push(projectPoints(sql, q, true));
337 + if (layer === "cloud" || layer === "connectivity" || q.cloud_regions) tasks.push(cloudRegionPoints(sql, q));
338 + if (layer === "ixps" || layer === "connectivity") tasks.push(ixpPoints(sql, q));
339 + if (layer === "power") clusters = powerEventClusters(sql, q);
340 + const [parts, cl] = await Promise.all([Promise.all(tasks), clusters]);
341 + return { overlay: parts.flat(), clusters: cl };
342 +}
343 +
153 344 export async function mapData(q: MapQuery): Promise<MapResponse> {
154 345 const sql = pg();
155 346 const zoom = Math.max(0, Math.min(22, Math.round(q.zoom)));
156 − const mode = mapMode(zoom);
157 − const cloud = q.cloud_regions ? cloudRegionPoints(sql, q) : Promise.resolve([] as MapPoint[]);
158 − if (mode === "country") {
159 − const [clusters, cr] = await Promise.all([countryClusters(sql, q), cloud]);
160 − const res: MapResponse = { zoom, mode: "clusters", clusters, total: clusters.reduce((a, c) => a + c.count, 0) };
161 − if (cr.length) res.points = cr;
347 + const layer: MapLayer = q.layer ?? "facilities";
348 + const withOverlay = (res: MapResponse, o: { overlay: MapPoint[]; clusters: MapCluster[] }): MapResponse => {
349 + if (o.overlay.length) res.overlay = o.overlay;
350 + // API 1.x compatibility: the legacy `cloud_regions=1` flag (no `layer`) returned cloud regions inside `points`
351 + if (o.overlay.length && q.cloud_regions && !q.layer) res.points = [...(res.points ?? []), ...o.overlay.filter((p) => p.k === "cloud_region")];
352 + if (o.clusters.length) res.clusters = [...(res.clusters ?? []), ...o.clusters];
162 353 return res;
354 + };
355 + const year = q.year ?? null;
356 + const cov = year != null ? yearCoverage(sql, q) : Promise.resolve(null);
357 + const overlayP = overlayFor(sql, q);
358 +
359 + if (q.density) {
360 + const [d, o, yc] = await Promise.all([densityCells(sql, q, zoom), overlayP, cov]);
361 + const res: MapResponse = { zoom, mode: "density", layer, density: { view: d.view, cells: d.cells, max: d.max }, total: d.total, year, yearCoverage: yc };
362 + return withOverlay(res, o);
163 363 }
164 − if (mode === "grid") {
165 − const [clusters, cr] = await Promise.all([gridClusters(sql, q, zoom), cloud]);
166 − const res: MapResponse = { zoom, mode: "clusters", clusters, total: clusters.reduce((a, c) => a + c.count, 0) };
167 − if (cr.length) res.points = cr;
168 − return res;
364 +
365 + if (!facilitiesDrawn(q)) {
366 + const [o, yc] = await Promise.all([overlayP, cov]);
367 + const res: MapResponse = { zoom, mode: "points", layer, points: [], total: o.overlay.length, year, yearCoverage: yc };
368 + return withOverlay(res, o);
169 369 }
170 − const [pts, cr] = await Promise.all([points(sql, q), cloud]);
370 +
371 + const mode = mapMode(zoom);
372 + if (mode === "country" || mode === "grid") {
373 + const [clusters, o, yc] = await Promise.all([mode === "country" ? countryClusters(sql, q) : gridClusters(sql, q, zoom), overlayP, cov]);
374 + const res: MapResponse = { zoom, mode: "clusters", layer, clusters, total: clusters.reduce((a, c) => a + c.count, 0), year, yearCoverage: yc };
375 + return withOverlay(res, o);
376 + }
377 + const [pts, o, yc] = await Promise.all([points(sql, q), overlayP, cov]);
171 378 if (pts === null) {
172 379 const clusters = await gridClusters(sql, q, 8);
173 − const res: MapResponse = { zoom, mode: "clusters", clusters, total: clusters.reduce((a, c) => a + c.count, 0) };
174 − if (cr.length) res.points = cr;
175 − return res;
380 + const res: MapResponse = { zoom, mode: "clusters", layer, clusters, total: clusters.reduce((a, c) => a + c.count, 0), degraded: true, year, yearCoverage: yc };
381 + return withOverlay(res, o);
176 382 }
177 − return { zoom, mode: "points", points: [...pts, ...cr], total: pts.length };
383 + const res: MapResponse = { zoom, mode: "points", layer, points: pts, total: pts.length, year, yearCoverage: yc };
384 + return withOverlay(res, o);
178 385 }
modified apps/api/src/repositories/metros.ts +58 −33
@@ -1,15 +1,18 @@
1 −import type { MetroDetail, MetroSummary } from "@dci/core";
2 −import { pg, mwExpr, plannedMwExpr, OPERATIONAL_SET, CONSTRUCTION_SET, PLANNED_SET, yearExpr, cloudRegionCols, likePattern, type Fragment, type Sql } from "../lib/sql.js";
1 +import type { GridConstraintDTO, MetroDetail, MetroSummary } from "@dci/core";
2 +import { pg, mwExpr, yearExpr, cloudRegionCols, likePattern, facilityView, knownMwAgg, countedAgg, projectLive, facilityJoins, facilitySummaryCols, eventCols, eventJoins, gridConstraintCols, gridConstraintJoins, AI_LEVELS, GRID_EVENT_TYPES, OPERATIONAL_SET, type Fragment, type Sql } from "../lib/sql.js";
3 3 import { int, iso, num, reqStr, str, strArray, type Row } from "../lib/rows.js";
4 −import { cloudRegionSummary, growthSeries } from "../lib/dto.js";
4 +import { cloudRegionSummary, facilitySummary, gridConstraintDto, gridConstraintFromEvent, growthSeries, round2, share } from "../lib/dto.js";
5 5 import { findBySlugOrId } from "../lib/resolve.js";
6 6 import { listFacilities } from "./facilities.js";
7 7 import { operatorsForScope } from "./operators.js";
8 8 import { projectsWhere } from "./projects.js";
9 −import { eventsForMetro } from "./events.js";
9 +import { eventsForMetro, toEventDtos } from "./events.js";
10 10 import { rankingPositions } from "./rankings.js";
11 +import { concentration, coverageRow, dimAggCols, momentum, pipelineBreakdown, scopeFor } from "../lib/pipeline.js";
12 +import { claimsFor } from "../lib/quality.js";
11 13
12 14 function metroSummary(r: Row): MetroSummary {
15 + const n = int(r.facility_count);
13 16 return {
14 17 id: reqStr(r.id),
15 18 slug: reqStr(r.slug),
@@ -19,44 +22,34 @@ function metroSummary(r: Row): MetroSummary {
19 22 regionName: str(r.region_name),
20 23 lat: num(r.lat) ?? 0,
21 24 lng: num(r.lng) ?? 0,
22 − facilityCount: int(r.facility_count),
25 + facilityCount: n,
23 26 operationalCount: int(r.operational),
24 27 constructionCount: int(r.construction),
25 28 plannedCount: int(r.planned),
26 − knownMw: num(r.known_mw),
27 − constructionMw: num(r.construction_mw),
28 − plannedMw: num(r.planned_mw),
29 + knownMw: round2(num(r.known_mw)),
30 + constructionMw: round2(num(r.construction_mw)),
31 + plannedMw: round2(num(r.planned_mw)),
29 32 operatorCount: int(r.operator_count),
30 33 cloudRegionCount: int(r.cloud_region_count),
31 34 ixpCount: int(r.ixp_count),
32 35 projectCount: int(r.project_count),
36 + projectPlannedMw: round2(num(r.project_planned_mw)),
37 + aiCount: int(r.ai),
38 + mwCoverage: n ? share(int(r.with_mw), n) : 0,
33 39 };
34 40 }
35 41
36 −const SELECT = (sql: Sql) => {
37 − const mw = mwExpr(sql);
38 − return sql`
42 +const SELECT = (sql: Sql) => sql`
39 43 select m.id, m.slug, m.name, m.country_iso2, c.name as country_name, m.region_name, m.lat, m.lng, m.aliases, m.description, m.updated_at,
40 − coalesce(fa.facility_count, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned,
41 − fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operator_count, 0) as operator_count,
42 − coalesce(cr.n, 0) as cloud_region_count, coalesce(ix.n, 0) as ixp_count, coalesce(pj.n, 0) as project_count
44 + coalesce(fa.facilities, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned,
45 + fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operators, 0) as operator_count, coalesce(fa.with_mw, 0) as with_mw, coalesce(fa.ai, 0) as ai,
46 + coalesce(cr.n, 0) as cloud_region_count, coalesce(ix.n, 0) as ixp_count, coalesce(pj.n, 0) as project_count, pj.mw as project_planned_mw
43 47 from metros m
44 48 left join countries c on c.iso2 = m.country_iso2
45 − left join (
46 − select f.metro_id, count(*)::int as facility_count,
47 − count(*) filter (where f.status = any(${OPERATIONAL_SET}))::int as operational,
48 − count(*) filter (where f.status = any(${CONSTRUCTION_SET}))::int as construction,
49 − count(*) filter (where f.status = any(${PLANNED_SET}))::int as planned,
50 − sum(${mw}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw,
51 − sum(${mw}) filter (where f.status = any(${CONSTRUCTION_SET}))::float as construction_mw,
52 − sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PLANNED_SET}))::float as planned_mw,
53 − count(distinct f.operator_id)::int as operator_count
54 − from facilities f where f.merged_into is null and f.metro_id is not null group by f.metro_id
55 − ) fa on fa.metro_id = m.id
56 − left join (select metro_id, count(*)::int as n from cloud_regions where metro_id is not null group by 1) cr on cr.metro_id = m.id
49 + left join (select f.metro_id, ${dimAggCols(sql)} from ${facilityView(sql)} f where f.metro_id is not null group by f.metro_id) fa on fa.metro_id = m.id
50 + left join (select metro_id, count(*)::int as n from cloud_regions where metro_id is not null and status <> 'retired' group by 1) cr on cr.metro_id = m.id
57 51 left join (select metro_id, count(*)::int as n from ixps where metro_id is not null group by 1) ix on ix.metro_id = m.id
58 − left join (select metro_id, count(*)::int as n from projects where metro_id is not null group by 1) pj on pj.metro_id = m.id`;
59 −};
52 + left join (select p.metro_id, count(*)::int as n, sum(p.planned_mw) filter (where p.status in ('rumored','proposed','announced','permitting','approved','delayed','under_construction'))::float as mw from projects p where p.metro_id is not null and ${projectLive(sql)} group by p.metro_id) pj on pj.metro_id = m.id`;
60 53
61 54 export async function listMetros(f: { country?: string; q?: string; limit?: number } = {}): Promise<MetroSummary[]> {
62 55 const sql = pg();
@@ -64,7 +57,7 @@ export async function listMetros(f: { country?: string; q?: string; limit?: numb
64 57 if (f.country) conds.push(sql`m.country_iso2 = ${f.country.toUpperCase()}`);
65 58 if (f.q) conds.push(sql`(m.name ilike ${likePattern(f.q)} or similarity(m.name, ${f.q}) > 0.35 or exists (select 1 from unnest(m.aliases) a where a ilike ${likePattern(f.q)}))`);
66 59 const rows = await sql<Row[]>`${SELECT(sql)} where ${conds.reduce<Fragment>((a, c) => sql`${a} and ${c}`, sql`true`)}
67 − order by coalesce(fa.facility_count, 0) desc, coalesce(cr.n, 0) desc, m.name limit ${f.limit ?? 1000}`;
60 + order by coalesce(fa.facilities, 0) desc, coalesce(cr.n, 0) desc, m.name limit ${f.limit ?? 1000}`;
68 61 return rows.map(metroSummary);
69 62 }
70 63
@@ -83,7 +76,7 @@ async function metroConstraints(metroId: string, name: string, aliases: string[]
83 76 const rows = await sql<Row[]>`
84 77 select n.title, n.summary, n.url, n.published_at, n.created_at, s.name as source_name
85 78 from news_items n left join sources s on s.id = n.source_id
86 − where (n.mentions->>'metroId' = ${metroId} or n.mentions->'metros' ? ${metroId} or n.mentions->'cities' ?| ${terms}::text[] or (${nameCond}))
79 + where (n.metro_id = ${metroId} or n.mentions->>'metroId' = ${metroId} or n.mentions->'metros' ? ${metroId} or n.mentions->'cities' ?| ${terms}::text[] or (${nameCond}))
87 80 and (n.title ~* '(moratorium|grid|power|electricity|substation|water|zoning|land|transmission|interconnection)' or n.summary ~* '(moratorium|grid (constraint|capacity)|power (shortage|constraint|crunch)|water (usage|shortage|restriction)|zoning|land (scarcity|shortage|constraint))')
88 81 order by n.published_at desc nulls last, n.created_at desc limit 40`;
89 82 const out: MetroDetail["constraints"] = [];
@@ -97,28 +90,60 @@ async function metroConstraints(metroId: string, name: string, aliases: string[]
97 90 return out;
98 91 }
99 92
93 +/** grid_constraints rows + grid / utility / power events for a metro or a country, deduplicated by event id. */
94 +export async function gridConstraintsFor(scope: { metroId?: string; countryIso2?: string }, limit = 30): Promise<GridConstraintDTO[]> {
95 + const sql = pg();
96 + const gCond = scope.metroId ? sql`g.metro_id = ${scope.metroId}` : scope.countryIso2 ? sql`g.country_iso2 = ${scope.countryIso2}` : sql`true`;
97 + const eCond = scope.metroId ? sql`(e.metro_id = ${scope.metroId} or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = ${scope.metroId})))` : scope.countryIso2 ? sql`e.country_iso2 = ${scope.countryIso2}` : sql`true`;
98 + const [rows, evRows] = await Promise.all([
99 + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} where ${gCond} order by g.effective_date desc nulls last, g.created_at desc limit ${limit}`,
100 + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where ${eCond} and e.event_type = any(${GRID_EVENT_TYPES}) and e.review_status <> 'rejected' order by e.detected_at desc limit ${limit}`,
101 + ]);
102 + const out = rows.map(gridConstraintDto);
103 + const seen = new Set(out.map((g) => g.eventId).filter(Boolean));
104 + for (const e of await toEventDtos(evRows)) { if (seen.has(e.id)) continue; seen.add(e.id); out.push(gridConstraintFromEvent(e)); }
105 + return out.slice(0, limit);
106 +}
107 +
100 108 export async function getMetroDetail(idOrSlug: string, fPage = 1): Promise<MetroDetail | null> {
101 109 const sql = pg();
102 110 const base = await findBySlugOrId("metros", idOrSlug);
103 111 if (!base) return null;
104 112 const id = String(base.id);
105 113 const aliases = strArray(base.aliases);
106 − const [sumRows, operators, cloudProviderRows, cloudRegionRows, ixpRows, facilities, projects, recentEvents, constraints, growthRows, rankings] = await Promise.all([
114 + const scope = scopeFor(sql, { metroId: id });
115 + const [sumRows, operators, cloudProviderRows, cloudRegionRows, ixpRows, facilities, projects, recentEvents, constraints, growthRows, rankings, conc, mom, pipeline, gridConstraints, aiRows, openingRows, coverage, claims] = await Promise.all([
107 116 sql<Row[]>`${SELECT(sql)} where m.id = ${id}`,
108 117 operatorsForScope({ metroId: id }, 15),
109 118 sql<Row[]>`select pr.id, pr.slug, pr.name, count(*)::int as n from cloud_regions r join operators pr on pr.id = r.provider_id where r.metro_id = ${id} group by 1, 2, 3 order by n desc, pr.name`,
110 119 sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.metro_id = ${id} order by pr.name, r.code`,
111 120 sql<Row[]>`select id, slug, name, network_count from ixps where metro_id = ${id} order by network_count desc nulls last, name`,
112 121 listFacilities({ metroId: id, page: fPage, per_page: 50, sort: "mw", order: "desc" }),
113 − projectsWhere(sql`p.metro_id = ${id}`, 20),
122 + projectsWhere(sql`${projectLive(sql)} and p.metro_id = ${id}`, 20),
114 123 eventsForMetro(id, 20),
115 124 metroConstraints(id, String(base.name), aliases),
116 − sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*)::int as n, sum(${mwExpr(sql)})::float as mw from facilities f where f.metro_id = ${id} and f.merged_into is null and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
125 + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${knownMwAgg(sql)})::float as mw from ${facilityView(sql)} f where f.metro_id = ${id} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
117 126 rankingPositions("metros", [id, String(base.slug)]),
127 + concentration(scope),
128 + momentum(scope),
129 + pipelineBreakdown(scope),
130 + gridConstraintsFor({ metroId: id }),
131 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.metro_id = ${id} and f.ai_evidence = any(${AI_LEVELS}) order by ${mwExpr(sql)} desc nulls last, f.name limit 20`,
132 + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${knownMwAgg(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw from ${facilityView(sql)} f where f.metro_id = ${id} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
133 + coverageRow(id, String(base.name), String(base.slug), scope),
134 + Promise.all([claimsFor("metro", id), claimsFor("market", id)]).then(([a, b]) => [...a, ...b]),
118 135 ]);
119 136 const summary = metroSummary(sumRows[0] ?? { ...base, facility_count: 0 });
120 137 return {
121 138 ...summary,
139 + concentration: conc,
140 + momentum: mom,
141 + pipeline,
142 + gridConstraints,
143 + aiFacilities: aiRows.map(facilitySummary),
144 + openingTimeline: openingRows.filter((r) => int(r.year) > 0).map((r) => ({ year: int(r.year), opened: int(r.n), openedMw: round2(num(r.mw)) })),
145 + coverage,
146 + claims,
122 147 aliases,
123 148 description: str(base.description),
124 149 operators,
modified apps/api/src/repositories/misc.ts +4 −3
@@ -1,6 +1,6 @@
1 1 /** Smaller public repositories: time series, sources, news, sitemap feeds. */
2 2 import type { SourceRef } from "@dci/core";
3 −import { pg, page, type Fragment } from "../lib/sql.js";
3 +import { pg, page, likePattern, type Fragment } from "../lib/sql.js";
4 4 import { int, iso, json, num, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js";
5 5 import { sourceRef } from "../lib/dto.js";
6 6
@@ -57,7 +57,7 @@ export async function listNews(f: { country?: string; operator?: string; since?:
57 57 if (f.country) c.push(sql`${f.country.toUpperCase()} = any(n.country_iso2s)`);
58 58 if (f.operator) c.push(sql`exists (select 1 from operators o where o.id = any(n.operator_ids) and (o.slug = ${f.operator} or o.id = ${f.operator}))`);
59 59 if (f.since) c.push(sql`coalesce(n.published_at, n.created_at) >= ${f.since}::timestamptz`);
60 − if (f.q) c.push(sql`(n.title ilike ${"%" + f.q + "%"} or n.summary ilike ${"%" + f.q + "%"})`);
60 + if (f.q) c.push(sql`(n.title ilike ${likePattern(f.q)} or n.summary ilike ${likePattern(f.q)})`);
61 61 const rows = await sql<Row[]>`
62 62 select n.*, s.name as source_name, s.kind as source_kind,
63 63 (select coalesce(jsonb_agg(jsonb_build_object('id', o.id, 'slug', o.slug, 'name', o.name)), '[]'::jsonb) from operators o where o.id = any(n.operator_ids)) as ops,
@@ -81,7 +81,8 @@ export async function sitemap(kind: SitemapKind, pageNo: number): Promise<{ item
81 81 const sql = pg();
82 82 const offset = Math.max(0, pageNo - 1) * SITEMAP_PAGE;
83 83 const table = kind === "cloud-regions" ? "cloud_regions" : kind;
84 − const where = kind === "facilities" ? sql`where merged_into is null` : sql``;
84 + // facilities: merged duplicates hidden; projects: false positives (hidden) and merged duplicates never listed
85 + const where = kind === "facilities" ? sql`where merged_into is null` : kind === "projects" ? sql`where hidden = false and merged_into is null` : sql``;
85 86 const rows = await sql<Row[]>`select slug, updated_at, count(*) over() as total from ${sql(table)} ${where} order by slug limit ${SITEMAP_PAGE} offset ${offset}`;
86 87 const total = rows.length ? int(rows[0]!.total) : int((await sql<Row[]>`select count(*)::int as n from ${sql(table)} ${where}`)[0]?.n);
87 88 return { items: rows.map((r) => ({ slug: reqStr(r.slug), updatedAt: reqIso(r.updated_at) })), total, page: pageNo, perPage: SITEMAP_PAGE };
modified apps/api/src/repositories/operators.ts +115 −43
@@ -1,18 +1,27 @@
1 −import type { OperatorDetail, OperatorSummary, SourceRef } from "@dci/core";
2 −import { pg, andAll, page, mwExpr, plannedMwExpr, OPERATIONAL_SET, PIPELINE_SET, likePattern, type Fragment, type Sql } from "../lib/sql.js";
3 −import { int, num, record, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js";
4 −import { statusBreakdown } from "../lib/dto.js";
1 +import type { EventDTO, OperatorDetail, OperatorSummary, SourceRef } from "@dci/core";
2 +import { pg, andAll, page, likePattern, facilityView, knownMwAgg, countedAgg, facilityJoins, facilitySummaryCols, mwExpr, cloudRegionCols, eventCols, eventJoins, projectLive, OPERATIONAL_SET, AI_LEVELS, CORPORATE_EVENT_TYPES, type Fragment, type Sql } from "../lib/sql.js";
3 +import { bool, int, num, record, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js";
4 +import { cloudRegionSummary, facilitySummary, round2, share, statusBreakdown } from "../lib/dto.js";
5 5 import { findBySlugOrId } from "../lib/resolve.js";
6 6 import { listFacilities, provenanceFor } from "./facilities.js";
7 7 import { projectsWhere } from "./projects.js";
8 −import { eventsForOperator } from "./events.js";
8 +import { eventsForOperator, toEventDtos } from "./events.js";
9 9 import { buildSourceHistory, documentVersionsFor, sourceIdsOf, sourceRefsFor } from "../lib/source-history.js";
10 +import { dimAggCols, expansionVelocity, pipelineBreakdown, scopeFor } from "../lib/pipeline.js";
11 +import { claimsFor, dataQualityFor } from "../lib/quality.js";
10 12
11 13 export interface OperatorFilters { q?: string; kind?: string; country?: string; sort?: "facilities" | "name" | "mw"; order?: "asc" | "desc"; page?: number; per_page?: number }
12 14
13 −function operatorSummary(r: Row): OperatorSummary {
15 +/**
16 + * OperatorSummary from a row carrying `stats` (worker-refreshed, containment-aware) and optional live aggregate
17 + * columns (facility_count, known_mw, …). Live columns win when present (they reflect the current database); the
18 + * stats jsonb fills what the live query did not compute.
19 + */
20 +export function operatorSummary(r: Row): OperatorSummary {
14 21 const stats = (r.stats as Record<string, unknown> | null) ?? {};
15 − const fc = int(r.facility_count);
22 + const live = r.facility_count != null;
23 + const fc = live ? int(r.facility_count) : int(stats.facilityCount);
24 + const withMw = live ? int(r.with_mw) : null;
16 25 return {
17 26 id: reqStr(r.id),
18 27 slug: reqStr(r.slug),
@@ -20,28 +29,44 @@ function operatorSummary(r: Row): OperatorSummary {
20 29 kind: (str(r.kind) as OperatorSummary["kind"]) ?? null,
21 30 website: str(r.website),
22 31 hqCountryIso2: str(r.hq_country_iso2),
23 − facilityCount: fc || int(stats.facilityCount),
24 − countryCount: int(r.country_count) || int(stats.countryCount),
25 − metroCount: int(r.metro_count) || int(stats.metroCount),
26 − knownMw: num(r.known_mw) ?? num(stats.knownMw),
27 − plannedMw: num(r.planned_mw) ?? num(stats.plannedMw),
28 − projectCount: int(r.project_count) || int(stats.projectCount),
32 + facilityCount: fc,
33 + countryCount: live ? int(r.country_count) : int(stats.countryCount),
34 + metroCount: live ? int(r.metro_count) : int(stats.metroCount),
35 + knownMw: live ? round2(num(r.known_mw)) : round2(num(stats.knownMw)),
36 + plannedMw: live ? round2(num(r.planned_mw)) : round2(num(stats.plannedMw)),
37 + constructionMw: live ? round2(num(r.construction_mw)) : round2(num(stats.constructionMw)),
38 + projectCount: r.project_count != null ? int(r.project_count) : int(stats.projectCount),
39 + projectPlannedMw: r.project_planned_mw != null ? round2(num(r.project_planned_mw)) : round2(num(stats.projectPlannedMw)),
40 + aiCount: live ? int(r.ai) : int(stats.aiCount),
41 + cloudRegionCount: r.cloud_region_count != null ? int(r.cloud_region_count) : int(stats.cloudRegionCount),
42 + mwCoverage: withMw != null ? (fc ? share(withMw, fc) : 0) : num(stats.mwCoverage),
43 + isCloudProvider: bool(r.is_cloud_provider),
44 + isCarrier: bool(r.is_carrier),
29 45 };
30 46 }
31 47
32 −/** Aggregate over live facilities per operator, optionally restricted (country / metro). */
48 +/** Containment-aware aggregate per operator over the facility view, optionally restricted to a country / metro. */
33 49 function aggFragment(sql: Sql, scope: { countryIso2?: string; metroId?: string } = {}): Fragment {
34 − const conds: Fragment[] = [sql`f.merged_into is null`, sql`f.operator_id is not null`];
50 + const conds: Fragment[] = [sql`f.operator_id is not null`];
35 51 if (scope.countryIso2) conds.push(sql`f.country_iso2 = ${scope.countryIso2}`);
36 52 if (scope.metroId) conds.push(sql`f.metro_id = ${scope.metroId}`);
37 53 return sql`(
38 − select f.operator_id, count(*)::int as facility_count, count(distinct f.country_iso2)::int as country_count, count(distinct f.metro_id)::int as metro_count,
39 − sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw,
40 − sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PIPELINE_SET}))::float as planned_mw
41 − from facilities f where ${andAll(sql, conds)} group by f.operator_id
54 + select f.operator_id, count(distinct f.country_iso2)::int as country_count, count(distinct f.metro_id)::int as metro_count, ${dimAggCols(sql)}
55 + from ${facilityView(sql)} f where ${andAll(sql, conds)} group by f.operator_id
42 56 )`;
43 57 }
44 58
59 +function projectAgg(sql: Sql): Fragment {
60 + return sql`(select p.operator_id, count(*)::int as n, sum(p.planned_mw) filter (where p.status in ('rumored','proposed','announced','permitting','approved','delayed','under_construction'))::float as mw from projects p where ${projectLive(sql)} group by p.operator_id)`;
61 +}
62 +
63 +const OPERATOR_COLS = (sql: Sql) => sql`
64 + o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, o.stats, o.is_cloud_provider, o.is_carrier,
65 + coalesce(a.facilities, 0) as facility_count, coalesce(a.country_count, 0) as country_count, coalesce(a.metro_count, 0) as metro_count,
66 + a.known_mw, a.planned_mw, a.construction_mw, coalesce(a.with_mw, 0) as with_mw, coalesce(a.ai, 0) as ai,
67 + coalesce(pc.n, 0) as project_count, pc.mw as project_planned_mw,
68 + (select count(*)::int from cloud_regions cr where cr.provider_id = o.id and cr.status <> 'retired') as cloud_region_count`;
69 +
45 70 export async function listOperators(f: OperatorFilters): Promise<{ items: OperatorSummary[]; total: number; page: number; perPage: number }> {
46 71 const sql = pg();
47 72 const pg_ = page(f.page, f.per_page, 100, 50);
@@ -50,66 +75,113 @@ export async function listOperators(f: OperatorFilters): Promise<{ items: Operat
50 75 if (f.kind) conds.push(sql`o.kind = ${f.kind}`);
51 76 if (f.country) { const iso = f.country.toUpperCase(); conds.push(sql`(o.hq_country_iso2 = ${iso} or exists (select 1 from facilities x where x.operator_id = o.id and x.country_iso2 = ${iso} and x.merged_into is null))`); }
52 77 const asc = f.order === "asc";
53 − const order = f.sort === "name" ? (asc ? sql`o.name asc` : sql`o.name desc`) : f.sort === "mw" ? (asc ? sql`a.known_mw asc nulls last, o.name` : sql`a.known_mw desc nulls last, o.name`) : asc ? sql`coalesce(a.facility_count, 0) asc, o.name` : sql`coalesce(a.facility_count, 0) desc, o.name`;
78 + const order = f.sort === "name" ? (asc ? sql`o.name asc` : sql`o.name desc`) : f.sort === "mw" ? (asc ? sql`a.known_mw asc nulls last, o.name` : sql`a.known_mw desc nulls last, o.name`) : asc ? sql`coalesce(a.facilities, 0) asc, o.name` : sql`coalesce(a.facilities, 0) desc, o.name`;
54 79 const rows = await sql<Row[]>`
55 − select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, o.stats,
56 − coalesce(a.facility_count, 0) as facility_count, coalesce(a.country_count, 0) as country_count, coalesce(a.metro_count, 0) as metro_count, a.known_mw, a.planned_mw,
57 − coalesce(pc.n, 0) as project_count, count(*) over() as total
80 + select ${OPERATOR_COLS(sql)}, count(*) over() as total
58 81 from operators o
59 82 left join ${aggFragment(sql)} a on a.operator_id = o.id
60 − left join (select operator_id, count(*)::int as n from projects group by operator_id) pc on pc.operator_id = o.id
83 + left join ${projectAgg(sql)} pc on pc.operator_id = o.id
61 84 where ${andAll(sql, conds)}
62 85 order by ${order}
63 86 limit ${pg_.perPage} offset ${pg_.offset}`;
64 87 return { items: rows.map(operatorSummary), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage };
65 88 }
66 89
67 −/** Operators ranked by facilities inside a scope (country or metro). */
90 +/** Operators ranked by facilities inside a scope (country or metro) — containment-aware counts. */
68 91 export async function operatorsForScope(scope: { countryIso2?: string; metroId?: string }, limit = 10): Promise<OperatorSummary[]> {
69 92 const sql = pg();
70 93 const rows = await sql<Row[]>`
71 − select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, '{}'::jsonb as stats,
72 − a.facility_count, a.country_count, a.metro_count, a.known_mw, a.planned_mw, 0 as project_count
94 + select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, '{}'::jsonb as stats, o.is_cloud_provider, o.is_carrier,
95 + a.facilities as facility_count, a.country_count, a.metro_count, a.known_mw, a.planned_mw, a.construction_mw, a.with_mw, a.ai, 0 as project_count, null::float as project_planned_mw, 0 as cloud_region_count
73 96 from ${aggFragment(sql, scope)} a join operators o on o.id = a.operator_id
74 − order by a.facility_count desc, a.known_mw desc nulls last, o.name limit ${limit}`;
97 + where a.facilities > 0
98 + order by a.facilities desc, a.known_mw desc nulls last, o.name limit ${limit}`;
75 99 return rows.map(operatorSummary);
76 100 }
77 101
102 +/** OperatorSummary for one id (live aggregates). null when unknown. */
103 +export async function operatorSummaryById(id: string): Promise<OperatorSummary | null> {
104 + const sql = pg();
105 + const rows = await sql<Row[]>`select ${OPERATOR_COLS(sql)} from operators o left join ${aggFragment(sql)} a on a.operator_id = o.id left join ${projectAgg(sql)} pc on pc.operator_id = o.id where o.id = ${id}`;
106 + return rows[0] ? operatorSummary(rows[0]) : null;
107 +}
108 +
109 +/** Corporate events (acquisitions, financing, partnerships, executive changes…) linked to an operator. */
110 +export async function corporateEventsForOperator(operatorId: string, limit = 30): Promise<EventDTO[]> {
111 + const sql = pg();
112 + const rows = await sql<Row[]>`
113 + select ${eventCols(sql)} from events e ${eventJoins(sql)}
114 + where (e.operator_id = ${operatorId} or (e.entity_type = 'operator' and e.entity_id = ${operatorId})) and e.event_type = any(${CORPORATE_EVENT_TYPES}) and e.review_status <> 'rejected'
115 + order by e.detected_at desc limit ${limit}`;
116 + return toEventDtos(rows);
117 +}
118 +
119 +/** Metros first entered by the operator within the last `months` months (by earliest facility opened_on / first_seen or project announcement). */
120 +export async function newMarkets(operatorId: string, months = 12): Promise<Array<{ id: string; slug: string; name: string }>> {
121 + const sql = pg();
122 + const rows = await sql<Row[]>`
123 + with fm as (
124 + select f.metro_id, min(case when f.opened_on ~ '^\\d{4}-\\d{2}-\\d{2}' then f.opened_on::date when f.opened_on ~ '^\\d{4}-\\d{2}$' then (f.opened_on || '-01')::date when f.opened_on ~ '^\\d{4}$' then (f.opened_on || '-01-01')::date else f.first_seen::date end) as first_d
125 + from facilities f where f.merged_into is null and f.operator_id = ${operatorId} and f.metro_id is not null group by 1
126 + union all
127 + select p.metro_id, min(case when p.announced_on ~ '^\\d{4}-\\d{2}-\\d{2}' then p.announced_on::date when p.announced_on ~ '^\\d{4}-\\d{2}$' then (p.announced_on || '-01')::date when p.announced_on ~ '^\\d{4}$' then (p.announced_on || '-01-01')::date else p.created_at::date end)
128 + from projects p where ${projectLive(sql)} and p.operator_id = ${operatorId} and p.metro_id is not null group by 1
129 + ), first as (select metro_id, min(first_d) as first_d from fm group by 1)
130 + select m.id, m.slug, m.name from first join metros m on m.id = first.metro_id where first.first_d >= current_date - make_interval(months => ${months}) order by first.first_d desc limit 25`;
131 + return rows.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }));
132 +}
133 +
78 134 export async function getOperatorDetail(idOrSlug: string, fPage = 1): Promise<{ detail: OperatorDetail; sources: SourceRef[] } | null> {
79 135 const sql = pg();
80 136 const row = await findBySlugOrId("operators", idOrSlug);
81 137 if (!row) return null;
82 138 const id = String(row.id);
83 − const [sumRows, countryRows, metroRows, statusRows, facilities, projects, recentEvents, provenance, versions, parentRows] = await Promise.all([
84 − sql<Row[]>`
85 − select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, o.stats,
86 − coalesce(a.facility_count, 0) as facility_count, coalesce(a.country_count, 0) as country_count, coalesce(a.metro_count, 0) as metro_count, a.known_mw, a.planned_mw,
87 − (select count(*)::int from projects p where p.operator_id = o.id) as project_count
88 − from operators o left join ${aggFragment(sql)} a on a.operator_id = o.id where o.id = ${id}`,
139 + const scope = scopeFor(sql, { operatorId: id });
140 + const known = knownMwAgg(sql), counted = countedAgg(sql);
141 + const [sumRows, countryRows, metroRows, statusRows, facilities, projects, recentEvents, provenance, versions, parentRows, pipeline, velocity, aiRows, cloudRows, corporateEvents, claims] = await Promise.all([
142 + sql<Row[]>`select ${OPERATOR_COLS(sql)} from operators o left join ${aggFragment(sql)} a on a.operator_id = o.id left join ${projectAgg(sql)} pc on pc.operator_id = o.id where o.id = ${id}`,
89 143 sql<Row[]>`
90 − select f.country_iso2 as iso2, c.name, c.slug, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw
91 − from facilities f left join countries c on c.iso2 = f.country_iso2
92 − where f.operator_id = ${id} and f.merged_into is null and f.country_iso2 is not null
144 + select f.country_iso2 as iso2, c.name, c.slug, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw
145 + from ${facilityView(sql)} f left join countries c on c.iso2 = f.country_iso2
146 + where f.operator_id = ${id} and f.country_iso2 is not null
93 147 group by 1, 2, 3 order by n desc, c.name`,
94 148 sql<Row[]>`
95 − select m.id, m.slug, m.name, m.country_iso2, count(*)::int as n
96 − from facilities f join metros m on m.id = f.metro_id
97 − where f.operator_id = ${id} and f.merged_into is null group by 1, 2, 3, 4 order by n desc, m.name limit 100`,
149 + select m.id, m.slug, m.name, m.country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw
150 + from ${facilityView(sql)} f join metros m on m.id = f.metro_id
151 + where f.operator_id = ${id} group by 1, 2, 3, 4 order by n desc, m.name limit 100`,
98 152 sql<Row[]>`select status, count(*)::int as n from facilities where operator_id = ${id} and merged_into is null group by status`,
99 153 listFacilities({ operatorId: id, page: fPage, per_page: 50, sort: "mw", order: "desc" }),
100 − projectsWhere(sql`p.operator_id = ${id}`, 20),
154 + projectsWhere(sql`${projectLive(sql)} and p.operator_id = ${id}`, 20),
101 155 eventsForOperator(id, 20),
102 156 provenanceFor("operator", id),
103 157 documentVersionsFor("operator", id),
104 158 row.parent_id ? sql<Row[]>`select id, slug, name from operators where id = ${String(row.parent_id)}` : Promise.resolve([] as Row[]),
159 + pipelineBreakdown(scope),
160 + expansionVelocity(scope),
161 + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and (f.operator_id = ${id} or f.owner_id = ${id}) and f.ai_evidence = any(${AI_LEVELS}) order by ${mwExpr(sql)} desc nulls last, f.name limit 20`,
162 + sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.provider_id = ${id} order by r.country_iso2, r.code limit 500`,
163 + corporateEventsForOperator(id, 30),
164 + claimsFor("operator", id),
105 165 ]);
106 166 const summary = operatorSummary(sumRows[0] ?? row);
167 + const completeness = Math.round(([row.website, row.hq_country_iso2, row.kind, row.description].filter((v) => v != null && v !== "").length / 4) * 100);
168 + const dataQuality = await dataQualityFor("operator", id, completeness, str(row.updated_at));
169 + const counted_total = summary.facilityCount || countryRows.reduce((a, c) => a + int(c.n), 0);
107 170 const detail: OperatorDetail = {
108 171 ...summary,
172 + pipeline,
173 + velocity,
174 + topCountries: countryRows.slice(0, 10).map((c) => ({ iso2: reqStr(c.iso2), name: reqStr(c.name, reqStr(c.iso2)), slug: reqStr(c.slug, reqStr(c.iso2).toLowerCase()), facilityCount: int(c.n), knownMw: round2(num(c.known_mw)), share: share(int(c.n), counted_total) })),
175 + topMetros: metroRows.slice(0, 10).map((m) => ({ id: reqStr(m.id), slug: reqStr(m.slug), name: reqStr(m.name), countryIso2: reqStr(m.country_iso2), facilityCount: int(m.n), knownMw: round2(num(m.known_mw)), share: share(int(m.n), counted_total) })),
176 + aiFacilities: aiRows.map(facilitySummary),
177 + cloudRegions: cloudRows.map(cloudRegionSummary),
178 + corporateEvents,
179 + claims,
180 + dataQuality,
109 181 aliases: strArray(row.aliases),
110 182 description: str(row.description),
111 183 parent: parentRows[0] ? { id: String(parentRows[0].id), slug: String(parentRows[0].slug), name: String(parentRows[0].name) } : null,
112 − countries: countryRows.map((c) => ({ iso2: reqStr(c.iso2), name: reqStr(c.name, reqStr(c.iso2)), slug: reqStr(c.slug, reqStr(c.iso2).toLowerCase()), facilityCount: int(c.n), knownMw: num(c.known_mw) })),
184 + countries: countryRows.map((c) => ({ iso2: reqStr(c.iso2), name: reqStr(c.name, reqStr(c.iso2)), slug: reqStr(c.slug, reqStr(c.iso2).toLowerCase()), facilityCount: int(c.n), knownMw: round2(num(c.known_mw)) })),
113 185 metros: metroRows.map((m) => ({ id: reqStr(m.id), slug: reqStr(m.slug), name: reqStr(m.name), countryIso2: reqStr(m.country_iso2), facilityCount: int(m.n) })),
114 186 statusBreakdown: statusBreakdown(statusRows),
115 187 facilities: facilities.items,
@@ -119,7 +191,7 @@ export async function getOperatorDetail(idOrSlug: string, fPage = 1): Promise<{
119 191 externalIds: record(row.external_ids),
120 192 updatedAt: reqIso(row.updated_at),
121 193 };
122 − const sources = await sourceRefsFor(sourceIdsOf(provenance, recentEvents, versions));
194 + const sources = await sourceRefsFor(sourceIdsOf(provenance, recentEvents, versions, claims, corporateEvents));
123 195 return { detail, sources };
124 196 }
125 197
added apps/api/src/repositories/power.ts +74 −0
@@ -0,0 +1,74 @@
1 +/**
2 + * /power — grid constraints, power / grid events, large loads (utility / grid MW published on facilities, current
3 + * grid claims, ≥ 200 MW planned projects) and national energy context. Utility / grid MW are supply-side figures,
4 + * never IT load; national grid averages never describe a facility's contracted electricity.
5 + */
6 +import type { EnergyContext, PowerOverview } from "@dci/core";
7 +import { pg, facilityView, knownMwAgg, countedAgg, gridConstraintCols, gridConstraintJoins, projectLive, POWER_EVENT_TYPES, GRID_EVENT_TYPES, OPERATIONAL_SET } from "../lib/sql.js";
8 +import { int, num, reqStr, str, type Row } from "../lib/rows.js";
9 +import { gridConstraintDto, gridConstraintFromEvent, round2 } from "../lib/dto.js";
10 +import { eventsWhere } from "./events.js";
11 +
12 +export const ENERGY_NOTE = "National grid averages (renewable share, total generation) describe the country's electricity system, not the electricity a given facility contracts (PPAs, on-site generation, utility tariffs). Grid carbon intensity is null until a reliable public source is connected.";
13 +export const POWER_NOTE = "utilityCapacityMw / gridConnectionMw are supply-side figures (power available or contracted from the utility / grid) — never IT load and never comparable with itCapacityMw. Large loads list published site-scoped figures only (utility capacity, grid connection, planned load ≥ 200 MW on live projects). Grid constraints combine curated public reports (grid_constraints) with detected grid / utility / power-agreement events. Substations, power plants and transmission lines have no connector yet and are not shown; `utilities` lists operators whose name says energy / power / electric AND that appear as tenants or in events — empty otherwise.";
14 +
15 +export function energyContext(r: Row): EnergyContext {
16 + return { renewableShare: num(r.renewable_share), electricityTwh: num(r.electricity_twh), statsYear: num(r.stats_year), gridCarbonIntensity: null, sourceName: str(r.energy_source_name), sourceUrl: str(r.energy_source_url), note: ENERGY_NOTE };
17 +}
18 +
19 +export async function powerOverview(): Promise<PowerOverview> {
20 + const sql = pg();
21 + const known = knownMwAgg(sql), counted = countedAgg(sql);
22 + const [gcRows, gridEvents, powerEvents, loadsF, loadsC, loadsP, energy, utilities] = await Promise.all([
23 + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} order by g.effective_date desc nulls last, g.created_at desc limit 20`,
24 + eventsWhere(sql`e.event_type = any(${GRID_EVENT_TYPES})`, 30),
25 + eventsWhere(sql`e.event_type = any(${POWER_EVENT_TYPES})`, 30),
26 + sql<Row[]>`select f.id, f.slug, f.name, f.country_iso2, f.utility_capacity_mw, f.grid_connection_mw, m.id as met_id, m.slug as met_slug, m.name as met_name,
27 + (select s.name from provenance p join sources s on s.id = p.source_id where p.entity_type = 'facility' and p.entity_id = f.id and p.is_current and p.field in ('utilityCapacityMw', 'gridConnectionMw') order by p.is_winner desc, p.last_observed desc limit 1) as source_name,
28 + (select p.url from provenance p where p.entity_type = 'facility' and p.entity_id = f.id and p.is_current and p.field in ('utilityCapacityMw', 'gridConnectionMw') order by p.is_winner desc, p.last_observed desc limit 1) as url
29 + from facilities f left join metros m on m.id = f.metro_id
30 + where f.merged_into is null and (f.utility_capacity_mw is not null or f.grid_connection_mw is not null)
31 + order by greatest(coalesce(f.utility_capacity_mw, 0), coalesce(f.grid_connection_mw, 0)) desc limit 50`,
32 + sql<Row[]>`select k.predicate, k.value, k.url, s.name as source_name, f.id, f.slug, f.name, f.country_iso2, m.id as met_id, m.slug as met_slug, m.name as met_name
33 + from claims k join facilities f on f.id = k.subject_id and k.subject_type = 'facility' left join metros m on m.id = f.metro_id left join sources s on s.id = k.source_id
34 + where k.status = 'current' and k.predicate in ('grid_connection_mw', 'utility_capacity_mw') and k.value is not null and f.merged_into is null
35 + and ((k.predicate = 'grid_connection_mw' and f.grid_connection_mw is null) or (k.predicate = 'utility_capacity_mw' and f.utility_capacity_mw is null))
36 + order by k.value desc limit 50`,
37 + sql<Row[]>`select p.id, p.slug, p.name, p.country_iso2, p.planned_mw, p.source_url, m.id as met_id, m.slug as met_slug, m.name as met_name,
38 + (select s.name from provenance pr join sources s on s.id = pr.source_id where pr.entity_type = 'project' and pr.entity_id = p.id and pr.is_current and pr.field = 'plannedMw' order by pr.is_winner desc, pr.last_observed desc limit 1) as source_name
39 + from projects p left join metros m on m.id = p.metro_id where ${projectLive(sql)} and p.planned_mw >= 200 order by p.planned_mw desc limit 50`,
40 + sql<Row[]>`select c.iso2, c.name, c.slug, c.renewable_share, c.electricity_twh, c.stats_year,
41 + (select s.name from provenance p join sources s on s.id = p.source_id where p.entity_type = 'country' and p.entity_id = c.iso2 and p.is_current and p.field in ('renewableShare', 'electricityTwh') limit 1) as energy_source_name,
42 + (select p.url from provenance p where p.entity_type = 'country' and p.entity_id = c.iso2 and p.is_current and p.field in ('renewableShare', 'electricityTwh') limit 1) as energy_source_url,
43 + fa.n as facilities, fa.known_mw
44 + from countries c join (select f.country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw from ${facilityView(sql)} f where f.country_iso2 is not null group by 1) fa on fa.country_iso2 = c.iso2
45 + where c.renewable_share is not null or c.electricity_twh is not null
46 + order by fa.n desc limit 60`,
47 + sql<Row[]>`select o.id, o.slug, o.name, o.kind, (select count(*)::int from facility_tenants t where t.operator_id = o.id) as tenancies, (select count(*)::int from facilities f where f.operator_id = o.id and f.merged_into is null) as facility_count
48 + from operators o where o.name ~* '(energy|power|electric|utility|utilities)'
49 + and (exists (select 1 from facility_tenants t where t.operator_id = o.id) or exists (select 1 from events e where e.operator_id = o.id or (e.entity_type = 'operator' and e.entity_id = o.id)))
50 + order by tenancies desc, o.name limit 50`,
51 + ]);
52 + const gridConstraints = [...gcRows.map(gridConstraintDto), ...gridEvents.map(gridConstraintFromEvent)];
53 + const seen = new Set<string>();
54 + const deduped = gridConstraints.filter((g) => { const k = g.eventId ?? g.id; if (seen.has(k)) return false; seen.add(k); return true; }).slice(0, 30);
55 + const metro = (r: Row) => (r.met_id ? { id: reqStr(r.met_id), slug: reqStr(r.met_slug), name: reqStr(r.met_name) } : null);
56 + const largeLoads: PowerOverview["largeLoads"] = [];
57 + for (const r of loadsF) {
58 + const fac = { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) };
59 + const u = num(r.utility_capacity_mw), g = num(r.grid_connection_mw);
60 + if (u != null) largeLoads.push({ facility: fac, project: null, kind: "utility_capacity", mw: u, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.url) });
61 + if (g != null) largeLoads.push({ facility: fac, project: null, kind: "grid_connection", mw: g, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.url) });
62 + }
63 + for (const r of loadsC) largeLoads.push({ facility: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, project: null, kind: reqStr(r.predicate) === "grid_connection_mw" ? "grid_connection" : "utility_capacity", mw: num(r.value) ?? 0, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.url) });
64 + for (const r of loadsP) largeLoads.push({ facility: null, project: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, kind: "planned_load", mw: num(r.planned_mw) ?? 0, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.source_url) });
65 + largeLoads.sort((a, b) => b.mw - a.mw);
66 + return {
67 + gridConstraints: deduped,
68 + powerEvents,
69 + largeLoads: largeLoads.slice(0, 50),
70 + countryEnergy: energy.map((r) => ({ iso2: reqStr(r.iso2), name: reqStr(r.name), slug: reqStr(r.slug), energy: energyContext(r), knownMw: round2(num(r.known_mw)), facilities: int(r.facilities) })),
71 + utilities: utilities.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), facilityCount: int(r.facility_count), kind: str(r.kind) })),
72 + note: POWER_NOTE,
73 + };
74 +}
modified apps/api/src/repositories/projects.ts +162 −20
@@ -1,9 +1,11 @@
1 −import type { FacilityStatus, ProjectDetail, ProjectSummary, SourceRef } from "@dci/core";
2 −import { partialDateSortKey } from "@dci/core";
3 −import { pg, andAll, projectJoins, projectSummaryCols, page, type Fragment } from "../lib/sql.js";
1 +import type { ClaimDTO, EntityHistory, EventDTO, FacilityStatus, HistoryPoint, ProjectDetail, ProjectStageDTO, ProjectSummary, SourceRef } from "@dci/core";
2 +import { PROJECT_STAGES, parsePartialDate, partialDateSortKey, projectTransition } from "@dci/core";
3 +import { pg, andAll, projectJoins, projectSummaryCols, projectLive, page, likePattern, yearExpr, AI_LEVELS, type Fragment } from "../lib/sql.js";
4 4 import { int, num, reqStr, str, type Row } from "../lib/rows.js";
5 5 import { asStatus, projectSummary } from "../lib/dto.js";
6 6 import { findBySlugOrId } from "../lib/resolve.js";
7 +import { nearby } from "../lib/nearby.js";
8 +import { capacityHistory, claimsFor, dataQualityFor, entityHistory, eventsForSubject, provenanceAll } from "../lib/quality.js";
7 9 import { eventsForProject } from "./events.js";
8 10 import { provenanceFor } from "./facilities.js";
9 11 import { buildSourceHistory, documentVersionsFor, sourceIdsOf, sourceRefsFor } from "../lib/source-history.js";
@@ -15,7 +17,13 @@ export interface ProjectFilters {
15 17 metro?: string; // slug or id
16 18 facilityId?: string;
17 19 min_mw?: number;
20 + max_mw?: number;
18 21 ai?: boolean;
22 + project_class?: string[];
23 + evidence_level?: string[];
24 + expected_from?: number;
25 + expected_to?: number;
26 + announced_since?: string;
19 27 q?: string;
20 28 sort?: "updated" | "mw" | "announced" | "opening";
21 29 order?: "asc" | "desc";
@@ -23,17 +31,24 @@ export interface ProjectFilters {
23 31 per_page?: number;
24 32 }
25 33
26 −function conds(f: ProjectFilters): Fragment[] {
34 +export function projectConds(f: ProjectFilters): Fragment[] {
27 35 const sql = pg();
28 − const c: Fragment[] = [sql`true`];
36 + const c: Fragment[] = [projectLive(sql)];
29 37 if (f.status?.length) c.push(sql`p.status = any(${f.status})`);
30 38 if (f.country) c.push(sql`p.country_iso2 = ${f.country.toUpperCase()}`);
31 39 if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`);
32 40 if (f.metro) c.push(sql`(m.slug = ${f.metro} or m.id = ${f.metro})`);
33 41 if (f.facilityId) c.push(sql`p.facility_id = ${f.facilityId}`);
34 42 if (f.min_mw != null) c.push(sql`p.planned_mw >= ${f.min_mw}`);
35 − if (f.ai === true) c.push(sql`p.is_ai`);
36 − if (f.q) c.push(sql`(p.name ilike ${"%" + f.q + "%"} or similarity(p.name, ${f.q}) > 0.3)`);
43 + if (f.max_mw != null) c.push(sql`p.planned_mw <= ${f.max_mw}`);
44 + if (f.ai === true) c.push(sql`(p.is_ai or p.ai_evidence = any(${AI_LEVELS}))`);
45 + if (f.ai === false) c.push(sql`(not p.is_ai and p.ai_evidence <> all(${AI_LEVELS}))`);
46 + if (f.project_class?.length) c.push(sql`p.project_class = any(${f.project_class})`);
47 + if (f.evidence_level?.length) c.push(sql`p.evidence_level = any(${f.evidence_level})`);
48 + if (f.expected_from != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} >= ${f.expected_from}`);
49 + if (f.expected_to != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} <= ${f.expected_to}`);
50 + if (f.announced_since) c.push(sql`(p.announced_on >= ${f.announced_since} or (p.announced_on is null and p.created_at >= ${f.announced_since}::timestamptz))`);
51 + if (f.q) { const t = f.q.trim(); if (t) c.push(sql`(p.name ilike ${likePattern(t)} or similarity(p.name, ${t}) > 0.3 or o.name ilike ${likePattern(t)})`); }
37 52 return c;
38 53 }
39 54
@@ -54,7 +69,7 @@ export async function listProjects(f: ProjectFilters): Promise<{ items: ProjectS
54 69 const rows = await sql<Row[]>`
55 70 select ${projectSummaryCols(sql)}, count(*) over() as total
56 71 from projects p ${projectJoins(sql)}
57 − where ${andAll(sql, conds(f))}
72 + where ${andAll(sql, projectConds(f))}
58 73 order by ${order(f.sort, f.order)}
59 74 limit ${pg_.perPage} offset ${pg_.offset}`;
60 75 return { items: rows.map(projectSummary), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage };
@@ -65,29 +80,126 @@ export async function projectsForFacility(facilityId: string, operatorId: string
65 80 const sql = pg();
66 81 const rows = await sql<Row[]>`
67 82 select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)}
68 − where p.facility_id = ${facilityId}
69 − ${operatorId && metroId ? sql`or (p.operator_id = ${operatorId} and p.metro_id = ${metroId})` : sql``}
83 + where ${projectLive(sql)} and (p.facility_id = ${facilityId}
84 + ${operatorId && metroId ? sql`or (p.operator_id = ${operatorId} and p.metro_id = ${metroId})` : sql``})
70 85 order by (p.facility_id = ${facilityId}) desc, p.last_update desc limit ${limit}`;
71 86 return rows.map(projectSummary);
72 87 }
73 88
89 +/** Live projects matching a condition over `projects p` (+ projectJoins aliases o / pf / m). */
74 90 export async function projectsWhere(where: Fragment, limit = 10, orderBy?: Fragment): Promise<ProjectSummary[]> {
75 91 const sql = pg();
76 − const rows = await sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${where} order by ${orderBy ?? sql`p.last_update desc`} limit ${limit}`;
92 + const rows = await sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and (${where}) order by ${orderBy ?? sql`p.last_update desc`} limit ${limit}`;
77 93 return rows.map(projectSummary);
78 94 }
79 95
80 −export async function getProjectDetail(idOrSlug: string): Promise<{ detail: ProjectDetail; sources: SourceRef[] } | null> {
96 +/** Resolve a project by slug or id; merged projects redirect to their survivor, hidden projects are never served. */
97 +export async function resolveProjectRow(idOrSlug: string): Promise<Row | null> {
81 98 const sql = pg();
82 − const row = await findBySlugOrId("projects", idOrSlug);
99 + const base = await findBySlugOrId("projects", idOrSlug);
100 + if (!base) return null;
101 + let row = base;
102 + for (let hops = 0; hops < 5 && str(row.merged_into); hops++) {
103 + const next = (await sql<Row[]>`select * from projects where id = ${String(row.merged_into)} limit 1`)[0];
104 + if (!next) break;
105 + row = next;
106 + }
107 + if (row.hidden === true) return null;
108 + return row;
109 +}
110 +
111 +// ─── lifecycle stages ──────────────────────────────────────────────────────────────────────────────
112 +
113 +type Stage = ProjectStageDTO["stage"];
114 +const STAGE_COLUMN: Partial<Record<Stage, string>> = { announced: "announced_on", permitting: "permit_filed_on", approved: "approved_on", under_construction: "construction_started_on", operational: "opened_on" };
115 +const TIMELINE_STAGE: Array<[RegExp, Stage]> = [
116 + [/cancel/i, "cancelled"], [/delay|on hold|paused/i, "delayed"], [/open|launch|live|commission|energi[sz]ed/i, "operational"], [/partial/i, "partially_operational"],
117 + [/construction|ground|topped/i, "under_construction"], [/approv|consent|permit(ted)?\b|green/i, "approved"], [/planning|filed|permit|zoning|application/i, "permitting"],
118 + [/announce|unveil|reveal|plan/i, "announced"], [/propos/i, "proposed"], [/rumo/i, "rumored"],
119 +];
120 +const EVENT_STAGE: Record<string, Stage> = { construction_started: "under_construction", planning_filed: "permitting", planning_approved: "approved", project_delayed: "delayed", project_cancelled: "cancelled", facility_opened: "operational", project_announced: "announced" };
121 +
122 +interface StageEvidence { date: string; url: string | null; sourceName: string | null; eventId: string | null }
123 +
124 +function isStage(s: unknown): s is Stage { return typeof s === "string" && (PROJECT_STAGES as readonly string[]).includes(s); }
125 +
126 +export function buildStages(row: Row, timeline: Array<{ date: string; type: string; url: string | null; sourceName: string | null }>, events: EventDTO[]): ProjectStageDTO[] {
127 + const current = asStatus(row.status);
128 + const ev = new Map<Stage, StageEvidence>();
129 + const put = (stage: Stage, e: StageEvidence) => {
130 + if (!e.date || !/^\d{4}/.test(e.date)) return;
131 + const prev = ev.get(stage);
132 + // earliest dated evidence wins; a dated column beats a detected_at fallback
133 + if (!prev || partialDateSortKey(e.date) < partialDateSortKey(prev.date)) ev.set(stage, e);
134 + };
135 + for (const [stage, col] of Object.entries(STAGE_COLUMN) as Array<[Stage, string]>) { const d = str(row[col]); if (d) put(stage, { date: d, url: str(row.source_url), sourceName: null, eventId: null }); }
136 + if (current === "operational" && !ev.has("operational") && str(row.expected_opening) && partialDateSortKey(str(row.expected_opening)) <= Date.now()) put("operational", { date: String(row.expected_opening), url: str(row.source_url), sourceName: null, eventId: null });
137 + for (const t of timeline) { const stage = TIMELINE_STAGE.find(([re]) => re.test(t.type))?.[1]; if (stage) put(stage, { date: t.date, url: t.url, sourceName: t.sourceName, eventId: null }); }
138 + for (const e of events) {
139 + let stage: Stage | undefined = EVENT_STAGE[e.eventType];
140 + if (e.eventType === "project_status_changed" || e.eventType === "status_changed") { const to = typeof e.newValue === "object" && e.newValue ? (e.newValue as Record<string, unknown>).status ?? (e.newValue as Record<string, unknown>).to : e.newValue; stage = isStage(to) ? to : undefined; }
141 + if (stage) put(stage, { date: e.effectiveDate ?? e.detectedAt.slice(0, 10), url: e.url || null, sourceName: e.sourceName, eventId: e.id });
142 + }
143 + // `expansion` is an operational site growing — it sits at the operational stage of the lifecycle view
144 + const currentStage: Stage | null = current === "expansion" ? "operational" : isStage(current) ? current : null;
145 + return PROJECT_STAGES.map((stage) => {
146 + const isCurrent = currentStage === stage;
147 + const side = stage === "delayed" || stage === "cancelled";
148 + // a stage is reached when it is the current one, when dated evidence placed the project there, or when the pipeline
149 + // moved forward past it (a project under construction was announced); side branches only when current or dated
150 + const forward = !side && currentStage != null && (currentStage === "delayed" || currentStage === "cancelled" ? ["rumored", "proposed", "announced"].includes(stage) : projectTransition(stage, currentStage) === "forward");
151 + const reached = isCurrent || ev.has(stage) || forward;
152 + const e = ev.get(stage);
153 + return { stage, date: e?.date ?? null, reached, current: isCurrent, url: e?.url ?? null, sourceName: e?.sourceName ?? null, eventId: e?.eventId ?? null };
154 + });
155 +}
156 +
157 +function daysBetween(a: string | null, b: string | null): number | null {
158 + if (!a || !b) return null;
159 + const pa = parsePartialDate(a), pb = parsePartialDate(b);
160 + if (!pa || !pb) return null;
161 + const da = new Date(pa.length === 4 ? `${pa}-01-01` : pa.length === 7 ? `${pa}-01` : pa.slice(0, 10));
162 + const db = new Date(pb.length === 4 ? `${pb}-01-01` : pb.length === 7 ? `${pb}-01` : pb.slice(0, 10));
163 + if (Number.isNaN(da.getTime()) || Number.isNaN(db.getTime())) return null;
164 + return Math.round((db.getTime() - da.getTime()) / 86_400_000);
165 +}
166 +
167 +export function velocityDays(stages: ProjectStageDTO[]): ProjectDetail["velocityDays"] {
168 + const d = (s: Stage) => stages.find((x) => x.stage === s)?.date ?? null;
169 + return {
170 + announcedToPermitting: daysBetween(d("announced"), d("permitting")),
171 + permittingToApproval: daysBetween(d("permitting"), d("approved")),
172 + approvalToConstruction: daysBetween(d("approved"), d("under_construction")),
173 + constructionToOpening: daysBetween(d("under_construction"), d("operational")),
174 + announcedToConstruction: daysBetween(d("announced"), d("under_construction")),
175 + };
176 +}
177 +
178 +const COMPLETENESS_COLS = ["operator_id", "country_iso2", "lat", "planned_mw", "expected_opening", "announced_on", "description"];
179 +export function projectCompleteness(row: Row): number {
180 + return Math.round((COMPLETENESS_COLS.filter((c) => row[c] != null && row[c] !== "").length / COMPLETENESS_COLS.length) * 100);
181 +}
182 +
183 +function investmentPoints(claims: ClaimDTO[]): HistoryPoint[] {
184 + return claims.filter((c) => /usd$/.test(c.predicate) && c.status !== "rejected").map((c) => ({ date: c.publishedAt && /^\d{4}/.test(c.publishedAt) ? c.publishedAt : c.firstObserved, field: "investmentUsd", predicate: c.predicate, value: c.value ?? c.valueText, sourceId: c.sourceId, sourceName: c.sourceName, sourceKind: c.sourceKind, url: c.url, claimId: c.id, kind: "claim" as const }));
185 +}
186 +
187 +export async function getProjectDetail(idOrSlug: string, opts: { radiusKm?: number } = {}): Promise<{ detail: ProjectDetail; sources: SourceRef[] } | null> {
188 + const sql = pg();
189 + const row = await resolveProjectRow(idOrSlug);
83 190 if (!row) return null;
84 191 const id = String(row.id);
85 − const [sumRows, timelineRows, provenance, events, versions] = await Promise.all([
192 + const lat = num(row.lat), lng = num(row.lng);
193 + const [sumRows, timelineRows, provenance, events, versions, campusRows, claims, nearbyInfra, dataQuality] = await Promise.all([
86 194 sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where p.id = ${id}`,
87 195 sql<Row[]>`select t.event_date, t.event_type, t.description, t.url, s.name as source_name from project_timeline t left join sources s on s.id = t.source_id where t.project_id = ${id}`,
88 196 provenanceFor("project", id),
89 197 eventsForProject(id, 50),
90 198 documentVersionsFor("project", id),
199 + row.campus_id ? sql<Row[]>`select id, slug, name from campuses where id = ${String(row.campus_id)}` : Promise.resolve([] as Row[]),
200 + claimsFor("project", id),
201 + lat != null && lng != null ? nearby({ lat, lng, radiusKm: opts.radiusKm ?? 25, excludeProjectId: id, limitPerType: 50 }) : Promise.resolve(null),
202 + dataQualityFor("project", id, projectCompleteness(row), null),
91 203 ]);
92 204 const summary = projectSummary(sumRows[0] ?? row);
93 205 const timeline = timelineRows
@@ -95,22 +207,51 @@ export async function getProjectDetail(idOrSlug: string): Promise<{ detail: Proj
95 207 .sort((a, b) => partialDateSortKey(a.date) - partialDateSortKey(b.date));
96 208 const statusHistory = events
97 209 .filter((e) => e.eventType === "project_status_changed" || e.eventType === "status_changed")
98 − .map((e) => ({ date: e.effectiveDate ?? e.detectedAt, from: e.oldValue != null ? asStatus(e.oldValue) : null, to: asStatus(e.newValue), url: e.url || null }))
210 + .map((e) => ({ date: e.effectiveDate ?? e.detectedAt, from: e.oldValue != null ? asStatus(typeof e.oldValue === "object" ? (e.oldValue as Record<string, unknown>).status : e.oldValue) : null, to: asStatus(typeof e.newValue === "object" && e.newValue ? (e.newValue as Record<string, unknown>).status : e.newValue), url: e.url || null }))
99 211 .sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0));
212 + const stages = buildStages(row, timeline, events);
213 + const relatedEvents = events.filter((e) => e.project && e.project.id === id && !(e.entityType === "project" && e.entityId === id));
214 + const history = [...capacityHistory(provenance, claims, events), ...investmentPoints(claims)].sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0));
100 215 const detail: ProjectDetail = {
101 216 ...summary,
102 217 description: str(row.description),
103 218 sourceUrl: str(row.source_url),
219 + campus: campusRows[0] ? { id: reqStr(campusRows[0].id), slug: reqStr(campusRows[0].slug), name: reqStr(campusRows[0].name) } : null,
220 + stages,
221 + velocityDays: velocityDays(stages),
104 222 timeline,
105 223 statusHistory,
224 + claims,
225 + history,
226 + nearbyInfrastructure: nearbyInfra,
227 + relatedEvents,
228 + dataQuality,
106 229 provenance,
107 230 events,
108 231 sourceHistory: buildSourceHistory(provenance, events, versions),
109 232 };
110 − const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions));
233 + const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions, claims));
111 234 return { detail, sources };
112 235 }
113 236
237 +/** /projects/:slug/history */
238 +export async function getProjectHistory(idOrSlug: string): Promise<{ history: EntityHistory; sources: SourceRef[] } | null> {
239 + const row = await resolveProjectRow(idOrSlug);
240 + if (!row) return null;
241 + const id = String(row.id);
242 + const [prov, claims, events] = await Promise.all([provenanceAll("project", id), claimsFor("project", id), eventsForSubject("project", id)]);
243 + return { history: entityHistory("project", id, prov, claims, events), sources: await sourceRefsFor(sourceIdsOf(prov, claims, events)) };
244 +}
245 +
246 +/** /projects/:slug/claims */
247 +export async function getProjectClaims(idOrSlug: string, opts: { status?: string[]; predicate?: string } = {}): Promise<{ id: string; claims: ClaimDTO[]; sources: SourceRef[] } | null> {
248 + const row = await resolveProjectRow(idOrSlug);
249 + if (!row) return null;
250 + const id = String(row.id);
251 + const claims = await claimsFor("project", id, opts);
252 + return { id, claims, sources: await sourceRefsFor(sourceIdsOf(claims)) };
253 +}
254 +
114 255 export interface PipelineAggregates {
115 256 byStatus: Array<{ status: FacilityStatus; count: number; mw: number | null; investmentUsd: number | null }>;
116 257 byYear: Array<{ year: string; count: number; mw: number | null }>;
@@ -120,11 +261,12 @@ export interface PipelineAggregates {
120 261
121 262 export async function projectPipeline(): Promise<PipelineAggregates> {
122 263 const sql = pg();
264 + const live = projectLive(sql);
123 265 const [st, yr, co, tot] = await Promise.all([
124 − sql<Row[]>`select status, count(*)::int as n, sum(planned_mw)::float as mw, sum(investment_usd)::float as inv from projects group by status order by n desc`,
125 − sql<Row[]>`select left(expected_opening, 4) as year, count(*)::int as n, sum(planned_mw)::float as mw from projects where expected_opening ~ '^\\d{4}' and status not in ('cancelled','closed') group by 1 order by 1`,
126 − sql<Row[]>`select p.country_iso2, c.name, c.slug, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p left join countries c on c.iso2 = p.country_iso2 where p.country_iso2 is not null group by 1,2,3 order by n desc, mw desc nulls last limit 15`,
127 − sql<Row[]>`select count(*)::int as n, sum(planned_mw)::float as mw, sum(investment_usd)::float as inv, count(*) filter (where is_ai)::int as ai from projects`,
266 + sql<Row[]>`select p.status, count(*)::int as n, sum(p.planned_mw)::float as mw, sum(p.investment_usd)::float as inv from projects p where ${live} group by p.status order by n desc`,
267 + sql<Row[]>`select left(p.expected_opening, 4) as year, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${live} and p.expected_opening ~ '^\\d{4}' and p.status not in ('cancelled','closed') group by 1 order by 1`,
268 + sql<Row[]>`select p.country_iso2, c.name, c.slug, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p left join countries c on c.iso2 = p.country_iso2 where ${live} and p.country_iso2 is not null group by 1,2,3 order by n desc, mw desc nulls last limit 15`,
269 + sql<Row[]>`select count(*)::int as n, sum(p.planned_mw)::float as mw, sum(p.investment_usd)::float as inv, count(*) filter (where p.is_ai or p.ai_evidence = any(${AI_LEVELS}))::int as ai from projects p where ${live}`,
128 270 ]);
129 271 return {
130 272 byStatus: st.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: num(r.mw), investmentUsd: num(r.inv) })),
added apps/api/src/repositories/pulse.ts +93 −0
@@ -0,0 +1,93 @@
1 +/**
2 + * Global infrastructure pulse — deterministic counts over a time window, computed from events and entity dates.
3 + * Nothing here is scored or weighted: every figure is a count or a sum of published MW figures.
4 + */
5 +import type { Pulse } from "@dci/core";
6 +import { pg, eventCols, eventJoins, projectLive, type Fragment, type Sql } from "../lib/sql.js";
7 +import { int, num, reqStr, str, type Row } from "../lib/rows.js";
8 +import { round2 } from "../lib/dto.js";
9 +import { toEventDtos } from "./events.js";
10 +
11 +export type PulseWindow = Pulse["window"];
12 +const WINDOW_INTERVAL: Record<PulseWindow, string> = { "24h": "24 hours", "7d": "7 days", "30d": "30 days" };
13 +
14 +export const PULSE_METHODOLOGY = "Deterministic counts over the window: events with review_status ≠ rejected; projects by created_at / construction_started_on; facilities by opened_on / first_seen; MW figures are sums of published site-scoped values on the records (no estimates). operatorsNewMarkets = operators whose first facility or project in a metro (or country when no metro) falls inside the window and who had none there before. Hidden (false-positive) and merged projects are excluded.";
15 +
16 +function partialDate(sql: Sql, col: Fragment, fallback: Fragment): Fragment {
17 + return sql`(case when ${col} ~ '^\\d{4}-\\d{2}-\\d{2}' then ${col}::date when ${col} ~ '^\\d{4}-\\d{2}$' then (${col} || '-01')::date when ${col} ~ '^\\d{4}$' then (${col} || '-01-01')::date else ${fallback} end)`;
18 +}
19 +
20 +export async function pulse(window: PulseWindow = "24h"): Promise<Pulse> {
21 + const sql = pg();
22 + const since = sql`(now() - ${WINDOW_INTERVAL[window]}::interval)`;
23 + const sinceDate = sql`(now() - ${WINDOW_INTERVAL[window]}::interval)::date`;
24 + const conDate = partialDate(sql, sql`p.construction_started_on`, sql`null::date`);
25 + const openedDate = partialDate(sql, sql`f.opened_on`, sql`null::date`);
26 + const fFirst = sql`coalesce(${partialDate(sql, sql`f.opened_on`, sql`null::date`)}, f.first_seen::date)`;
27 + const pFirst = partialDate(sql, sql`p.announced_on`, sql`p.created_at::date`);
28 + const [ev, pr, fa, markets, major, byCountry, byOperator, sinceRow] = await Promise.all([
29 + sql<Row[]>`select count(*)::int as total,
30 + count(*) filter (where e.event_type = 'facility_opened')::int as opened_ev,
31 + count(*) filter (where e.event_type = 'cloud_region_announced')::int as cloud_ann,
32 + count(*) filter (where e.event_type = 'power_agreement')::int as power,
33 + count(*) filter (where e.event_type in ('grid_constraint', 'utility_event'))::int as grid,
34 + count(*) filter (where e.event_type = 'acquisition')::int as acq,
35 + count(*) filter (where e.event_type = 'investment_announced')::int as fin,
36 + count(*) filter (where e.event_type = 'facility_discovered')::int as discovered,
37 + count(*) filter (where e.event_type in ('capacity_changed', 'planned_capacity_changed'))::int as cap,
38 + count(distinct e.project_id) filter (where e.project_id is not null and (e.event_type = 'construction_started' or (e.event_type = 'project_status_changed' and e.new_value::text ilike '%under_construction%')))::int as con_ev
39 + from events e where e.review_status <> 'rejected' and e.detected_at >= ${since}`,
40 + sql<Row[]>`select
41 + count(*) filter (where p.created_at >= ${since})::int as new_n, sum(p.planned_mw) filter (where p.created_at >= ${since})::float as new_mw,
42 + count(*) filter (where ${conDate} >= ${sinceDate})::int as con_n, sum(p.planned_mw) filter (where ${conDate} >= ${sinceDate})::float as con_mw
43 + from projects p where ${projectLive(sql)}`,
44 + sql<Row[]>`select count(*) filter (where ${openedDate} >= ${sinceDate} and f.status in ('operational', 'partially_operational', 'expansion'))::int as opened, count(*) filter (where f.first_seen >= ${since})::int as indexed from facilities f where f.merged_into is null`,
45 + sql<Row[]>`
46 + with entries as (
47 + select f.operator_id, f.metro_id, f.country_iso2, ${fFirst} as d from facilities f where f.merged_into is null and f.operator_id is not null and (f.metro_id is not null or f.country_iso2 is not null)
48 + union all
49 + select p.operator_id, p.metro_id, p.country_iso2, ${pFirst} from projects p where ${projectLive(sql)} and p.operator_id is not null and (p.metro_id is not null or p.country_iso2 is not null)
50 + ), firsts as (
51 + select operator_id, metro_id, country_iso2, min(d) as first_d from entries group by 1, 2, 3
52 + )
53 + select o.id, o.slug, o.name, m.id as met_id, m.slug as met_slug, m.name as met_name, fs.country_iso2, fs.first_d
54 + from firsts fs join operators o on o.id = fs.operator_id left join metros m on m.id = fs.metro_id
55 + where fs.first_d >= ${sinceDate}
56 + and not exists (select 1 from firsts f2 where f2.operator_id = fs.operator_id and f2.first_d < ${sinceDate} and (f2.metro_id is not distinct from fs.metro_id) and (fs.metro_id is not null or f2.country_iso2 is not distinct from fs.country_iso2))
57 + order by fs.first_d desc, o.name limit 25`,
58 + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.review_status <> 'rejected' and e.detected_at >= ${since} and e.significance >= 75 order by e.significance desc, e.detected_at desc limit 15`,
59 + sql<Row[]>`select c.iso2, c.name, c.slug, coalesce(ev.n, 0) as events, coalesce(pj.n, 0) as new_projects
60 + from countries c
61 + left join (select country_iso2, count(*)::int as n from events where review_status <> 'rejected' and detected_at >= ${since} and country_iso2 is not null group by 1) ev on ev.country_iso2 = c.iso2
62 + left join (select p.country_iso2, count(*)::int as n from projects p where ${projectLive(sql)} and p.created_at >= ${since} and p.country_iso2 is not null group by 1) pj on pj.country_iso2 = c.iso2
63 + where coalesce(ev.n, 0) > 0 or coalesce(pj.n, 0) > 0 order by events desc, new_projects desc, c.name limit 15`,
64 + sql<Row[]>`select o.id, o.slug, o.name, coalesce(ev.n, 0) as events, coalesce(pj.n, 0) as new_projects
65 + from operators o
66 + left join (select operator_id, count(*)::int as n from events where review_status <> 'rejected' and detected_at >= ${since} and operator_id is not null group by 1) ev on ev.operator_id = o.id
67 + left join (select p.operator_id, count(*)::int as n from projects p where ${projectLive(sql)} and p.created_at >= ${since} and p.operator_id is not null group by 1) pj on pj.operator_id = o.id
68 + where coalesce(ev.n, 0) > 0 or coalesce(pj.n, 0) > 0 order by events desc, new_projects desc, o.name limit 15`,
69 + sql<Row[]>`select ${since} as since`,
70 + ]);
71 + const e = ev[0] ?? {}, p = pr[0] ?? {}, f = fa[0] ?? {};
72 + return {
73 + window,
74 + since: new Date(String(sinceRow[0]?.since ?? Date.now())).toISOString(),
75 + newProjects: int(p.new_n),
76 + projectsEnteredConstruction: Math.max(int(p.con_n), int(e.con_ev)),
77 + facilitiesOpened: Math.max(int(f.opened), int(e.opened_ev)),
78 + newlyAnnouncedMw: round2(num(p.new_mw)),
79 + constructionStartedMw: round2(num(p.con_mw)),
80 + operatorsNewMarkets: markets.map((r) => ({ operator: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, market: r.met_id ? { id: reqStr(r.met_id), slug: reqStr(r.met_slug), name: reqStr(r.met_name) } : null, countryIso2: str(r.country_iso2) })),
81 + cloudRegionsAnnounced: int(e.cloud_ann),
82 + powerAgreements: int(e.power),
83 + gridConstraintEvents: int(e.grid),
84 + acquisitions: int(e.acq),
85 + financingEvents: int(e.fin),
86 + newFacilitiesIndexed: Math.max(int(f.indexed), int(e.discovered)),
87 + capacityChanges: int(e.cap),
88 + eventsTotal: int(e.total),
89 + majorEvents: await toEventDtos(major),
90 + byCountry: byCountry.map((r) => ({ iso2: reqStr(r.iso2), name: reqStr(r.name), slug: reqStr(r.slug), events: int(r.events), newProjects: int(r.new_projects) })),
91 + byOperator: byOperator.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), events: int(r.events), newProjects: int(r.new_projects) })),
92 + };
93 +}
modified apps/api/src/repositories/search.ts +41 −7
@@ -1,10 +1,11 @@
1 1 import type { SearchHit, SearchResponse } from "@dci/core";
2 −import { pg, likePattern } from "../lib/sql.js";
2 +import { pg, likePattern, projectLive } from "../lib/sql.js";
3 3 import { int, num, reqStr, str, type Row } from "../lib/rows.js";
4 4 import { hrefFor } from "../lib/resolve.js";
5 −import { hasFilters, interpretQuery } from "../search/interpret.js";
5 +import { hasFilters, interpretQuery, type InterpretOptions } from "../search/interpret.js";
6 6 import { operatorCandidates } from "./operators.js";
7 7 import { listFacilities, searchCities } from "./facilities.js";
8 +import { listProjects } from "./projects.js";
8 9 import { searchCountries } from "./countries.js";
9 10
10 11 const MAX_HITS = 30;
@@ -57,7 +58,7 @@ async function projectHits(text: string, limit = 5): Promise<SearchHit[]> {
57 58 const rows = await sql<Row[]>`
58 59 select p.id, p.slug, p.name, p.status, p.country_iso2, p.planned_mw, o.name as op_name, greatest(similarity(p.name, ${text}), case when p.name ilike ${likePattern(text)} then 0.7 else 0 end) as score
59 60 from projects p left join operators o on o.id = p.operator_id
60 − where p.name % ${text} or p.name ilike ${likePattern(text)}
61 + where ${projectLive(sql)} and (p.name % ${text} or p.name ilike ${likePattern(text)})
61 62 order by score desc, p.last_update desc limit ${limit}`;
62 63 return rows.map((r) => ({ type: "project", id: reqStr(r.id), slug: reqStr(r.slug), title: reqStr(r.name), subtitle: [str(r.op_name), str(r.status)?.replace(/_/g, " "), num(r.planned_mw) != null ? `${num(r.planned_mw)} MW` : null, str(r.country_iso2)].filter(Boolean).join(" · ") || null, href: hrefFor("project", reqStr(r.slug)), score: Math.min(1, 0.45 + 0.4 * (num(r.score) ?? 0)) }));
63 64 }
@@ -71,11 +72,30 @@ async function ixpHits(text: string, limit = 3): Promise<SearchHit[]> {
71 72 return rows.map((r) => ({ type: "ixp", id: reqStr(r.id), slug: reqStr(r.slug), title: reqStr(r.name), subtitle: [str(r.name_long), str(r.city), str(r.country_iso2)].filter(Boolean).join(" · ") || null, href: hrefFor("ixp", reqStr(r.slug)), score: Math.min(1, 0.4 + 0.4 * (num(r.score) ?? 0)) }));
72 73 }
73 74
75 +/** Metro candidates for query interpretation (trigram on name, substring on aliases). */
76 +async function metroCandidates(q: string, limit = 5): Promise<NonNullable<InterpretOptions["metros"]>> {
77 + const sql = pg();
78 + const rows = await sql<Row[]>`
79 + select m.slug, m.name, m.aliases from metros m
80 + where position(lower(m.name) in lower(${q})) > 0 or (m.name % ${q} and similarity(m.name, ${q}) >= 0.5)
81 + or exists (select 1 from unnest(m.aliases) a where length(a) >= 3 and position(lower(a) in lower(${q})) > 0)
82 + order by similarity(m.name, ${q}) desc limit ${limit}`;
83 + return rows.map((r) => ({ slug: reqStr(r.slug), name: reqStr(r.name), aliases: Array.isArray(r.aliases) ? (r.aliases as string[]) : [] }));
84 +}
85 +
86 +/** Facility ids behind a search response (hits + list), for the envelope `sources`. */
87 +export function facilityIdsOf(res: SearchResponse): string[] {
88 + const ids = new Set<string>();
89 + for (const h of res.hits) if (h.type === "facility") ids.add(h.id);
90 + for (const f of res.facilities?.items ?? []) ids.add(f.id);
91 + return [...ids];
92 +}
93 +
74 94 export async function search(q: string): Promise<SearchResponse> {
75 95 const query = q.replace(/\s+/g, " ").trim().slice(0, 200);
76 96 if (!query) return { query: "", hits: [] };
77 − const ops = await operatorCandidates(query, 5).catch(() => []);
78 − const interpreted = interpretQuery(query, { operators: ops });
97 + const [ops, metros] = await Promise.all([operatorCandidates(query, 5).catch(() => []), metroCandidates(query, 5).catch(() => [])]);
98 + const interpreted = interpretQuery(query, { operators: ops, metros });
79 99 const text = interpreted.text ?? "";
80 100 const hits: SearchHit[] = [];
81 101 const tasks: Promise<SearchHit[]>[] = [];
@@ -87,15 +107,29 @@ export async function search(q: string): Promise<SearchResponse> {
87 107 // structured hits for the interpreted filters (country / operator)
88 108 if (interpreted.countryIso2) tasks.push(searchCountries(interpreted.countryIso2, 1).then((cs) => cs.map((c) => ({ type: "country" as const, id: c.iso2, slug: c.slug, title: c.name, subtitle: `${c.facilityCount} facilities`, href: hrefFor("country", c.slug), score: 0.99, meta: { facilityCount: c.facilityCount } }))));
89 109 if (interpreted.operator) { const op = ops.find((o) => o.slug === interpreted.operator); if (op) hits.push({ type: "operator", id: op.id, slug: op.slug, title: op.name, subtitle: "Operator", href: hrefFor("operator", op.slug), score: 0.97 }); }
110 + if (interpreted.metro) { const m = metros.find((x) => x.slug === interpreted.metro); if (m) hits.push({ type: "metro", id: `metro:${m.slug}`, slug: m.slug, title: m.name, subtitle: "Market", href: hrefFor("metro", m.slug), score: 0.96 }); }
90 111 const results = await Promise.allSettled(tasks);
91 112 for (const r of results) if (r.status === "fulfilled") hits.push(...r.value);
92 113 // dedupe by type+id
93 114 const seen = new Set<string>();
94 115 const merged = hits.filter((h) => { const k = `${h.type}:${h.id}`; if (seen.has(k)) return false; seen.add(k); return true; }).sort((a, b) => b.score - a.score).slice(0, MAX_HITS);
95 116 const response: SearchResponse = { query, interpreted, hits: merged.map((h) => ({ ...h, score: Math.round(h.score * 1000) / 1000 })) };
117 + const yearFrom = interpreted.year != null && interpreted.yearOp !== "before" ? interpreted.year : undefined;
118 + const yearTo = interpreted.year != null && interpreted.yearOp !== "after" ? interpreted.year : undefined;
119 + if (interpreted.entity === "project") {
120 + // project intent: rank project rows instead of facilities (planned MW, status, country, AI, expected year)
121 + const pl = await listProjects({ q: text || undefined, country: interpreted.countryIso2, operator: interpreted.operator, metro: interpreted.metro, status: interpreted.status ? [interpreted.status] : undefined, min_mw: interpreted.minMw, ai: interpreted.ai ? true : undefined, per_page: 20, page: 1, sort: text ? "updated" : "mw", order: "desc" }).catch(() => ({ items: [], total: 0 }));
122 + const filtered = pl.items.filter((p) => (interpreted.maxMw == null || p.plannedMw == null || p.plannedMw <= interpreted.maxMw) && (yearFrom == null || !p.expectedOpening || Number(p.expectedOpening.slice(0, 4)) >= yearFrom) && (yearTo == null || !p.expectedOpening || Number(p.expectedOpening.slice(0, 4)) <= yearTo));
123 + const seenP = new Set(response.hits.filter((h) => h.type === "project").map((h) => h.id));
124 + for (const p of filtered) {
125 + if (seenP.has(p.id)) continue;
126 + response.hits.push({ type: "project", id: p.id, slug: p.slug, title: p.name, subtitle: [p.operator?.name, p.status.replace(/_/g, " "), p.plannedMw != null ? `${p.plannedMw} MW` : null, p.expectedOpening ? `opening ${p.expectedOpening}` : null, p.countryIso2].filter(Boolean).join(" · ") || null, href: hrefFor("project", p.slug), score: 0.9, meta: { status: p.status, plannedMw: p.plannedMw, countryIso2: p.countryIso2 } });
127 + }
128 + response.hits = response.hits.sort((a, b) => b.score - a.score).slice(0, MAX_HITS);
129 + return response;
130 + }
96 131 if (hasFilters(interpreted) || text) {
97 − const fl = await listFacilities({ q: text || undefined, country: interpreted.countryIso2 ? [interpreted.countryIso2] : undefined, operator: interpreted.operator, status: interpreted.status ? [interpreted.status] : undefined, type: interpreted.facilityType && interpreted.facilityType !== "cloud_region" ? [interpreted.facilityType] : undefined, ai: interpreted.facilityType === "ai" ? true : undefined, min_mw: interpreted.minMw, per_page: 20, page: 1, sort: text ? "completeness" : "mw", order: "desc" });
98 − // when the type is "ai" we already matched is_ai OR type = ai above; drop the strict type filter result if it is empty and retry without it
132 + const fl = await listFacilities({ q: text || undefined, country: interpreted.countryIso2 ? [interpreted.countryIso2] : undefined, operator: interpreted.operator, metro: interpreted.metro, status: interpreted.status ? [interpreted.status] : undefined, type: interpreted.facilityType && interpreted.facilityType !== "cloud_region" && interpreted.facilityType !== "ai" ? [interpreted.facilityType] : undefined, ai: interpreted.ai || interpreted.facilityType === "ai" ? true : undefined, min_mw: interpreted.minMw, max_mw: interpreted.maxMw, opened_from: yearFrom, opened_to: yearTo, per_page: 20, page: 1, sort: text ? "completeness" : "mw", order: "desc" });
99 133 response.facilities = { total: fl.total, items: fl.items };
100 134 }
101 135 return response;
added apps/api/src/repositories/time-machine.ts +43 −0
@@ -0,0 +1,43 @@
1 +/**
2 + * /time-machine — yearly frames of the index as it would have looked from published opening dates. Frames only
3 + * include facilities WITH an opening date, so every year is a lower bound; coverage is stated.
4 + */
5 +import type { TimeMachine, TimeMachineFrame } from "@dci/core";
6 +import { pg, facilityView, knownMwAgg, countedAgg, projectLive, yearExpr } from "../lib/sql.js";
7 +import { int, num, type Row } from "../lib/rows.js";
8 +import { round2, share } from "../lib/dto.js";
9 +
10 +export const TIME_MACHINE_NOTE = "Frames only include facilities with a published opening date (coverage given); facilities without a date are absent from every frame, so early years are lower bounds. knownMw is the cumulative known operational MW of dated facilities (containment-aware, published figures only). announced / construction count live projects by announcement / construction-start year.";
11 +export const RELIABLE_MIN_FACILITIES = 20;
12 +
13 +export async function timeMachine(): Promise<TimeMachine> {
14 + const sql = pg();
15 + const known = knownMwAgg(sql), counted = countedAgg(sql);
16 + const openedYear = yearExpr(sql, sql`f.opened_on`);
17 + const [byYear, cov, ann, con] = await Promise.all([
18 + sql<Row[]>`select ${openedYear} as year, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational', 'partially_operational', 'expansion'))::float as mw from ${facilityView(sql)} f where f.opened_on ~ '^\\d{4}' group by 1 order by 1`,
19 + sql<Row[]>`select count(*) filter (where ${counted})::int as n, count(*) filter (where ${counted} and f.opened_on ~ '^\\d{4}')::int as dated from ${facilityView(sql)} f`,
20 + sql<Row[]>`select ${yearExpr(sql, sql`p.announced_on`)} as year, count(*)::int as n from projects p where ${projectLive(sql)} and p.announced_on ~ '^\\d{4}' group by 1`,
21 + sql<Row[]>`select ${yearExpr(sql, sql`p.construction_started_on`)} as year, count(*)::int as n from projects p where ${projectLive(sql)} and p.construction_started_on ~ '^\\d{4}' group by 1`,
22 + ]);
23 + const opened = new Map<number, { n: number; mw: number | null }>();
24 + for (const r of byYear) { const y = int(r.year); if (y) opened.set(y, { n: int(r.n), mw: num(r.mw) }); }
25 + const annMap = new Map<number, number>(ann.map((r) => [int(r.year), int(r.n)]));
26 + const conMap = new Map<number, number>(con.map((r) => [int(r.year), int(r.n)]));
27 + const nowYear = new Date().getUTCFullYear();
28 + const years = [...opened.keys(), ...annMap.keys(), ...conMap.keys()].filter((y) => y >= 1950 && y <= nowYear + 10);
29 + const frames: TimeMachineFrame[] = [];
30 + let earliestReliableYear = 0;
31 + if (years.length) {
32 + const from = Math.min(...years);
33 + const to = Math.max(nowYear, ...years.filter((y) => y <= nowYear));
34 + let cumN = 0, cumMw = 0, sawMw = false;
35 + for (let y = from; y <= to; y++) {
36 + const o = opened.get(y);
37 + if (o) { cumN += o.n; if (o.mw != null) { cumMw += o.mw; sawMw = true; } }
38 + if (!earliestReliableYear && cumN >= RELIABLE_MIN_FACILITIES) earliestReliableYear = y;
39 + frames.push({ year: y, facilities: cumN, knownMw: sawMw ? round2(cumMw) : null, announced: annMap.get(y) ?? 0, construction: conMap.get(y) ?? 0, opened: o?.n ?? 0 });
40 + }
41 + }
42 + return { frames, earliestReliableYear: earliestReliableYear || (frames.at(-1)?.year ?? nowYear), openingDateCoverage: share(int(cov[0]?.dated), int(cov[0]?.n)), note: TIME_MACHINE_NOTE };
43 +}
added apps/api/src/repositories/watchlist.ts +99 −0
@@ -0,0 +1,99 @@
1 +/**
2 + * Private watchlists — no accounts. The owner is an opaque random token stored in the httpOnly `dci_watch` cookie;
3 + * rows are only ever read / deleted through that token.
4 + */
5 +import { randomBytes } from "node:crypto";
6 +import type { EventDTO, WatchlistItem } from "@dci/core";
7 +import { newId } from "@dci/core";
8 +import { pg, eventCols, eventJoins, page } from "../lib/sql.js";
9 +import { reqIso, reqStr, str, type Row } from "../lib/rows.js";
10 +import { findBySlugOrId, findCountry, resolveEntityRefs } from "../lib/resolve.js";
11 +import { toEventDtos } from "./events.js";
12 +
13 +export const WATCH_COOKIE = "dci_watch";
14 +export const WATCH_MAX_ITEMS = 200;
15 +export const WATCH_ENTITY_TYPES = ["operator", "metro", "country", "project", "facility"] as const;
16 +export type WatchEntityType = (typeof WATCH_ENTITY_TYPES)[number];
17 +
18 +export function newWatchToken(): string {
19 + return randomBytes(24).toString("base64url");
20 +}
21 +
22 +export function parseCookies(header: string | undefined): Record<string, string> {
23 + const out: Record<string, string> = {};
24 + if (!header) return out;
25 + for (const part of header.split(";")) {
26 + const i = part.indexOf("=");
27 + if (i < 0) continue;
28 + const k = part.slice(0, i).trim();
29 + const v = part.slice(i + 1).trim();
30 + if (k && /^[A-Za-z0-9_-]{8,128}$/.test(v)) out[k] = v;
31 + }
32 + return out;
33 +}
34 +
35 +const TABLE: Record<WatchEntityType, "operators" | "metros" | "projects" | "facilities" | null> = { operator: "operators", metro: "metros", project: "projects", facility: "facilities", country: null };
36 +
37 +/** Resolve a slug or id to the canonical entity id (null when unknown). */
38 +export async function resolveWatchEntity(type: WatchEntityType, idOrSlug: string): Promise<string | null> {
39 + if (type === "country") { const c = await findCountry(idOrSlug); return c ? reqStr(c.iso2) : null; }
40 + const row = await findBySlugOrId(TABLE[type]!, idOrSlug);
41 + if (!row) return null;
42 + if (type === "project" && (row.hidden === true || row.merged_into)) return str(row.merged_into) ?? null;
43 + return reqStr(row.id);
44 +}
45 +
46 +async function toItems(rows: Row[]): Promise<WatchlistItem[]> {
47 + const refs = await resolveEntityRefs(rows.map((r) => ({ type: reqStr(r.entity_type), id: reqStr(r.entity_id) })));
48 + return rows.map((r) => {
49 + const ref = refs.get(`${reqStr(r.entity_type)}:${reqStr(r.entity_id)}`);
50 + return { id: reqStr(r.id), entityType: reqStr(r.entity_type) as WatchEntityType, entityId: reqStr(r.entity_id), slug: ref?.slug ?? reqStr(r.entity_id), name: ref?.name ?? reqStr(r.entity_id), createdAt: reqIso(r.created_at) };
51 + });
52 +}
53 +
54 +export async function listWatchlist(token: string): Promise<WatchlistItem[]> {
55 + const sql = pg();
56 + const rows = await sql<Row[]>`select * from watchlists where owner_token = ${token} order by created_at desc limit ${WATCH_MAX_ITEMS}`;
57 + return toItems(rows);
58 +}
59 +
60 +export async function addWatch(token: string, type: WatchEntityType, entityId: string): Promise<{ item: WatchlistItem; created: boolean } | { error: "full" }> {
61 + const sql = pg();
62 + const n = await sql<Row[]>`select count(*)::int as n from watchlists where owner_token = ${token}`;
63 + const existing = await sql<Row[]>`select * from watchlists where owner_token = ${token} and entity_type = ${type} and entity_id = ${entityId}`;
64 + if (existing[0]) return { item: (await toItems(existing))[0]!, created: false };
65 + if (Number(n[0]?.n ?? 0) >= WATCH_MAX_ITEMS) return { error: "full" };
66 + const rows = await sql<Row[]>`insert into watchlists (id, owner_token, entity_type, entity_id) values (${newId("watch")}, ${token}, ${type}, ${entityId}) on conflict (owner_token, entity_type, entity_id) do update set entity_id = excluded.entity_id returning *`;
67 + return { item: (await toItems(rows))[0]!, created: true };
68 +}
69 +
70 +export async function removeWatch(token: string, id: string): Promise<boolean> {
71 + const sql = pg();
72 + const rows = await sql`delete from watchlists where owner_token = ${token} and id = ${id} returning id`;
73 + return rows.length > 0;
74 +}
75 +
76 +/** Events for the watched entities, newest first, one row per cluster. */
77 +export async function watchFeed(token: string, pageNo?: number, perPage?: number): Promise<{ items: EventDTO[]; total: number; page: number; perPage: number }> {
78 + const sql = pg();
79 + const pg_ = page(pageNo, perPage, 100, 50);
80 + const rows = await sql<Row[]>`
81 + with w as (select entity_type, entity_id from watchlists where owner_token = ${token}),
82 + matched as (
83 + select distinct on (coalesce(e.cluster_id, e.id)) e.id
84 + from events e
85 + where e.review_status <> 'rejected' and (
86 + exists (select 1 from w where w.entity_type = 'operator' and (e.operator_id = w.entity_id or (e.entity_type = 'operator' and e.entity_id = w.entity_id)))
87 + or exists (select 1 from w where w.entity_type = 'country' and e.country_iso2 = w.entity_id)
88 + or exists (select 1 from w where w.entity_type = 'project' and (e.project_id = w.entity_id or (e.entity_type = 'project' and e.entity_id = w.entity_id)))
89 + or exists (select 1 from w where w.entity_type = 'facility' and e.entity_type = 'facility' and e.entity_id = w.entity_id)
90 + or exists (select 1 from w where w.entity_type = 'metro' and (e.metro_id = w.entity_id or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = w.entity_id))))
91 + )
92 + order by coalesce(e.cluster_id, e.id), e.significance desc, e.detected_at asc
93 + )
94 + select ${eventCols(sql)}, count(*) over() as total from events e ${eventJoins(sql)}
95 + where e.id in (select id from matched)
96 + order by e.detected_at desc, e.id desc limit ${pg_.perPage} offset ${pg_.offset}`;
97 + const total = rows.length ? Number(rows[0]!.total) : 0;
98 + return { items: await toEventDtos(rows), total, page: pg_.page, perPage: pg_.perPage };
99 +}
added apps/api/src/routes/admin/claims.ts +24 −0
@@ -0,0 +1,24 @@
1 +/** Admin claim workbench: list + status changes. */
2 +import type { FastifyInstance } from "fastify";
3 +import { z } from "zod";
4 +import { envelope, notFound, parseBody, parseQuery } from "../../lib/http.js";
5 +import { intParam, pageParam, strParam } from "../../lib/params.js";
6 +import { listClaims, setClaimStatus } from "../../repositories/admin/claims.js";
7 +import { invalidate } from "../../cache.js";
8 +
9 +export async function claimAdminRoutes(app: FastifyInstance): Promise<void> {
10 + app.get("/claims", { schema: { summary: "Claims (ClaimDTO[]) ?subject_type=&subject_id=&status=current|superseded|rejected|review|unscoped|all&predicate=&page=&per_page=" } }, async (req) => {
11 + const q = parseQuery(z.object({ subject_type: strParam, subject_id: strParam, status: strParam, predicate: strParam, page: pageParam, per_page: intParam }), req.query);
12 + const res = await listClaims({ subjectType: q.subject_type, subjectId: q.subject_id, status: q.status, predicate: q.predicate, page: q.page, perPage: q.per_page });
13 + return envelope(res.items, { total: res.total, page: res.page, perPage: res.perPage });
14 + });
15 +
16 + app.post("/claims/:id/status", { schema: { summary: "Set a claim status {status: current|rejected|review, reason?}. Never rewrites the entity column: when a rejected claim was backing a displayed facility MW value a quality flag claim_rejected_backing_value is raised for the next reconciliation" } }, async (req) => {
17 + const { id } = req.params as { id: string };
18 + const body = parseBody(z.object({ status: z.enum(["current", "rejected", "review"]), reason: z.string().max(500).optional() }), req.body);
19 + const res = await setClaimStatus(id, body.status, body.reason ?? null);
20 + if (!res) throw notFound("claim");
21 + await Promise.all([invalidate("/datacenters"), invalidate("/projects")]);
22 + return envelope(res);
23 + });
24 +}
modified apps/api/src/routes/admin/connectors.ts +25 −0
@@ -10,6 +10,7 @@ import { strParam, intParam } from "../../lib/params.js";
10 10 import { pg } from "../../lib/sql.js";
11 11 import { enqueueCrawl } from "../../queues.js";
12 12 import { getConnectorAdmin, getRun, listConnectorHealth, listRuns } from "../../repositories/admin/connectors.js";
13 +import { rollbackRun, runChanges } from "../../repositories/admin/runs.js";
13 14 import { invalidate } from "../../cache.js";
14 15
15 16 const runBody = z.object({ task: z.enum(["crawl", "discover", "full", "reprocess"]).default("crawl"), group: z.string().min(1).max(64).optional(), limit: z.number().int().min(1).max(100000).optional(), force: z.boolean().optional() });
@@ -89,6 +90,15 @@ export async function connectorAdminRoutes(app: FastifyInstance): Promise<void>
89 90 return envelope(items, { total: items.length });
90 91 });
91 92
93 + app.post("/connectors/:id/quarantine", { schema: { summary: "Quarantine {on: boolean}: the connector extracts and previews but publishes nothing; health = quarantine while on" } }, async (req) => {
94 + const { id } = req.params as { id: string };
95 + const body = parseBody(z.object({ on: z.boolean() }), req.body);
96 + const sql = pg();
97 + const rows = await sql`update connectors set quarantine = ${body.on}, health = case when ${body.on} then 'quarantine' when paused then 'paused' when last_run_at is null then 'never_run' else 'ok' end, updated_at = now() where id = ${id} returning id, quarantine, health`;
98 + if (!rows.length) throw notFound("connector");
99 + return envelope(rows[0]);
100 + });
101 +
92 102 app.get("/runs/:id", { schema: { summary: "Run detail with log" } }, async (req) => {
93 103 const { id } = req.params as { id: string };
94 104 const r = await getRun(id);
@@ -96,6 +106,21 @@ export async function connectorAdminRoutes(app: FastifyInstance): Promise<void>
96 106 return envelope(r);
97 107 });
98 108
109 + app.get("/runs/:id/changes", { schema: { summary: "Everything a run wrote: provenance rows, claims, events, document versions (run_id = id; 2 000 rows per list)" } }, async (req) => {
110 + const { id } = req.params as { id: string };
111 + const r = await runChanges(id);
112 + if (!r) throw notFound("run");
113 + return envelope(r, { ...r.counts });
114 + });
115 +
116 + app.post("/runs/:id/rollback", { schema: { summary: "Roll a run back: its claims → rejected (rollback), its provenance rows → not current (winner restored to the latest remaining row per field), its events → rejected. Entity columns are re-derived by the worker's next reconciliation" } }, async (req) => {
117 + const { id } = req.params as { id: string };
118 + const r = await rollbackRun(id);
119 + if (!r) throw notFound("run");
120 + await Promise.all([invalidate("/datacenters"), invalidate("/projects"), invalidate("/events"), invalidate("/dashboard")]);
121 + return envelope(r);
122 + });
123 +
99 124 app.post("/cache/invalidate", { schema: { summary: "Invalidate API cache entries by route prefix (body {prefix}) — default all" } }, async (req) => {
100 125 const body = parseBody(z.object({ prefix: z.string().max(200).default("") }), req.body);
101 126 const n = await invalidate(body.prefix);
modified apps/api/src/routes/admin/curation.ts +9 −1
@@ -5,7 +5,7 @@ import { FACILITY_STATUSES, FACILITY_TYPES, GEO_PRECISIONS, newId, normalizeName
5 5 import { envelope, notFound, parseBody, badRequest } from "../../lib/http.js";
6 6 import { pg } from "../../lib/sql.js";
7 7 import { str, type Row } from "../../lib/rows.js";
8 −import { ensureManualSource, mergeFacility } from "../../repositories/admin/merge.js";
8 +import { ensureManualSource, mergeFacility, setFacilityParent } from "../../repositories/admin/merge.js";
9 9 import { getFacilityDetail } from "../../repositories/facilities.js";
10 10 import { invalidate } from "../../cache.js";
11 11
@@ -100,6 +100,14 @@ export async function curationAdminRoutes(app: FastifyInstance): Promise<void> {
100 100 return envelope(res);
101 101 });
102 102
103 + app.post("/facilities/:id/parent", { schema: { summary: "Containment link {parentId | null}: the facility becomes a building of the parent campus (record_scope building / campus); aggregates never count both" } }, async (req) => {
104 + const { id } = req.params as { id: string };
105 + const body = parseBody(z.object({ parentId: z.string().min(1).nullable() }), req.body);
106 + const res = await setFacilityParent(id, body.parentId);
107 + await Promise.all([invalidate("/datacenters"), invalidate("/dashboard"), invalidate("/map")]);
108 + return envelope(res);
109 + });
110 +
103 111 app.patch("/events/:id", { schema: { summary: "Review an event {reviewStatus: auto|pending|approved|rejected}" } }, async (req) => {
104 112 const { id } = req.params as { id: string };
105 113 const body = parseBody(z.object({ reviewStatus: z.enum(["auto", "pending", "approved", "rejected"]) }), req.body);
modified apps/api/src/routes/admin/documents.ts +2 −2
@@ -2,7 +2,7 @@ import type { FastifyInstance } from "fastify";
2 2 import { z } from "zod";
3 3 import { envelope, notFound, parseBody, parseQuery } from "../../lib/http.js";
4 4 import { intParam, pageParam, strParam } from "../../lib/params.js";
5 −import { pg, page, andAll, type Fragment } from "../../lib/sql.js";
5 +import { pg, page, andAll, likePattern, type Fragment } from "../../lib/sql.js";
6 6 import { int, iso, json, num, reqStr, str, type Row } from "../../lib/rows.js";
7 7 import { resolveEntityRefs } from "../../lib/resolve.js";
8 8 import { enqueueCrawl, manualJobId } from "../../queues.js";
@@ -34,7 +34,7 @@ export async function documentAdminRoutes(app: FastifyInstance): Promise<void> {
34 34 if (q.status === "error") c.push(sql`d.error is not null`);
35 35 if (q.status === "changed") c.push(sql`d.change_count > 0`);
36 36 if (q.status === "quarantined") c.push(sql`d.quarantined`);
37 − if (q.q) c.push(sql`(d.url ilike ${"%" + q.q + "%"} or d.title ilike ${"%" + q.q + "%"})`);
37 + if (q.q) c.push(sql`(d.url ilike ${likePattern(q.q)} or d.title ilike ${likePattern(q.q)})`);
38 38 const rows = await sql<Row[]>`select d.*, count(*) over() as total from documents d where ${andAll(sql, c)} order by coalesce(d.last_changed, d.last_fetched, d.first_seen) desc limit ${pg_.perPage} offset ${pg_.offset}`;
39 39 return envelope(rows.map(docDto), { total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage });
40 40 });
modified apps/api/src/routes/admin/index.ts +9 −1
@@ -12,6 +12,10 @@ import { matchAdminRoutes } from "./matches.js";
12 12 import { curationAdminRoutes } from "./curation.js";
13 13 import { devtoolAdminRoutes } from "./devtool.js";
14 14 import { opsAdminRoutes } from "./ops.js";
15 +import { qualityAdminRoutes } from "./quality.js";
16 +import { proxyAdminRoutes } from "./proxy.js";
17 +import { claimAdminRoutes } from "./claims.js";
18 +import { projectAdminRoutes } from "./projects.js";
15 19
16 20 export function tokenMatches(provided: string | undefined, expected: string | null): boolean {
17 21 if (!expected || !provided) return false;
@@ -22,7 +26,7 @@ export function tokenMatches(provided: string | undefined, expected: string | nu
22 26
23 27 export function adminAuth(req: FastifyRequest, _reply: FastifyReply, done: (err?: Error) => void): void {
24 28 const expected = getEnv().adminToken;
25 − if (!expected) return done(new HttpError(503, "admin API disabled: DCI_ADMIN_TOKEN is not configured"));
29 + if (!expected) return done(new HttpError(503, "admin API disabled: DCI_ADMIN_TOKEN is not configured (missing, empty or the \"change-me\" placeholder)"));
26 30 const header = req.headers["x-dci-admin-token"];
27 31 const provided = Array.isArray(header) ? header[0] : header;
28 32 if (!tokenMatches(provided, expected)) return done(new HttpError(401, "unauthorized"));
@@ -43,4 +47,8 @@ export async function adminRoutes(app: FastifyInstance): Promise<void> {
43 47 await app.register(curationAdminRoutes);
44 48 await app.register(devtoolAdminRoutes);
45 49 await app.register(opsAdminRoutes);
50 + await app.register(qualityAdminRoutes);
51 + await app.register(proxyAdminRoutes);
52 + await app.register(claimAdminRoutes);
53 + await app.register(projectAdminRoutes);
46 54 }
modified apps/api/src/routes/admin/matches.ts +101 −17
@@ -1,42 +1,97 @@
1 +/**
2 + * Entity match workbench: pending candidates with BOTH facility rows side by side (name, operator, city, address,
3 + * coordinates + distance, facility codes, external ids, source counts) and the matcher's score / reasons.
4 + * Decisions: approve (merge), reject (keep separate), related-campus (candidate is a building of the matched campus),
5 + * defer.
6 + */
1 7 import type { FastifyInstance } from "fastify";
2 8 import { z } from "zod";
3 −import { envelope, notFound, parseQuery, badRequest } from "../../lib/http.js";
9 +import { extractFacilityCodes } from "@dci/core";
10 +import { envelope, notFound, parseQuery, parseBody, badRequest } from "../../lib/http.js";
4 11 import { intParam, pageParam, strParam } from "../../lib/params.js";
5 12 import { pg, page } from "../../lib/sql.js";
6 −import { int, iso, json, num, reqStr, str, strArray, type Row } from "../../lib/rows.js";
13 +import { int, iso, json, num, record, reqStr, str, strArray, type Row } from "../../lib/rows.js";
7 14 import { facilitiesByIds } from "../../repositories/facilities.js";
8 −import { ensureManualSource, mergeFacility } from "../../repositories/admin/merge.js";
15 +import { ensureManualSource, mergeFacility, setFacilityParent } from "../../repositories/admin/merge.js";
9 16 import { invalidate } from "../../cache.js";
10 17
18 +const STATUSES = ["pending", "approved", "rejected", "related_campus", "deferred", "auto_merged", "auto_created", "all"] as const;
19 +
11 20 function matchDto(r: Row): Record<string, unknown> {
12 21 return { id: reqStr(r.id), connectorId: reqStr(r.connector_id), candidateKey: reqStr(r.candidate_key), candidate: json(r.candidate, {}), matchedFacilityId: str(r.matched_facility_id), score: num(r.score), reasons: strArray(r.reasons), status: reqStr(r.status), decidedBy: str(r.decided_by), decidedAt: iso(r.decided_at), createdAt: iso(r.created_at), candidateFacilityId: str(r.candidate_facility_id) };
13 22 }
14 23
24 +interface Side { id: string; slug: string; name: string; operator: string | null; city: string | null; address: string | null; lat: number | null; lng: number | null; codes: string[]; externalIds: Record<string, string | number>; sourceCount: number; status: string; recordScope: string; parentFacilityId: string | null }
25 +
26 +function side(r: Row): Side {
27 + return { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), operator: str(r.op_name), city: str(r.city), address: str(r.address), lat: num(r.lat), lng: num(r.lng), codes: extractFacilityCodes(reqStr(r.name)), externalIds: record(r.external_ids), sourceCount: int(r.source_count), status: reqStr(r.status), recordScope: reqStr(r.record_scope, "facility"), parentFacilityId: str(r.parent_facility_id) };
28 +}
29 +
30 +function haversineKm(a: Side, b: Side): number | null {
31 + if (a.lat == null || a.lng == null || b.lat == null || b.lng == null) return null;
32 + const R = 6371.0088, toRad = (d: number) => (d * Math.PI) / 180;
33 + const dLat = toRad(b.lat - a.lat), dLng = toRad(b.lng - a.lng);
34 + const h = Math.sin(dLat / 2) ** 2 + Math.cos(toRad(a.lat)) * Math.cos(toRad(b.lat)) * Math.sin(dLng / 2) ** 2;
35 + return Math.round(2 * R * Math.asin(Math.sqrt(Math.min(1, h))) * 1000) / 1000;
36 +}
37 +
38 +/** The facility created from the candidate (entity key or candidate.createdFacilityId) — for a pending row. */
39 +function candidateFacilityId(r: Row): string | null {
40 + return str(r.candidate_facility_id) ?? str((json<Record<string, unknown>>(r.candidate, {}) as Record<string, unknown>).createdFacilityId);
41 +}
42 +
15 43 export async function matchAdminRoutes(app: FastifyInstance): Promise<void> {
16 − app.get("/matches", { schema: { summary: "Entity match queue (?status=pending) with candidate + matched facility summaries" } }, async (req) => {
17 − const q = parseQuery(z.object({ status: strParam, connector: strParam, page: pageParam, per_page: intParam }), req.query);
44 + app.get("/matches", { schema: { summary: "Entity match queue (?status=pending|approved|rejected|related_campus|deferred|auto_merged|auto_created|all&connector=) with both facility rows (pair: name, operator, city, address, coords + distance, codes, external ids, source counts), score and reasons" } }, async (req) => {
45 + const q = parseQuery(z.object({ status: z.preprocess((v) => (v === "" || v == null ? undefined : v), z.enum(STATUSES).optional()), connector: strParam, page: pageParam, per_page: intParam }), req.query);
18 46 const sql = pg();
19 47 const pg_ = page(q.page, q.per_page, 200, 50);
20 48 const status = q.status ?? "pending";
21 49 const rows = await sql<Row[]>`
22 − select m.*, k.entity_id as candidate_facility_id, count(*) over() as total
50 + select m.*, coalesce(k.entity_id, m.candidate->>'createdFacilityId') as candidate_facility_id, count(*) over() as total
23 51 from entity_matches m
24 52 left join entity_keys k on k.key = m.candidate_key and k.entity_type = 'facility'
25 53 where ${status === "all" ? sql`true` : sql`m.status = ${status}`} and ${q.connector ? sql`m.connector_id = ${q.connector}` : sql`true`}
26 54 order by m.score desc, m.created_at desc limit ${pg_.perPage} offset ${pg_.offset}`;
27 − const ids = [...new Set(rows.flatMap((r) => [str(r.matched_facility_id), str(r.candidate_facility_id)]).filter((x): x is string => Boolean(x)))];
28 − const facs = await facilitiesByIds(ids);
55 + const ids = [...new Set(rows.flatMap((r) => [str(r.matched_facility_id), candidateFacilityId(r)]).filter((x): x is string => Boolean(x)))];
56 + const [facs, sides] = await Promise.all([
57 + facilitiesByIds(ids),
58 + ids.length ? sql<Row[]>`select f.id, f.slug, f.name, f.city, f.address, f.lat, f.lng, f.external_ids, f.source_count, f.status, f.record_scope, f.parent_facility_id, o.name as op_name from facilities f left join operators o on o.id = f.operator_id where f.id = any(${ids})` : Promise.resolve([] as Row[]),
59 + ]);
29 60 const by = new Map(facs.map((f) => [f.id, f]));
30 − return envelope(rows.map((r) => ({ ...matchDto(r), matched: r.matched_facility_id ? by.get(String(r.matched_facility_id)) ?? null : null, candidateFacility: r.candidate_facility_id ? by.get(String(r.candidate_facility_id)) ?? null : null })), { total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage });
61 + const sideBy = new Map(sides.map((r) => [reqStr(r.id), side(r)]));
62 + const items = rows.map((r) => {
63 + const matchedId = str(r.matched_facility_id);
64 + const candId = candidateFacilityId(r);
65 + const cand = json<Record<string, unknown>>(r.candidate, {});
66 + const a = candId ? sideBy.get(candId) ?? null : null;
67 + const b = matchedId ? sideBy.get(matchedId) ?? null : null;
68 + // when the candidate was never persisted as a facility, describe it from the candidate payload itself
69 + const candidateSide: Partial<Side> | null = a ?? (Object.keys(cand).length ? { id: "", slug: "", name: str(cand.name) ?? "", operator: str(cand.operatorName ?? cand.operator), city: str(cand.city), address: str(cand.address), lat: num((cand.geo as Record<string, unknown> | undefined)?.lat ?? cand.lat), lng: num((cand.geo as Record<string, unknown> | undefined)?.lng ?? cand.lng), codes: extractFacilityCodes(str(cand.name) ?? ""), externalIds: record(cand.externalIds), sourceCount: 0, status: str(cand.status) ?? "unknown", recordScope: "facility", parentFacilityId: null } : null);
70 + const distanceKm = candidateSide && b && candidateSide.lat != null && candidateSide.lng != null ? haversineKm(candidateSide as Side, b) : null;
71 + return {
72 + ...matchDto(r),
73 + candidateFacilityId: candId,
74 + matched: matchedId ? by.get(matchedId) ?? null : null,
75 + candidateFacility: candId ? by.get(candId) ?? null : null,
76 + pair: { candidate: candidateSide, matched: b, distanceKm, sameCodes: candidateSide && b ? candidateSide.codes!.filter((c) => b.codes.includes(c)) : [] },
77 + };
78 + });
79 + return envelope(items, { total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage });
31 80 });
32 81
33 − app.post("/matches/:id/approve", { schema: { summary: "Approve: point the candidate key at the matched facility and merge any separately-created facility into it" } }, async (req) => {
34 − const { id } = req.params as { id: string };
82 + async function pendingMatch(id: string): Promise<Row> {
35 83 const sql = pg();
36 − const rows = await sql<Row[]>`select * from entity_matches where id = ${id}`;
84 + const rows = await sql<Row[]>`select m.*, coalesce(k.entity_id, m.candidate->>'createdFacilityId') as candidate_facility_id from entity_matches m left join entity_keys k on k.key = m.candidate_key and k.entity_type = 'facility' where m.id = ${id}`;
37 85 const m = rows[0];
38 86 if (!m) throw notFound("match");
39 − if (m.status !== "pending") throw badRequest(`match is already ${String(m.status)}`);
87 + if (m.status !== "pending" && m.status !== "deferred") throw badRequest(`match is already ${String(m.status)}`);
88 + return m;
89 + }
90 +
91 + app.post("/matches/:id/approve", { schema: { summary: "Approve: point the candidate key at the matched facility and merge any separately-created facility into it" } }, async (req) => {
92 + const { id } = req.params as { id: string };
93 + const sql = pg();
94 + const m = await pendingMatch(id);
40 95 const matched = str(m.matched_facility_id);
41 96 if (!matched) throw badRequest("match has no matched_facility_id");
42 97 const target = await sql<Row[]>`select id, merged_into from facilities where id = ${matched}`;
@@ -44,11 +99,11 @@ export async function matchAdminRoutes(app: FastifyInstance): Promise<void> {
44 99 const survivor = str(target[0].merged_into) ?? matched;
45 100 await ensureManualSource();
46 101 const key = reqStr(m.candidate_key);
47 − const existing = await sql<Row[]>`select entity_id from entity_keys where key = ${key}`;
48 − const separate = str(existing[0]?.entity_id);
102 + const existing = await sql<Row[]>`select entity_id from entity_keys where key = ${key} and entity_type = 'facility'`;
103 + const separate = str(existing[0]?.entity_id) ?? candidateFacilityId(m);
49 104 let merge: Awaited<ReturnType<typeof mergeFacility>> | null = null;
50 105 if (separate && separate !== survivor) merge = await mergeFacility(separate, survivor);
51 − await sql`insert into entity_keys (key, entity_type, entity_id, connector_id) values (${key}, 'facility', ${survivor}, ${reqStr(m.connector_id)}) on conflict (key) do update set entity_id = ${survivor}`;
106 + await sql`insert into entity_keys (key, entity_type, entity_id, connector_id) values (${key}, 'facility', ${survivor}, ${reqStr(m.connector_id)}) on conflict (key, entity_type) do update set entity_id = ${survivor}`;
52 107 await sql`update entity_matches set status = 'approved', decided_by = 'admin', decided_at = now() where id = ${id}`;
53 108 await invalidate("/datacenters");
54 109 return envelope({ id, status: "approved", facilityId: survivor, candidateKey: key, merged: merge });
@@ -57,7 +112,7 @@ export async function matchAdminRoutes(app: FastifyInstance): Promise<void> {
57 112 app.post("/matches/:id/reject", { schema: { summary: "Reject: keep the candidate as a separate facility" } }, async (req) => {
58 113 const { id } = req.params as { id: string };
59 114 const sql = pg();
60 − const rows = await sql<Row[]>`update entity_matches set status = 'rejected', decided_by = 'admin', decided_at = now() where id = ${id} and status = 'pending' returning id`;
115 + const rows = await sql<Row[]>`update entity_matches set status = 'rejected', decided_by = 'admin', decided_at = now() where id = ${id} and status in ('pending', 'deferred') returning id`;
61 116 if (!rows[0]) {
62 117 const exists = await sql`select status from entity_matches where id = ${id}`;
63 118 if (!exists.length) throw notFound("match");
@@ -65,4 +120,33 @@ export async function matchAdminRoutes(app: FastifyInstance): Promise<void> {
65 120 }
66 121 return envelope({ id, status: "rejected" });
67 122 });
123 +
124 + app.post("/matches/:id/related-campus", { schema: { summary: "Related campus: the candidate facility is a BUILDING of the matched campus (parent_facility_id set, record_scope building / campus); both rows stay, aggregates never count both" } }, async (req) => {
125 + const { id } = req.params as { id: string };
126 + const sql = pg();
127 + const m = await pendingMatch(id);
128 + const matched = str(m.matched_facility_id);
129 + const building = candidateFacilityId(m);
130 + if (!matched) throw badRequest("match has no matched_facility_id");
131 + if (!building) throw badRequest("the candidate was never persisted as a facility — approve or reject instead");
132 + const parentRow = (await sql<Row[]>`select id, merged_into from facilities where id = ${matched}`)[0];
133 + if (!parentRow) throw badRequest("matched facility no longer exists");
134 + const campus = str(parentRow.merged_into) ?? matched;
135 + const res = await setFacilityParent(building, campus);
136 + await sql`update entity_matches set status = 'related_campus', decided_by = 'admin', decided_at = now() where id = ${id}`;
137 + await Promise.all([invalidate("/datacenters"), invalidate("/dashboard"), invalidate("/map")]);
138 + return envelope({ id, status: "related_campus", building, campus, containment: res });
139 + });
140 +
141 + app.post("/matches/:id/defer", { schema: { summary: "Defer the decision (status deferred, keeps the row in the workbench under ?status=deferred)" } }, async (req) => {
142 + const { id } = req.params as { id: string };
143 + const sql = pg();
144 + const rows = await sql<Row[]>`update entity_matches set status = 'deferred', decided_by = 'admin', decided_at = now() where id = ${id} and status = 'pending' returning id`;
145 + if (!rows[0]) {
146 + const exists = await sql`select status from entity_matches where id = ${id}`;
147 + if (!exists.length) throw notFound("match");
148 + throw badRequest(`match is already ${String(exists[0]!.status)}`);
149 + }
150 + return envelope({ id, status: "deferred" });
151 + });
68 152 }
modified apps/api/src/routes/admin/ops.ts +7 −5
@@ -89,12 +89,14 @@ export async function opsAdminRoutes(app: FastifyInstance): Promise<void> {
89 89 });
90 90 });
91 91
92 − app.post("/maintenance/:task", { schema: { summary: "Enqueue a maintenance job (rankings | metrics | refresh-stats) on dci:maintenance" } }, async (req) => {
92 + // Note for deploy: the compose healthcheck of the api service should probe GET /api/ready (Postgres ping), not /api/health.
93 + const maintenanceBody = z.object({ limit: z.number().int().min(1).max(100000).optional(), dryRun: z.boolean().optional(), connector: z.string().max(64).optional(), reason: z.string().max(500).optional() }).strict();
94 + app.post("/maintenance/:task", { schema: { summary: "Enqueue a maintenance job (rankings | metrics | refresh-stats | quality | snapshot | cleanup) on dci:maintenance; body {limit?, dryRun?, connector?, reason?} (strict)" } }, async (req) => {
93 95 const { task } = req.params as { task: string };
94 − const t = z.enum(["rankings", "metrics", "refresh-stats"]).safeParse(task);
95 − if (!t.success) return envelope({ enqueued: false, error: "unknown task (rankings | metrics | refresh-stats)" });
96 − const body = parseBody(z.record(z.string(), z.unknown()).optional(), req.body ?? {});
97 − const job = await enqueueMaintenance(t.data, body ?? {});
96 + const t = z.enum(["rankings", "metrics", "refresh-stats", "quality", "snapshot", "cleanup"]).safeParse(task);
97 + if (!t.success) return envelope({ enqueued: false, error: "unknown task (rankings | metrics | refresh-stats | quality | snapshot | cleanup)" });
98 + const body = parseBody(maintenanceBody, req.body ?? {});
99 + const job = await enqueueMaintenance(t.data, body);
98 100 return envelope({ enqueued: true, job });
99 101 });
100 102
added apps/api/src/routes/admin/projects.ts +88 −0
@@ -0,0 +1,88 @@
1 +/** Project curation routes: hide / unhide / merge / PATCH. */
2 +import type { FastifyInstance } from "fastify";
3 +import { z } from "zod";
4 +import { AI_EVIDENCE_LEVELS, FACILITY_STATUSES, GEO_PRECISIONS, PROJECT_CLASSES, parsePartialDate } from "@dci/core";
5 +import { envelope, notFound, parseBody, badRequest } from "../../lib/http.js";
6 +import { pg } from "../../lib/sql.js";
7 +import { invalidate } from "../../cache.js";
8 +import { mergeProject, patchProject, setProjectHidden } from "../../repositories/admin/projects.js";
9 +
10 +const partialDate = z.string().max(20).nullable().refine((v) => v == null || parsePartialDate(v) != null, "not a (partial) date");
11 +
12 +const patchSchema = z.object({
13 + name: z.string().min(2).max(200).optional(),
14 + status: z.enum(FACILITY_STATUSES).optional(),
15 + planned_mw: z.number().positive().max(20000).nullable().optional(),
16 + investment_usd: z.number().positive().nullable().optional(),
17 + expected_opening: partialDate.optional(),
18 + announced_on: partialDate.optional(),
19 + construction_started_on: partialDate.optional(),
20 + approved_on: partialDate.optional(),
21 + permit_filed_on: partialDate.optional(),
22 + opened_on: partialDate.optional(),
23 + operator_id: z.string().nullable().optional(),
24 + country_iso2: z.string().length(2).nullable().optional(),
25 + metro_id: z.string().nullable().optional(),
26 + city: z.string().max(200).nullable().optional(),
27 + lat: z.number().min(-90).max(90).nullable().optional(),
28 + lng: z.number().min(-180).max(180).nullable().optional(),
29 + geo_precision: z.enum(GEO_PRECISIONS).optional(),
30 + project_class: z.enum(PROJECT_CLASSES).nullable().optional(),
31 + ai_evidence: z.enum(AI_EVIDENCE_LEVELS).optional(),
32 + is_ai: z.boolean().optional(),
33 + description: z.string().max(5000).nullable().optional(),
34 + note: z.string().max(500).optional(),
35 +}).strict();
36 +
37 +const CACHE_PREFIXES = ["/projects", "/dashboard", "/map", "/explore", "/pulse"];
38 +async function bust(): Promise<void> { await Promise.all(CACHE_PREFIXES.map((p) => invalidate(p))); }
39 +
40 +export async function projectAdminRoutes(app: FastifyInstance): Promise<void> {
41 + app.post("/projects/:id/hide", { schema: { summary: "Hide a project (false positive kept for audit, never listed) {reason}; writes src_manual provenance + a project_status_changed event {hidden}" } }, async (req) => {
42 + const { id } = req.params as { id: string };
43 + const body = parseBody(z.object({ reason: z.string().min(1).max(500) }), req.body);
44 + const res = await setProjectHidden(id, true, body.reason);
45 + if (!res) throw notFound("project");
46 + await bust();
47 + return envelope(res);
48 + });
49 +
50 + app.post("/projects/:id/unhide", { schema: { summary: "Restore a hidden project {reason?}" } }, async (req) => {
51 + const { id } = req.params as { id: string };
52 + const body = parseBody(z.object({ reason: z.string().max(500).optional() }), req.body ?? {});
53 + const res = await setProjectHidden(id, false, body.reason ?? null);
54 + if (!res) throw notFound("project");
55 + await bust();
56 + return envelope(res);
57 + });
58 +
59 + app.post("/projects/:id/merge", { schema: { summary: "Merge this project into another {into}: events, timeline, claims, provenance, entity keys move to the survivor; the duplicate keeps merged_into" } }, async (req) => {
60 + const { id } = req.params as { id: string };
61 + const body = parseBody(z.object({ into: z.string().min(1) }), req.body);
62 + let res: Awaited<ReturnType<typeof mergeProject>>;
63 + try { res = await mergeProject(id, body.into); } catch (e) {
64 + const msg = (e as Error).message;
65 + if (msg === "project not found") throw notFound("project");
66 + throw badRequest(msg);
67 + }
68 + await bust();
69 + return envelope(res);
70 + });
71 +
72 + app.patch("/projects/:id", { schema: { summary: "Curate project fields (strict body); src_manual provenance on every field; events only for status (project_status_changed) and planned_mw (planned_capacity_changed)" } }, async (req) => {
73 + const { id } = req.params as { id: string };
74 + const body = parseBody(patchSchema, req.body);
75 + const { note, ...fields } = body;
76 + const entries = Object.entries(fields).filter(([, v]) => v !== undefined);
77 + if (!entries.length) throw badRequest("no fields to update");
78 + if ((fields.lat === undefined) !== (fields.lng === undefined) || (fields.lat == null) !== (fields.lng == null)) throw badRequest("lat and lng must be set together");
79 + const sql = pg();
80 + if (fields.operator_id) { const r = await sql`select id from operators where id = ${fields.operator_id}`; if (!r.length) throw badRequest("operator_id does not exist"); }
81 + if (fields.country_iso2) { fields.country_iso2 = fields.country_iso2.toUpperCase(); const r = await sql`select iso2 from countries where iso2 = ${fields.country_iso2}`; if (!r.length) throw badRequest("country_iso2 does not exist"); }
82 + if (fields.metro_id) { const r = await sql`select id from metros where id = ${fields.metro_id}`; if (!r.length) throw badRequest("metro_id does not exist"); }
83 + const res = await patchProject(id, fields, note ?? null);
84 + if (!res) throw notFound("project");
85 + if (res.changed.length) await bust();
86 + return envelope({ ...res, message: res.changed.length ? undefined : "no changes" });
87 + });
88 +}
added apps/api/src/routes/admin/proxy.ts +39 −0
@@ -0,0 +1,39 @@
1 +/**
2 + * Proxies to the worker's internal HTTP server (DCI_WORKER_URL, default http://127.0.0.1:8320; compose:
3 + * http://worker-maint:8320): GET /data-gaps and GET /trace/<documentId> (extraction debugger). 10 s timeout, 502 on failure.
4 + */
5 +import type { FastifyInstance } from "fastify";
6 +import { z } from "zod";
7 +import { getEnv } from "../../env.js";
8 +import { envelope, HttpError, parseQuery } from "../../lib/http.js";
9 +import { strParam } from "../../lib/params.js";
10 +
11 +async function workerGet(path: string): Promise<unknown> {
12 + const base = getEnv().workerUrl.replace(/\/+$/, "");
13 + const url = `${base}${path}`;
14 + let host = base;
15 + try { host = new URL(base).host; } catch { /* keep raw */ }
16 + try {
17 + const res = await fetch(url, { headers: { accept: "application/json" }, signal: AbortSignal.timeout(10_000) });
18 + const text = await res.text();
19 + let body: unknown = text;
20 + try { body = JSON.parse(text); } catch { /* keep text */ }
21 + if (!res.ok) throw new HttpError(res.status === 404 ? 404 : 502, `worker answered HTTP ${res.status}`, { worker: host, path, body });
22 + return body;
23 + } catch (e) {
24 + if (e instanceof HttpError) throw e;
25 + const msg = (e as Error).name === "TimeoutError" ? "timeout after 10 s" : (e as Error).message;
26 + throw new HttpError(502, `worker unreachable: ${msg}`, { worker: host, path });
27 + }
28 +}
29 +
30 +export async function proxyAdminRoutes(app: FastifyInstance): Promise<void> {
31 + app.get("/data-gaps", { schema: { summary: "Data gaps counters (proxied from the worker GET /data-gaps; 502 when the worker is unreachable)" } }, async () => envelope(await workerGet("/data-gaps"), { source: getEnv().workerUrl }));
32 +
33 + app.get("/documents/:id/trace", { schema: { summary: "Extraction debugger trace for a document (proxied from the worker GET /trace/<id>?live=1 re-fetches the page)" } }, async (req) => {
34 + const { id } = req.params as { id: string };
35 + const q = parseQuery(z.object({ live: strParam }), req.query);
36 + const live = q.live === "1" || q.live === "true" ? "1" : "";
37 + return envelope(await workerGet(`/trace/${encodeURIComponent(id)}${live ? "?live=1" : ""}`), { source: getEnv().workerUrl, live: live === "1" });
38 + });
39 +}
added apps/api/src/routes/admin/quality.ts +32 −0
@@ -0,0 +1,32 @@
1 +/** Admin data-quality dashboard: overview, flag list, resolve / dismiss. */
2 +import type { FastifyInstance } from "fastify";
3 +import { z } from "zod";
4 +import { envelope, notFound, parseBody, parseQuery } from "../../lib/http.js";
5 +import { intParam, pageParam, strParam } from "../../lib/params.js";
6 +import { listFlags, qualityOverview, setFlagStatus } from "../../repositories/admin/quality.js";
7 +
8 +export async function qualityAdminRoutes(app: FastifyInstance): Promise<void> {
9 + app.get("/quality", { schema: { summary: "Data-quality overview (QualityOverview): open flags by severity / code, largest published values, review queue, alerts" } }, async () => envelope(await qualityOverview(), { methodology: "Counts are over open quality_flags, claims by status and live (not hidden / merged) projects. 'Largest' lists are plain ORDER BY on published figures; pipeline MW per operator / metro is containment-aware (campus vs buildings never both) plus site-scoped project MW." }));
10 +
11 + app.get("/quality/flags", { schema: { summary: "Quality flags (QualityFlagDTO[]) ?status=open|resolved|dismissed|all&code=&entity_type=&min_priority=&page=&per_page=" } }, async (req) => {
12 + const q = parseQuery(z.object({ status: strParam, code: strParam, entity_type: strParam, min_priority: intParam, page: pageParam, per_page: intParam }), req.query);
13 + const res = await listFlags({ status: q.status ?? "open", code: q.code, entityType: q.entity_type, minPriority: q.min_priority, page: q.page, perPage: q.per_page });
14 + return envelope(res.items, { total: res.total, page: res.page, perPage: res.perPage });
15 + });
16 +
17 + app.post("/quality/flags/:id/resolve", { schema: { summary: "Resolve a quality flag {resolution}" } }, async (req) => {
18 + const { id } = req.params as { id: string };
19 + const body = parseBody(z.object({ resolution: z.string().min(1).max(500) }), req.body);
20 + const flag = await setFlagStatus(id, "resolved", body.resolution);
21 + if (!flag) throw notFound("quality flag");
22 + return envelope(flag);
23 + });
24 +
25 + app.post("/quality/flags/:id/dismiss", { schema: { summary: "Dismiss a quality flag {reason?}" } }, async (req) => {
26 + const { id } = req.params as { id: string };
27 + const body = parseBody(z.object({ reason: z.string().max(500).optional() }), req.body ?? {});
28 + const flag = await setFlagStatus(id, "dismissed", body.reason ?? null);
29 + if (!flag) throw notFound("quality flag");
30 + return envelope(flag);
31 + });
32 +}
modified apps/api/src/routes/public/activity.ts +34 −18
@@ -4,44 +4,60 @@ import { z } from "zod";
4 4 import { publicGet } from "../../lib/route.js";
5 5 import { TTL, csv, notFound } from "../../lib/http.js";
6 6 import { boolParam, csvParam, intParam, numParam, orderParam, pageParam, strParam } from "../../lib/params.js";
7 −import { getProjectDetail, listProjects, projectPipeline } from "../../repositories/projects.js";
8 −import { getEvent, listEvents } from "../../repositories/events.js";
7 +import { sourcesForEntities, sourcesForIds } from "../../lib/source-history.js";
8 +import { getProjectClaims, getProjectDetail, getProjectHistory, listProjects, projectPipeline } from "../../repositories/projects.js";
9 +import { EVENTS_DEDUPE_METHODOLOGY, getEvent, listEvents } from "../../repositories/events.js";
9 10 import { listNews } from "../../repositories/misc.js";
10 11
11 12 const empty = (v: unknown) => (v === "" || v === null ? undefined : v);
13 +const PROJECTS_NOTE = "Project records only (never facilities). False positives hidden by review and merged duplicates are excluded from every list, count and sum. plannedMw / investmentUsd are site-scoped published figures; company-wide or portfolio totals stay as claims (see /projects/:slug/claims).";
12 14
13 15 export async function activityRoutes(app: FastifyInstance): Promise<void> {
14 16 // ---- projects (static /pipeline before /:slug — find-my-way prefers static anyway)
15 − publicGet(app, { url: "/projects/pipeline", ttl: TTL.list, summary: "Project pipeline aggregates: by status, by expected year, top 15 countries", tags: ["projects"] }, async () => {
17 + publicGet(app, { url: "/projects/pipeline", ttl: TTL.list, summary: "Project pipeline aggregates: by status, by expected year, top 15 countries", tags: ["projects"], response: { type: "PipelineAggregates" } }, async () => {
16 18 const data = await projectPipeline();
17 − return { data, meta: { methodology: "count and SUM(planned_mw) over projects; byYear uses the year prefix of expected_opening and excludes cancelled/closed" } };
19 + return { data, meta: { methodology: `count and SUM(planned_mw) over live projects (hidden / merged excluded); byYear uses the year prefix of expected_opening and excludes cancelled/closed. ${PROJECTS_NOTE}` } };
18 20 });
19 21
20 − const projectsQuery = z.object({ status: csvParam, country: strParam, operator: strParam, metro: strParam, min_mw: numParam, ai: boolParam, q: strParam, sort: z.preprocess(empty, z.enum(["updated", "mw", "announced", "opening"]).optional()), order: orderParam, page: pageParam, per_page: intParam });
21 − publicGet(app, { url: "/projects", ttl: TTL.list, query: projectsQuery, summary: "List projects (ProjectSummary[])", tags: ["projects"] }, async (q) => {
22 − const res = await listProjects({ ...q, status: csv(q.status) });
23 − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage } };
22 + const projectsQuery = z.object({ status: csvParam, project_status: csvParam, country: strParam, operator: strParam, metro: strParam, min_mw: numParam, max_mw: numParam, ai: boolParam, project_class: csvParam, evidence_level: csvParam, expected_from: intParam, expected_to: intParam, announced_since: strParam, q: strParam, sort: z.preprocess(empty, z.enum(["updated", "mw", "announced", "opening"]).optional()), order: orderParam, page: pageParam, per_page: intParam });
23 + publicGet(app, { url: "/projects", ttl: TTL.list, query: projectsQuery, summary: "List projects (ProjectSummary[]) — status/project_status, country, operator, metro, min/max_mw, ai, project_class, evidence_level, expected_from/to, announced_since, q", tags: ["projects"], response: { type: "ProjectSummary[]" } }, async (q) => {
24 + const res = await listProjects({ ...q, status: [...csv(q.status), ...csv(q.project_status)], project_class: csv(q.project_class), evidence_level: csv(q.evidence_level) });
25 + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: PROJECTS_NOTE }, sources: await sourcesForEntities("project", res.items.map((p) => p.id)) };
26 + });
27 + publicGet(app, { url: "/projects/:slug", ttl: TTL.detail, query: z.object({ radius_km: numParam }), summary: "Project detail (ProjectDetail): lifecycle stages + velocity, claims, history, nearby infrastructure (?radius_km), related events, data quality, provenance, sourceHistory", tags: ["projects"], params: { slug: "project slug or prj_… id" }, response: { type: "ProjectDetail" } }, async (q, params) => {
28 + const res = await getProjectDetail(params.slug!, { radiusKm: q.radius_km != null ? Math.min(200, Math.max(0.1, q.radius_km)) : undefined });
29 + if (!res) throw notFound("project");
30 + return { data: res.detail, sources: res.sources, meta: { methodology: `${PROJECTS_NOTE} Stages are dated from the project's dated columns, its timeline and status-change events (earliest evidence wins); velocityDays are day counts between dated stages, null when a date is unknown.` } };
24 31 });
25 − publicGet(app, { url: "/projects/:slug", ttl: TTL.detail, summary: "Project detail (ProjectDetail) with timeline, status history, provenance, events, sourceHistory", tags: ["projects"], params: { slug: "project slug or prj_… id" } }, async (_q, params) => {
26 − const res = await getProjectDetail(params.slug!);
32 + publicGet(app, { url: "/projects/:slug/history", ttl: TTL.detail, summary: "Project field history (EntityHistory)", tags: ["projects"], params: { slug: "project slug or prj_… id" }, response: { type: "EntityHistory" } }, async (_q, params) => {
33 + const res = await getProjectHistory(params.slug!);
27 34 if (!res) throw notFound("project");
28 − return { data: res.detail, sources: res.sources };
35 + return { data: res.history, sources: res.sources };
36 + });
37 + publicGet(app, { url: "/projects/:slug/claims", ttl: TTL.detail, query: z.object({ status: csvParam, predicate: strParam }), summary: "Claims about a project (ClaimDTO[]) — capacity and investment figures with scope and evidence", tags: ["projects"], params: { slug: "project slug or prj_… id" }, response: { type: "ClaimDTO[]" } }, async (q, params) => {
38 + const res = await getProjectClaims(params.slug!, { status: csv(q.status), predicate: q.predicate });
39 + if (!res) throw notFound("project");
40 + return { data: res.claims, sources: res.sources, meta: { total: res.claims.length, projectId: res.id } };
29 41 });
30 42
31 43 // ---- events
32 − const eventsQuery = z.object({ type: csvParam, country: strParam, operator: strParam, metro: strParam, project: strParam, entity_type: strParam, entity_id: strParam, min_significance: intParam, since: strParam, page: pageParam, per_page: intParam });
33 − publicGet(app, { url: "/events", ttl: TTL.events, query: eventsQuery, summary: "Change feed (EventDTO[]) — newest first", tags: ["events"] }, async (q) => {
34 − const res = await listEvents({ type: csv(q.type), country: q.country, operator: q.operator, metro: q.metro, project: q.project, entityType: q.entity_type, entityId: q.entity_id, minSignificance: q.min_significance, since: q.since, page: q.page, perPage: q.per_page });
35 − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage } };
44 + const eventsQuery = z.object({
45 + type: csvParam, country: strParam, metro: strParam, operator: strParam, project: strParam, entity_type: strParam, entity_id: strParam,
46 + min_significance: intParam, significance: z.preprocess(empty, z.enum(["major", "medium", "minor"]).optional()), confidence: csvParam, source_kind: csvParam, ai: boolParam,
47 + since: strParam, until: strParam, q: strParam, dedupe: boolParam, page: pageParam, per_page: intParam,
48 + });
49 + publicGet(app, { url: "/events", ttl: TTL.events, query: eventsQuery, summary: "Change feed (EventDTO[]) — newest first; filters: type, country, metro, operator, project, entity, significance band, confidence, source_kind, ai, since/until, q; dedupe=true collapses clustered coverage", tags: ["events"], response: { type: "EventDTO[]" } }, async (q) => {
50 + const res = await listEvents({ type: csv(q.type), country: q.country, operator: q.operator, metro: q.metro, project: q.project, entityType: q.entity_type, entityId: q.entity_id, minSignificance: q.min_significance, significance: q.significance, confidence: csv(q.confidence), sourceKind: csv(q.source_kind), ai: q.ai, since: q.since, until: q.until, q: q.q, dedupe: q.dedupe !== false, page: q.page, perPage: q.per_page });
51 + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, dedupe: q.dedupe !== false, methodology: EVENTS_DEDUPE_METHODOLOGY }, sources: await sourcesForIds(res.items.map((e) => e.sourceId)) };
36 52 });
37 − publicGet(app, { url: "/events/:id", ttl: TTL.detail, summary: "Single event (EventDTO)", tags: ["events"], params: { id: "evt_… id" } }, async (_q, params) => {
53 + publicGet(app, { url: "/events/:id", ttl: TTL.detail, summary: "Single event (EventDTO) with otherSources", tags: ["events"], params: { id: "evt_… id" }, response: { type: "EventDTO" } }, async (_q, params) => {
38 54 const e = await getEvent(params.id!);
39 55 if (!e) throw notFound("event");
40 − return { data: e };
56 + return { data: e, sources: await sourcesForIds([e.sourceId]) };
41 57 });
42 58
43 59 // ---- news
44 − publicGet(app, { url: "/news", ttl: TTL.events, query: z.object({ country: strParam, operator: strParam, since: strParam, q: strParam, page: pageParam, per_page: intParam }), summary: "News items (crawled announcements) — newest first", tags: ["events"] }, async (q) => {
60 + publicGet(app, { url: "/news", ttl: TTL.events, query: z.object({ country: strParam, operator: strParam, since: strParam, q: strParam, page: pageParam, per_page: intParam }), summary: "News items (crawled announcements) — newest first", tags: ["events"], response: { type: "NewsItemDTO[]" } }, async (q) => {
45 61 const res = await listNews(q);
46 62 return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage } };
47 63 });
modified apps/api/src/routes/public/discovery.ts +21 −8
@@ -2,11 +2,15 @@
2 2 import type { FastifyInstance } from "fastify";
3 3 import { z } from "zod";
4 4 import { publicGet } from "../../lib/route.js";
5 +import { DENSITY_VIEWS, MAP_LAYERS } from "@dci/core";
5 6 import { TTL, badRequest, csv, notFound } from "../../lib/http.js";
6 7 import { boolParam, csvParam, intParam, numParam, pageParam, strParam } from "../../lib/params.js";
7 8 import { METHODOLOGY_MW } from "../../lib/sql.js";
8 −import { mapData, MAX_POINTS, type Bbox } from "../../repositories/map.js";
9 −import { search } from "../../repositories/search.js";
9 +import { sourcesForEntities } from "../../lib/source-history.js";
10 +import { mapData, MAP_METHODOLOGY, MAX_POINTS, type Bbox } from "../../repositories/map.js";
11 +import { facilityIdsOf, search } from "../../repositories/search.js";
12 +
13 +const empty = (v: unknown) => (v === "" || v === null ? undefined : v);
10 14 import { getRanking, listRankings } from "../../repositories/rankings.js";
11 15 import { dashboard } from "../../repositories/dashboard.js";
12 16 import { availableMetrics, getSource, listSources, sitemap, SITEMAP_KINDS, timeseries, type SitemapKind } from "../../repositories/misc.js";
@@ -21,15 +25,24 @@ function parseBbox(s: string | undefined): Bbox | undefined {
21 25 }
22 26
23 27 export async function discoveryRoutes(app: FastifyInstance): Promise<void> {
24 − const mapQuery = z.object({ zoom: numParam, bbox: strParam, status: csvParam, type: csvParam, operator: strParam, country: csvParam, min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, cloud_regions: boolParam });
25 − publicGet(app, { url: "/map", ttl: TTL.map, query: mapQuery, summary: "Map data: country clusters (zoom < 5), grid clusters (5–8), points (≥ 9, capped 5 000 → clusters)", tags: ["map"] }, async (q) => {
26 − const data = await mapData({ zoom: q.zoom ?? 2, bbox: parseBbox(q.bbox), status: csv(q.status), type: csv(q.type), operator: q.operator, country: csv(q.country).map((c) => c.toUpperCase()), min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, cloud_regions: q.cloud_regions });
27 − return { data, meta: { maxPoints: MAX_POINTS, methodology: "Only facilities with coordinates; `p` is the geo precision (city-level points are approximate). mw = best known figure." } };
28 + const mapQuery = z.object({
29 + zoom: numParam, bbox: strParam.describe("w,s,e,n"), status: csvParam, type: csvParam, operator: strParam, metro: strParam, country: csvParam, q: strParam, min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, has_mw: boolParam, confidence: csvParam, opened_from: intParam, opened_to: intParam, cloud_regions: boolParam.describe("legacy: add the cloud-region overlay"),
30 + layer: z.preprocess(empty, z.enum(MAP_LAYERS).optional()).describe(`thematic layer: ${MAP_LAYERS.join(" | ")} (default facilities)`),
31 + density: z.preprocess(empty, z.enum(DENSITY_VIEWS).optional()).describe(`density grid weighted by ${DENSITY_VIEWS.join(" | ")}`),
32 + year: intParam.describe("time machine: facilities with an opening date ≤ year"),
33 + project_status: csvParam, expected_from: intParam, expected_to: intParam, location_precision: csvParam, ai_evidence: csvParam.describe("confirmed | likely | associated | unknown"),
34 + });
35 + publicGet(app, { url: "/map", ttl: TTL.map, query: mapQuery, summary: "Map data (MapResponse): country clusters (zoom < 5), grid clusters (5–8), points (≥ 9, capped 5 000 → clusters + degraded); layers (capacity, projects, ai, cloud, ixps, connectivity, power, pipeline), density grids and a year filter", tags: ["map"], response: { type: "MapResponse", example: { zoom: 4, mode: "clusters", layer: "facilities", clusters: [{ key: "US", lat: 39.8, lng: -98.6, count: 2600, mw: 12000, statuses: { operational: 2400 }, label: "United States" }], total: 2600, year: null, yearCoverage: null } } }, async (q) => {
36 + const data = await mapData({ zoom: q.zoom ?? 2, bbox: parseBbox(q.bbox), status: csv(q.status), type: csv(q.type), operator: q.operator, metro: q.metro, country: csv(q.country).map((c) => c.toUpperCase()), q: q.q, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, confidence: csv(q.confidence), opened_from: q.opened_from, opened_to: q.opened_to, cloud_regions: q.cloud_regions, layer: q.layer, density: q.density, year: q.year, project_status: csv(q.project_status), expected_from: q.expected_from, expected_to: q.expected_to, location_precision: csv(q.location_precision), ai_evidence: csv(q.ai_evidence) });
37 + const facilityIds = data.mode === "points" ? (data.points ?? []).filter((p) => !p.k).map((p) => p.id) : [];
38 + const sources = facilityIds.length ? await sourcesForEntities("facility", facilityIds) : [];
39 + return { data, meta: { maxPoints: MAX_POINTS, layer: data.layer, methodology: MAP_METHODOLOGY }, sources };
28 40 });
29 41
30 − publicGet(app, { url: "/search", ttl: TTL.search, query: z.object({ q: strParam, type: strParam }), summary: "Universal search: interprets the query (country, operator, status, MW, type) and returns ranked hits + matching facilities", tags: ["search"] }, async (q) => {
42 + publicGet(app, { url: "/search", ttl: TTL.search, query: z.object({ q: strParam, type: strParam }), summary: "Universal search (SearchResponse): interprets the query (country, metro, operator, status, min / max MW, type, AI, projects, opening year) and returns ranked hits + matching facilities or projects", tags: ["search"], response: { type: "SearchResponse", example: { query: "projects over 500 MW in texas", interpreted: { entity: "project", minMw: 500, text: "texas" }, hits: [] } } }, async (q) => {
31 43 const data = await search(q.q ?? "");
32 − return { data, meta: { total: data.hits.length } };
44 + const sources = await sourcesForEntities("facility", facilityIdsOf(data));
45 + return { data, meta: { total: data.hits.length, interpreted: data.interpreted }, sources };
33 46 });
34 47
35 48 publicGet(app, { url: "/rankings", ttl: TTL.rankings, summary: "Current rankings (key, label, scope, unit, methodology, computedAt, total)", tags: ["rankings"] }, async () => {
modified apps/api/src/routes/public/docs-meta.ts +39 −3
@@ -1,6 +1,42 @@
1 −/** /docs-meta — endpoint catalogue generated from the route registry for the web API docs page. */
1 +/** /docs-meta — endpoint catalogue (ApiEndpointDoc[]) generated from the live route registry for the web API docs page. */
2 2 import type { FastifyInstance } from "fastify";
3 +import type { ApiEndpointDoc } from "@dci/core";
4 +import { getEnv } from "../../env.js";
5 +import { publicGet, routeRegistry, type RegisteredRoute } from "../../lib/route.js";
6 +import { TTL } from "../../lib/http.js";
3 7
4 −export async function docsMetaRoutes(_app: FastifyInstance): Promise<void> {
5 − // filled in by the intelligence implementation
8 +const SAMPLE: Record<string, string> = { idOrSlug: "equinix-dc2", slug: "equinix", slugOrIso2: "us", key: "facilities", id: "evt_example", kind: "facilities", lat: "39.04", lng: "-77.49", radius_km: "25", country: "US", q: "hyperscale texas", zoom: "4", window: "7d", slugs: "equinix,digital-realty", format: "csv", entity: "facilities", status: "operational", per_page: "10", page: "1", layer: "facilities", year: "2020", types: "facilities,ixps", metric: "facilities_total" };
9 +
10 +function sampleFor(p: RegisteredRoute["params"][number]): string {
11 + if (p.example) return p.example;
12 + if (SAMPLE[p.name]) return SAMPLE[p.name]!;
13 + if (p.type === "integer" || p.type === "number") return "1";
14 + if (p.type === "boolean") return "true";
15 + return "value";
16 +}
17 +
18 +export function endpointDoc(r: RegisteredRoute, siteUrl: string): ApiEndpointDoc {
19 + let path = r.path;
20 + for (const p of r.params.filter((x) => x.in === "path")) path = path.replace(`:${p.name}`, encodeURIComponent(sampleFor(p)));
21 + const qs = r.params.filter((x) => x.in === "query").slice(0, 2).map((p) => `${p.name}=${encodeURIComponent(sampleFor(p))}`).join("&");
22 + const url = `${siteUrl}${path}${qs && r.method === "GET" ? `?${qs}` : ""}`;
23 + const method: ApiEndpointDoc["method"] = r.method === "PATCH" ? "POST" : r.method;
24 + const bodyObj = r.method === "POST" ? Object.fromEntries(r.params.filter((x) => x.in === "query").slice(0, 2).map((p) => [p.name, sampleFor(p)])) : null;
25 + const body = bodyObj ? JSON.stringify(bodyObj) : null;
26 + const curl = method === "GET" ? `curl -s "${url}"` : method === "DELETE" ? `curl -s -X DELETE "${url}"` : `curl -s -X POST "${url}" -H 'content-type: application/json' -d '${body}'`;
27 + const js = method === "GET"
28 + ? `const res = await fetch("${url}");\nconst { data, meta, sources } = await res.json();`
29 + : `const res = await fetch("${url}", { method: "${method}"${body ? `, headers: { "content-type": "application/json" }, body: JSON.stringify(${body})` : ""}, credentials: "include" });\nconst { data } = await res.json();`;
30 + const python = method === "GET"
31 + ? `import requests\nr = requests.get("${url}")\npayload = r.json()\ndata, meta, sources = payload["data"], payload.get("meta"), payload.get("sources")`
32 + : `import requests\nr = requests.${method.toLowerCase()}("${url}"${body ? `, json=${body}` : ""})\ndata = r.json()["data"]`;
33 + return { method, path: r.path, summary: r.summary, group: r.group, params: r.params.map((p) => ({ name: p.name, in: p.in, type: p.type, description: p.description, ...(p.example ? { example: p.example } : {}) })), example: { curl, js, python }, responseSchema: r.responseType + (r.responseDescription ? ` — ${r.responseDescription}` : "") };
34 +}
35 +
36 +export async function docsMetaRoutes(app: FastifyInstance): Promise<void> {
37 + publicGet(app, { url: "/docs-meta", ttl: TTL.sitemap, tags: ["docs"], summary: "Endpoint catalogue (ApiEndpointDoc[]) generated from the live route table: method, path, params, curl / JS / Python examples, response type", response: { type: "ApiEndpointDoc[]", example: [{ method: "GET", path: "/api/v1/datacenters", summary: "List facilities", group: "facilities", params: [{ name: "country", in: "query", type: "string", description: "ISO-3166 alpha-2" }], example: { curl: "curl -s https://www.datacenterindex.io/api/v1/datacenters?country=US", js: "await fetch(…)", python: "requests.get(…)" }, responseSchema: "FacilitySummary[]" }] } }, async () => {
38 + const siteUrl = getEnv().siteUrl.replace(/\/$/, "");
39 + const data = routeRegistry.map((r) => endpointDoc(r, siteUrl)).sort((a, b) => a.group.localeCompare(b.group) || a.path.localeCompare(b.path) || a.method.localeCompare(b.method));
40 + return { data, meta: { total: data.length, methodology: "Generated from the Fastify route table at boot; every 200 response is the envelope { data, meta, sources } unless the summary says otherwise (downloads stream raw rows)." } };
41 + });
6 42 }
modified apps/api/src/routes/public/download.ts +41 −3
@@ -1,6 +1,44 @@
1 −/** Dataset downloads (CSV / JSON / GeoJSON) with license gating. */
1 +/** Dataset downloads (CSV / JSON / GeoJSON) with license gating — streamed, no envelope. */
2 +import { Readable } from "node:stream";
2 3 import type { FastifyInstance } from "fastify";
4 +import { z } from "zod";
5 +import { publicGet, registerRoute } from "../../lib/route.js";
6 +import { TTL, HttpError, parseQuery } from "../../lib/http.js";
7 +import { boolParam, csvParam, intParam, numParam, strParam } from "../../lib/params.js";
8 +import { contentType, DOWNLOAD_KEYS, DOWNLOAD_LICENSE, listDatasets, MAX_ROWS, restrictedSources, specFor, streamDataset, type DownloadFormat, type DownloadKey } from "../../repositories/download.js";
3 9
4 −export async function downloadRoutes(_app: FastifyInstance): Promise<void> {
5 − // filled in by the intelligence implementation
10 +const empty = (v: unknown) => (v === "" || v === null ? undefined : v);
11 +const downloadQuery = z.object({
12 + format: z.preprocess(empty, z.enum(["csv", "json", "geojson"]).optional()),
13 + country: csvParam, status: csvParam, project_status: csvParam, type: csvParam, operator: strParam, metro: strParam,
14 + min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, has_mw: boolParam,
15 + expected_from: intParam, expected_to: intParam, event_type: csvParam, type_: strParam.optional(),
16 + since: strParam, until: strParam, min_significance: intParam,
17 +});
18 +
19 +export async function downloadRoutes(app: FastifyInstance): Promise<void> {
20 + publicGet(app, { url: "/download/datasets", ttl: TTL.list, tags: ["download"], summary: "Downloadable datasets (DownloadDataset[]): formats, accepted filters, live row counts, licence and excluded (non-redistributable) sources", response: { type: "DownloadDataset[]", example: [{ key: "facilities", label: "Facilities", description: "…", formats: ["csv", "json", "geojson"], filters: ["country", "status"], rows: 8469, license: "…", attribution: [], excludedSources: [] }] } }, async () => {
21 + const data = await listDatasets();
22 + return { data, meta: { total: data.length, maxRows: MAX_ROWS, methodology: DOWNLOAD_LICENSE } };
23 + });
24 +
25 + registerRoute({ method: "GET", path: "/api/v1/download/:key", summary: `Stream a dataset (${DOWNLOAD_KEYS.join(" | ")}) as CSV, JSON or GeoJSON (≤ ${MAX_ROWS} rows, license-gated, Content-Disposition attachment; no envelope)`, group: "download", params: [{ name: "key", in: "path", type: "string", description: DOWNLOAD_KEYS.join(" | "), example: "facilities" }, { name: "format", in: "query", type: "csv|json|geojson", description: "output format (default csv)", example: "csv" }, { name: "country", in: "query", type: "string", description: "ISO-3166 alpha-2 (comma list)", example: "US" }], responseType: "text/csv | { data: rows[], meta } | GeoJSON FeatureCollection" });
26 + app.get("/download/:key", { schema: { summary: `Stream a dataset (${DOWNLOAD_KEYS.join(" | ")}) as csv | json | geojson — ≤ ${MAX_ROWS} rows, license-gated (rows whose only sources forbid redistribution are excluded and listed in x-dci-excluded-sources / meta.excludedSources), no envelope`, tags: ["download"], params: { type: "object", properties: { key: { type: "string", enum: [...DOWNLOAD_KEYS] } } }, querystring: { type: "object", properties: { format: { type: "string", enum: ["csv", "json", "geojson"] }, country: { type: "string" }, status: { type: "string" }, operator: { type: "string" }, since: { type: "string" } } }, response: { 200: { description: "Dataset rows (CSV text, JSON { data, meta } or GeoJSON FeatureCollection)", type: "string" } } } }, async (req, reply) => {
27 + const { key } = req.params as { key: string };
28 + const spec = specFor(key);
29 + if (!spec) throw new HttpError(404, "dataset not found");
30 + const q = parseQuery(downloadQuery, req.query);
31 + const format: DownloadFormat = q.format ?? "csv";
32 + if (!spec.formats.includes(format)) throw new HttpError(400, `format ${format} is not available for ${key} (${spec.formats.join(", ")})`);
33 + const excluded = spec.gated ? await restrictedSources() : [];
34 + const filters = { country: q.country, status: q.status, project_status: q.project_status, type: q.type, operator: q.operator, metro: q.metro, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, expected_from: q.expected_from, expected_to: q.expected_to, event_type: q.event_type, since: q.since, until: q.until, min_significance: q.min_significance };
35 + const day = new Date().toISOString().slice(0, 10);
36 + reply.header("content-type", contentType(format));
37 + reply.header("content-disposition", `attachment; filename="dci-${key}-${day}.${format}"`);
38 + reply.header("cache-control", "public, max-age=300");
39 + reply.header("x-dci-license", "attribution required; see /api/v1/download/datasets");
40 + reply.header("x-dci-excluded-sources", excluded.map((s) => s.id).join(",") || "none");
41 + reply.header("x-dci-max-rows", String(MAX_ROWS));
42 + return reply.send(Readable.from(streamDataset(spec.key as DownloadKey, format, filters, excluded)));
43 + });
6 44 }
modified apps/api/src/routes/public/facilities.ts +33 −9
@@ -1,19 +1,43 @@
1 1 import type { FastifyInstance } from "fastify";
2 +import { z } from "zod";
2 3 import { publicGet } from "../../lib/route.js";
3 −import { TTL, notFound } from "../../lib/http.js";
4 −import { facilityFiltersSchema } from "../../lib/params.js";
5 −import { METHODOLOGY_MW } from "../../lib/sql.js";
6 −import { getFacilityDetail, listFacilities, toFacilityQuery } from "../../repositories/facilities.js";
4 +import { TTL, csv, notFound } from "../../lib/http.js";
5 +import { facilityFiltersSchema, csvParam, numParam, strParam } from "../../lib/params.js";
6 +import { METHODOLOGY_MW, METHODOLOGY_CONTAINMENT } from "../../lib/sql.js";
7 +import { sourcesForEntities } from "../../lib/source-history.js";
8 +import { getFacilityClaims, getFacilityDetail, getFacilityHistory, getFacilityProvenance, listFacilities, toFacilityQuery } from "../../repositories/facilities.js";
9 +
10 +const LIST_NOTE = `${METHODOLOGY_MW} ${METHODOLOGY_CONTAINMENT} List rows may show both a campus (recordScope=campus) and its buildings (parentFacility set) — aggregates never count both.`;
11 +const radiusQuery = z.object({ radius_km: numParam });
12 +const claimsQuery = z.object({ status: csvParam, predicate: strParam });
7 13
8 14 export async function facilityRoutes(app: FastifyInstance): Promise<void> {
9 − publicGet(app, { url: "/datacenters", ttl: TTL.list, query: facilityFiltersSchema, summary: "List facilities (FacilitySummary[]) with filters, sorting and pagination", tags: ["facilities"] }, async (q) => {
15 + publicGet(app, { url: "/datacenters", ttl: TTL.list, query: facilityFiltersSchema, summary: "List facilities (FacilitySummary[]) with filters, sorting and pagination", tags: ["facilities"], response: { type: "FacilitySummary[]", description: "Facilities matching the filters; `sources` = distinct sources behind the returned rows (≤ 30).", example: [{ id: "fac_…", slug: "example-dc1", name: "Example DC1", status: "operational", itCapacityMw: 12.5, recordScope: "facility", aiEvidence: "unknown" }] } }, async (q) => {
10 16 const res = await listFacilities(toFacilityQuery(q));
11 − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: METHODOLOGY_MW } };
17 + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: LIST_NOTE }, sources: await sourcesForEntities("facility", res.items.map((f) => f.id)) };
18 + });
19 +
20 + publicGet(app, { url: "/datacenters/:idOrSlug", ttl: TTL.detail, query: radiusQuery, summary: "Facility detail by slug or id (FacilityDetail): buildings, tenants, claims, capacity history, nearby infrastructure (?radius_km ≤ 200, default 25), power context, data quality", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "FacilityDetail" } }, async (q, params) => {
21 + const res = await getFacilityDetail(params.idOrSlug!, { radiusKm: q.radius_km != null ? Math.min(200, Math.max(0.1, q.radius_km)) : undefined });
22 + if (!res) throw notFound("facility");
23 + return { data: res.detail, sources: res.sources, meta: { methodology: `${METHODOLOGY_MW} Claims list every figure published about this site with its scope; only site-scoped, non-rejected claims may back the displayed values (isWinner).` } };
24 + });
25 +
26 + publicGet(app, { url: "/datacenters/:idOrSlug/history", ttl: TTL.detail, summary: "Field history (EntityHistory): dated observations, claims and value changes per field", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "EntityHistory", example: { entityType: "facility", entityId: "fac_…", fields: { itCapacityMw: [{ date: "2025-03-01T00:00:00.000Z", field: "itCapacityMw", value: 12.5, kind: "observed", sourceId: "src_…", sourceName: "Operator site", sourceKind: "operator", url: "https://…" }] }, changes: [] } } }, async (_q, params) => {
27 + const res = await getFacilityHistory(params.idOrSlug!);
28 + if (!res) throw notFound("facility");
29 + return { data: res.history, sources: res.sources, meta: { methodology: "observed = a source page stated the value (first observation date); claim = a figure asserted by a document (published date when known); changed = a detected value change with old → new. Dates are ISO or partial (YYYY, YYYY-MM)." } };
30 + });
31 +
32 + publicGet(app, { url: "/datacenters/:idOrSlug/claims", ttl: TTL.detail, query: claimsQuery, summary: "Claims about a facility (ClaimDTO[]) — every published figure with scope, evidence sentence, authority tier and status (?status=current,unscoped&predicate=)", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "ClaimDTO[]" } }, async (q, params) => {
33 + const res = await getFacilityClaims(params.idOrSlug!, { status: csv(q.status), predicate: q.predicate });
34 + if (!res) throw notFound("facility");
35 + return { data: res.claims, sources: res.sources, meta: { total: res.claims.length, facilityId: res.id, methodology: "A claim is one figure asserted by one document about this site. Scope building/facility/campus may back the displayed value (isWinner); portfolio/company/country/metro/unknown scopes are kept as claims only (status unscoped)." } };
12 36 });
13 37
14 − publicGet(app, { url: "/datacenters/:idOrSlug", ttl: TTL.detail, summary: "Facility detail by slug or id (FacilityDetail)", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" } }, async (_q, params) => {
15 − const res = await getFacilityDetail(params.idOrSlug!);
38 + publicGet(app, { url: "/datacenters/:idOrSlug/provenance", ttl: TTL.detail, summary: "All provenance observations for a facility (ProvenanceDTO[]), current and superseded, with winner flag", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "ProvenanceDTO[]" } }, async (_q, params) => {
39 + const res = await getFacilityProvenance(params.idOrSlug!);
16 40 if (!res) throw notFound("facility");
17 − return { data: res.detail, sources: res.sources, meta: { methodology: METHODOLOGY_MW } };
41 + return { data: res.provenance, sources: res.sources, meta: { total: res.provenance.length, current: res.current, superseded: res.provenance.length - res.current, facilityId: res.id } };
18 42 });
19 43 }
modified apps/api/src/routes/public/graph.ts +21 −19
@@ -4,7 +4,7 @@ import { z } from "zod";
4 4 import { publicGet } from "../../lib/route.js";
5 5 import { TTL, notFound } from "../../lib/http.js";
6 6 import { boolParam, intParam, orderParam, pageParam, strParam } from "../../lib/params.js";
7 −import { METHODOLOGY_MW } from "../../lib/sql.js";
7 +import { METHODOLOGY_MW, METHODOLOGY_CONTAINMENT } from "../../lib/sql.js";
8 8 import { getOperatorDetail, listOperators } from "../../repositories/operators.js";
9 9 import { getCountryDetail, listCountries } from "../../repositories/countries.js";
10 10 import { getMetroDetail, listMetros } from "../../repositories/metros.js";
@@ -13,61 +13,63 @@ import { getIxp, listIxps } from "../../repositories/ixps.js";
13 13
14 14 const empty = (v: unknown) => (v === "" || v === null ? undefined : v);
15 15 const fPage = z.object({ fPage: pageParam });
16 +const METHODOLOGY = `${METHODOLOGY_MW} ${METHODOLOGY_CONTAINMENT} mwCoverage (0..1) is the share of counted facilities with a published MW figure — compare MW totals only when coverage is comparable.`;
17 +const HHI_NOTE = "concentration.hhi = Σ (share × 100)² over operators (0–10 000), computed on facility counts and separately on known MW with its coverage; momentum components are shown separately and never collapsed into a score.";
16 18
17 19 export async function graphRoutes(app: FastifyInstance): Promise<void> {
18 20 // ---- operators
19 21 const operatorsQuery = z.object({ q: strParam, kind: strParam, country: strParam, sort: z.preprocess(empty, z.enum(["facilities", "name", "mw"]).optional()), order: orderParam, page: pageParam, per_page: intParam });
20 − publicGet(app, { url: "/operators", ttl: TTL.list, query: operatorsQuery, summary: "List operators (OperatorSummary[]) with live facility aggregates", tags: ["operators"] }, async (q) => {
22 + publicGet(app, { url: "/operators", ttl: TTL.list, query: operatorsQuery, summary: "List operators (OperatorSummary[]) with containment-aware facility aggregates", tags: ["operators"], response: { type: "OperatorSummary[]", example: [{ id: "op_x", slug: "equinix", name: "Equinix", kind: "colocation", facilityCount: 262, countryCount: 33, metroCount: 71, knownMw: 1180.5, plannedMw: null, constructionMw: 60, projectCount: 4, aiCount: 0, cloudRegionCount: 0, mwCoverage: 0.61, isCloudProvider: false, isCarrier: false }] } }, async (q) => {
21 23 const res = await listOperators(q);
22 − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: METHODOLOGY_MW } };
24 + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: METHODOLOGY } };
23 25 });
24 − publicGet(app, { url: "/operators/:slug", ttl: TTL.detail, query: fPage, summary: "Operator detail (OperatorDetail); facilities paginated with ?fPage (50 per page)", tags: ["operators"], params: { slug: "operator slug or op_… id" } }, async (q, params) => {
26 + publicGet(app, { url: "/operators/:slug", ttl: TTL.detail, query: fPage, summary: "Operator detail (OperatorDetail): pipeline, expansion velocity, top countries / metros, AI facilities, cloud regions, corporate events, claims, data quality; facilities paginated with ?fPage (50 per page)", tags: ["operators"], params: { slug: "operator slug or op_… id" }, response: { type: "OperatorDetail", description: "OperatorSummary + pipeline (PipelineBreakdown), velocity (12m / 3y / 5y windows, countriesOverTime with basis), topCountries / topMetros with share, aiFacilities, cloudRegions, corporateEvents, claims, dataQuality and the legacy lists." } }, async (q, params) => {
25 27 const res = await getOperatorDetail(params.slug!, q.fPage);
26 28 if (!res) throw notFound("operator");
27 − return { data: res.detail, sources: res.sources, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: res.detail.facilityCount } };
29 + return { data: res.detail, sources: res.sources, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: res.detail.facilityCount, methodology: `${METHODOLOGY} Velocity windows use opened_on (else first indexed) for facilities and announced_on (else first indexed) for projects — the basis is labelled per point. Corporate events (acquisitions, financing, partnerships, executive changes) are never mixed with physical projects.` } };
28 30 });
29 31
30 32 // ---- countries
31 − publicGet(app, { url: "/countries", ttl: TTL.list, query: z.object({ all: boolParam, region: strParam }), summary: "Countries with ≥1 facility or cloud region (CountrySummary[]); ?all=1 for every country", tags: ["countries"] }, async (q) => {
33 + publicGet(app, { url: "/countries", ttl: TTL.list, query: z.object({ all: boolParam, region: strParam }), summary: "Countries with ≥1 facility or cloud region (CountrySummary[]); ?all=1 for every country", tags: ["countries"], response: { type: "CountrySummary[]", example: [{ iso2: "US", iso3: "USA", slug: "united-states", name: "United States", facilityCount: 3200, operationalCount: 2900, constructionCount: 60, plannedCount: 110, knownMw: 9800, projectCount: 120, projectPlannedMw: 14000, ixpCount: 140, mwCoverage: 0.22 }] } }, async (q) => {
32 34 const items = await listCountries({ all: q.all === true, region: q.region });
33 − return { data: items, meta: { total: items.length, methodology: METHODOLOGY_MW } };
35 + return { data: items, meta: { total: items.length, methodology: METHODOLOGY } };
34 36 });
35 − publicGet(app, { url: "/countries/:slugOrIso2", ttl: TTL.detail, query: fPage, summary: "Country detail (CountryDetail); facilities paginated with ?fPage", tags: ["countries"], params: { slugOrIso2: "country slug, ISO-3166 alpha-2 or alpha-3" } }, async (q, params) => {
37 + publicGet(app, { url: "/countries/:slugOrIso2", ttl: TTL.detail, query: fPage, summary: "Country detail (CountryDetail): IXPs, grid constraints, energy context, AI facilities / projects, pipeline, coverage, claims; facilities paginated with ?fPage", tags: ["countries"], params: { slugOrIso2: "country slug, ISO-3166 alpha-2 or alpha-3" }, response: { type: "CountryDetail", description: "CountrySummary + ixps, gridConstraints, energy (national grid averages with a note — they do not describe a facility's contracted electricity), aiFacilities, aiProjects, pipeline, coverage (CoverageRow), claims and the legacy lists." } }, async (q, params) => {
36 38 const d = await getCountryDetail(params.slugOrIso2!, q.fPage);
37 39 if (!d) throw notFound("country");
38 − return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: METHODOLOGY_MW } };
40 + return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: `${METHODOLOGY} announcedInvestmentUsd sums site-scoped project investments only (company capex, deal values and national programmes are kept as claims).` } };
39 41 });
40 42
41 43 // ---- metros
42 − publicGet(app, { url: "/metros", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Metros / markets with counts (MetroSummary[])", tags: ["metros"] }, async (q) => {
44 + publicGet(app, { url: "/metros", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Metros / markets with containment-aware counts (MetroSummary[])", tags: ["metros"], response: { type: "MetroSummary[]", example: [{ id: "met_x", slug: "northern-virginia", name: "Northern Virginia", countryIso2: "US", lat: 39.04, lng: -77.49, facilityCount: 310, operationalCount: 280, constructionCount: 20, plannedCount: 10, knownMw: 4200, operatorCount: 40, cloudRegionCount: 6, ixpCount: 4, projectCount: 25, projectPlannedMw: 5600, aiCount: 12, mwCoverage: 0.48 }] } }, async (q) => {
43 45 const items = await listMetros({ country: q.country, q: q.q });
44 − return { data: items, meta: { total: items.length, methodology: METHODOLOGY_MW } };
46 + return { data: items, meta: { total: items.length, methodology: METHODOLOGY } };
45 47 });
46 − publicGet(app, { url: "/metros/:slug", ttl: TTL.detail, query: fPage, summary: "Metro detail (MetroDetail); facilities paginated with ?fPage", tags: ["metros"], params: { slug: "metro slug or met_… id" } }, async (q, params) => {
48 + publicGet(app, { url: "/metros/:slug", ttl: TTL.detail, query: fPage, summary: "Metro detail (MetroDetail): concentration (HHI), 12-month momentum components, pipeline, grid constraints, AI facilities, opening timeline, coverage, claims; facilities paginated with ?fPage", tags: ["metros"], params: { slug: "metro slug or met_… id" }, response: { type: "MetroDetail", description: "MetroSummary + concentration (MarketConcentration), momentum (MarketMomentum, window 12m), pipeline, gridConstraints (grid_constraints rows + grid / utility / power events), aiFacilities, openingTimeline, coverage, claims and the legacy lists." } }, async (q, params) => {
47 49 const d = await getMetroDetail(params.slug!, q.fPage);
48 50 if (!d) throw notFound("metro");
49 − return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: METHODOLOGY_MW } };
51 + return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: `${METHODOLOGY} ${HHI_NOTE}` } };
50 52 });
51 53
52 54 // ---- cloud regions
53 − publicGet(app, { url: "/cloud-regions", ttl: TTL.list, query: z.object({ provider: strParam, country: strParam, status: strParam }), summary: "Cloud regions (CloudRegionSummary[])", tags: ["cloud-regions"] }, async (q) => {
55 + publicGet(app, { url: "/cloud-regions", ttl: TTL.list, query: z.object({ provider: strParam, country: strParam, status: strParam }), summary: "Cloud regions (CloudRegionSummary[])", tags: ["cloud-regions"], response: { type: "CloudRegionSummary[]" } }, async (q) => {
54 56 const items = await listCloudRegions({ provider: q.provider, country: q.country, status: q.status });
55 57 return { data: items, meta: { total: items.length } };
56 58 });
57 − publicGet(app, { url: "/cloud-regions/:slug", ttl: TTL.detail, summary: "Cloud region detail (CloudRegionSummary + metro, siblings, facilities in metro)", tags: ["cloud-regions"], params: { slug: "cloud region slug or cr_… id" } }, async (_q, params) => {
59 + publicGet(app, { url: "/cloud-regions/:slug", ttl: TTL.detail, summary: "Cloud region detail (CloudRegionDetail): host facilities (publicly verified only) or market facilities, siblings, events, provenance", tags: ["cloud-regions"], params: { slug: "cloud region slug or cr_… id" }, response: { type: "CloudRegionDetail" } }, async (_q, params) => {
58 60 const d = await getCloudRegion(params.slug!);
59 61 if (!d) throw notFound("cloud region");
60 − return { data: d };
62 + return { data: d, meta: { methodology: "hostFacilities lists only facilities with an explicit public tenancy link to the provider; otherwise the region is associated with its market (marketFacilities) and never pinned to a building." } };
61 63 });
62 64
63 65 // ---- ixps
64 − publicGet(app, { url: "/ixps", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Internet exchange points (IxpSummary[])", tags: ["ixps"] }, async (q) => {
66 + publicGet(app, { url: "/ixps", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Internet exchange points (IxpSummary[])", tags: ["ixps"], response: { type: "IxpSummary[]" } }, async (q) => {
65 67 const items = await listIxps({ country: q.country, q: q.q });
66 68 return { data: items, meta: { total: items.length } };
67 69 });
68 − publicGet(app, { url: "/ixps/:slug", ttl: TTL.detail, summary: "IXP detail (IxpSummary + facilities)", tags: ["ixps"], params: { slug: "ixp slug or ix_… id" } }, async (_q, params) => {
70 + publicGet(app, { url: "/ixps/:slug", ttl: TTL.detail, summary: "IXP detail (IxpDetail): facilities, operators, nearby facilities (metro coordinates), source history", tags: ["ixps"], params: { slug: "ixp slug or ix_… id" }, response: { type: "IxpDetail" } }, async (_q, params) => {
69 71 const d = await getIxp(params.slug!);
70 72 if (!d) throw notFound("ixp");
71 − return { data: d };
73 + return { data: d, meta: { methodology: "IXPs carry no published coordinates; nearby facilities are computed from the metro reference point when the IXP is assigned to a metro." } };
72 74 });
73 75 }
modified apps/api/src/routes/public/intelligence.ts +76 −2
@@ -1,6 +1,80 @@
1 1 /** Nearby, explore, AI index, power, connectivity, time machine. */
2 2 import type { FastifyInstance } from "fastify";
3 +import { z } from "zod";
4 +import { publicGet } from "../../lib/route.js";
5 +import { TTL, badRequest } from "../../lib/http.js";
6 +import { boolParam, csvParam, intParam, numParam, orderParam, pageParam, strParam } from "../../lib/params.js";
7 +import { METHODOLOGY_CONTAINMENT } from "../../lib/sql.js";
8 +import { nearby, parseNearbyTypes, NEARBY_MAX_RADIUS_KM, NEARBY_TYPES } from "../../lib/nearby.js";
9 +import { explore, EXPLORE_METHODOLOGY } from "../../repositories/explore.js";
10 +import { aiIndex, AI_EVIDENCE_NOTE } from "../../repositories/ai.js";
11 +import { powerOverview, POWER_NOTE } from "../../repositories/power.js";
12 +import { connectivityOverview, CONNECTIVITY_NOTE } from "../../repositories/connectivity.js";
13 +import { timeMachine, TIME_MACHINE_NOTE } from "../../repositories/time-machine.js";
3 14
4 −export async function intelligenceRoutes(_app: FastifyInstance): Promise<void> {
5 − // filled in by the intelligence implementation
15 +const empty = (v: unknown) => (v === "" || v === null ? undefined : v);
16 +
17 +const nearbyQuery = z.object({ lat: numParam.describe("latitude (-90..90)"), lng: numParam.describe("longitude (-180..180)"), radius_km: numParam.describe(`radius in km (default 25, max ${NEARBY_MAX_RADIUS_KM})`), types: csvParam.describe(`comma list of ${NEARBY_TYPES.join("|")} (default all)`) });
18 +
19 +export const exploreQuery = z.object({
20 + entity: z.preprocess(empty, z.enum(["facilities", "projects"]).optional()),
21 + status: csvParam,
22 + project_status: csvParam,
23 + country: csvParam,
24 + metro: strParam,
25 + operator: strParam,
26 + type: csvParam,
27 + min_mw: numParam,
28 + max_mw: numParam,
29 + ai: z.preprocess(empty, z.enum(["confirmed", "likely", "associated", "any"]).optional()),
30 + hyperscale: boolParam,
31 + expected_before: intParam,
32 + expected_after: intParam,
33 + opened_from: intParam,
34 + opened_to: intParam,
35 + announced_since: strParam,
36 + confidence: csvParam,
37 + location_precision: csvParam,
38 + has_mw: boolParam,
39 + project_class: csvParam,
40 + q: strParam,
41 + sort: z.preprocess(empty, z.enum(["name", "mw", "updated", "opened", "completeness", "announced", "opening"]).optional()),
42 + order: orderParam,
43 + page: pageParam,
44 + per_page: intParam,
45 + view: z.preprocess(empty, z.enum(["table", "map", "charts"]).optional()),
46 +}).strict();
47 +
48 +export async function intelligenceRoutes(app: FastifyInstance): Promise<void> {
49 + publicGet(app, { url: "/nearby", ttl: TTL.detail, query: nearbyQuery, tags: ["nearby"], summary: "Infrastructure around a point (NearbyInfrastructure): facilities, live projects, IXPs, cloud regions, metros within radius_km", description: "Great-circle distances on a bbox pre-filter. Layers without a connector (landing stations, substations, power plants) are empty with a note.", response: { type: "NearbyInfrastructure", example: { center: { lat: 39.04, lng: -77.49 }, radiusKm: 25, facilities: [], projects: [], ixps: [], cloudRegions: [], metros: [], landingStations: [], substations: [], powerPlants: [], note: "…" } } }, async (q) => {
50 + if (q.lat == null || q.lng == null || Math.abs(q.lat) > 90 || Math.abs(q.lng) > 180) throw badRequest("lat and lng are required (lat -90..90, lng -180..180)");
51 + if (q.radius_km != null && (q.radius_km <= 0 || q.radius_km > NEARBY_MAX_RADIUS_KM)) throw badRequest(`radius_km must be between 0 and ${NEARBY_MAX_RADIUS_KM}`);
52 + const data = await nearby({ lat: q.lat, lng: q.lng, radiusKm: q.radius_km, types: parseNearbyTypes(q.types) });
53 + return { data, meta: { total: data.facilities.length + data.projects.length + data.ixps.length + data.cloudRegions.length + data.metros.length, methodology: data.note } };
54 + });
55 +
56 + publicGet(app, { url: "/explore", ttl: TTL.list, query: exploreQuery, tags: ["explore"], summary: "Structured explorer (ExploreResponse): facilities or projects with facets, charts and map points; unknown query keys are rejected (400)", response: { type: "ExploreResponse", example: { query: { entity: "facilities", country: "US" }, total: 0, items: [], facets: { status: [], country: [], operator: [], type: [], ai: [] }, charts: { byStatus: [], byCountry: [], byYear: [] }, map: { points: [], total: 0, degraded: false }, mwCoverage: 0 } } }, async (q) => {
57 + const data = await explore(q);
58 + return { data, meta: { total: data.total, page: q.page, perPage: q.per_page ?? 24, mwCoverage: data.mwCoverage, methodology: `${EXPLORE_METHODOLOGY} ${METHODOLOGY_CONTAINMENT}` } };
59 + });
60 +
61 + publicGet(app, { url: "/ai-infrastructure", ttl: TTL.detail, tags: ["ai"], summary: "AI / HPC infrastructure index (AiIndex): confirmed + likely facilities and live projects, top operators / metros / countries, pipeline by stage", response: { type: "AiIndex", example: { stats: { facilities: 0, confirmed: 0, likely: 0, associated: 0, projects: 0, plannedMw: null, constructionMw: null, countries: 0, operators: 0, mwCoverage: 0 }, topOperators: [], topMetros: [], topCountries: [], pipelineByStage: [], recentAnnouncements: [], recentProjects: [], facilities: [], evidenceNote: "…" } } }, async () => {
62 + const data = await aiIndex();
63 + return { data, meta: { total: data.stats.facilities, mwCoverage: data.stats.mwCoverage, methodology: AI_EVIDENCE_NOTE } };
64 + });
65 +
66 + publicGet(app, { url: "/power", ttl: TTL.detail, tags: ["power"], summary: "Power overview (PowerOverview): grid constraints, power / grid events, large loads (utility, grid, planned ≥ 200 MW), national energy context, utilities", response: { type: "PowerOverview", example: { gridConstraints: [], powerEvents: [], largeLoads: [], countryEnergy: [], utilities: [], note: "…" } } }, async () => {
67 + const data = await powerOverview();
68 + return { data, meta: { total: data.largeLoads.length, methodology: POWER_NOTE } };
69 + });
70 +
71 + publicGet(app, { url: "/connectivity", ttl: TTL.detail, tags: ["connectivity"], summary: "Connectivity overview (ConnectivityOverview): IXPs, cloud regions, carrier hotels, per-metro interconnection density", response: { type: "ConnectivityOverview", example: { ixps: [], cloudRegions: [], carrierHotels: [], landingStations: [], byMetro: [], note: "…" } } }, async () => {
72 + const data = await connectivityOverview();
73 + return { data, meta: { total: data.ixps.length, methodology: CONNECTIVITY_NOTE } };
74 + });
75 +
76 + publicGet(app, { url: "/time-machine", ttl: TTL.detail, tags: ["time-machine"], summary: "Yearly frames of the index from published opening dates (TimeMachine) with earliestReliableYear and openingDateCoverage", response: { type: "TimeMachine", example: { frames: [{ year: 2020, facilities: 120, knownMw: 950.5, announced: 12, construction: 4, opened: 9 }], earliestReliableYear: 2005, openingDateCoverage: 0.12, note: "…" } } }, async () => {
77 + const data = await timeMachine();
78 + return { data, meta: { total: data.frames.length, openingDateCoverage: data.openingDateCoverage, methodology: TIME_MACHINE_NOTE } };
79 + });
6 80 }
modified apps/api/src/routes/public/markets.ts +51 −3
@@ -1,6 +1,54 @@
1 −/** Pulse, operator comparison, coverage report (owned by the aggregates group). */
1 +/** Pulse, operator comparison, coverage report. */
2 2 import type { FastifyInstance } from "fastify";
3 +import { z } from "zod";
4 +import { publicGet } from "../../lib/route.js";
5 +import { TTL, badRequest, csv } from "../../lib/http.js";
6 +import { csvParam } from "../../lib/params.js";
7 +import { pulse, PULSE_METHODOLOGY } from "../../repositories/pulse.js";
8 +import { compareOperators, COMPARE_METHODOLOGY } from "../../repositories/compare.js";
9 +import { coverageReport, sourceCoverage, COVERAGE_METHODOLOGY } from "../../repositories/coverage.js";
3 10
4 −export async function marketRoutes(_app: FastifyInstance): Promise<void> {
5 − // filled in by the aggregates implementation
11 +const empty = (v: unknown) => (v === "" || v === null ? undefined : v);
12 +
13 +export async function marketRoutes(app: FastifyInstance): Promise<void> {
14 + publicGet(app, {
15 + url: "/pulse", ttl: TTL.dashboard, tags: ["pulse"],
16 + query: z.object({ window: z.preprocess(empty, z.enum(["24h", "7d", "30d"]).default("24h")) }),
17 + summary: "Infrastructure pulse: what measurably changed in the last 24h / 7d / 30d (Pulse)",
18 + description: "Deterministic counts (events, new projects, construction starts, openings, new markets per operator, power / grid events) — no scoring.",
19 + response: { type: "Pulse", description: "Counts over the window plus the major events and per-country / per-operator breakdowns.", example: { window: "7d", since: "2026-09-05T00:00:00.000Z", newProjects: 12, projectsEnteredConstruction: 3, facilitiesOpened: 2, newlyAnnouncedMw: 850, eventsTotal: 310, operatorsNewMarkets: [{ operator: { id: "op_x", slug: "example", name: "Example" }, market: { id: "met_x", slug: "ashburn", name: "Northern Virginia" }, countryIso2: "US" }] } },
20 + }, async (q) => {
21 + const data = await pulse(q.window);
22 + return { data, meta: { window: q.window, methodology: PULSE_METHODOLOGY } };
23 + });
24 +
25 + publicGet(app, {
26 + url: "/compare/operators", ttl: TTL.detail, tags: ["compare"],
27 + query: z.object({ slugs: csvParam }),
28 + summary: "Compare 2–5 operators side by side (OperatorComparison)",
29 + response: { type: "OperatorComparison", description: "One column per operator: summary, pipeline, velocity windows, new markets (12 m), recent projects, MW coverage.", example: { operators: [{ id: "op_x", slug: "equinix", name: "Equinix", facilityCount: 260, countryCount: 33, knownMw: 1200, mwCoverage: 0.62, pipeline: { operational: { count: 240, mw: 1200 } } }], generatedAt: "2026-09-12T00:00:00.000Z" } },
30 + }, async (q) => {
31 + const slugs = [...new Set(csv(q.slugs))];
32 + if (slugs.length < 2 || slugs.length > 5) throw badRequest("slugs must list 2 to 5 operator slugs (comma-separated)");
33 + const data = await compareOperators(slugs);
34 + return { data, meta: { methodology: COMPARE_METHODOLOGY } };
35 + });
36 +
37 + publicGet(app, {
38 + url: "/coverage", ttl: TTL.detail, tags: ["coverage"],
39 + summary: "Coverage report: share of known fields per country / metro / operator (≥ 5 facilities) / field / source kind (CoverageReport)",
40 + response: { type: "CoverageReport", description: "Containment-aware shares 0..1; a low share means unknown, not zero.", example: { global: { key: "global", name: "Global", slug: "global", facilities: 8400, capacityCoverage: 0.18, operatorCoverage: 0.9, preciseLocationCoverage: 0.55 }, fields: [{ field: "it_capacity_mw", label: "IT capacity (MW)", coverage: 0.12, count: 1000 }] } },
41 + }, async () => {
42 + const data = await coverageReport();
43 + return { data, meta: { methodology: COVERAGE_METHODOLOGY } };
44 + });
45 +
46 + publicGet(app, {
47 + url: "/coverage/sources", ttl: TTL.detail, tags: ["coverage"],
48 + summary: "Per-source contribution, freshness, authority tier, failure rate and licence (SourceCoverage[])",
49 + response: { type: "SourceCoverage[]", description: "One row per source; uniqueRecords = entities only this source documents.", example: [{ id: "src_x", name: "Equinix", kind: "operator", recordsContributed: 262, fieldsContributed: 3100, uniqueRecords: 40, lastSuccessfulCrawl: "2026-09-11T02:00:00.000Z", freshnessDays: 1, authority: "A", failureRate: 0, license: "Terms of use", redistribution: "attribution" }] },
50 + }, async () => {
51 + const items = await sourceCoverage();
52 + return { data: items, meta: { total: items.length, methodology: "recordsContributed / fieldsContributed count current provenance rows; authority is the identity-field tier of the source kind (packages/core claims.ts FIELD_AUTHORITY); failureRate = failed or aborted runs ÷ finished runs over 30 days." } };
53 + });
6 54 }
modified apps/api/src/routes/public/watchlist.ts +63 −4
@@ -1,6 +1,65 @@
1 −/** Private cookie-scoped watchlist (no accounts). */
2 −import type { FastifyInstance } from "fastify";
1 +/** Private cookie-scoped watchlist (no accounts): GET / POST / DELETE /watchlist, GET /watchlist/feed. Never cached. */
2 +import type { FastifyInstance, FastifyReply, FastifyRequest } from "fastify";
3 +import { z } from "zod";
4 +import { getEnv } from "../../env.js";
5 +import { envelope, HttpError, parseBody, parseQuery } from "../../lib/http.js";
6 +import { intParam, pageParam } from "../../lib/params.js";
7 +import { registerRoute } from "../../lib/route.js";
8 +import { addWatch, listWatchlist, newWatchToken, parseCookies, removeWatch, resolveWatchEntity, watchFeed, WATCH_COOKIE, WATCH_ENTITY_TYPES, WATCH_MAX_ITEMS } from "../../repositories/watchlist.js";
3 9
4 −export async function watchlistRoutes(_app: FastifyInstance): Promise<void> {
5 − // filled in by the intelligence implementation
10 +const addBody = z.object({ entityType: z.enum(WATCH_ENTITY_TYPES), entityId: z.string().min(1).max(120).optional(), slug: z.string().min(1).max(200).optional() }).strict().refine((b) => b.entityId || b.slug, "entityId or slug is required");
11 +
12 +/** Token from the cookie, minting (and setting) a new one when absent. */
13 +function tokenFor(req: FastifyRequest, reply: FastifyReply, create: boolean): string | null {
14 + const existing = parseCookies(req.headers.cookie)[WATCH_COOKIE];
15 + if (existing) return existing;
16 + if (!create) return null;
17 + const token = newWatchToken();
18 + const secure = getEnv().siteUrl.startsWith("https://") ? "; Secure" : "";
19 + reply.header("set-cookie", `${WATCH_COOKIE}=${token}; Path=/; HttpOnly; SameSite=Lax; Max-Age=31536000${secure}`);
20 + return token;
21 +}
22 +
23 +const NO_STORE = { "cache-control": "no-store", vary: "cookie" };
24 +
25 +export async function watchlistRoutes(app: FastifyInstance): Promise<void> {
26 + registerRoute({ method: "GET", path: "/api/v1/watchlist", summary: "Your private watchlist (WatchlistItem[]) — keyed by the httpOnly dci_watch cookie, created on first use; no accounts", group: "watchlist", params: [], responseType: "WatchlistItem[]" });
27 + app.get("/watchlist", { schema: { summary: "Private watchlist (WatchlistItem[]) keyed by the dci_watch cookie", tags: ["watchlist"] } }, async (req, reply) => {
28 + reply.headers(NO_STORE);
29 + const token = tokenFor(req, reply, true)!;
30 + const items = await listWatchlist(token);
31 + return envelope(items, { total: items.length, max: WATCH_MAX_ITEMS });
32 + });
33 +
34 + registerRoute({ method: "POST", path: "/api/v1/watchlist", summary: "Watch an entity: body { entityType: operator|metro|country|project|facility, entityId | slug } → WatchlistItem (201 when created)", group: "watchlist", params: [{ name: "entityType", in: "query", type: "enum", description: WATCH_ENTITY_TYPES.join(" | "), example: "operator" }, { name: "slug", in: "query", type: "string", description: "entity slug (or entityId)", example: "equinix" }], responseType: "WatchlistItem" });
35 + app.post("/watchlist", { schema: { summary: "Add an entity to the private watchlist { entityType, entityId | slug }", tags: ["watchlist"], body: { type: "object", properties: { entityType: { type: "string", enum: [...WATCH_ENTITY_TYPES] }, entityId: { type: "string" }, slug: { type: "string" } }, required: ["entityType"] } } }, async (req, reply) => {
36 + reply.headers(NO_STORE);
37 + const body = parseBody(addBody, req.body);
38 + const id = await resolveWatchEntity(body.entityType, body.entityId ?? body.slug!);
39 + if (!id) throw new HttpError(404, `${body.entityType} not found`);
40 + const token = tokenFor(req, reply, true)!;
41 + const res = await addWatch(token, body.entityType, id);
42 + if ("error" in res) throw new HttpError(409, `watchlist is full (${WATCH_MAX_ITEMS} items)`);
43 + reply.code(res.created ? 201 : 200);
44 + return envelope(res.item, { created: res.created });
45 + });
46 +
47 + registerRoute({ method: "DELETE", path: "/api/v1/watchlist/:id", summary: "Stop watching (only your own rows) → { deleted }", group: "watchlist", params: [{ name: "id", in: "path", type: "string", description: "watchlist item id (wtc_…)" }], responseType: "{ deleted: boolean }" });
48 + app.delete("/watchlist/:id", { schema: { summary: "Remove a watchlist item (owner only)", tags: ["watchlist"], params: { type: "object", properties: { id: { type: "string" } } } } }, async (req, reply) => {
49 + reply.headers(NO_STORE);
50 + const token = tokenFor(req, reply, false);
51 + const { id } = req.params as { id: string };
52 + const deleted = token ? await removeWatch(token, id) : false;
53 + return envelope({ id, deleted });
54 + });
55 +
56 + registerRoute({ method: "GET", path: "/api/v1/watchlist/feed", summary: "Events for your watched entities (EventDTO[]), newest first, one row per announcement cluster", group: "watchlist", params: [{ name: "page", in: "query", type: "integer", description: "page (default 1)" }, { name: "per_page", in: "query", type: "integer", description: "rows per page (default 50, max 100)" }], responseType: "EventDTO[]" });
57 + app.get("/watchlist/feed", { schema: { summary: "Change feed for the watched entities (EventDTO[])", tags: ["watchlist"], querystring: { type: "object", properties: { page: { type: "integer" }, per_page: { type: "integer" } } } } }, async (req, reply) => {
58 + reply.headers(NO_STORE);
59 + const q = parseQuery(z.object({ page: pageParam, per_page: intParam }), req.query);
60 + const token = tokenFor(req, reply, false);
61 + if (!token) return envelope([], { total: 0, page: q.page, perPage: q.per_page ?? 50 });
62 + const res = await watchFeed(token, q.page, q.per_page);
63 + return envelope(res.items, { total: res.total, page: res.page, perPage: res.perPage });
64 + });
6 65 }
modified apps/api/src/search/interpret.test.ts +43 −0
@@ -24,12 +24,53 @@ describe("interpretQuery", () => {
24 24 expect(interpretQuery("at least 50mw germany")).toMatchObject({ minMw: 50, countryIso2: "DE" });
25 25 });
26 26
27 + it("extracts maximum MW thresholds", () => {
28 + expect(interpretQuery("<500 MW")).toMatchObject({ maxMw: 500 });
29 + expect(interpretQuery("under 500 MW announced")).toMatchObject({ maxMw: 500, status: "announced" });
30 + expect(interpretQuery("below 1 GW").maxMw).toBe(1000);
31 + expect(interpretQuery("less than 50mw in france")).toMatchObject({ maxMw: 50, countryIso2: "FR" });
32 + expect(interpretQuery("less than 50mw").minMw).toBeUndefined();
33 + expect(interpretQuery(">500 MW announced")).toMatchObject({ minMw: 500, status: "announced" });
34 + expect(interpretQuery("500+ MW").minMw).toBe(500);
35 + });
36 +
27 37 it("extracts status and type words", () => {
28 38 expect(interpretQuery("under construction data centers")).toMatchObject({ status: "under_construction" });
29 39 expect(interpretQuery("planned AI data center")).toMatchObject({ status: "announced", facilityType: "ai" });
30 40 expect(interpretQuery("operational colocation in Canada")).toMatchObject({ status: "operational", facilityType: "colocation", countryIso2: "CA" });
31 41 });
32 42
43 + it("flags AI queries", () => {
44 + expect(interpretQuery("ai data centers in texas")).toMatchObject({ ai: true, facilityType: "ai" });
45 + expect(interpretQuery("gpu campus")).toMatchObject({ ai: true });
46 + expect(interpretQuery("colocation in Paris").ai).toBeUndefined();
47 + });
48 +
49 + it("switches to projects when the query says projects", () => {
50 + const i = interpretQuery("projects over 500 MW in texas");
51 + expect(i).toMatchObject({ entity: "project", minMw: 500 });
52 + expect(i.text).toBe("texas");
53 + expect(interpretQuery("announced data centers").entity).toBeUndefined();
54 + expect(interpretQuery("planned campus in Ohio").entity).toBeUndefined();
55 + });
56 +
57 + it("extracts opening year phrases", () => {
58 + expect(interpretQuery("projects opening 2028")).toMatchObject({ entity: "project", year: 2028, yearOp: "in" });
59 + expect(interpretQuery("before 2028")).toMatchObject({ year: 2028, yearOp: "before" });
60 + expect(interpretQuery("after 2027 hyperscale")).toMatchObject({ year: 2027, yearOp: "after", facilityType: "hyperscale" });
61 + expect(interpretQuery("opens in 2026").year).toBe(2026);
62 + expect(interpretQuery("by 2030")).toMatchObject({ year: 2030, yearOp: "before" });
63 + expect(interpretQuery("500 MW").year).toBeUndefined();
64 + });
65 +
66 + it("matches metro names and aliases supplied by the caller", () => {
67 + const metros = [{ slug: "northern-virginia", name: "Northern Virginia", aliases: ["NoVA", "Ashburn"] }, { slug: "dallas-fort-worth", name: "Dallas-Fort Worth", aliases: ["DFW"] }];
68 + expect(interpretQuery("hyperscale in Northern Virginia", { metros })).toMatchObject({ metro: "northern-virginia", facilityType: "hyperscale" });
69 + expect(interpretQuery("ashburn colocation", { metros })).toMatchObject({ metro: "northern-virginia" });
70 + expect(interpretQuery("DFW over 50 MW", { metros })).toMatchObject({ metro: "dallas-fort-worth", minMw: 50 });
71 + expect(interpretQuery("virginia", { metros }).metro).toBeUndefined();
72 + });
73 +
33 74 it("recognises an operator supplied by the caller and keeps the rest as text", () => {
34 75 const i = interpretQuery("Equinix Paris operational", { operators: [{ slug: "equinix", name: "Equinix", similarity: 0.5 }] });
35 76 expect(i.operator).toBe("equinix");
@@ -53,5 +94,7 @@ describe("interpretQuery", () => {
53 94 const i = interpretQuery("equinix operational 100 MW in france", { operators: [{ slug: "equinix", name: "Equinix", similarity: 0.4 }] });
54 95 expect(i).toEqual({ minMw: 100, operator: "equinix", countryIso2: "FR", status: "operational" });
55 96 expect(hasFilters(i)).toBe(true);
97 + expect(hasFilters({ entity: "project" })).toBe(true);
98 + expect(hasFilters({ year: 2028 })).toBe(true);
56 99 });
57 100 });
modified apps/api/src/search/interpret.ts +54 −14
@@ -1,6 +1,7 @@
1 1 /**
2 − * Query interpretation for /search: pulls structured filters (country, operator, status, min MW, facility
3 − * type) out of free text and returns the remaining text. Pure function — unit-tested.
2 + * Query interpretation for /search: pulls structured filters (country, operator, metro, status, min / max MW,
3 + * facility type, AI, entity = project, opening year) out of free text and returns the remaining text.
4 + * Pure function — unit-tested.
4 5 */
5 6 import type { FacilityStatus, FacilityType, SearchResponse } from "@dci/core";
6 7 import { COUNTRY_ALIASES, normalizeStatus, parseMw } from "@dci/core";
@@ -12,6 +13,8 @@ export interface InterpretOptions {
12 13 operators?: Array<{ slug: string; name: string; similarity?: number }>;
13 14 /** extra country names from the countries table: [{ iso2, name }] */
14 15 countries?: Array<{ iso2: string; name: string }>;
16 + /** metro / market names (and aliases) from the metros table */
17 + metros?: Array<{ slug: string; name: string; aliases?: string[] }>;
15 18 }
16 19
17 20 const STATUS_PHRASES: Array<[RegExp, FacilityStatus]> = [
@@ -44,7 +47,14 @@ const TYPE_PHRASES: Array<[RegExp, FacilityType]> = [
44 47 [/\b(government)\b/i, "government"],
45 48 ];
46 49
47 −const STOP = new Set(["data", "center", "centers", "centre", "centres", "datacenter", "datacenters", "datacentre", "datacentres", "dc", "facility", "facilities", "in", "at", "near", "the", "of", "and", "with", "over", "above", "more", "than", "least", "min", "minimum", "campus", "campuses", "site", "sites"]);
50 +const STOP = new Set(["data", "center", "centers", "centre", "centres", "datacenter", "datacenters", "datacentre", "datacentres", "dc", "facility", "facilities", "in", "at", "near", "the", "of", "and", "with", "over", "above", "more", "than", "least", "min", "minimum", "campus", "campuses", "site", "sites", "under", "below", "less", "max", "maximum", "up", "to", "by", "opening", "opens", "open", "project", "projects"]);
51 +
52 +const MW_UNIT = String.raw`(\d+(?:[.,]\d+)?\s*\+?\s*(?:gw|gigawatts?|mw|megawatts?))\b`;
53 +const MAX_RE = new RegExp(String.raw`(?:<=?|under|below|less than|at most|max(?:imum)?|up to|smaller than|no more than)\s*${MW_UNIT}`, "i");
54 +const MIN_RE = new RegExp(String.raw`(?:>=?|over|above|more than|at least|min(?:imum)?|\+|larger than|bigger than)?\s*${MW_UNIT}`, "i");
55 +const AI_RE = /\b(ai|artificial intelligence|gpu|gpus|ai[- ]ready|ai[- ]factory|ai[- ]factories)\b/i;
56 +const PROJECT_RE = /\b(projects?|pipeline projects?)\b/i;
57 +const YEAR_RE = /\b(before|by|until|prior to|pre|after|from|since|post|opening|opens|open(?:ed|ing)? in|in|for|due)?\s*((?:19|20)\d{2})\b/i;
48 58
49 59 function esc(s: string): string {
50 60 return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
@@ -59,15 +69,33 @@ export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interp
59 69 let text = raw.replace(/\s+/g, " ").trim();
60 70 if (!text) return out;
61 71
62 − // 1. Power figure: "100 MW", "over 50mw", "> 1 GW", "100+ MW"
63 − const mwMatch = text.match(/(?:>=?|over|above|more than|at least|min(?:imum)?|\+)?\s*(\d+(?:[.,]\d+)?\s*\+?\s*(?:gw|gigawatts?|mw|megawatts?))\b/i);
72 + // 1. Power figures: "<500 MW" / "under 500 MW" → maxMw; "100 MW", "over 50mw", "> 1 GW", "100+ MW" → minMw
73 + const maxMatch = text.match(MAX_RE);
74 + if (maxMatch) {
75 + const mw = parseMw(maxMatch[1]!.replace("+", ""));
76 + if (mw != null && mw > 0) out.maxMw = mw;
77 + text = strip(text, new RegExp(esc(maxMatch[0]), "i"));
78 + }
79 + const mwMatch = text.match(MIN_RE);
64 80 if (mwMatch) {
65 81 const mw = parseMw(mwMatch[1]!.replace("+", ""));
66 82 if (mw != null && mw > 0) out.minMw = mw;
67 83 text = strip(text, new RegExp(esc(mwMatch[0]), "i"));
68 84 }
69 85
70 − // 2. Operator (candidates supplied by the caller from a trigram lookup on the full query)
86 + // 2. Opening year phrases: "opening 2028", "before 2028", "after 2027" (a year right after a MW figure was consumed above)
87 + const yearMatch = text.match(YEAR_RE);
88 + if (yearMatch) {
89 + const y = Number(yearMatch[2]);
90 + const cue = (yearMatch[1] ?? "").toLowerCase();
91 + if (y >= 1990 && y <= 2060) {
92 + out.year = y;
93 + out.yearOp = /^(before|by|until|prior to|pre)$/.test(cue) ? "before" : /^(after|from|since|post)$/.test(cue) ? "after" : "in";
94 + text = strip(text, new RegExp(esc(yearMatch[0]), "i"));
95 + }
96 + }
97 +
98 + // 3. Operator (candidates supplied by the caller from a trigram lookup on the full query)
71 99 for (const op of opts.operators ?? []) {
72 100 const re = new RegExp(`(^|[^a-z0-9])${esc(op.name)}([^a-z0-9]|$)`, "i");
73 101 if (re.test(text) || (op.similarity ?? 0) >= 0.8) {
@@ -78,11 +106,17 @@ export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interp
78 106 }
79 107 }
80 108
81 − // 3. Country: bare ISO2 token in caps, known aliases, or the countries table
82 − const iso2Token = raw.match(/(^|\s)([A-Z]{2})(\s|$)/);
83 − if (iso2Token && !/^(AI|DC|MW|GW|IX|HQ|IT|AZ|US)$/.test(iso2Token[2]!)) {
84 − // AZ/IT/US ambiguous in lower-case; treat "US"/"IT" explicitly below via aliases
109 + // 4. Metro / market names (longest first, word boundaries, aliases included)
110 + if (opts.metros?.length) {
111 + const cands = opts.metros.flatMap((m) => [m.name, ...(m.aliases ?? [])].filter((n) => n && n.length >= 3).map((n) => ({ slug: m.slug, n }))).sort((a, b) => b.n.length - a.n.length);
112 + for (const c of cands) {
113 + const re = new RegExp(`(^|[^\\p{L}\\p{N}])${esc(c.n)}([^\\p{L}\\p{N}]|$)`, "iu");
114 + if (re.test(text)) { out.metro = c.slug; text = strip(text, re); break; }
115 + }
85 116 }
117 +
118 + // 5. Country: bare ISO2 token in caps, known aliases, or the countries table
119 + const iso2Token = raw.match(/(^|\s)([A-Z]{2})(\s|$)/);
86 120 if (iso2Token && /^(US|UK|CA|DE|FR|NL|GB|IE|SG|JP|AU|IN|BR|ES|IT|SE|NO|FI|DK|PL|CH|AT|BE|PT|MX|ZA|AE|SA|KR|CN|HK|TW|MY|ID|TH|VN|PH|NZ|CL|AR|CO|IL|TR|QA|KE|NG|EG)$/.test(iso2Token[2]!)) {
87 121 const code = iso2Token[2]! === "UK" ? "GB" : iso2Token[2]!;
88 122 out.countryIso2 = code;
@@ -103,22 +137,28 @@ export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interp
103 137 }
104 138 }
105 139
106 − // 4. Status
140 + // 6. Entity: "projects" (plural is the strongest signal) — "announced" / "planned" alone only set the status
141 + if (PROJECT_RE.test(text)) { out.entity = "project"; text = strip(text, PROJECT_RE); }
142 +
143 + // 7. Status
107 144 for (const [re, st] of STATUS_PHRASES) {
108 145 if (re.test(text)) { out.status = normalizeStatus(st) ?? st; text = strip(text, re); break; }
109 146 }
110 147
111 − // 5. Facility type
148 + // 8. AI flag (kept alongside the facility type so callers can match is_ai / ai_evidence rather than the type column)
149 + if (AI_RE.test(text)) out.ai = true;
150 +
151 + // 9. Facility type
112 152 for (const [re, ty] of TYPE_PHRASES) {
113 153 if (re.test(text)) { out.facilityType = ty; text = strip(text, re); break; }
114 154 }
115 155
116 − // 6. Remaining text (drop generic words)
156 + // 10. Remaining text (drop generic words)
117 157 const rest = text.split(" ").filter((t) => t && !STOP.has(t.toLowerCase())).join(" ").trim();
118 158 if (rest) out.text = rest;
119 159 return out;
120 160 }
121 161
122 162 export function hasFilters(i: Interpreted): boolean {
123 − return Boolean(i.countryIso2 || i.operator || i.status || i.minMw != null || i.facilityType);
163 + return Boolean(i.countryIso2 || i.operator || i.metro || i.status || i.minMw != null || i.maxMw != null || i.facilityType || i.ai || i.entity || i.year != null);
124 164 }
modified apps/web/src/lib/labels.ts +10 −0
@@ -176,6 +176,16 @@ export const EVENT_LABEL: Record<EventType, string> = {
176 176 incident: "Incident",
177 177 page_changed: "Page changed",
178 178 news: "News",
179 + land_acquired: "Land acquired",
180 + grid_connection: "Grid connection",
181 + grid_constraint: "Grid constraint",
182 + utility_event: "Utility event",
183 + operator_expansion: "Operator expansion",
184 + customer_agreement: "Customer agreement",
185 + partnership: "Partnership",
186 + executive_change: "Executive change",
187 + project_delayed: "Project delayed",
188 + project_cancelled: "Project cancelled",
179 189 };
180 190
181 191 /** Event families → tone, so the feed reads by colour without a rainbow. */
182 192