API 2.0: claims/history/provenance/nearby/pulse/compare/explore/coverage/ai-infrastructure/power/connectivity/time-machine/download/watchlist/docs-meta endpoints, map layers + density + time filter, event dedupe by cluster, list envelopes carry sources with redistribution; admin quality/data-gaps/trace proxy/claims/project hide+merge/related-campus/quarantine/run rollback; containment-aware detail payloads; env placeholder token rejected; 59 API tests
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
54 changed files +3,981 −509
modified
apps/api/src/app.test.ts
+552 −44
@@ -1,54 +1,91 @@ | ||
| 1 | 1 | /** |
| 2 | − * Integration tests against the local Postgres (repo-root .env). A small fixture (country ZZ, operator, two | |
| 3 | − * facilities, one project, one event) is inserted before and removed after; every id is prefixed `*_apitest`. | |
| 2 | + * Integration tests against the local Postgres (repo-root .env). A small fixture (countries ZZ/ZY, operator, three | |
| 3 | + * facilities, one project, events, a connector run with provenance / claim / event, sources with licences, one quality | |
| 4 | + * flag) is inserted before and removed after; every id is prefixed `*_apitest`. | |
| 4 | 5 | */ |
| 5 | 6 | import { afterAll, beforeAll, describe, expect, it } from "vitest"; |
| 6 | 7 | import type { FastifyInstance } from "fastify"; |
| 7 | 8 | import { loadEnvFile, resetEnv } from "./env.js"; |
| 8 | 9 | |
| 9 | 10 | loadEnvFile(); |
| 10 | −process.env.DCI_ADMIN_TOKEN = process.env.DCI_ADMIN_TOKEN || "test-admin-token"; | |
| 11 | +process.env.DCI_ADMIN_TOKEN = process.env.DCI_ADMIN_TOKEN && process.env.DCI_ADMIN_TOKEN !== "change-me" ? process.env.DCI_ADMIN_TOKEN : "test-admin-token"; | |
| 11 | 12 | process.env.DCI_API_CACHE = "0"; // deterministic responses while fixtures change |
| 12 | 13 | process.env.DCI_API_CH_LOG = "0"; |
| 14 | +process.env.DCI_WORKER_URL = "http://127.0.0.1:1"; // nothing listens: proxy routes must answer 502 | |
| 13 | 15 | resetEnv(); |
| 14 | 16 | |
| 15 | 17 | const TOKEN = process.env.DCI_ADMIN_TOKEN!; |
| 18 | +const ADMIN = { "x-dci-admin-token": TOKEN }; | |
| 16 | 19 | const F1 = "fac_apitest000001"; |
| 17 | 20 | const F2 = "fac_apitest000002"; |
| 21 | +const F3 = "fac_apitest000003"; | |
| 18 | 22 | const OP = "op_apitest00000001"; |
| 19 | 23 | const PRJ = "prj_apitest0000001"; |
| 20 | 24 | const EVT = "evt_apitest0000001"; |
| 25 | +const EVT_RUN = "evt_apitest0000002"; | |
| 21 | 26 | const MET = "met_apitest0000001"; |
| 27 | +const RUN = "run_apitest0000001"; | |
| 28 | +const CLAIM = "clm_apitest0000001"; | |
| 29 | +const PROV_OLD = "prv_apitest0000001"; | |
| 30 | +const PROV_RUN = "prv_apitest0000002"; | |
| 31 | +const PROV_F3 = "prv_apitest0000003"; | |
| 32 | +const FLAG = "flg_apitest0000001"; | |
| 33 | +const SRC = "src_apitest"; | |
| 34 | +const SRC_R = "src_apitest_restricted"; | |
| 22 | 35 | |
| 23 | 36 | let app: FastifyInstance; |
| 24 | 37 | let sql: Awaited<ReturnType<typeof import("./lib/sql.js").pg>>; |
| 25 | 38 | |
| 26 | 39 | async function cleanup(): Promise<void> { |
| 27 | − await sql`delete from events where id = ${EVT} or entity_id in (${F1}, ${F2})`; | |
| 28 | − await sql`delete from provenance where entity_id in (${F1}, ${F2}, ${OP}, ${PRJ})`; | |
| 29 | − await sql`delete from facility_aliases where facility_id in (${F1}, ${F2})`; | |
| 30 | − await sql`delete from entity_keys where entity_id in (${F1}, ${F2})`; | |
| 40 | + await sql`delete from watchlists where entity_id in (${F1}, ${F2}, ${OP}, ${PRJ})`; | |
| 41 | + await sql`delete from quality_flags where dedupe_key like 'apitest%' or entity_id in (${F1}, ${F2}, ${F3}, ${PRJ})`; | |
| 42 | + await sql`delete from claims where subject_id in (${F1}, ${F2}, ${F3}, ${PRJ}) or run_id = ${RUN}`; | |
| 43 | + await sql`delete from events where id in (${EVT}, ${EVT_RUN}) or entity_id in (${F1}, ${F2}, ${F3}, ${PRJ}) or project_id = ${PRJ} or run_id = ${RUN}`; | |
| 44 | + await sql`delete from provenance where entity_id in (${F1}, ${F2}, ${F3}, ${OP}, ${PRJ}) or run_id = ${RUN}`; | |
| 45 | + await sql`delete from project_timeline where project_id = ${PRJ}`; | |
| 46 | + await sql`delete from facility_aliases where facility_id in (${F1}, ${F2}, ${F3})`; | |
| 47 | + await sql`delete from entity_keys where entity_id in (${F1}, ${F2}, ${F3}, ${PRJ})`; | |
| 48 | + await sql`delete from connector_runs where id = ${RUN}`; | |
| 31 | 49 | await sql`delete from projects where id = ${PRJ}`; |
| 32 | − await sql`delete from facilities where id in (${F1}, ${F2})`; | |
| 50 | + await sql`update facilities set parent_facility_id = null where parent_facility_id in (${F1}, ${F2}, ${F3})`; | |
| 51 | + await sql`delete from facilities where id in (${F1}, ${F2}, ${F3})`; | |
| 33 | 52 | await sql`delete from metros where id = ${MET}`; |
| 34 | 53 | await sql`delete from operators where id = ${OP}`; |
| 35 | − await sql`delete from countries where iso2 = 'ZZ'`; | |
| 54 | + await sql`delete from countries where iso2 in ('ZZ', 'ZY')`; | |
| 55 | + await sql`delete from sources where id in (${SRC}, ${SRC_R})`; | |
| 36 | 56 | } |
| 37 | 57 | |
| 38 | 58 | beforeAll(async () => { |
| 39 | 59 | const { pg } = await import("./lib/sql.js"); |
| 40 | 60 | sql = pg(); |
| 41 | 61 | await cleanup(); |
| 42 | − await sql`insert into countries (iso2, iso3, slug, name, region, subregion, lat, lng) values ('ZZ', 'ZZZ', 'apitest-land', 'Apitest Land', 'Testing', 'Unit', 45.5, -73.6)`; | |
| 62 | + await sql`insert into sources (id, connector_id, name, domain, kind, priority, url, license, attribution, redistribution) values | |
| 63 | + (${SRC}, 'apitest', 'Apitest Source', 'example.test', 'operator', 2, 'https://example.test', 'CC-BY-4.0', 'Apitest Source', 'attribution'), | |
| 64 | + (${SRC_R}, 'apitest-restricted', 'Apitest Restricted Source', 'restricted.test', 'dataset', 3, 'https://restricted.test', 'proprietary', 'Restricted', 'restricted')`; | |
| 65 | + await sql`insert into countries (iso2, iso3, slug, name, region, subregion, lat, lng, renewable_share, electricity_twh, stats_year) values ('ZZ', 'ZZZ', 'apitest-land', 'Apitest Land', 'Testing', 'Unit', 45.5, -73.6, 0.42, 123.4, 2024), ('ZY', 'ZYY', 'apitest-land-two', 'Apitest Land Two', 'Testing', 'Unit', 10, 10, null, null, null)`; | |
| 43 | 66 | await sql`insert into operators (id, slug, name, normalized_name, kind, hq_country_iso2, website) values (${OP}, 'apitest-operator', 'Apitest Operator', 'apitest operator', 'colocation', 'ZZ', 'https://example.test')`; |
| 44 | − await sql`insert into metros (id, slug, name, country_iso2, lat, lng) values (${MET}, 'apitest-metro', 'Apitest Metro', 'ZZ', 45.5, -73.6)`; | |
| 45 | − await sql`insert into facilities (id, slug, name, normalized_name, operator_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, it_capacity_mw, total_power_mw, is_ai, confidence, completeness, opened_on, last_verified) | |
| 46 | − values (${F1}, 'apitest-one', 'Apitest One Data Center', 'apitest one', ${OP}, ${MET}, 'ZZ', 'Montréal-Test', 45.5017, -73.5673, 'exact', 'operational', 'colocation', 12.5, 20, true, 'high', 80, '2019-05', now())`; | |
| 67 | + await sql`insert into metros (id, slug, name, country_iso2, lat, lng, aliases) values (${MET}, 'apitest-metro', 'Apitest Metro', 'ZZ', 45.5, -73.6, '{"Apitestville"}')`; | |
| 68 | + await sql`insert into facilities (id, slug, name, normalized_name, operator_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, it_capacity_mw, total_power_mw, utility_capacity_mw, is_ai, ai_evidence, confidence, completeness, opened_on, last_verified, source_count) | |
| 69 | + values (${F1}, 'apitest-one', 'Apitest One Data Center', 'apitest one', ${OP}, ${MET}, 'ZZ', 'Montréal-Test', 45.5017, -73.5673, 'exact', 'operational', 'colocation', 12.5, 20, 40, true, 'confirmed', 'high', 80, '2019-05', now(), 2)`; | |
| 47 | 70 | await sql`insert into facilities (id, slug, name, normalized_name, operator_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, planned_power_mw, confidence, completeness, opened_on) |
| 48 | 71 | values (${F2}, 'apitest-two', 'Apitest Two Campus', 'apitest two', ${OP}, ${MET}, 'ZZ', 'Montréal-Test', 45.52, -73.58, 'city', 'under_construction', 'hyperscale', 100, 'moderate', 40, '2027')`; |
| 49 | − await sql`insert into projects (id, slug, name, normalized_name, operator_id, facility_id, metro_id, country_iso2, status, expected_opening, planned_mw, is_ai) values (${PRJ}, 'apitest-project', 'Apitest Expansion', 'apitest expansion', ${OP}, ${F2}, ${MET}, 'ZZ', 'under_construction', '2027-Q2', 100, true)`; | |
| 50 | − await sql`insert into events (id, entity_type, entity_id, event_type, old_value, new_value, source_id, url, title, significance, confidence, review_status, country_iso2, operator_id, fingerprint) | |
| 51 | − values (${EVT}, 'facility', ${F1}, 'capacity_changed', '10'::jsonb, '12.5'::jsonb, 'src_apitest', 'https://example.test/one', 'Apitest One capacity 10 → 12.5 MW', 80, 'high', 'auto', 'ZZ', ${OP}, 'apitest:evt:1')`; | |
| 72 | + await sql`insert into facilities (id, slug, name, normalized_name, operator_id, country_iso2, city, lat, lng, geo_precision, status, facility_type, confidence, completeness, source_count) | |
| 73 | + values (${F3}, 'apitest-three', 'Apitest Three Restricted', 'apitest three', ${OP}, 'ZY', 'Restricted City', 10.01, 10.01, 'city', 'operational', 'colocation', 'moderate', 20, 1)`; | |
| 74 | + await sql`insert into projects (id, slug, name, normalized_name, operator_id, facility_id, metro_id, country_iso2, city, lat, lng, geo_precision, status, announced_on, construction_started_on, expected_opening, planned_mw, is_ai, ai_evidence, project_class, evidence_level) | |
| 75 | + values (${PRJ}, 'apitest-project', 'Apitest Expansion', 'apitest expansion', ${OP}, ${F2}, ${MET}, 'ZZ', 'Montréal-Test', 45.51, -73.57, 'city', 'under_construction', '2025-01', '2026-03', '2027-Q2', 100, true, 'likely', 'EXPANSION', 'strong')`; | |
| 76 | + await sql`insert into events (id, entity_type, entity_id, event_type, old_value, new_value, source_id, url, title, significance, confidence, review_status, country_iso2, operator_id, metro_id, fingerprint, is_ai) | |
| 77 | + values (${EVT}, 'facility', ${F1}, 'capacity_changed', '10'::jsonb, '12.5'::jsonb, ${SRC}, 'https://example.test/one', 'Apitest One capacity 10 → 12.5 MW', 80, 'high', 'auto', 'ZZ', ${OP}, ${MET}, 'apitest:evt:1', true)`; | |
| 78 | + // a connector run with one provenance row (winner), one claim and one event — for /runs/:id/changes and rollback | |
| 79 | + await sql`insert into connector_runs (id, connector_id, task, started_at, finished_at, status, stats) values (${RUN}, 'apitest', 'crawl', now() - interval '1 hour', now() - interval '50 minutes', 'ok', '{"discovered": 3, "fetched": 3, "created": 1, "changed": 1, "updated": 0, "failed": 0}'::jsonb)`; | |
| 80 | + await sql`insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, url, first_observed, last_observed, retrieved_at, confidence, is_current, is_winner, run_id) | |
| 81 | + values (${PROV_OLD}, 'facility', ${F1}, 'itCapacityMw', '10'::jsonb, ${SRC}, 'apitest', 'https://example.test/one-old', now() - interval '30 days', now() - interval '2 days', now() - interval '2 days', 'moderate', true, false, null), | |
| 82 | + (${PROV_RUN}, 'facility', ${F1}, 'itCapacityMw', '12.5'::jsonb, ${SRC}, 'apitest', 'https://example.test/one', now() - interval '1 hour', now() - interval '1 hour', now() - interval '1 hour', 'high', true, true, ${RUN}), | |
| 83 | + (${PROV_F3}, 'facility', ${F3}, 'name', '"Apitest Three Restricted"'::jsonb, ${SRC_R}, 'apitest-restricted', 'https://restricted.test/three', now(), now(), now(), 'moderate', true, true, null)`; | |
| 84 | + await sql`insert into claims (id, subject_type, subject_id, predicate, value, unit, scope, scope_reason, source_id, connector_id, url, published_at, confidence, authority_tier, evidence_text, status, run_id) | |
| 85 | + values (${CLAIM}, 'facility', ${F1}, 'it_capacity_mw', 12.5, 'MW', 'facility', 'regex:facility', ${SRC}, 'apitest', 'https://example.test/one', '2026-09', 'high', 'B', 'The facility offers 12.5 MW of IT capacity.', 'current', ${RUN})`; | |
| 86 | + await sql`insert into events (id, entity_type, entity_id, event_type, new_value, source_id, url, title, significance, confidence, review_status, country_iso2, operator_id, fingerprint, run_id) | |
| 87 | + values (${EVT_RUN}, 'facility', ${F1}, 'facility_updated', '{"itCapacityMw": 12.5}'::jsonb, ${SRC}, 'https://example.test/one', 'Apitest One updated', 30, 'moderate', 'auto', 'ZZ', ${OP}, 'apitest:evt:2', ${RUN})`; | |
| 88 | + await sql`insert into quality_flags (id, entity_type, entity_id, code, severity, field, message, priority, status, dedupe_key) values (${FLAG}, 'facility', ${F1}, 'mw_change_5x', 'warn', 'itCapacityMw', 'apitest flag', 999, 'open', 'apitest:flag:1')`; | |
| 52 | 89 | const { buildApp } = await import("./app.js"); |
| 53 | 90 | app = await buildApp({ logger: false, underPressure: false }); |
| 54 | 91 | await app.ready(); |
@@ -63,6 +100,8 @@ afterAll(async () => { | ||
| 63 | 100 | await Promise.allSettled([closeQueues(), closeRedis(), closeDb()]); |
| 64 | 101 | }); |
| 65 | 102 | |
| 103 | +const slugs = (arr: Array<{ slug: string }>) => arr.map((f) => f.slug).filter((s) => s.startsWith("apitest")); | |
| 104 | + | |
| 66 | 105 | describe("system", () => { |
| 67 | 106 | it("health and ready", async () => { |
| 68 | 107 | const h = await app.inject({ url: "/api/health" }); |
@@ -77,15 +116,21 @@ describe("system", () => { | ||
| 77 | 116 | expect(m.statusCode).toBe(200); |
| 78 | 117 | expect(m.body).toContain("dci_api_requests_total"); |
| 79 | 118 | }); |
| 80 | − it("openapi document is served", async () => { | |
| 119 | + it("openapi document is served with response schemas", async () => { | |
| 81 | 120 | const o = await app.inject({ url: "/api/v1/openapi.json" }); |
| 82 | 121 | expect(o.statusCode).toBe(200); |
| 83 | − expect(o.json().paths["/api/v1/datacenters"]).toBeDefined(); | |
| 122 | + const paths = o.json().paths; | |
| 123 | + expect(paths["/api/v1/datacenters"]).toBeDefined(); | |
| 124 | + expect(paths["/api/v1/nearby"]).toBeDefined(); | |
| 125 | + const ok = paths["/api/v1/datacenters"].get.responses["200"]; | |
| 126 | + expect(ok).toBeDefined(); | |
| 127 | + const schema = ok.content?.["application/json"]?.schema ?? ok; | |
| 128 | + expect(schema.properties?.data ?? schema["x-example"] ?? schema.description).toBeDefined(); | |
| 84 | 129 | }); |
| 85 | 130 | }); |
| 86 | 131 | |
| 87 | 132 | describe("envelope, ETag and 404", () => { |
| 88 | − it("wraps list responses and sets caching headers", async () => { | |
| 133 | + it("wraps list responses and sets caching headers + sources", async () => { | |
| 89 | 134 | const res = await app.inject({ url: "/api/v1/datacenters?country=ZZ&sort=mw" }); |
| 90 | 135 | expect(res.statusCode).toBe(200); |
| 91 | 136 | const body = res.json(); |
@@ -94,6 +139,9 @@ describe("envelope, ETag and 404", () => { | ||
| 94 | 139 | expect(body.meta.generatedAt).toBeTruthy(); |
| 95 | 140 | expect(body.data[0].slug).toBe("apitest-two"); // planned 100 MW sorts first (mw = COALESCE(it,total,planned)) |
| 96 | 141 | expect(body.data[1].operator).toEqual({ id: OP, slug: "apitest-operator", name: "Apitest Operator" }); |
| 142 | + expect(body.data[1]).toMatchObject({ recordScope: "facility", aiEvidence: "confirmed", utilityCapacityMw: 40, sourceCount: 2 }); | |
| 143 | + expect(Array.isArray(body.sources)).toBe(true); | |
| 144 | + expect(body.sources.some((s: { id: string; redistribution: string }) => s.id === SRC && s.redistribution === "attribution")).toBe(true); | |
| 97 | 145 | expect(res.headers.etag).toMatch(/^W\//); |
| 98 | 146 | expect(res.headers["cache-control"]).toContain("s-maxage=120"); |
| 99 | 147 | const again = await app.inject({ url: "/api/v1/datacenters?country=ZZ&sort=mw", headers: { "if-none-match": String(res.headers.etag) } }); |
@@ -110,7 +158,7 @@ describe("envelope, ETag and 404", () => { | ||
| 110 | 158 | expect(d.json().data.map((f: { slug: string }) => f.slug)).toEqual(["apitest-two"]); |
| 111 | 159 | }); |
| 112 | 160 | it("returns a consistent 404 JSON", async () => { |
| 113 | − for (const url of ["/api/v1/datacenters/nope", "/api/v1/operators/nope", "/api/v1/countries/nope", "/api/v1/metros/nope", "/api/v1/projects/nope", "/api/v1/events/nope", "/api/v1/rankings/nope", "/api/v1/sitemap/nope", "/api/nothing-here"]) { | |
| 161 | + for (const url of ["/api/v1/datacenters/nope", "/api/v1/operators/nope", "/api/v1/countries/nope", "/api/v1/metros/nope", "/api/v1/projects/nope", "/api/v1/events/nope", "/api/v1/rankings/nope", "/api/v1/sitemap/nope", "/api/v1/download/nope", "/api/nothing-here"]) { | |
| 114 | 162 | const r = await app.inject({ url }); |
| 115 | 163 | expect(r.statusCode, url).toBe(404); |
| 116 | 164 | expect(typeof r.json().error, url).toBe("string"); |
@@ -130,39 +178,124 @@ describe("details", () => { | ||
| 130 | 178 | const d = r.json().data; |
| 131 | 179 | expect(d.id).toBe(F1); |
| 132 | 180 | expect(d.metro.slug).toBe("apitest-metro"); |
| 133 | − expect(d.nearby.map((f: { slug: string }) => f.slug)).toEqual(["apitest-two"]); | |
| 181 | + // legacy `nearby` is capped at the 8 closest rows: real Montréal facilities may outrank the fixture — shape only | |
| 182 | + expect(Array.isArray(d.nearby)).toBe(true); | |
| 183 | + expect(d.nearby.length).toBeLessThanOrEqual(8); | |
| 134 | 184 | expect(d.projects.map((p: { slug: string }) => p.slug)).toEqual(["apitest-project"]); // same operator + metro |
| 135 | − expect(d.events.length).toBe(1); | |
| 136 | − expect(d.sourceHistory[0].kind).toBe("changed"); | |
| 185 | + expect(d.events.length).toBe(2); | |
| 186 | + expect(d.sourceHistory[0].kind).toBeTruthy(); | |
| 137 | 187 | expect(Array.isArray(d.provenance)).toBe(true); |
| 188 | + expect(d.provenance.find((p: { isWinner: boolean; field: string }) => p.field === "itCapacityMw" && p.isWinner).value).toBe(12.5); | |
| 138 | 189 | expect(d.certifications).toEqual([]); |
| 190 | + // API 2.0 fields | |
| 191 | + expect(d.buildings).toEqual([]); | |
| 192 | + expect(d.tenants).toEqual([]); | |
| 193 | + expect(d.claims.length).toBe(1); | |
| 194 | + expect(d.claims[0]).toMatchObject({ id: CLAIM, predicate: "it_capacity_mw", label: "IT capacity", scope: "facility", isWinner: true, authorityTier: "B" }); | |
| 195 | + expect(d.capacityHistory.length).toBeGreaterThanOrEqual(3); // 2 observations + 1 claim + capacity_changed event | |
| 196 | + expect(d.capacityHistory.some((h: { kind: string }) => h.kind === "changed")).toBe(true); | |
| 197 | + expect(d.nearbyInfrastructure.radiusKm).toBe(25); | |
| 198 | + expect(slugs(d.nearbyInfrastructure.facilities)).toEqual(["apitest-two"]); | |
| 199 | + expect(slugs(d.nearbyInfrastructure.projects)).toEqual(["apitest-project"]); | |
| 200 | + expect(d.nearbyInfrastructure.landingStations).toEqual([]); | |
| 201 | + expect(typeof d.nearbyInfrastructure.note).toBe("string"); | |
| 202 | + expect(d.powerContext.utilityCapacityMw).toBe(40); | |
| 203 | + expect(d.powerContext.countryEnergy).toMatchObject({ renewableShare: 0.42, electricityTwh: 123.4, statsYear: 2024 }); | |
| 204 | + expect(typeof d.powerContext.note).toBe("string"); | |
| 205 | + expect(d.dataQuality).toMatchObject({ sourceCount: 1, primarySourceCount: 1, claimsTotal: 1, claimsCurrent: 1, pendingDuplicate: false }); | |
| 206 | + expect(d.dataQuality.openFlags.some((f: { code: string }) => f.code === "mw_change_5x")).toBe(true); | |
| 139 | 207 | const byId = await app.inject({ url: `/api/v1/datacenters/${F1}` }); |
| 140 | 208 | expect(byId.json().data.slug).toBe("apitest-one"); |
| 209 | + const small = await app.inject({ url: "/api/v1/datacenters/apitest-one?radius_km=1" }); | |
| 210 | + expect(small.json().data.nearbyInfrastructure.radiusKm).toBe(1); | |
| 211 | + }); | |
| 212 | + it("facility history, claims and provenance endpoints", async () => { | |
| 213 | + const h = (await app.inject({ url: "/api/v1/datacenters/apitest-one/history" })).json().data; | |
| 214 | + expect(h.entityType).toBe("facility"); | |
| 215 | + expect(h.entityId).toBe(F1); | |
| 216 | + expect(h.fields.itCapacityMw.length).toBeGreaterThanOrEqual(2); | |
| 217 | + expect(h.changes.some((c: { eventId: string }) => c.eventId === EVT)).toBe(true); | |
| 218 | + const c = (await app.inject({ url: "/api/v1/datacenters/apitest-one/claims" })).json().data; | |
| 219 | + expect(c.map((x: { id: string }) => x.id)).toEqual([CLAIM]); | |
| 220 | + const p = (await app.inject({ url: `/api/v1/datacenters/${F1}/provenance` })).json().data; | |
| 221 | + expect(p.filter((x: { field: string }) => x.field === "itCapacityMw").length).toBe(2); | |
| 222 | + expect((await app.inject({ url: "/api/v1/datacenters/nope/history" })).statusCode).toBe(404); | |
| 141 | 223 | }); |
| 142 | 224 | it("operator, country, metro and project details aggregate the fixture", async () => { |
| 143 | 225 | const op = (await app.inject({ url: "/api/v1/operators/apitest-operator" })).json().data; |
| 144 | − expect(op.facilityCount).toBe(2); | |
| 145 | − expect(op.countries[0]).toMatchObject({ iso2: "ZZ", facilityCount: 2, knownMw: 12.5 }); | |
| 146 | − expect(op.statusBreakdown).toEqual({ operational: 1, under_construction: 1 }); | |
| 147 | − expect(op.recentEvents.length).toBe(1); | |
| 226 | + expect(op.facilityCount).toBe(3); | |
| 227 | + expect(op.countries.find((c: { iso2: string }) => c.iso2 === "ZZ")).toMatchObject({ iso2: "ZZ", facilityCount: 2, knownMw: 12.5 }); | |
| 228 | + expect(op.statusBreakdown).toEqual({ operational: 2, under_construction: 1 }); | |
| 229 | + expect(op.recentEvents.length).toBe(2); | |
| 230 | + expect(op.pipeline.operational).toMatchObject({ count: 2, mw: 12.5 }); | |
| 231 | + expect(op.pipeline.construction).toMatchObject({ count: 1, mw: 100 }); | |
| 232 | + expect(op.pipeline.projects).toEqual([{ status: "under_construction", count: 1, mw: 100 }]); | |
| 233 | + expect(op.velocity.windows.map((w: { label: string }) => w.label)).toEqual(["12m", "3y", "5y"]); | |
| 234 | + expect(op.velocity.countriesOverTime.length).toBeGreaterThan(0); | |
| 235 | + expect(op.topCountries[0]).toMatchObject({ iso2: "ZZ", facilityCount: 2 }); | |
| 236 | + expect(op.topCountries[0].share).toBeCloseTo(2 / 3, 2); | |
| 237 | + expect(slugs(op.aiFacilities)).toEqual(["apitest-one"]); | |
| 238 | + expect(Array.isArray(op.corporateEvents)).toBe(true); | |
| 239 | + expect(op.dataQuality.sourceCount).toBe(0); | |
| 148 | 240 | const co = (await app.inject({ url: "/api/v1/countries/zz" })).json().data; |
| 149 | − expect(co).toMatchObject({ iso2: "ZZ", facilityCount: 2, operationalCount: 1, constructionCount: 1, knownMw: 12.5, constructionMw: 100, mwCoverage: 1 }); | |
| 150 | − expect(co.growth).toEqual([{ year: 2019, facilities: 1, knownMw: 12.5 }, { year: 2027, facilities: 2, knownMw: 112.5 }]); | |
| 241 | + expect(co).toMatchObject({ iso2: "ZZ", facilityCount: 2, operationalCount: 1, constructionCount: 1, knownMw: 12.5, constructionMw: 100, mwCoverage: 1, projectCount: 1 }); | |
| 242 | + // known MW = operational figures only: the 100 MW planned for the 2027 campus is pipeline, not known capacity | |
| 243 | + expect(co.growth).toEqual([{ year: 2019, facilities: 1, knownMw: 12.5 }, { year: 2027, facilities: 2, knownMw: 12.5 }]); | |
| 151 | 244 | expect(co.topOperators[0].slug).toBe("apitest-operator"); |
| 245 | + expect(co.energy).toMatchObject({ renewableShare: 0.42, electricityTwh: 123.4, statsYear: 2024, gridCarbonIntensity: null }); | |
| 246 | + expect(co.energy.note).toMatch(/national/i); | |
| 247 | + expect(co.pipeline.construction.count).toBe(1); | |
| 248 | + expect(co.coverage).toMatchObject({ key: "ZZ", facilities: 2, capacityCoverage: 1, anyLocationCoverage: 1, openingDateCoverage: 1 }); | |
| 249 | + expect(slugs(co.aiFacilities)).toEqual(["apitest-one"]); | |
| 250 | + expect(co.aiProjects.map((p: { slug: string }) => p.slug)).toEqual(["apitest-project"]); | |
| 251 | + expect(co.gridConstraints).toEqual([]); | |
| 252 | + expect(co.ixps).toEqual([]); | |
| 152 | 253 | const me = (await app.inject({ url: "/api/v1/metros/apitest-metro" })).json().data; |
| 153 | 254 | expect(me.facilityCount).toBe(2); |
| 154 | 255 | expect(me.projectCount).toBe(1); |
| 155 | 256 | expect(me.constraints).toEqual([]); |
| 257 | + expect(me.concentration.operatorCount).toBe(1); | |
| 258 | + expect(me.concentration.facilities.hhi).toBe(10000); | |
| 259 | + expect(me.concentration.knownMw.hhi).toBe(10000); | |
| 260 | + expect(me.momentum.window).toBe("12m"); | |
| 261 | + expect(me.momentum.projectsEnteredConstruction).toBe(1); | |
| 262 | + expect(me.pipeline.operational.count).toBe(1); | |
| 263 | + expect(me.openingTimeline).toEqual([{ year: 2019, opened: 1, openedMw: 12.5 }, { year: 2027, opened: 1, openedMw: null }]); | |
| 264 | + expect(me.coverage.facilities).toBe(2); | |
| 265 | + expect(me.gridConstraints).toEqual([]); | |
| 156 | 266 | const pr = (await app.inject({ url: "/api/v1/projects/apitest-project" })).json().data; |
| 157 | 267 | expect(pr.facility.slug).toBe("apitest-two"); |
| 158 | 268 | expect(pr.timeline).toEqual([]); |
| 269 | + expect(pr).toMatchObject({ projectClass: "EXPANSION", evidenceLevel: "strong", aiEvidence: "likely", isAi: true, geoPrecision: "city", constructionStartedOn: "2026-03" }); | |
| 270 | + const stages = Object.fromEntries(pr.stages.map((s: { stage: string; date: string | null; reached: boolean; current: boolean }) => [s.stage, s])); | |
| 271 | + expect(stages.announced).toMatchObject({ date: "2025-01", reached: true, current: false }); | |
| 272 | + expect(stages.under_construction).toMatchObject({ date: "2026-03", reached: true, current: true }); | |
| 273 | + expect(stages.operational).toMatchObject({ reached: false, current: false }); | |
| 274 | + expect(pr.velocityDays.announcedToConstruction).toBe(Math.round((Date.UTC(2026, 2, 1) - Date.UTC(2025, 0, 1)) / 86_400_000)); | |
| 275 | + expect(pr.velocityDays.constructionToOpening).toBeNull(); | |
| 276 | + expect(pr.claims).toEqual([]); | |
| 277 | + expect(pr.nearbyInfrastructure).not.toBeNull(); | |
| 278 | + expect(slugs(pr.nearbyInfrastructure.facilities)).toEqual(["apitest-one", "apitest-two"]); | |
| 279 | + expect(Array.isArray(pr.relatedEvents)).toBe(true); | |
| 280 | + expect(pr.dataQuality.claimsTotal).toBe(0); | |
| 281 | + expect(pr.campus).toBeNull(); | |
| 282 | + const ph = (await app.inject({ url: "/api/v1/projects/apitest-project/history" })).json().data; | |
| 283 | + expect(ph.entityType).toBe("project"); | |
| 284 | + expect((await app.inject({ url: "/api/v1/projects/apitest-project/claims" })).json().data).toEqual([]); | |
| 159 | 285 | }); |
| 160 | − it("list endpoints include the fixture", async () => { | |
| 286 | + it("list endpoints include the fixture and carry sources", async () => { | |
| 161 | 287 | expect((await app.inject({ url: "/api/v1/countries" })).json().data.some((c: { iso2: string }) => c.iso2 === "ZZ")).toBe(true); |
| 162 | − expect((await app.inject({ url: "/api/v1/operators?q=apitest" })).json().data[0].facilityCount).toBe(2); | |
| 163 | − expect((await app.inject({ url: "/api/v1/projects?country=ZZ&ai=1" })).json().meta.total).toBe(1); | |
| 288 | + expect((await app.inject({ url: "/api/v1/operators?q=apitest" })).json().data[0].facilityCount).toBe(3); | |
| 289 | + const pj = (await app.inject({ url: "/api/v1/projects?country=ZZ&ai=1" })).json(); | |
| 290 | + expect(pj.meta.total).toBe(1); | |
| 291 | + expect(Array.isArray(pj.sources)).toBe(true); | |
| 164 | 292 | const ev = (await app.inject({ url: "/api/v1/events?country=ZZ&type=capacity_changed" })).json(); |
| 165 | − expect(ev.data[0]).toMatchObject({ id: EVT, entity: { slug: "apitest-one", name: "Apitest One Data Center" }, operator: { slug: "apitest-operator" } }); | |
| 293 | + expect(ev.data[0]).toMatchObject({ id: EVT, entity: { slug: "apitest-one", name: "Apitest One Data Center" }, operator: { slug: "apitest-operator" }, metro: { slug: "apitest-metro" }, isAi: true, significanceBand: "major", evidenceCount: 1, sourceName: "Apitest Source", sourceKind: "operator" }); | |
| 294 | + expect(ev.sources.map((s: { id: string }) => s.id)).toContain(SRC); | |
| 295 | + const major = (await app.inject({ url: "/api/v1/events?country=ZZ&significance=major&ai=1&source_kind=operator&q=capacity" })).json(); | |
| 296 | + expect(major.data.map((e: { id: string }) => e.id)).toEqual([EVT]); | |
| 297 | + const none = (await app.inject({ url: "/api/v1/events?country=ZZ&significance=minor" })).json(); | |
| 298 | + expect(none.data.map((e: { id: string }) => e.id)).toEqual([EVT_RUN]); | |
| 166 | 299 | expect((await app.inject({ url: "/api/v1/sitemap/facilities" })).json().data.some((x: { slug: string }) => x.slug === "apitest-one")).toBe(true); |
| 167 | 300 | }); |
| 168 | 301 | }); |
@@ -191,10 +324,29 @@ describe("map", () => { | ||
| 191 | 324 | const p = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=-74,45,-73,46&country=ZZ" })).json().data; |
| 192 | 325 | expect(p.mode).toBe("points"); |
| 193 | 326 | expect(p.points.map((x: { slug: string }) => x.slug)).toEqual(["apitest-two", "apitest-one"]); |
| 194 | − expect(p.points[1]).toMatchObject({ n: "Apitest One Data Center", o: "Apitest Operator", s: "operational", t: "colocation", mw: 12.5, p: "exact", ai: 1, c: "ZZ" }); | |
| 327 | + expect(p.points[1]).toMatchObject({ n: "Apitest One Data Center", o: "Apitest Operator", s: "operational", t: "colocation", mw: 12.5, p: "exact", ai: 1, c: "ZZ", y: 2019 }); | |
| 195 | 328 | const outside = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=10,10,11,11&country=ZZ" })).json().data; |
| 196 | 329 | expect(outside.points).toEqual([]); |
| 197 | 330 | }); |
| 331 | + it("layers, density and time machine", async () => { | |
| 332 | + const pr = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=-74,45,-73,46&country=ZZ&layer=projects" })).json(); | |
| 333 | + expect(pr.statusCode ?? 200).toBe(200); | |
| 334 | + expect(pr.data.layer).toBe("projects"); | |
| 335 | + const overlay = pr.data.overlay ?? []; | |
| 336 | + expect(overlay.find((x: { slug: string }) => x.slug === "apitest-project")).toMatchObject({ k: "project", s: "under_construction", p: "city" }); | |
| 337 | + const dens = (await app.inject({ url: "/api/v1/map?zoom=6&country=ZZ&density=facilities" })).json().data; | |
| 338 | + expect(dens.mode).toBe("density"); | |
| 339 | + expect(dens.density.view).toBe("facilities"); | |
| 340 | + expect(dens.density.cells.length).toBeGreaterThanOrEqual(1); | |
| 341 | + expect(dens.density.max).toBeGreaterThanOrEqual(1); | |
| 342 | + expect(dens.density.cells.reduce((a: number, c: { n: number }) => a + c.n, 0)).toBe(2); | |
| 343 | + const yr = (await app.inject({ url: "/api/v1/map?zoom=3&country=ZZ&year=2020" })).json().data; | |
| 344 | + expect(yr.year).toBe(2020); | |
| 345 | + expect(yr.yearCoverage).toBe(1); | |
| 346 | + expect(yr.clusters[0].count).toBe(1); // apitest-one opened 2019; apitest-two opens 2027 | |
| 347 | + const bad = await app.inject({ url: "/api/v1/map?zoom=3&layer=nope" }); | |
| 348 | + expect(bad.statusCode).toBe(400); | |
| 349 | + }); | |
| 198 | 350 | }); |
| 199 | 351 | |
| 200 | 352 | describe("search", () => { |
@@ -205,26 +357,266 @@ describe("search", () => { | ||
| 205 | 357 | expect(d.interpreted.status).toBe("operational"); |
| 206 | 358 | expect(d.hits.some((h: { type: string; slug: string }) => h.type === "facility" && h.slug === "apitest-one")).toBe(true); |
| 207 | 359 | expect(d.hits.some((h: { type: string }) => h.type === "operator")).toBe(true); |
| 208 | − expect(d.facilities.items.map((f: { slug: string }) => f.slug)).toEqual(["apitest-one"]); | |
| 360 | + expect(d.facilities.items.map((f: { slug: string }) => f.slug).filter((s: string) => s.startsWith("apitest"))).toEqual(["apitest-one", "apitest-three"]); | |
| 209 | 361 | const city = d.hits.find((h: { type: string }) => h.type === "city"); |
| 210 | 362 | expect(city).toBeUndefined(); // "apitest" is not a city |
| 363 | + expect(Array.isArray(r.json().sources)).toBe(true); | |
| 211 | 364 | }); |
| 212 | 365 | it("city hits link to the facility list", async () => { |
| 213 | 366 | const d = (await app.inject({ url: "/api/v1/search?q=Montr%C3%A9al-Test" })).json().data; |
| 214 | 367 | const city = d.hits.find((h: { type: string }) => h.type === "city"); |
| 215 | 368 | expect(city.href).toContain("/datacenters?q="); |
| 216 | 369 | }); |
| 370 | + it("interprets project, max MW and year phrases", async () => { | |
| 371 | + const d = (await app.inject({ url: "/api/v1/search?q=projects%20under%20500%20MW%20opening%202027" })).json().data; | |
| 372 | + expect(d.interpreted).toMatchObject({ entity: "project", maxMw: 500, year: 2027 }); | |
| 373 | + }); | |
| 374 | +}); | |
| 375 | + | |
| 376 | +describe("intelligence endpoints", () => { | |
| 377 | + it("/nearby returns layers around a point (empty layers are honest)", async () => { | |
| 378 | + const r = await app.inject({ url: "/api/v1/nearby?lat=45.5017&lng=-73.5673&radius_km=10" }); | |
| 379 | + expect(r.statusCode).toBe(200); | |
| 380 | + const d = r.json().data; | |
| 381 | + expect(d.center).toEqual({ lat: 45.5017, lng: -73.5673 }); | |
| 382 | + expect(d.radiusKm).toBe(10); | |
| 383 | + expect(slugs(d.facilities)).toEqual(["apitest-one", "apitest-two"]); | |
| 384 | + expect(d.facilities[0].distanceKm).toBe(0); | |
| 385 | + expect(slugs(d.projects)).toEqual(["apitest-project"]); | |
| 386 | + expect(d.metros.some((m: { slug: string }) => m.slug === "apitest-metro")).toBe(true); | |
| 387 | + expect(d.landingStations).toEqual([]); | |
| 388 | + expect(d.substations).toEqual([]); | |
| 389 | + expect(d.powerPlants).toEqual([]); | |
| 390 | + expect(d.note).toMatch(/never/i); | |
| 391 | + expect((await app.inject({ url: "/api/v1/nearby?lat=95&lng=0" })).statusCode).toBe(400); | |
| 392 | + expect((await app.inject({ url: "/api/v1/nearby?lng=0" })).statusCode).toBe(400); | |
| 393 | + expect((await app.inject({ url: "/api/v1/nearby?lat=45.5&lng=-73.6&radius_km=500" })).statusCode).toBe(400); // > 200 km is rejected, not clamped | |
| 394 | + const typed = (await app.inject({ url: "/api/v1/nearby?lat=45.5&lng=-73.6&radius_km=200&types=metros" })).json().data; | |
| 395 | + expect(typed.radiusKm).toBe(200); | |
| 396 | + expect(typed.facilities).toEqual([]); | |
| 397 | + expect(typed.metros.some((m: { slug: string }) => m.slug === "apitest-metro")).toBe(true); | |
| 398 | + }); | |
| 399 | + it("/pulse is deterministic and validates the window", async () => { | |
| 400 | + const r = await app.inject({ url: "/api/v1/pulse?window=7d" }); | |
| 401 | + expect(r.statusCode).toBe(200); | |
| 402 | + const d = r.json().data; | |
| 403 | + expect(d.window).toBe("7d"); | |
| 404 | + expect(typeof d.since).toBe("string"); | |
| 405 | + expect(d.newProjects).toBeGreaterThanOrEqual(1); // the fixture project was created now | |
| 406 | + expect(d.eventsTotal).toBeGreaterThanOrEqual(2); | |
| 407 | + expect(d.capacityChanges).toBeGreaterThanOrEqual(1); | |
| 408 | + for (const k of ["projectsEnteredConstruction", "facilitiesOpened", "cloudRegionsAnnounced", "powerAgreements", "gridConstraintEvents", "acquisitions", "financingEvents", "newFacilitiesIndexed"]) expect(typeof d[k], k).toBe("number"); | |
| 409 | + expect(Array.isArray(d.majorEvents)).toBe(true); | |
| 410 | + // top-15 lists: the fixture cannot be expected to rank in an 11 000-event database — check shape only | |
| 411 | + expect(Array.isArray(d.byCountry)).toBe(true); | |
| 412 | + if (d.byCountry.length) expect(d.byCountry[0]).toMatchObject({ iso2: expect.any(String), slug: expect.any(String), events: expect.any(Number), newProjects: expect.any(Number) }); | |
| 413 | + expect(Array.isArray(d.byOperator)).toBe(true); | |
| 414 | + if (d.byOperator.length) expect(d.byOperator[0]).toMatchObject({ slug: expect.any(String), events: expect.any(Number) }); | |
| 415 | + expect((await app.inject({ url: "/api/v1/pulse?window=1y" })).statusCode).toBe(400); | |
| 416 | + expect((await app.inject({ url: "/api/v1/pulse" })).json().data.window).toBe("24h"); | |
| 417 | + }); | |
| 418 | + it("/explore filters facilities and projects, rejects unknown keys", async () => { | |
| 419 | + const f = (await app.inject({ url: "/api/v1/explore?entity=facilities&country=ZZ&per_page=10" })).json().data; | |
| 420 | + expect(f.total).toBe(2); | |
| 421 | + expect(f.items.map((x: { slug: string }) => x.slug).sort()).toEqual(["apitest-one", "apitest-two"]); | |
| 422 | + expect(Object.fromEntries(f.facets.status.map((s: { key: string; count: number }) => [s.key, s.count]))).toEqual({ operational: 1, under_construction: 1 }); | |
| 423 | + expect(f.facets.country[0]).toMatchObject({ key: "ZZ", name: "Apitest Land", count: 2 }); | |
| 424 | + expect(f.facets.operator[0]).toMatchObject({ key: "apitest-operator", count: 2 }); | |
| 425 | + expect(f.facets.ai.some((a: { key: string; count: number }) => a.key === "confirmed" && a.count === 1)).toBe(true); | |
| 426 | + expect(f.charts.byStatus.find((s: { key: string }) => s.key === "operational")).toMatchObject({ count: 1, mw: 12.5 }); | |
| 427 | + expect(f.charts.byYear.map((y: { year: number }) => y.year)).toEqual([2019, 2027]); | |
| 428 | + expect(f.map.total).toBe(2); | |
| 429 | + expect(f.map.degraded).toBe(false); | |
| 430 | + expect(f.mwCoverage).toBe(1); | |
| 431 | + const ai = (await app.inject({ url: "/api/v1/explore?country=ZZ&ai=confirmed" })).json().data; | |
| 432 | + expect(ai.items.map((x: { slug: string }) => x.slug)).toEqual(["apitest-one"]); | |
| 433 | + const p = (await app.inject({ url: "/api/v1/explore?entity=projects&country=ZZ&min_mw=50" })).json().data; | |
| 434 | + expect(p.total).toBe(1); | |
| 435 | + expect(p.items[0].slug).toBe("apitest-project"); | |
| 436 | + expect(p.map.points[0]).toMatchObject({ k: "project" }); | |
| 437 | + const bad = await app.inject({ url: "/api/v1/explore?bogus=1" }); | |
| 438 | + expect(bad.statusCode).toBe(400); | |
| 439 | + expect((await app.inject({ url: "/api/v1/explore?entity=nope" })).statusCode).toBe(400); | |
| 440 | + }); | |
| 441 | + it("/coverage reports per scope and per source", async () => { | |
| 442 | + const r = await app.inject({ url: "/api/v1/coverage" }); | |
| 443 | + expect(r.statusCode).toBe(200); | |
| 444 | + const d = r.json().data; | |
| 445 | + expect(d.global.facilities).toBeGreaterThan(2); | |
| 446 | + expect(d.global.capacityCoverage).toBeGreaterThan(0); | |
| 447 | + expect(d.global.capacityCoverage).toBeLessThanOrEqual(1); | |
| 448 | + const zz = d.countries.find((c: { key: string }) => c.key === "ZZ"); | |
| 449 | + expect(zz).toMatchObject({ facilities: 2, capacityCoverage: 1, operatorCoverage: 1, preciseLocationCoverage: 0.5, statusCoverage: 1, openingDateCoverage: 1, projects: 1, projectsWithLocation: 1 }); | |
| 450 | + expect(d.fields.some((f: { field: string }) => /it.?capacity|itCapacityMw/i.test(f.field))).toBe(true); | |
| 451 | + expect(d.bySourceKind.some((k: { kind: string }) => k.kind === "operator")).toBe(true); | |
| 452 | + const s = (await app.inject({ url: "/api/v1/coverage/sources" })).json().data; | |
| 453 | + const src = s.find((x: { id: string }) => x.id === SRC); | |
| 454 | + expect(src).toMatchObject({ kind: "operator", recordsContributed: 1, fieldsContributed: 2, uniqueRecords: 1, redistribution: "attribution", license: "CC-BY-4.0" }); | |
| 455 | + expect(src.lastSuccessfulCrawl).toBeTruthy(); | |
| 456 | + }); | |
| 457 | + it("/ai-infrastructure, /power, /connectivity and /time-machine", async () => { | |
| 458 | + const ai = (await app.inject({ url: "/api/v1/ai-infrastructure" })).json().data; | |
| 459 | + expect(ai.stats.confirmed).toBeGreaterThanOrEqual(1); | |
| 460 | + expect(ai.stats.facilities).toBeGreaterThanOrEqual(ai.stats.confirmed); | |
| 461 | + expect(ai.stats.projects).toBeGreaterThanOrEqual(1); | |
| 462 | + expect(ai.evidenceNote).toMatch(/confirmed/); | |
| 463 | + expect(ai.topOperators.length).toBeGreaterThan(0); | |
| 464 | + expect(ai.topOperators[0]).toHaveProperty("plannedMw"); | |
| 465 | + expect(Array.isArray(ai.pipelineByStage)).toBe(true); | |
| 466 | + const pw = (await app.inject({ url: "/api/v1/power" })).json().data; | |
| 467 | + // top-50 by MW: a 40 MW fixture is outranked by the ≥ 200 MW projects in the database — check shape + ordering | |
| 468 | + expect(pw.largeLoads.length).toBeGreaterThan(0); | |
| 469 | + expect(pw.largeLoads.every((l: { kind: string; mw: number }) => ["utility_capacity", "grid_connection", "planned_load"].includes(l.kind) && l.mw > 0)).toBe(true); | |
| 470 | + expect(pw.largeLoads.every((l: { mw: number }, i: number, a: Array<{ mw: number }>) => i === 0 || a[i - 1]!.mw >= l.mw)).toBe(true); | |
| 471 | + expect(Array.isArray(pw.gridConstraints)).toBe(true); | |
| 472 | + // countryEnergy lists the 60 largest markets: honest energy context, never an estimated carbon intensity | |
| 473 | + expect(pw.countryEnergy.length).toBeGreaterThan(0); | |
| 474 | + expect(pw.countryEnergy.every((c: { energy: { gridCarbonIntensity: unknown; note: string }; facilities: number }) => c.energy.gridCarbonIntensity === null && typeof c.energy.note === "string" && c.facilities > 0)).toBe(true); | |
| 475 | + expect(typeof pw.note).toBe("string"); | |
| 476 | + expect(Array.isArray(pw.utilities)).toBe(true); | |
| 477 | + const cn = (await app.inject({ url: "/api/v1/connectivity" })).json().data; | |
| 478 | + expect(cn.landingStations).toEqual([]); | |
| 479 | + expect(typeof cn.note).toBe("string"); | |
| 480 | + expect(Array.isArray(cn.ixps)).toBe(true); | |
| 481 | + expect(Array.isArray(cn.byMetro)).toBe(true); | |
| 482 | + const tm = (await app.inject({ url: "/api/v1/time-machine" })).json().data; | |
| 483 | + expect(tm.frames.length).toBeGreaterThan(0); | |
| 484 | + const y2019 = tm.frames.find((f: { year: number }) => f.year === 2019); | |
| 485 | + expect(y2019.opened).toBeGreaterThanOrEqual(1); | |
| 486 | + expect(typeof tm.earliestReliableYear).toBe("number"); | |
| 487 | + expect(tm.openingDateCoverage).toBeGreaterThan(0); | |
| 488 | + expect(tm.openingDateCoverage).toBeLessThanOrEqual(1); | |
| 489 | + expect(tm.note).toMatch(/opening date/i); | |
| 490 | + }); | |
| 491 | + it("/compare/operators validates and compares", async () => { | |
| 492 | + expect((await app.inject({ url: "/api/v1/compare/operators?slugs=apitest-operator" })).statusCode).toBe(400); | |
| 493 | + expect((await app.inject({ url: "/api/v1/compare/operators?slugs=apitest-operator,does-not-exist" })).statusCode).toBe(404); | |
| 494 | + const other = (await app.inject({ url: "/api/v1/operators?per_page=1&sort=facilities" })).json().data[0]; | |
| 495 | + const r = await app.inject({ url: `/api/v1/compare/operators?slugs=apitest-operator,${other.slug}` }); | |
| 496 | + expect(r.statusCode).toBe(200); | |
| 497 | + const d = r.json().data; | |
| 498 | + expect(d.operators.length).toBe(2); | |
| 499 | + const me = d.operators.find((o: { slug: string }) => o.slug === "apitest-operator"); | |
| 500 | + expect(me.pipeline.construction.count).toBe(1); | |
| 501 | + expect(me.countriesList.sort()).toEqual(["ZY", "ZZ"]); | |
| 502 | + expect(me.aiCount).toBe(1); | |
| 503 | + expect(me.velocity.windows.length).toBe(3); | |
| 504 | + }); | |
| 505 | + it("/docs-meta lists endpoints with examples", async () => { | |
| 506 | + const d = (await app.inject({ url: "/api/v1/docs-meta" })).json().data; | |
| 507 | + const nearby = d.find((e: { path: string }) => e.path === "/api/v1/nearby"); | |
| 508 | + expect(nearby).toMatchObject({ method: "GET", group: "nearby" }); | |
| 509 | + expect(nearby.params.some((p: { name: string }) => p.name === "lat")).toBe(true); | |
| 510 | + expect(nearby.example.curl).toContain("/api/v1/nearby"); | |
| 511 | + expect(typeof nearby.example.python).toBe("string"); | |
| 512 | + expect(d.some((e: { path: string }) => e.path.startsWith("/api/v1/download/"))).toBe(true); | |
| 513 | + expect(d.some((e: { path: string; method: string }) => e.path === "/api/v1/watchlist" && e.method === "POST")).toBe(true); | |
| 514 | + }); | |
| 515 | +}); | |
| 516 | + | |
| 517 | +describe("dashboard", () => { | |
| 518 | + it("has the API 2.0 sections", async () => { | |
| 519 | + const r = await app.inject({ url: "/api/v1/dashboard" }); | |
| 520 | + expect(r.statusCode).toBe(200); | |
| 521 | + const d = r.json().data; | |
| 522 | + expect(d.stats.facilities).toBeGreaterThan(0); | |
| 523 | + expect(d.pulse.window).toBe("24h"); | |
| 524 | + expect(Array.isArray(d.fastestGrowingMetros)).toBe(true); | |
| 525 | + expect(d.majorProjects.length).toBeGreaterThan(0); | |
| 526 | + expect(d.majorProjects.every((p: { plannedMw: number }) => p.plannedMw >= 100)).toBe(true); | |
| 527 | + expect(Array.isArray(d.operatorExpansion)).toBe(true); | |
| 528 | + expect(Array.isArray(d.powerEvents)).toBe(true); | |
| 529 | + expect(Array.isArray(d.gridConstraints)).toBe(true); | |
| 530 | + expect(d.coverage.key).toBe("global"); | |
| 531 | + expect(typeof d.ingestion.runs24h).toBe("number"); | |
| 532 | + expect(d.ingestion.connectorsTotal).toBeGreaterThanOrEqual(0); | |
| 533 | + }); | |
| 534 | +}); | |
| 535 | + | |
| 536 | +describe("download", () => { | |
| 537 | + it("lists datasets", async () => { | |
| 538 | + const r = await app.inject({ url: "/api/v1/download/datasets" }); | |
| 539 | + expect(r.statusCode).toBe(200); | |
| 540 | + const d = r.json().data; | |
| 541 | + const fac = d.find((x: { key: string }) => x.key === "facilities"); | |
| 542 | + expect(fac.formats).toEqual(expect.arrayContaining(["csv", "json", "geojson"])); | |
| 543 | + expect(fac.rows).toBeGreaterThan(0); | |
| 544 | + expect(fac.excludedSources.some((s: { id: string }) => s.id === SRC_R)).toBe(true); | |
| 545 | + expect(d.map((x: { key: string }) => x.key)).toEqual(expect.arrayContaining(["projects", "operators", "events", "cloud-regions", "ixps", "countries", "markets"])); | |
| 546 | + }); | |
| 547 | + it("streams CSV with a header row, license gating and content-disposition", async () => { | |
| 548 | + const r = await app.inject({ url: "/api/v1/download/facilities?format=csv&country=ZZ" }); | |
| 549 | + expect(r.statusCode).toBe(200); | |
| 550 | + expect(String(r.headers["content-type"])).toContain("text/csv"); | |
| 551 | + expect(String(r.headers["content-disposition"])).toMatch(/attachment; filename="?dci-facilities-\d{4}-\d{2}-\d{2}\.csv"?/); | |
| 552 | + const lines = r.body.trim().split(/\r?\n/); | |
| 553 | + expect(lines[0]!.startsWith("id,slug,name")).toBe(true); | |
| 554 | + expect(lines.length).toBe(3); | |
| 555 | + expect(r.body).toContain("apitest-one"); | |
| 556 | + // the only source of apitest-three forbids redistribution → the row is excluded and the source listed | |
| 557 | + const zy = await app.inject({ url: "/api/v1/download/facilities?format=csv&country=ZY" }); | |
| 558 | + expect(zy.statusCode).toBe(200); | |
| 559 | + expect(zy.body.trim().split(/\r?\n/).length).toBe(1); | |
| 560 | + expect(String(zy.headers["x-dci-excluded-sources"] ?? "")).toContain(SRC_R); | |
| 561 | + const js = await app.inject({ url: "/api/v1/download/facilities?format=json&country=ZY" }); | |
| 562 | + expect(js.json().meta.excludedSources.some((s: { id: string }) => s.id === SRC_R)).toBe(true); | |
| 563 | + const gj = await app.inject({ url: "/api/v1/download/projects?format=geojson&country=ZZ" }); | |
| 564 | + expect(gj.statusCode).toBe(200); | |
| 565 | + expect(gj.json().type).toBe("FeatureCollection"); | |
| 566 | + expect(gj.json().features[0].geometry).toEqual({ type: "Point", coordinates: [-73.57, 45.51] }); | |
| 567 | + expect((await app.inject({ url: "/api/v1/download/facilities?format=xml" })).statusCode).toBe(400); | |
| 568 | + }); | |
| 569 | +}); | |
| 570 | + | |
| 571 | +describe("watchlist", () => { | |
| 572 | + it("is private to an httpOnly cookie token and feeds events", async () => { | |
| 573 | + const first = await app.inject({ url: "/api/v1/watchlist" }); | |
| 574 | + expect(first.statusCode).toBe(200); | |
| 575 | + expect(first.json().data).toEqual([]); | |
| 576 | + const setCookie = String(first.headers["set-cookie"] ?? ""); | |
| 577 | + expect(setCookie).toMatch(/dci_watch=/); | |
| 578 | + expect(setCookie).toMatch(/HttpOnly/i); | |
| 579 | + expect(first.headers["cache-control"]).toBe("no-store"); | |
| 580 | + const cookie = setCookie.split(";")[0]!; | |
| 581 | + const add = await app.inject({ method: "POST", url: "/api/v1/watchlist", headers: { cookie }, payload: { entityType: "facility", slug: "apitest-one" } }); | |
| 582 | + expect([200, 201]).toContain(add.statusCode); | |
| 583 | + expect(add.json().data).toMatchObject({ entityType: "facility", entityId: F1, slug: "apitest-one", name: "Apitest One Data Center" }); | |
| 584 | + const id = add.json().data.id as string; | |
| 585 | + const list = (await app.inject({ url: "/api/v1/watchlist", headers: { cookie } })).json().data; | |
| 586 | + expect(list.map((w: { entityId: string }) => w.entityId)).toEqual([F1]); | |
| 587 | + const other = (await app.inject({ url: "/api/v1/watchlist" })).json().data; | |
| 588 | + expect(other).toEqual([]); // a different visitor sees nothing | |
| 589 | + const feed = (await app.inject({ url: "/api/v1/watchlist/feed", headers: { cookie } })).json().data; | |
| 590 | + expect(feed.map((e: { id: string }) => e.id).sort()).toEqual([EVT, EVT_RUN].sort()); | |
| 591 | + expect((await app.inject({ method: "POST", url: "/api/v1/watchlist", headers: { cookie }, payload: { entityType: "nope", slug: "x" } })).statusCode).toBe(400); | |
| 592 | + expect((await app.inject({ method: "POST", url: "/api/v1/watchlist", headers: { cookie }, payload: { entityType: "facility", slug: "does-not-exist" } })).statusCode).toBe(404); | |
| 593 | + const del = await app.inject({ method: "DELETE", url: `/api/v1/watchlist/${id}`, headers: { cookie } }); | |
| 594 | + expect(del.json().data.deleted).toBe(true); | |
| 595 | + expect((await app.inject({ url: "/api/v1/watchlist", headers: { cookie } })).json().data).toEqual([]); | |
| 596 | + }); | |
| 217 | 597 | }); |
| 218 | 598 | |
| 219 | 599 | describe("admin", () => { |
| 220 | 600 | it("requires the token", async () => { |
| 221 | 601 | expect((await app.inject({ url: "/api/admin/connectors" })).statusCode).toBe(401); |
| 222 | 602 | expect((await app.inject({ url: "/api/admin/connectors", headers: { "x-dci-admin-token": "wrong" } })).statusCode).toBe(401); |
| 223 | − const ok = await app.inject({ url: "/api/admin/connectors", headers: { "x-dci-admin-token": TOKEN } }); | |
| 603 | + const ok = await app.inject({ url: "/api/admin/connectors", headers: ADMIN }); | |
| 224 | 604 | expect(ok.statusCode).toBe(200); |
| 225 | 605 | expect(Array.isArray(ok.json().data)).toBe(true); |
| 226 | 606 | expect(ok.headers["cache-control"]).toBe("no-store"); |
| 227 | 607 | }); |
| 608 | + it("rejects the placeholder token with 503", async () => { | |
| 609 | + const { getEnv } = await import("./env.js"); | |
| 610 | + const saved = process.env.DCI_ADMIN_TOKEN; | |
| 611 | + process.env.DCI_ADMIN_TOKEN = "change-me"; | |
| 612 | + resetEnv(); | |
| 613 | + expect(getEnv().adminToken).toBeNull(); | |
| 614 | + const r = await app.inject({ url: "/api/admin/connectors", headers: { "x-dci-admin-token": "change-me" } }); | |
| 615 | + expect(r.statusCode).toBe(503); | |
| 616 | + process.env.DCI_ADMIN_TOKEN = saved; | |
| 617 | + resetEnv(); | |
| 618 | + expect(getEnv().adminToken).toBe(TOKEN); | |
| 619 | + }); | |
| 228 | 620 | it("constant-time token compare helper", async () => { |
| 229 | 621 | const { tokenMatches } = await import("./routes/admin/index.js"); |
| 230 | 622 | expect(tokenMatches("abc", "abc")).toBe(true); |
@@ -232,21 +624,133 @@ describe("admin", () => { | ||
| 232 | 624 | expect(tokenMatches(undefined, "abc")).toBe(false); |
| 233 | 625 | expect(tokenMatches("abc", null)).toBe(false); |
| 234 | 626 | }); |
| 627 | + it("connector health carries the full DTO", async () => { | |
| 628 | + const rows = (await app.inject({ url: "/api/admin/connectors", headers: ADMIN })).json().data; | |
| 629 | + if (rows.length) { | |
| 630 | + const c = rows[0]; | |
| 631 | + for (const k of ["quarantine", "consecutiveFailures", "urlsDiscovered", "urlsFetched", "newDocs", "changedDocs", "recordsCreated", "recordsModified", "rejectedClaims", "httpErrors", "antiBotEscalations", "scrapflyRequests", "firecrawlRequests"]) expect(c, k).toHaveProperty(k); | |
| 632 | + expect(["ok", "degraded", "failing", "paused", "never_run", "blocked", "schema_change", "no_new_content", "quarantine"]).toContain(c.health); | |
| 633 | + } | |
| 634 | + }); | |
| 635 | + it("quality overview, flags and review queue", async () => { | |
| 636 | + const r = await app.inject({ url: "/api/admin/quality", headers: ADMIN }); | |
| 637 | + expect(r.statusCode).toBe(200); | |
| 638 | + const d = r.json().data; | |
| 639 | + expect(d.openFlags).toBeGreaterThanOrEqual(1); | |
| 640 | + expect(d.bySeverity.warn).toBeGreaterThanOrEqual(1); | |
| 641 | + expect(d.byCode.some((c: { code: string }) => c.code === "mw_change_5x")).toBe(true); | |
| 642 | + expect(d.reviewQueue[0]).toMatchObject({ id: FLAG, entity: { slug: "apitest-one", name: "Apitest One Data Center" }, priority: 999 }); | |
| 643 | + expect(d.largest.operationalFacilityMw.length).toBeGreaterThan(0); | |
| 644 | + expect(d.largest.operationalFacilityMw[0]).toHaveProperty("scope"); | |
| 645 | + expect(d.largest.projectMw.length).toBeGreaterThan(0); | |
| 646 | + for (const k of ["suspiciousMw", "suspiciousInvestment", "unscopedClaims", "claimsInReview", "possibleDuplicates", "projectFalsePositiveCandidates", "unverifiedLargeProjects", "facilitiesMissingCountry", "projectsWithoutFacility", "orphanOperators"]) expect(typeof d[k], k).toBe("number"); | |
| 647 | + const flags = (await app.inject({ url: "/api/admin/quality/flags?status=open&code=mw_change_5x&min_priority=900", headers: ADMIN })).json(); | |
| 648 | + expect(flags.data.map((f: { id: string }) => f.id)).toContain(FLAG); | |
| 649 | + const res = await app.inject({ method: "POST", url: `/api/admin/quality/flags/${FLAG}/resolve`, headers: ADMIN, payload: { resolution: "checked against the operator page" } }); | |
| 650 | + expect(res.statusCode).toBe(200); | |
| 651 | + expect(res.json().data).toMatchObject({ id: FLAG, status: "resolved", resolution: "checked against the operator page" }); | |
| 652 | + const dq = (await app.inject({ url: "/api/v1/datacenters/apitest-one" })).json().data.dataQuality; | |
| 653 | + expect(dq.openFlags.some((f: { code: string }) => f.code === "mw_change_5x")).toBe(false); | |
| 654 | + expect((await app.inject({ method: "POST", url: "/api/admin/quality/flags/flg_nope/dismiss", headers: ADMIN, payload: {} })).statusCode).toBe(404); | |
| 655 | + }); | |
| 656 | + it("claims listing and status change", async () => { | |
| 657 | + const list = (await app.inject({ url: `/api/admin/claims?subject_type=facility&subject_id=${F1}`, headers: ADMIN })).json(); | |
| 658 | + expect(list.data.map((c: { id: string }) => c.id)).toEqual([CLAIM]); | |
| 659 | + const r = await app.inject({ method: "POST", url: `/api/admin/claims/${CLAIM}/status`, headers: ADMIN, payload: { status: "review", reason: "double-check scope" } }); | |
| 660 | + expect(r.statusCode).toBe(200); | |
| 661 | + expect(r.json().data.claim).toMatchObject({ id: CLAIM, status: "review" }); | |
| 662 | + expect(r.json().data.flagged).toBe(false); | |
| 663 | + await app.inject({ method: "POST", url: `/api/admin/claims/${CLAIM}/status`, headers: ADMIN, payload: { status: "current" } }); | |
| 664 | + expect((await app.inject({ method: "POST", url: `/api/admin/claims/${CLAIM}/status`, headers: ADMIN, payload: { status: "bogus" } })).statusCode).toBe(400); | |
| 665 | + }); | |
| 666 | + it("worker proxies answer 502 when the worker is unreachable", async () => { | |
| 667 | + const g = await app.inject({ url: "/api/admin/data-gaps", headers: ADMIN }); | |
| 668 | + expect(g.statusCode).toBe(502); | |
| 669 | + expect(typeof g.json().error).toBe("string"); | |
| 670 | + const t = await app.inject({ url: "/api/admin/documents/doc_nope/trace", headers: ADMIN }); | |
| 671 | + expect(t.statusCode).toBe(502); | |
| 672 | + }); | |
| 673 | + it("hides a project and excludes it everywhere, then unhides", async () => { | |
| 674 | + const hide = await app.inject({ method: "POST", url: `/api/admin/projects/${PRJ}/hide`, headers: ADMIN, payload: { reason: "executive appointment, not a project" } }); | |
| 675 | + expect(hide.statusCode).toBe(200); | |
| 676 | + expect((await app.inject({ url: "/api/v1/projects?country=ZZ" })).json().meta.total).toBe(0); | |
| 677 | + expect((await app.inject({ url: "/api/v1/projects/apitest-project" })).statusCode).toBe(404); | |
| 678 | + expect((await app.inject({ url: "/api/v1/countries/zz" })).json().data.projectCount).toBe(0); | |
| 679 | + expect((await app.inject({ url: "/api/v1/metros/apitest-metro" })).json().data.projectCount).toBe(0); | |
| 680 | + expect((await app.inject({ url: "/api/v1/explore?entity=projects&country=ZZ" })).json().data.total).toBe(0); | |
| 681 | + const overlay = (await app.inject({ url: "/api/v1/map?zoom=12&bbox=-74,45,-73,46&country=ZZ&layer=projects" })).json().data.overlay ?? []; | |
| 682 | + expect(overlay.some((x: { slug: string }) => x.slug === "apitest-project")).toBe(false); | |
| 683 | + expect((await app.inject({ url: "/api/v1/pulse?window=7d" })).json().data.newProjects).toBe(0 + (await sql`select count(*)::int as n from projects where hidden = false and merged_into is null and created_at >= now() - interval '7 days'`)[0]!.n); | |
| 684 | + const prov = await sql`select value from provenance where entity_type = 'project' and entity_id = ${PRJ} and field = 'hidden' and source_id = 'src_manual'`; | |
| 685 | + expect(prov.length).toBe(1); | |
| 686 | + const unhide = await app.inject({ method: "POST", url: `/api/admin/projects/${PRJ}/unhide`, headers: ADMIN, payload: {} }); | |
| 687 | + expect(unhide.statusCode).toBe(200); | |
| 688 | + expect((await app.inject({ url: "/api/v1/projects?country=ZZ" })).json().meta.total).toBe(1); | |
| 689 | + expect((await app.inject({ method: "POST", url: `/api/admin/projects/${PRJ}/hide`, headers: ADMIN, payload: {} })).statusCode).toBe(400); | |
| 690 | + }); | |
| 691 | + it("patches a project with provenance", async () => { | |
| 692 | + const r = await app.inject({ method: "PATCH", url: `/api/admin/projects/${PRJ}`, headers: ADMIN, payload: { planned_mw: 120, expected_opening: "2027-Q3", note: "operator update" } }); | |
| 693 | + expect(r.statusCode).toBe(200); | |
| 694 | + const d = (await app.inject({ url: "/api/v1/projects/apitest-project" })).json().data; | |
| 695 | + expect(d.plannedMw).toBe(120); | |
| 696 | + expect(d.expectedOpening).toBe("2027-Q3"); | |
| 697 | + const prov = await sql`select field from provenance where entity_type = 'project' and entity_id = ${PRJ} and source_id = 'src_manual' and is_current order by field`; | |
| 698 | + expect(prov.map((p) => p.field)).toEqual(expect.arrayContaining(["expectedOpening", "plannedMw"])); | |
| 699 | + expect((await app.inject({ method: "PATCH", url: `/api/admin/projects/${PRJ}`, headers: ADMIN, payload: { expected_opening: "someday" } })).statusCode).toBe(400); | |
| 700 | + expect((await app.inject({ method: "PATCH", url: `/api/admin/projects/${PRJ}`, headers: ADMIN, payload: { bogus: 1 } })).statusCode).toBe(400); | |
| 701 | + }); | |
| 702 | + it("links a building to its campus (containment)", async () => { | |
| 703 | + const r = await app.inject({ method: "POST", url: `/api/admin/facilities/${F1}/parent`, headers: ADMIN, payload: { parentId: F2 } }); | |
| 704 | + expect(r.statusCode).toBe(200); | |
| 705 | + const child = (await app.inject({ url: "/api/v1/datacenters/apitest-one" })).json().data; | |
| 706 | + expect(child.recordScope).toBe("building"); | |
| 707 | + expect(child.parentFacility).toEqual({ id: F2, slug: "apitest-two", name: "Apitest Two Campus" }); | |
| 708 | + const parent = (await app.inject({ url: "/api/v1/datacenters/apitest-two" })).json().data; | |
| 709 | + expect(parent.recordScope).toBe("campus"); | |
| 710 | + expect(parent.buildings.map((b: { slug: string }) => b.slug)).toEqual(["apitest-one"]); | |
| 711 | + // containment-aware aggregate: the campus row with a building is not counted; the building keeps its own MW | |
| 712 | + const co = (await app.inject({ url: "/api/v1/countries/zz" })).json().data; | |
| 713 | + expect(co.pipeline.operational).toMatchObject({ count: 1, mw: 12.5 }); | |
| 714 | + expect(co.pipeline.construction.count).toBe(0); | |
| 715 | + expect((await app.inject({ method: "POST", url: `/api/admin/facilities/${F2}/parent`, headers: ADMIN, payload: { parentId: F1 } })).statusCode).toBe(400); // cycle | |
| 716 | + const undo = await app.inject({ method: "POST", url: `/api/admin/facilities/${F1}/parent`, headers: ADMIN, payload: { parentId: null } }); | |
| 717 | + expect(undo.statusCode).toBe(200); | |
| 718 | + expect((await app.inject({ url: "/api/v1/datacenters/apitest-one" })).json().data.recordScope).toBe("facility"); | |
| 719 | + }); | |
| 720 | + it("run changes and rollback", async () => { | |
| 721 | + const ch = (await app.inject({ url: `/api/admin/runs/${RUN}/changes`, headers: ADMIN })).json().data; | |
| 722 | + expect(ch.provenance.length).toBe(1); | |
| 723 | + expect(ch.claims.map((c: { id: string }) => c.id)).toEqual([CLAIM]); | |
| 724 | + expect(ch.events.map((e: { id: string }) => e.id)).toEqual([EVT_RUN]); | |
| 725 | + expect((await app.inject({ url: "/api/admin/runs/run_nope/changes", headers: ADMIN })).statusCode).toBe(404); | |
| 726 | + const rb = await app.inject({ method: "POST", url: `/api/admin/runs/${RUN}/rollback`, headers: ADMIN, payload: {} }); | |
| 727 | + expect(rb.statusCode).toBe(200); | |
| 728 | + const claim = await sql`select status, rejection_reason from claims where id = ${CLAIM}`; | |
| 729 | + expect(claim[0]).toMatchObject({ status: "rejected", rejection_reason: "rollback" }); | |
| 730 | + const prov = await sql`select id, is_current, is_winner from provenance where entity_id = ${F1} and field = 'itCapacityMw' order by id`; | |
| 731 | + expect(prov.find((p) => p.id === PROV_RUN)).toMatchObject({ is_current: false, is_winner: false }); | |
| 732 | + expect(prov.find((p) => p.id === PROV_OLD)).toMatchObject({ is_current: true, is_winner: true }); | |
| 733 | + const ev = await sql`select review_status from events where id = ${EVT_RUN}`; | |
| 734 | + expect(ev[0]!.review_status).toBe("rejected"); | |
| 735 | + const ids = (await app.inject({ url: "/api/v1/events?country=ZZ&per_page=100" })).json().data.map((e: { id: string }) => e.id); | |
| 736 | + expect(ids).toContain(EVT); | |
| 737 | + expect(ids).not.toContain(EVT_RUN); | |
| 738 | + }); | |
| 235 | 739 | it("curates a facility with provenance + event", async () => { |
| 236 | − const r = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: { "x-dci-admin-token": TOKEN }, payload: { total_power_mw: 25, status: "expansion", note: "test" } }); | |
| 740 | + const r = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: ADMIN, payload: { total_power_mw: 25, status: "expansion", note: "test" } }); | |
| 237 | 741 | expect(r.statusCode).toBe(200); |
| 238 | 742 | const body = r.json().data; |
| 239 | 743 | expect(body.changed.map((c: { field: string }) => c.field).sort()).toEqual(["status", "totalPowerMw"]); |
| 240 | 744 | expect(body.facility.totalPowerMw).toBe(25); |
| 241 | 745 | const prov = await sql`select field, value, source_id from provenance where entity_id = ${F1} and source_id = 'src_manual' order by field`; |
| 242 | − expect(prov.map((p) => p.field)).toEqual(["status", "totalPowerMw"]); | |
| 243 | − const ev = await sql`select event_type, significance from events where entity_id = ${F1} and source_id = 'src_manual'`; | |
| 746 | + expect(prov.map((p) => p.field)).toEqual(expect.arrayContaining(["status", "totalPowerMw"])); | |
| 747 | + const ev = await sql`select event_type, significance from events where entity_id = ${F1} and source_id = 'src_manual' and event_type = 'status_changed'`; | |
| 244 | 748 | expect(ev[0]).toMatchObject({ event_type: "status_changed", significance: 90 }); |
| 245 | − const bad = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: { "x-dci-admin-token": TOKEN }, payload: { status: "not-a-status" } }); | |
| 749 | + const bad = await app.inject({ method: "PATCH", url: `/api/admin/facilities/${F1}`, headers: ADMIN, payload: { status: "not-a-status" } }); | |
| 246 | 750 | expect(bad.statusCode).toBe(400); |
| 247 | 751 | }); |
| 248 | 752 | it("merges a duplicate facility", async () => { |
| 249 | − const r = await app.inject({ method: "POST", url: `/api/admin/facilities/${F2}/merge`, headers: { "x-dci-admin-token": TOKEN }, payload: { into: "apitest-one" } }); | |
| 753 | + const r = await app.inject({ method: "POST", url: `/api/admin/facilities/${F2}/merge`, headers: ADMIN, payload: { into: "apitest-one" } }); | |
| 250 | 754 | expect(r.statusCode).toBe(200); |
| 251 | 755 | expect(r.json().data).toMatchObject({ into: F1, merged: F2 }); |
| 252 | 756 | const list = (await app.inject({ url: "/api/v1/datacenters?country=ZZ" })).json(); |
@@ -258,15 +762,15 @@ describe("admin", () => { | ||
| 258 | 762 | expect(prj.facility.id).toBe(F1); |
| 259 | 763 | }); |
| 260 | 764 | it("validates connector yaml", async () => { |
| 261 | − const good = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: { "x-dci-admin-token": TOKEN }, payload: { yaml: "id: apitest\nname: Apitest\ndomain: example.test\nkind: operator\n" } }); | |
| 765 | + const good = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: ADMIN, payload: { yaml: "id: apitest\nname: Apitest\ndomain: example.test\nkind: operator\n" } }); | |
| 262 | 766 | expect(good.json().data.ok).toBe(true); |
| 263 | − const bad = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: { "x-dci-admin-token": TOKEN }, payload: { yaml: "id: Bad Id\nname: x\n" } }); | |
| 767 | + const bad = await app.inject({ method: "POST", url: "/api/admin/devtool/validate-config", headers: ADMIN, payload: { yaml: "id: Bad Id\nname: x\n" } }); | |
| 264 | 768 | expect(bad.json().data.ok).toBe(false); |
| 265 | 769 | expect(bad.json().data.errors.length).toBeGreaterThan(0); |
| 266 | 770 | }); |
| 267 | 771 | it("previews an extractor against inline html", async () => { |
| 268 | 772 | const html = `<html><head><title>Apitest DC1</title><script type="application/ld+json">{"@type":"Place","name":"Apitest DC1","address":{"streetAddress":"1 Rue Test","addressLocality":"Montréal","addressCountry":"CA"},"geo":{"latitude":45.5,"longitude":-73.6}}</script></head><body><h1>Apitest DC1</h1><div class="specs">Total power 36 MW</div></body></html>`; |
| 269 | − const r = await app.inject({ method: "POST", url: "/api/admin/devtool/preview", headers: { "x-dci-admin-token": TOKEN }, payload: { html, url: "https://example.test/data-centers/dc1", extractor: { kind: "facility", key: "apitest:{code}", fields: { name: "h1", code: { selector: "h1", regex: "\\b(DC\\d+)\\b" }, city: { jsonld: "Place", jsonPath: "address.addressLocality" }, countryIso2: { jsonld: "Place", jsonPath: "address.addressCountry", transform: "country" }, lat: { jsonld: "Place", jsonPath: "geo.latitude", transform: "float" }, lng: { jsonld: "Place", jsonPath: "geo.longitude", transform: "float" }, totalPowerMw: { selector: ".specs", regex: "(\\d+(?:\\.\\d+)?\\s*MW)", transform: "mw" } } } } }); | |
| 773 | + const r = await app.inject({ method: "POST", url: "/api/admin/devtool/preview", headers: ADMIN, payload: { html, url: "https://example.test/data-centers/dc1", extractor: { kind: "facility", key: "apitest:{code}", fields: { name: "h1", code: { selector: "h1", regex: "\\b(DC\\d+)\\b" }, city: { jsonld: "Place", jsonPath: "address.addressLocality" }, countryIso2: { jsonld: "Place", jsonPath: "address.addressCountry", transform: "country" }, lat: { jsonld: "Place", jsonPath: "geo.latitude", transform: "float" }, lng: { jsonld: "Place", jsonPath: "geo.longitude", transform: "float" }, totalPowerMw: { selector: ".specs", regex: "(\\d+(?:\\.\\d+)?\\s*MW)", transform: "mw" } } } } }); | |
| 270 | 774 | expect(r.statusCode).toBe(200); |
| 271 | 775 | const d = r.json().data; |
| 272 | 776 | expect(d.records.length).toBe(1); |
@@ -275,4 +779,8 @@ describe("admin", () => { | ||
| 275 | 779 | expect(d.entities[0].geo).toMatchObject({ lat: 45.5, lng: -73.6 }); |
| 276 | 780 | expect(d.validation).toMatchObject({ total: 1, valid: 1, rejected: 0 }); |
| 277 | 781 | }); |
| 782 | + it("maintenance body is strict", async () => { | |
| 783 | + const bad = await app.inject({ method: "POST", url: "/api/admin/maintenance/rankings", headers: ADMIN, payload: { evil: "x" } }); | |
| 784 | + expect(bad.statusCode).toBe(400); | |
| 785 | + }); | |
| 278 | 786 | }); |
modified
apps/api/src/app.ts
+2 −2
@@ -60,9 +60,9 @@ export async function buildApp(opts: BuildOptions = {}): Promise<FastifyInstance | ||
| 60 | 60 | await app.register(swagger, { |
| 61 | 61 | openapi: { |
| 62 | 62 | openapi: "3.1.0", |
| 63 | − info: { title: "DataCenterIndex API", version: "1.0.0", description: "Public read API (`/api/v1`, envelope `{ data, meta, sources }`) and token-protected admin API (`/api/admin`, header `x-dci-admin-token`). See packages/core/src/api-types.ts for response types." }, | |
| 63 | + info: { title: "DataCenterIndex API", version: "2.0.0", description: "Public read API (`/api/v1`, envelope `{ data, meta, sources }`, weak ETags, `s-maxage` caching) and token-protected admin API (`/api/admin`, header `x-dci-admin-token`). Response `data` types are named after packages/core/src/api-types.ts (the contract); every 200 response documents the envelope and an `x-example`. Figures are published values only — coverage and methodology are stated in `meta.methodology`." }, | |
| 64 | 64 | servers: [{ url: env.siteUrl }, { url: `http://127.0.0.1:${env.port}` }], |
| 65 | − tags: [{ name: "facilities" }, { name: "operators" }, { name: "countries" }, { name: "metros" }, { name: "cloud-regions" }, { name: "ixps" }, { name: "projects" }, { name: "events" }, { name: "map" }, { name: "search" }, { name: "rankings" }, { name: "dashboard" }, { name: "stats" }, { name: "sources" }, { name: "sitemap" }, { name: "system" }, { name: "admin" }], | |
| 65 | + tags: [{ name: "facilities" }, { name: "operators" }, { name: "countries" }, { name: "metros" }, { name: "cloud-regions" }, { name: "ixps" }, { name: "projects" }, { name: "events" }, { name: "map" }, { name: "search" }, { name: "rankings" }, { name: "dashboard" }, { name: "stats" }, { name: "sources" }, { name: "sitemap" }, { name: "nearby" }, { name: "pulse" }, { name: "compare" }, { name: "explore" }, { name: "coverage" }, { name: "ai" }, { name: "power" }, { name: "connectivity" }, { name: "time-machine" }, { name: "download" }, { name: "watchlist" }, { name: "docs" }, { name: "system" }, { name: "admin" }], | |
| 66 | 66 | components: { securitySchemes: { adminToken: { type: "apiKey", in: "header", name: "x-dci-admin-token" } } }, |
| 67 | 67 | }, |
| 68 | 68 | }); |
modified
apps/api/src/env.ts
+5 −1
@@ -19,7 +19,10 @@ export interface ApiEnv { | ||
| 19 | 19 | s3AccessKey: string; |
| 20 | 20 | s3SecretKey: string; |
| 21 | 21 | s3Region: string; |
| 22 | + /** null when DCI_ADMIN_TOKEN is missing, empty or the "change-me" placeholder → admin API answers 503 */ | |
| 22 | 23 | adminToken: string | null; |
| 24 | + /** internal HTTP server of the worker (GET /trace/<documentId>, GET /data-gaps) */ | |
| 25 | + workerUrl: string; | |
| 23 | 26 | /** directory holding connector YAML files */ |
| 24 | 27 | configDir: string; |
| 25 | 28 | /** optional path to the worker connectors registry (parsers) for the dev tool; null = generic only */ |
@@ -79,7 +82,8 @@ export function getEnv(): ApiEnv { | ||
| 79 | 82 | s3AccessKey: e.S3_ACCESS_KEY ?? "dci", |
| 80 | 83 | s3SecretKey: e.S3_SECRET_KEY ?? "", |
| 81 | 84 | s3Region: e.S3_REGION ?? "us-east-1", |
| 82 | − adminToken: e.DCI_ADMIN_TOKEN && e.DCI_ADMIN_TOKEN !== "change-me" ? e.DCI_ADMIN_TOKEN : e.DCI_ADMIN_TOKEN ?? null, | |
| 85 | + adminToken: e.DCI_ADMIN_TOKEN && e.DCI_ADMIN_TOKEN.trim() !== "" && e.DCI_ADMIN_TOKEN !== "change-me" ? e.DCI_ADMIN_TOKEN : null, | |
| 86 | + workerUrl: (e.DCI_WORKER_URL && e.DCI_WORKER_URL.trim()) || "http://127.0.0.1:8320", | |
| 83 | 87 | configDir: e.DCI_CONFIG_DIR ? resolve(e.DCI_CONFIG_DIR) : join(root, "config", "connectors"), |
| 84 | 88 | workerConnectorsDir: existsSync(workerDir) ? workerDir : null, |
| 85 | 89 | logLevel: (e.DCI_LOG_LEVEL ?? "info").toLowerCase(), |
modified
apps/api/src/queues.ts
+1 −1
@@ -24,7 +24,7 @@ export interface CrawlJobData { | ||
| 24 | 24 | requestedBy?: string; |
| 25 | 25 | } |
| 26 | 26 | |
| 27 | −export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup"; | |
| 27 | +export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup" | "quality" | "snapshot"; | |
| 28 | 28 | export interface MaintenanceJobData { kind: MaintenanceKind; task: MaintenanceKind; requestedBy?: string; [k: string]: unknown } |
| 29 | 29 | |
| 30 | 30 | const queues = new Map<string, Queue>(); |
added
apps/api/src/repositories/admin/claims.ts
+47 −0
@@ -0,0 +1,47 @@ | ||
| 1 | +/** Admin claim reads / status changes. Changing a claim's status never rewrites an entity column: the worker's next reconciliation does. */ | |
| 2 | +import type { ClaimDTO } from "@dci/core"; | |
| 3 | +import { CAPACITY_COLUMN } from "@dci/core"; | |
| 4 | +import { pg, claimCols, andAll, page, type Fragment } from "../../lib/sql.js"; | |
| 5 | +import { int, num, reqStr, str, type Row } from "../../lib/rows.js"; | |
| 6 | +import { claimDto } from "../../lib/dto.js"; | |
| 7 | + | |
| 8 | +export interface ClaimFilters { subjectType?: string; subjectId?: string; status?: string; predicate?: string; page?: number; perPage?: number } | |
| 9 | + | |
| 10 | +export async function listClaims(f: ClaimFilters): Promise<{ items: ClaimDTO[]; total: number; page: number; perPage: number }> { | |
| 11 | + const sql = pg(); | |
| 12 | + const pg_ = page(f.page, f.perPage, 500, 50); | |
| 13 | + const c: Fragment[] = []; | |
| 14 | + if (f.subjectType) c.push(sql`k.subject_type = ${f.subjectType}`); | |
| 15 | + if (f.subjectId) c.push(sql`k.subject_id = ${f.subjectId}`); | |
| 16 | + if (f.status && f.status !== "all") c.push(sql`k.status = ${f.status}`); | |
| 17 | + if (f.predicate) c.push(sql`k.predicate = ${f.predicate}`); | |
| 18 | + const rows = await sql<Row[]>`select ${claimCols(sql)}, count(*) over() as total from claims k left join sources s on s.id = k.source_id where ${andAll(sql, c)} order by k.last_observed desc, k.id limit ${pg_.perPage} offset ${pg_.offset}`; | |
| 19 | + return { items: rows.map((r) => claimDto(r, false)), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }; | |
| 20 | +} | |
| 21 | + | |
| 22 | +const FACILITY_COL: Record<string, string> = { itCapacityMw: "it_capacity_mw", totalPowerMw: "total_power_mw", plannedPowerMw: "planned_power_mw", utilityCapacityMw: "utility_capacity_mw", gridConnectionMw: "grid_connection_mw", ultimateCampusMw: "ultimate_campus_mw" }; | |
| 23 | + | |
| 24 | +/** Set a claim status; when a rejected claim was backing a displayed facility MW column, a quality flag is raised (column untouched). */ | |
| 25 | +export async function setClaimStatus(id: string, status: "current" | "rejected" | "review", reason: string | null): Promise<{ claim: ClaimDTO; flagged: boolean } | null> { | |
| 26 | + const sql = pg(); | |
| 27 | + const rows = await sql<Row[]>`update claims set status = ${status}, rejection_reason = ${status === "rejected" ? reason : null} where id = ${id} returning id`; | |
| 28 | + if (!rows[0]) return null; | |
| 29 | + const full = (await sql<Row[]>`select ${claimCols(sql)} from claims k left join sources s on s.id = k.source_id where k.id = ${id}`)[0]!; | |
| 30 | + let flagged = false; | |
| 31 | + if (status === "rejected" && (str(full.subject_type) === "facility" || str(full.subject_type) === "campus")) { | |
| 32 | + const camel = (CAPACITY_COLUMN as Record<string, string | null>)[reqStr(full.predicate)] ?? null; | |
| 33 | + const col = camel ? FACILITY_COL[camel] : null; | |
| 34 | + const v = num(full.value); | |
| 35 | + if (col && v != null) { | |
| 36 | + const fac = (await sql<Row[]>`select ${sql(col)} as v, name from facilities where id = ${reqStr(full.subject_id)}`)[0]; | |
| 37 | + const cur = num(fac?.v); | |
| 38 | + if (fac && cur != null && Math.abs(cur - v) < 1e-9) { | |
| 39 | + await sql`insert into quality_flags (id, entity_type, entity_id, claim_id, code, severity, field, message, details, priority, status, dedupe_key) | |
| 40 | + values (${"flg_" + id.replace(/^clm_/, "").slice(0, 24)}, ${reqStr(full.subject_type)}, ${reqStr(full.subject_id)}, ${id}, 'claim_rejected_backing_value', 'warn', ${camel}, ${`rejected claim (${reqStr(full.predicate)} = ${v}) is still the displayed ${camel} value — re-reconcile`}, ${JSON.stringify({ claimId: id, value: v, reason })}::jsonb, 60, 'open', ${"claim_rejected:" + id}) | |
| 41 | + on conflict (dedupe_key) do update set status = 'open', message = excluded.message, details = excluded.details, updated_at = now()`; | |
| 42 | + flagged = true; | |
| 43 | + } | |
| 44 | + } | |
| 45 | + } | |
| 46 | + return { claim: claimDto(full, false), flagged }; | |
| 47 | +} | |
modified
apps/api/src/repositories/admin/connectors.ts
+82 −15
@@ -1,26 +1,32 @@ | ||
| 1 | +/** | |
| 2 | + * Connector health (ConnectorHealthDTO) from the connectors row, its latest run, the last-24 h run stats, document | |
| 3 | + * counters, the source licence and (best effort) ClickHouse crawl_log latency / premium request counts. | |
| 4 | + */ | |
| 1 | 5 | import type { ConnectorHealthDTO } from "@dci/core"; |
| 2 | 6 | import { chQuery } from "@dci/db/clickhouse"; |
| 3 | 7 | import { pg } from "../../lib/sql.js"; |
| 4 | 8 | import { bool, int, iso, json, num, reqStr, str, type Row } from "../../lib/rows.js"; |
| 5 | 9 | import { asSourceKind } from "../../lib/dto.js"; |
| 6 | 10 | |
| 7 | −interface ChCost { connector_id: string; fetcher: string; n: string | number; avg_ms: string | number | null; credits: string | number | null } | |
| 11 | +interface ChCost { connector_id: string; fetcher: string; n: string | number; avg_ms: string | number | null; credits: string | number | null; n24: string | number | null } | |
| 12 | +interface ChStats { avgMs: number | null; cost: ConnectorHealthDTO["cost"]; requests24h: { scrapfly: number; firecrawl: number } } | |
| 8 | 13 | |
| 9 | −/** Per-connector cost / latency from ClickHouse crawl_log over the last 7 days (null when CH unavailable). */ | |
| 10 | −async function crawlLogStats(): Promise<Map<string, { avgMs: number | null; cost: ConnectorHealthDTO["cost"] }> | null> { | |
| 14 | +/** Per-connector cost / latency from ClickHouse crawl_log over the last 7 days + premium request counts over 24 h (null when CH unavailable). */ | |
| 15 | +async function crawlLogStats(): Promise<Map<string, ChStats> | null> { | |
| 11 | 16 | try { |
| 12 | 17 | const rows = await Promise.race([ |
| 13 | − chQuery<ChCost>("select connector_id, fetcher, count() as n, avg(duration_ms) as avg_ms, sum(credits) as credits from crawl_log where ts >= now() - interval 7 day group by connector_id, fetcher"), | |
| 18 | + chQuery<ChCost>("select connector_id, fetcher, count() as n, avg(duration_ms) as avg_ms, sum(credits) as credits, countIf(ts >= now() - interval 24 hour) as n24 from crawl_log where ts >= now() - interval 7 day group by connector_id, fetcher"), | |
| 14 | 19 | new Promise<never>((_, rej) => setTimeout(() => rej(new Error("clickhouse timeout")), 3_000)), |
| 15 | 20 | ]); |
| 16 | − const out = new Map<string, { avgMs: number | null; cost: ConnectorHealthDTO["cost"]; _n: number; _sum: number }>(); | |
| 21 | + const out = new Map<string, ChStats & { _n: number; _sum: number }>(); | |
| 17 | 22 | for (const r of rows) { |
| 18 | 23 | let e = out.get(r.connector_id); |
| 19 | − if (!e) { e = { avgMs: null, cost: { direct: 0, firecrawl: 0, scrapfly: 0, credits: 0 }, _n: 0, _sum: 0 }; out.set(r.connector_id, e); } | |
| 24 | + if (!e) { e = { avgMs: null, cost: { direct: 0, firecrawl: 0, scrapfly: 0, credits: 0 }, requests24h: { scrapfly: 0, firecrawl: 0 }, _n: 0, _sum: 0 }; out.set(r.connector_id, e); } | |
| 20 | 25 | const n = int(r.n); |
| 21 | 26 | const f = r.fetcher === "firecrawl" ? "firecrawl" : r.fetcher === "scrapfly" ? "scrapfly" : "direct"; |
| 22 | 27 | e.cost[f] += n; |
| 23 | 28 | e.cost.credits += num(r.credits) ?? 0; |
| 29 | + if (f !== "direct") e.requests24h[f] += int(r.n24); | |
| 24 | 30 | e._n += n; |
| 25 | 31 | e._sum += (num(r.avg_ms) ?? 0) * n; |
| 26 | 32 | e.avgMs = e._n ? Math.round(e._sum / e._n) : null; |
@@ -36,17 +42,36 @@ function runStat(stats: Record<string, unknown>, ...keys: string[]): number | nu | ||
| 36 | 42 | return null; |
| 37 | 43 | } |
| 38 | 44 | |
| 39 | −export function connectorHealth(r: Row, ch: Map<string, { avgMs: number | null; cost: ConnectorHealthDTO["cost"] }> | null): ConnectorHealthDTO { | |
| 45 | +export function connectorHealth(r: Row, ch: Map<string, ChStats> | null): ConnectorHealthDTO { | |
| 40 | 46 | const stats = json<Record<string, unknown>>(r.stats, {}); |
| 41 | 47 | const runStats = json<Record<string, unknown>>(r.run_stats, {}); |
| 48 | + const day = json<Record<string, unknown>>(r.day_stats, {}); | |
| 42 | 49 | const merged = { ...stats, ...runStats }; |
| 43 | 50 | const paused = bool(r.paused); |
| 51 | + const quarantine = bool(r.quarantine); | |
| 52 | + const consecutiveFailures = int(r.consecutive_failures); | |
| 53 | + const blockedSince = iso(r.blocked_since); | |
| 44 | 54 | const healthCol = str(r.health) ?? "never_run"; |
| 45 | − const health: ConnectorHealthDTO["health"] = paused ? "paused" : !r.last_run_at && !r.run_started ? "never_run" : healthCol === "ok" || healthCol === "degraded" || healthCol === "failing" ? healthCol : healthCol === "never_run" ? "never_run" : "ok"; | |
| 55 | + const lastError = str(r.last_error) ?? ""; | |
| 46 | 56 | const docsFetched = int(r.doc_fetched); |
| 47 | 57 | const extractTried = int(r.extract_tried); |
| 58 | + const extractionSuccess = extractTried ? Math.round((int(r.extract_ok) / extractTried) * 1000) / 1000 : runStat(runStats, "extractionSuccess"); | |
| 59 | + const neverRun = !r.last_run_at && !r.run_started; | |
| 60 | + const weekRuns = int(r.week_runs); | |
| 61 | + const weekNew = int(r.week_created) + int(r.week_changed); | |
| 62 | + let health: ConnectorHealthDTO["health"]; | |
| 63 | + if (paused) health = "paused"; | |
| 64 | + else if (quarantine) health = "quarantine"; | |
| 65 | + else if (neverRun) health = "never_run"; | |
| 66 | + else if (blockedSince) health = "blocked"; | |
| 67 | + else if ((extractionSuccess != null && extractionSuccess < 0.2 && extractTried >= 10) || /schema|selector|parser/i.test(lastError)) health = "schema_change"; | |
| 68 | + else if (consecutiveFailures >= 3) health = "failing"; | |
| 69 | + else if (weekRuns >= 3 && weekNew === 0 && (str(r.run_status) === "ok" || str(r.last_status) === "ok")) health = "no_new_content"; | |
| 70 | + else if (healthCol === "ok" || healthCol === "degraded" || healthCol === "failing") health = healthCol; | |
| 71 | + else health = str(r.run_status) === "failed" ? "failing" : "ok"; // stored health stale (e.g. never_run) but a run exists | |
| 48 | 72 | const chStats = ch?.get(reqStr(r.id)); |
| 49 | 73 | const cost: ConnectorHealthDTO["cost"] = chStats?.cost ?? { direct: runStat(merged, "direct", "fetch_direct") ?? 0, firecrawl: runStat(merged, "firecrawl", "fetch_firecrawl") ?? 0, scrapfly: runStat(merged, "scrapfly", "fetch_scrapfly") ?? 0, credits: runStat(merged, "credits", "creditsSpent") ?? 0 }; |
| 74 | + const d = (k: string, ...alts: string[]) => runStat(day, k, ...alts) ?? 0; | |
| 50 | 75 | return { |
| 51 | 76 | id: reqStr(r.id), |
| 52 | 77 | sourceName: reqStr(r.source_name), |
@@ -57,6 +82,8 @@ export function connectorHealth(r: Row, ch: Map<string, { avgMs: number | null; | ||
| 57 | 82 | health, |
| 58 | 83 | parserVersion: reqStr(r.parser_version, "v1"), |
| 59 | 84 | lastRunAt: iso(r.last_run_at) ?? iso(r.run_started), |
| 85 | + lastSuccessAt: iso(r.last_success_at), | |
| 86 | + lastFailureAt: iso(r.last_failure_at), | |
| 60 | 87 | nextRunAt: iso(r.next_run_at), |
| 61 | 88 | lastStatus: str(r.last_status) ?? str(r.run_status), |
| 62 | 89 | discovered: runStat(runStats, "discovered", "urls") ?? int(r.doc_total), |
@@ -64,19 +91,59 @@ export function connectorHealth(r: Row, ch: Map<string, { avgMs: number | null; | ||
| 64 | 91 | changed: runStat(runStats, "changed") ?? int(r.doc_changed), |
| 65 | 92 | failed: runStat(runStats, "failed", "errors") ?? int(r.doc_failed), |
| 66 | 93 | extracted: runStat(runStats, "extracted", "entities", "received") ?? int(r.doc_extracted), |
| 67 | − extractionSuccess: extractTried ? Math.round((int(r.extract_ok) / extractTried) * 1000) / 1000 : runStat(runStats, "extractionSuccess"), | |
| 94 | + extractionSuccess, | |
| 68 | 95 | avgResponseMs: chStats?.avgMs ?? runStat(merged, "avgMs", "avg_ms", "avgResponseMs"), |
| 69 | 96 | cost, |
| 70 | 97 | schedule: json<Record<string, string>>(r.schedule, {}), |
| 98 | + quarantine, | |
| 99 | + consecutiveFailures, | |
| 100 | + blockedSince, | |
| 101 | + urlsDiscovered: d("discovered"), | |
| 102 | + urlsFetched: d("fetched"), | |
| 103 | + newDocs: d("created"), | |
| 104 | + changedDocs: d("changed"), | |
| 105 | + recordsCreated: d("created"), | |
| 106 | + recordsModified: d("updated"), | |
| 107 | + rejectedClaims: d("unscopedClaims"), | |
| 108 | + httpErrors: d("failed"), | |
| 109 | + antiBotEscalations: d("blockedFetches", "robotsBlocked"), | |
| 110 | + scrapflyRequests: chStats ? chStats.requests24h.scrapfly : d("scrapfly", "fetch_scrapfly"), | |
| 111 | + firecrawlRequests: chStats ? chStats.requests24h.firecrawl : d("firecrawl", "fetch_firecrawl"), | |
| 112 | + priorityScore: num(r.priority_score), | |
| 113 | + license: str(r.src_license), | |
| 114 | + redistribution: str(r.src_redistribution), | |
| 71 | 115 | }; |
| 72 | 116 | } |
| 73 | 117 | |
| 118 | +const DAY_KEYS = ["discovered", "fetched", "created", "changed", "updated", "unscopedClaims", "failed", "blockedFetches", "robotsBlocked", "scrapfly", "firecrawl", "fetch_scrapfly", "fetch_firecrawl"]; | |
| 119 | + | |
| 74 | 120 | const CONNECTOR_SELECT = (sql: ReturnType<typeof pg>) => sql` |
| 75 | 121 | select c.*, r.id as run_id, r.status as run_status, r.stats as run_stats, r.started_at as run_started, r.finished_at as run_finished, |
| 122 | + ls.finished_at as last_success_at, lf.finished_at as last_failure_at, | |
| 76 | 123 | coalesce(d.total, 0) as doc_total, coalesce(d.fetched, 0) as doc_fetched, coalesce(d.changed, 0) as doc_changed, coalesce(d.failed, 0) as doc_failed, coalesce(d.extracted, 0) as doc_extracted, |
| 77 | − coalesce(d.extract_ok, 0) as extract_ok, coalesce(d.extract_tried, 0) as extract_tried, coalesce(d.quarantined, 0) as doc_quarantined | |
| 124 | + coalesce(d.extract_ok, 0) as extract_ok, coalesce(d.extract_tried, 0) as extract_tried, coalesce(d.quarantined, 0) as doc_quarantined, | |
| 125 | + coalesce(s24.day_stats, '{}'::jsonb) as day_stats, | |
| 126 | + coalesce(w.runs, 0) as week_runs, coalesce(w.created, 0) as week_created, coalesce(w.changed, 0) as week_changed, | |
| 127 | + src.license as src_license, src.redistribution as src_redistribution | |
| 78 | 128 | from connectors c |
| 79 | 129 | left join lateral (select id, status, stats, started_at, finished_at from connector_runs where connector_id = c.id order by started_at desc limit 1) r on true |
| 130 | + left join lateral (select finished_at from connector_runs where connector_id = c.id and status = 'ok' order by started_at desc limit 1) ls on true | |
| 131 | + left join lateral (select finished_at from connector_runs where connector_id = c.id and status = 'failed' order by started_at desc limit 1) lf on true | |
| 132 | + left join lateral (select license, redistribution from sources where connector_id = c.id order by priority, id limit 1) src on true | |
| 133 | + left join ( | |
| 134 | + select connector_id, count(*)::int as runs, coalesce(sum((stats->>'created')::numeric), 0) as created, coalesce(sum((stats->>'changed')::numeric), 0) as changed | |
| 135 | + from connector_runs where started_at >= now() - interval '7 days' and status <> 'running' group by connector_id | |
| 136 | + ) w on w.connector_id = c.id | |
| 137 | + left join ( | |
| 138 | + select connector_id, jsonb_object_agg(key, total) as day_stats | |
| 139 | + from ( | |
| 140 | + select cr.connector_id, kv.key, sum(kv.value) as total | |
| 141 | + from connector_runs cr | |
| 142 | + cross join lateral (select key, (value #>> '{}')::numeric as value from jsonb_each(cr.stats) where key = any(${DAY_KEYS}) and jsonb_typeof(value) = 'number') kv | |
| 143 | + where cr.started_at >= now() - interval '24 hours' | |
| 144 | + group by cr.connector_id, kv.key | |
| 145 | + ) x group by connector_id | |
| 146 | + ) s24 on s24.connector_id = c.id | |
| 80 | 147 | left join ( |
| 81 | 148 | select connector_id, count(*)::int as total, count(*) filter (where last_fetched is not null)::int as fetched, count(*) filter (where change_count > 0)::int as changed, |
| 82 | 149 | count(*) filter (where error_count > 0)::int as failed, coalesce(sum(extract_count), 0)::int as extracted, count(*) filter (where extract_ok)::int as extract_ok, |
@@ -98,7 +165,7 @@ export async function getConnectorAdmin(id: string): Promise<ConnectorAdminDetai | ||
| 98 | 165 | const r = rows[0]; |
| 99 | 166 | if (!r) return null; |
| 100 | 167 | const [runs, byType, errors] = await Promise.all([ |
| 101 | − sql<Row[]>`select id, task, started_at, finished_at, status, stats, error from connector_runs where connector_id = ${id} order by started_at desc limit 30`, | |
| 168 | + sql<Row[]>`select id, task, started_at, finished_at, status, stats, error, quarantined from connector_runs where connector_id = ${id} order by started_at desc limit 30`, | |
| 102 | 169 | sql<Row[]>`select page_type, count(*)::int as n, count(*) filter (where change_count > 0)::int as changed, count(*) filter (where error_count > 0)::int as failed, count(*) filter (where quarantined)::int as quarantined from documents where connector_id = ${id} group by page_type order by n desc`, |
| 103 | 170 | sql<Row[]>`select id, url, error, status_code, error_count, last_checked from documents where connector_id = ${id} and error is not null order by last_checked desc nulls last limit 10`, |
| 104 | 171 | ]); |
@@ -107,7 +174,7 @@ export async function getConnectorAdmin(id: string): Promise<ConnectorAdminDetai | ||
| 107 | 174 | config: json<Record<string, unknown>>(r.config, {}), |
| 108 | 175 | paused: bool(r.paused), |
| 109 | 176 | lastError: str(r.last_error), |
| 110 | − runs: runs.map((x) => ({ id: reqStr(x.id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error) })), | |
| 177 | + runs: runs.map((x) => ({ id: reqStr(x.id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), quarantined: bool(x.quarantined) })), | |
| 111 | 178 | documentsByPageType: byType.map((x) => ({ pageType: reqStr(x.page_type), count: int(x.n), changed: int(x.changed), failed: int(x.failed), quarantined: int(x.quarantined) })), |
| 112 | 179 | errorSamples: errors.map((x) => ({ id: reqStr(x.id), url: reqStr(x.url), error: str(x.error), statusCode: num(x.status_code), errorCount: int(x.error_count), lastChecked: iso(x.last_checked) })), |
| 113 | 180 | createdAt: iso(r.created_at), |
@@ -117,8 +184,8 @@ export async function getConnectorAdmin(id: string): Promise<ConnectorAdminDetai | ||
| 117 | 184 | |
| 118 | 185 | export async function listRuns(connectorId: string | undefined, limit = 100): Promise<Array<Record<string, unknown>>> { |
| 119 | 186 | const sql = pg(); |
| 120 | − const rows = await sql<Row[]>`select id, connector_id, task, started_at, finished_at, status, stats, error from connector_runs where ${connectorId ? sql`connector_id = ${connectorId}` : sql`true`} order by started_at desc limit ${limit}`; | |
| 121 | − return rows.map((x) => ({ id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error) })); | |
| 187 | + const rows = await sql<Row[]>`select id, connector_id, task, started_at, finished_at, status, stats, error, quarantined from connector_runs where ${connectorId ? sql`connector_id = ${connectorId}` : sql`true`} order by started_at desc limit ${limit}`; | |
| 188 | + return rows.map((x) => ({ id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), quarantined: bool(x.quarantined) })); | |
| 122 | 189 | } |
| 123 | 190 | |
| 124 | 191 | export async function getRun(id: string): Promise<Record<string, unknown> | null> { |
@@ -126,5 +193,5 @@ export async function getRun(id: string): Promise<Record<string, unknown> | null | ||
| 126 | 193 | const rows = await sql<Row[]>`select * from connector_runs where id = ${id}`; |
| 127 | 194 | const x = rows[0]; |
| 128 | 195 | if (!x) return null; |
| 129 | − return { id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), log: json(x.log, []) }; | |
| 196 | + return { id: reqStr(x.id), connectorId: reqStr(x.connector_id), task: reqStr(x.task), startedAt: iso(x.started_at), finishedAt: iso(x.finished_at), status: reqStr(x.status), stats: json(x.stats, {}), error: str(x.error), log: json(x.log, []), quarantined: bool(x.quarantined) }; | |
| 130 | 197 | } |
modified
apps/api/src/repositories/admin/merge.ts
+55 −0
@@ -4,6 +4,7 @@ | ||
| 4 | 4 | * detail lookups follow the pointer). |
| 5 | 5 | */ |
| 6 | 6 | import { pg } from "../../lib/sql.js"; |
| 7 | +import { HttpError } from "../../lib/http.js"; | |
| 7 | 8 | import type { Row } from "../../lib/rows.js"; |
| 8 | 9 | |
| 9 | 10 | export interface MergeResult { into: string; merged: string; moved: Record<string, number> } |
@@ -48,6 +49,60 @@ export async function mergeFacility(dupId: string, intoId: string, decidedBy = " | ||
| 48 | 49 | }); |
| 49 | 50 | } |
| 50 | 51 | |
| 52 | +export interface ParentResult { id: string; parentId: string | null; recordScope: string; parentRecordScope: string | null; changed: boolean } | |
| 53 | + | |
| 54 | +/** | |
| 55 | + * Containment link: `childId` becomes a building of `parentId` (null detaches). Guards: parent exists, is not merged, | |
| 56 | + * is not the child, and the parent's own chain never leads back to the child (no cycles, checked 5 levels up). | |
| 57 | + * record_scope: child → building (or facility when detached), parent → campus (stays campus while it has children). | |
| 58 | + */ | |
| 59 | +export async function setFacilityParent(childIdOrSlug: string, parentIdOrSlug: string | null, decidedBy = "admin"): Promise<ParentResult> { | |
| 60 | + const sql = pg(); | |
| 61 | + const child = (await sql<Row[]>`select id, slug, name, parent_facility_id, record_scope, merged_into from facilities where id = ${childIdOrSlug} or slug = ${childIdOrSlug} order by (id = ${childIdOrSlug}) desc limit 1`)[0]; | |
| 62 | + if (!child) throw new HttpError(404, "facility not found"); | |
| 63 | + const childId = String(child.id); | |
| 64 | + let parentId: string | null = null; | |
| 65 | + if (parentIdOrSlug != null) { | |
| 66 | + const parent = (await sql<Row[]>`select id, merged_into from facilities where id = ${parentIdOrSlug} or slug = ${parentIdOrSlug} order by (id = ${parentIdOrSlug}) desc limit 1`)[0]; | |
| 67 | + if (!parent) throw new HttpError(400, "parent facility not found"); | |
| 68 | + if (parent.merged_into) throw new HttpError(400, `parent is merged into ${String(parent.merged_into)}`); | |
| 69 | + parentId = String(parent.id); | |
| 70 | + if (parentId === childId) throw new HttpError(400, "a facility cannot be its own parent"); | |
| 71 | + // cycle check: walk up from the parent | |
| 72 | + let cur: string | null = parentId; | |
| 73 | + for (let i = 0; i < 5 && cur; i++) { | |
| 74 | + const up: Row | undefined = (await sql<Row[]>`select parent_facility_id from facilities where id = ${cur}`)[0]; | |
| 75 | + cur = up?.parent_facility_id == null ? null : String(up.parent_facility_id); | |
| 76 | + if (cur === childId) throw new HttpError(400, "containment cycle: the parent is already contained by this facility"); | |
| 77 | + } | |
| 78 | + } | |
| 79 | + const previous = child.parent_facility_id == null ? null : String(child.parent_facility_id); | |
| 80 | + const scope = parentId ? "building" : "facility"; | |
| 81 | + if (previous === parentId && String(child.record_scope) === scope) return { id: childId, parentId, recordScope: scope, parentRecordScope: null, changed: false }; | |
| 82 | + await ensureManualSource(); | |
| 83 | + const now = new Date().toISOString(); | |
| 84 | + const url = `https://www.datacenterindex.io/admin/facilities/${childId}`; | |
| 85 | + let parentScope: string | null = null; | |
| 86 | + await sql.begin(async (tx) => { | |
| 87 | + await tx`update facilities set parent_facility_id = ${parentId}, record_scope = ${scope}, updated_at = now() where id = ${childId}`; | |
| 88 | + for (const [field, value] of [["parentFacilityId", parentId], ["recordScope", scope]] as Array<[string, unknown]>) { | |
| 89 | + await tx`insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, document_id, url, first_observed, last_observed, retrieved_at, confidence, is_estimate, method, extractor_version, is_current, is_winner, note) | |
| 90 | + values (${"prov_" + Math.random().toString(36).slice(2, 14)}, 'facility', ${childId}, ${field}, ${JSON.stringify(value)}::jsonb, 'src_manual', 'manual', null, ${url}, ${now}, ${now}, ${now}, 'high', false, 'manual', 'admin', true, true, ${"containment set by " + decidedBy}) | |
| 91 | + on conflict (entity_type, entity_id, field, source_id, url) do update set value = excluded.value, last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, is_current = true, is_winner = true, note = excluded.note`; | |
| 92 | + } | |
| 93 | + if (parentId) { | |
| 94 | + await tx`update facilities set record_scope = 'campus', updated_at = now() where id = ${parentId} and record_scope <> 'campus'`; | |
| 95 | + parentScope = "campus"; | |
| 96 | + } | |
| 97 | + // a former parent stays a campus only while it still has building rows | |
| 98 | + if (previous && previous !== parentId) { | |
| 99 | + const r = await tx`update facilities set record_scope = 'facility', updated_at = now() where id = ${previous} and record_scope = 'campus' and not exists (select 1 from facilities c where c.parent_facility_id = ${previous} and c.merged_into is null)`; | |
| 100 | + if (r.count && !parentId) parentScope = "facility"; | |
| 101 | + } | |
| 102 | + }); | |
| 103 | + return { id: childId, parentId, recordScope: scope, parentRecordScope: parentScope, changed: true }; | |
| 104 | +} | |
| 105 | + | |
| 51 | 106 | /** Ensure the manual-curation source exists (kind registry, "Manual curation"). */ |
| 52 | 107 | export async function ensureManualSource(): Promise<void> { |
| 53 | 108 | const sql = pg(); |
added
apps/api/src/repositories/admin/projects.ts
+126 −0
@@ -0,0 +1,126 @@ | ||
| 1 | +/** | |
| 2 | + * Project curation: hide / unhide (false positives are never deleted), merge (duplicate folded into a survivor), | |
| 3 | + * PATCH curated fields. Every change writes src_manual provenance; only lifecycle-relevant changes emit an event | |
| 4 | + * (status → project_status_changed, planned MW → planned_capacity_changed, hide/unhide → project_status_changed with | |
| 5 | + * {hidden} values). Everything else is provenance-only. | |
| 6 | + */ | |
| 7 | +import { createHash } from "node:crypto"; | |
| 8 | +import type postgres from "postgres"; | |
| 9 | +import { newId, normalizeName } from "@dci/core"; | |
| 10 | +import { pg } from "../../lib/sql.js"; | |
| 11 | +import { str, type Row } from "../../lib/rows.js"; | |
| 12 | +import { ensureManualSource } from "./merge.js"; | |
| 13 | + | |
| 14 | +const ADMIN_URL = (id: string) => `https://www.datacenterindex.io/admin/projects/${id}`; | |
| 15 | + | |
| 16 | +export async function findProject(idOrSlug: string): Promise<Row | null> { | |
| 17 | + const sql = pg(); | |
| 18 | + return (await sql<Row[]>`select * from projects where id = ${idOrSlug} or slug = ${idOrSlug} order by (id = ${idOrSlug}) desc limit 1`)[0] ?? null; | |
| 19 | +} | |
| 20 | + | |
| 21 | +type Tx = postgres.TransactionSql; | |
| 22 | + | |
| 23 | +async function manualProvenance(tx: Tx, projectId: string, field: string, value: unknown, now: string, note: string | null, scope: string | null = null): Promise<void> { | |
| 24 | + await tx`insert into provenance (id, entity_type, entity_id, field, value, source_id, connector_id, document_id, url, first_observed, last_observed, retrieved_at, confidence, is_estimate, method, extractor_version, is_current, is_winner, scope, note) | |
| 25 | + values (${newId("provenance")}, 'project', ${projectId}, ${field}, ${JSON.stringify(value)}::jsonb, 'src_manual', 'manual', null, ${ADMIN_URL(projectId)}, ${now}, ${now}, ${now}, 'high', false, 'manual', 'admin', true, true, ${scope}, ${note}) | |
| 26 | + on conflict (entity_type, entity_id, field, source_id, url) do update set value = excluded.value, last_observed = excluded.last_observed, retrieved_at = excluded.retrieved_at, is_current = true, is_winner = true, scope = excluded.scope, note = excluded.note`; | |
| 27 | + // the manual value is the displayed one: other current rows for the field are no longer winners | |
| 28 | + await tx`update provenance set is_winner = false where entity_type = 'project' and entity_id = ${projectId} and field = ${field} and source_id <> 'src_manual' and is_winner`; | |
| 29 | +} | |
| 30 | + | |
| 31 | +export async function setProjectHidden(idOrSlug: string, hidden: boolean, reason: string | null): Promise<{ id: string; hidden: boolean; changed: boolean } | null> { | |
| 32 | + const sql = pg(); | |
| 33 | + const cur = await findProject(idOrSlug); | |
| 34 | + if (!cur) return null; | |
| 35 | + const id = String(cur.id); | |
| 36 | + if (Boolean(cur.hidden) === hidden) return { id, hidden, changed: false }; | |
| 37 | + await ensureManualSource(); | |
| 38 | + const now = new Date().toISOString(); | |
| 39 | + await sql.begin(async (tx) => { | |
| 40 | + await tx`update projects set hidden = ${hidden}, updated_at = now() where id = ${id}`; | |
| 41 | + await manualProvenance(tx, id, "hidden", hidden, now, reason); | |
| 42 | + await tx`insert into events (id, entity_type, entity_id, event_type, detected_at, effective_date, old_value, new_value, source_id, document_id, url, title, summary, significance, confidence, review_status, country_iso2, operator_id, project_id, fingerprint) | |
| 43 | + values (${newId("event")}, 'project', ${id}, 'project_status_changed', ${now}, ${now.slice(0, 10)}, ${JSON.stringify({ hidden: !hidden })}::jsonb, ${JSON.stringify({ hidden })}::jsonb, 'src_manual', null, ${ADMIN_URL(id)}, ${`${String(cur.name)}: ${hidden ? "hidden by review" : "restored by review"}${reason ? ` (${reason})` : ""}`}, ${reason}, 10, 'high', 'approved', ${str(cur.country_iso2)}, ${str(cur.operator_id)}, ${id}, ${`manual:${hidden ? "hide" : "unhide"}:${id}:${now.slice(0, 10)}`}) | |
| 44 | + on conflict (fingerprint) do nothing`; | |
| 45 | + }); | |
| 46 | + return { id, hidden, changed: true }; | |
| 47 | +} | |
| 48 | + | |
| 49 | +export interface ProjectMergeResult { into: string; merged: string; moved: Record<string, number> } | |
| 50 | + | |
| 51 | +export async function mergeProject(dupIdOrSlug: string, intoIdOrSlug: string): Promise<ProjectMergeResult> { | |
| 52 | + const sql = pg(); | |
| 53 | + const [dup, into] = await Promise.all([findProject(dupIdOrSlug), findProject(intoIdOrSlug)]); | |
| 54 | + if (!dup) throw new Error("project not found"); | |
| 55 | + if (!into) throw new Error("target project not found"); | |
| 56 | + const dupId = String(dup.id), intoId = String(into.id); | |
| 57 | + if (dupId === intoId) throw new Error("cannot merge a project into itself"); | |
| 58 | + if (into.hidden === true) throw new Error("target project is hidden"); | |
| 59 | + if (into.merged_into) throw new Error(`target is itself merged into ${String(into.merged_into)}`); | |
| 60 | + if (dup.merged_into) throw new Error(`project is already merged into ${String(dup.merged_into)}`); | |
| 61 | + await ensureManualSource(); | |
| 62 | + return sql.begin(async (tx) => { | |
| 63 | + const moved: Record<string, number> = {}; | |
| 64 | + const count = (r: { count: number }) => r.count; | |
| 65 | + moved.events = count(await tx`update events set project_id = ${intoId} where project_id = ${dupId}`); | |
| 66 | + moved.eventsEntity = count(await tx`update events set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${dupId}`); | |
| 67 | + moved.timeline = count(await tx`update project_timeline set project_id = ${intoId} where project_id = ${dupId}`); | |
| 68 | + moved.claims = count(await tx`update claims k set subject_id = ${intoId} where k.subject_type = 'project' and k.subject_id = ${dupId} and not exists (select 1 from claims q where q.subject_type = 'project' and q.subject_id = ${intoId} and q.predicate = k.predicate and q.source_id = k.source_id and q.url = k.url and coalesce(q.value, 0) = coalesce(k.value, 0) and coalesce(q.value_text, '') = coalesce(k.value_text, ''))`); | |
| 69 | + await tx`delete from claims where subject_type = 'project' and subject_id = ${dupId}`; | |
| 70 | + moved.provenance = count(await tx`update provenance p set entity_id = ${intoId} where p.entity_type = 'project' and p.entity_id = ${dupId} and not exists (select 1 from provenance q where q.entity_type = 'project' and q.entity_id = ${intoId} and q.field = p.field and q.source_id = p.source_id and q.url = p.url)`); | |
| 71 | + await tx`delete from provenance where entity_type = 'project' and entity_id = ${dupId}`; | |
| 72 | + moved.entityKeys = count(await tx`update entity_keys set entity_id = ${intoId} where entity_type = 'project' and entity_id = ${dupId}`); | |
| 73 | + moved.qualityFlags = count(await tx`update quality_flags set status = 'resolved', resolution = ${"merged into " + intoId}, resolved_by = 'admin', resolved_at = now(), updated_at = now() where entity_type = 'project' and entity_id = ${dupId} and status = 'open'`); | |
| 74 | + await tx`update projects set merged_into = ${intoId}, updated_at = now() where id = ${dupId}`; | |
| 75 | + await tx`update projects set last_update = now(), updated_at = now() where id = ${intoId}`; | |
| 76 | + await tx`insert into events (id, entity_type, entity_id, event_type, detected_at, old_value, new_value, source_id, url, title, summary, significance, confidence, review_status, country_iso2, operator_id, project_id, fingerprint) | |
| 77 | + values (${newId("event")}, 'project', ${intoId}, 'facility_updated', now(), ${JSON.stringify({ mergedProject: dupId, name: dup.name })}::jsonb, ${JSON.stringify({ into: intoId })}::jsonb, 'src_manual', ${ADMIN_URL(intoId)}, ${"Duplicate project merged: " + String(dup.name)}, 'Merged by admin', 20, 'high', 'approved', ${str(into.country_iso2)}, ${str(into.operator_id)}, ${intoId}, ${"merge:project:" + dupId + ":" + intoId}) | |
| 78 | + on conflict (fingerprint) do nothing`; | |
| 79 | + return { into: intoId, merged: dupId, moved }; | |
| 80 | + }); | |
| 81 | +} | |
| 82 | + | |
| 83 | +/** snake_case column → provenance field name */ | |
| 84 | +const COLUMN_TO_FIELD: Record<string, string> = { planned_mw: "plannedMw", investment_usd: "investmentUsd", expected_opening: "expectedOpening", announced_on: "announcedOn", construction_started_on: "constructionStartedOn", approved_on: "approvedOn", permit_filed_on: "permitFiledOn", opened_on: "openedOn", operator_id: "operatorId", country_iso2: "countryIso2", metro_id: "metroId", geo_precision: "geoPrecision", project_class: "projectClass", ai_evidence: "aiEvidence", is_ai: "isAi" }; | |
| 85 | + | |
| 86 | +export interface PatchChange { field: string; column: string; oldValue: unknown; newValue: unknown } | |
| 87 | + | |
| 88 | +export async function patchProject(idOrSlug: string, fields: Record<string, unknown>, note: string | null): Promise<{ id: string; changed: PatchChange[] } | null> { | |
| 89 | + const sql = pg(); | |
| 90 | + const cur = await findProject(idOrSlug); | |
| 91 | + if (!cur) return null; | |
| 92 | + const id = String(cur.id); | |
| 93 | + const changed: PatchChange[] = []; | |
| 94 | + const set: Record<string, unknown> = {}; | |
| 95 | + for (const [col, val] of Object.entries(fields)) { | |
| 96 | + if (val === undefined) continue; | |
| 97 | + const before = cur[col] ?? null; | |
| 98 | + if (JSON.stringify(before) === JSON.stringify(val)) continue; | |
| 99 | + set[col] = val; | |
| 100 | + changed.push({ field: COLUMN_TO_FIELD[col] ?? col, column: col, oldValue: before, newValue: val }); | |
| 101 | + } | |
| 102 | + if (!changed.length) return { id, changed }; | |
| 103 | + await ensureManualSource(); | |
| 104 | + const now = new Date().toISOString(); | |
| 105 | + if (set.name) set.normalized_name = normalizeName(String(set.name)); | |
| 106 | + if (set.planned_mw !== undefined) { set.capacity_scope = "facility"; set.capacity_semantics = "planned_power_mw"; } | |
| 107 | + if (set.investment_usd !== undefined) { set.investment_scope = "facility"; set.investment_semantics = "project_investment_usd"; } | |
| 108 | + if (set.lat !== undefined && set.lat !== null && set.geo_precision === undefined) set.geo_precision = "approximate"; | |
| 109 | + set.updated_at = now; | |
| 110 | + set.last_update = now; | |
| 111 | + set.confidence = "high"; | |
| 112 | + await sql.begin(async (tx) => { | |
| 113 | + await tx`update projects set ${tx(set)} where id = ${id}`; | |
| 114 | + for (const c of changed) await manualProvenance(tx, id, c.field, c.newValue, now, note, /mw$|investment/i.test(c.column) ? "facility" : null); | |
| 115 | + const statusChange = changed.find((c) => c.column === "status"); | |
| 116 | + const mwChange = changed.find((c) => c.column === "planned_mw"); | |
| 117 | + const evt = statusChange ? { type: "project_status_changed", old: statusChange.oldValue, nw: statusChange.newValue, sig: 85 } : mwChange ? { type: "planned_capacity_changed", old: mwChange.oldValue, nw: mwChange.newValue, sig: 75 } : null; | |
| 118 | + if (evt) { | |
| 119 | + const digest = createHash("sha256").update(JSON.stringify([evt.type, evt.nw])).digest("hex").slice(0, 16); | |
| 120 | + await tx`insert into events (id, entity_type, entity_id, event_type, detected_at, effective_date, old_value, new_value, source_id, document_id, url, title, summary, significance, confidence, review_status, country_iso2, operator_id, project_id, fingerprint) | |
| 121 | + values (${newId("event")}, 'project', ${id}, ${evt.type}, ${now}, ${now.slice(0, 10)}, ${JSON.stringify(evt.old)}::jsonb, ${JSON.stringify(evt.nw)}::jsonb, 'src_manual', null, ${"https://www.datacenterindex.io/projects/" + String(cur.slug)}, ${`${String(set.name ?? cur.name)}: ${statusChange ? "status" : "planned capacity"} updated by curation`}, ${note}, ${evt.sig}, 'high', 'approved', ${(set.country_iso2 as string | undefined) ?? str(cur.country_iso2)}, ${(set.operator_id as string | undefined) ?? str(cur.operator_id)}, ${id}, ${"manual:project:" + id + ":" + now.slice(0, 10) + ":" + digest}) | |
| 122 | + on conflict (fingerprint) do nothing`; | |
| 123 | + } | |
| 124 | + }); | |
| 125 | + return { id, changed }; | |
| 126 | +} | |
added
apps/api/src/repositories/admin/quality.ts
+140 −0
@@ -0,0 +1,140 @@ | ||
| 1 | +/** | |
| 2 | + * Admin data-quality reads: QualityOverview (largest values, review queue, counters) and the quality_flags list. | |
| 3 | + * Every "largest" query is a plain ORDER BY on published figures — the point is to surface outliers for review. | |
| 4 | + */ | |
| 5 | +import type { QualityFlagDTO, QualityOverview } from "@dci/core"; | |
| 6 | +import { pg, facilityView, pipelineMwAgg, projectLive, andAll, page, PLANNED_SET, CONSTRUCTION_SET, type Fragment } from "../../lib/sql.js"; | |
| 7 | +import { int, iso, num, reqStr, str, type Row } from "../../lib/rows.js"; | |
| 8 | +import { qualityFlagDto } from "../../lib/dto.js"; | |
| 9 | +import { entityKey, resolveEntityRefs } from "../../lib/resolve.js"; | |
| 10 | + | |
| 11 | +const CODE_LABELS: Record<string, string> = { | |
| 12 | + mw_single_site_gt_1000: "Single site > 1 000 MW", | |
| 13 | + mw_building_gt_500: "One building > 500 MW", | |
| 14 | + mw_market_statistic: "MW figure is a market statistic", | |
| 15 | + mw_change_5x: "MW figure changed > 5×", | |
| 16 | + mw_money_collision: "MW / money figure collision", | |
| 17 | + mw_semantics_default: "MW semantics defaulted", | |
| 18 | + mw_utility_not_it: "Utility / grid MW (not IT load)", | |
| 19 | + mw_project_gt_2000: "Project > 2 000 MW", | |
| 20 | + mw_density_implausible: "Implausible power density", | |
| 21 | + mw_invalid: "Invalid MW figure", | |
| 22 | + inv_single_site_gt_50b: "Single site investment > $50B", | |
| 23 | + inv_gt_500b: "Investment > $500B (industry statistic)", | |
| 24 | + inv_change_5x: "Investment changed > 5×", | |
| 25 | + inv_money_mw_collision: "Investment / MW figure collision", | |
| 26 | + inv_invalid: "Invalid investment figure", | |
| 27 | + project_false_positive_candidate: "Project false-positive candidate", | |
| 28 | + project_title_like_name: "Project named after a headline", | |
| 29 | + project_no_location: "Project without a location", | |
| 30 | + project_unknown_scope: "Project figure with unknown scope", | |
| 31 | + facility_no_country: "Facility without a country", | |
| 32 | + operator_bad_slug: "Operator with a broken slug", | |
| 33 | + claim_rejected_backing_value: "Rejected claim was backing the displayed value", | |
| 34 | +}; | |
| 35 | + | |
| 36 | +export function flagLabel(code: string): string { | |
| 37 | + if (CODE_LABELS[code]) return CODE_LABELS[code]!; | |
| 38 | + if (code.startsWith("scope_")) return `Figure scope: ${code.slice(6).replace(/_/g, " ")}`; | |
| 39 | + if (code.startsWith("inv_scope_")) return `Investment scope: ${code.slice(10).replace(/_/g, " ")}`; | |
| 40 | + if (code.startsWith("inv_")) return `Investment: ${code.slice(4).replace(/_/g, " ")}`; | |
| 41 | + if (code.startsWith("duplicate")) return "Possible duplicate"; | |
| 42 | + return code.replace(/_/g, " ").replace(/^\w/, (c) => c.toUpperCase()); | |
| 43 | +} | |
| 44 | + | |
| 45 | +function named(r: Row): { id: string; slug: string; name: string } { | |
| 46 | + return { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }; | |
| 47 | +} | |
| 48 | + | |
| 49 | +/** Attach {slug,name} refs to quality flag rows (one lookup per entity type). */ | |
| 50 | +export async function flagsWithRefs(rows: Row[]): Promise<QualityFlagDTO[]> { | |
| 51 | + const refs = await resolveEntityRefs(rows.map((r) => ({ type: reqStr(r.entity_type), id: str(r.entity_id) }))); | |
| 52 | + return rows.map((r) => qualityFlagDto(r, refs.get(entityKey(r.entity_type, r.entity_id)) ?? null)); | |
| 53 | +} | |
| 54 | + | |
| 55 | +export async function qualityOverview(): Promise<QualityOverview> { | |
| 56 | + const sql = pg(); | |
| 57 | + const live = projectLive(sql); | |
| 58 | + const [counts, sev, codes, largeOps, largeCon, largePrj, largeInv, opPipe, metPipe, queue, alerts] = await Promise.all([ | |
| 59 | + sql<Row[]>`select | |
| 60 | + (select count(*) from quality_flags where status = 'open')::int as open_flags, | |
| 61 | + (select count(*) from entity_matches where status = 'pending')::int + (select count(*) from quality_flags where status = 'open' and code like 'duplicate%')::int as possible_duplicates, | |
| 62 | + (select count(*) from quality_flags where status = 'open' and code like 'mw\\_%')::int as suspicious_mw, | |
| 63 | + (select count(*) from quality_flags where status = 'open' and code like 'inv\\_%')::int as suspicious_inv, | |
| 64 | + (select count(*) from quality_flags where status = 'open' and code = 'project_false_positive_candidate')::int as project_fp, | |
| 65 | + (select count(*) from claims where status = 'unscoped')::int as unscoped_claims, | |
| 66 | + (select count(*) from claims where status = 'review')::int as claims_review, | |
| 67 | + (select count(*) from projects p where ${live} and p.planned_mw >= 500 and (coalesce(p.evidence_level, 'none') <> 'strong' or p.confidence in ('moderate', 'estimated', 'unverified')))::int as unverified_large, | |
| 68 | + (select count(*) from facilities where merged_into is null and country_iso2 is null)::int as fac_no_country, | |
| 69 | + (select count(*) from projects p where ${live} and p.facility_id is null)::int as prj_no_facility, | |
| 70 | + (select count(*) from operators o where not exists (select 1 from facilities f where (f.operator_id = o.id or f.owner_id = o.id) and f.merged_into is null) | |
| 71 | + and not exists (select 1 from projects p where p.operator_id = o.id and ${live}) | |
| 72 | + and not exists (select 1 from cloud_regions r where r.provider_id = o.id) | |
| 73 | + and not exists (select 1 from facility_tenants t where t.operator_id = o.id))::int as orphan_operators`, | |
| 74 | + sql<Row[]>`select severity, count(*)::int as n from quality_flags where status = 'open' group by 1`, | |
| 75 | + sql<Row[]>`select code, count(*)::int as n from quality_flags where status = 'open' group by 1 order by n desc`, | |
| 76 | + sql<Row[]>`select id, slug, name, coalesce(it_capacity_mw, total_power_mw) as mw, capacity_scope as scope, capacity_semantics as semantics from facilities where merged_into is null and status in ('operational', 'partially_operational', 'expansion') and coalesce(it_capacity_mw, total_power_mw) is not null order by 4 desc limit 15`, | |
| 77 | + sql<Row[]>`select id, slug, name, coalesce(planned_power_mw, it_capacity_mw, total_power_mw) as mw from facilities where merged_into is null and status = 'under_construction' and coalesce(planned_power_mw, it_capacity_mw, total_power_mw) is not null order by 4 desc limit 15`, | |
| 78 | + sql<Row[]>`select p.id, p.slug, p.name, p.planned_mw as mw, p.capacity_scope as scope from projects p where ${live} and p.planned_mw is not null order by p.planned_mw desc limit 15`, | |
| 79 | + sql<Row[]>`select p.id, p.slug, p.name, p.investment_usd as inv, p.investment_scope as scope from projects p where ${live} and p.investment_usd is not null order by p.investment_usd desc limit 15`, | |
| 80 | + sql<Row[]>`select o.id, o.slug, o.name, | |
| 81 | + coalesce((select sum(${pipelineMwAgg(sql)}) from ${facilityView(sql)} f where f.operator_id = o.id and f.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0) | |
| 82 | + + coalesce((select sum(p.planned_mw) from projects p where p.operator_id = o.id and ${live} and p.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0) as mw | |
| 83 | + from operators o order by mw desc nulls last limit 15`, | |
| 84 | + sql<Row[]>`select m.id, m.slug, m.name, | |
| 85 | + coalesce((select sum(${pipelineMwAgg(sql)}) from ${facilityView(sql)} f where f.metro_id = m.id and f.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0) | |
| 86 | + + coalesce((select sum(p.planned_mw) from projects p where p.metro_id = m.id and ${live} and p.status = any(${[...PLANNED_SET, ...CONSTRUCTION_SET]})), 0) as mw | |
| 87 | + from metros m order by mw desc nulls last limit 15`, | |
| 88 | + sql<Row[]>`select * from quality_flags where status = 'open' order by priority desc, created_at desc limit 50`, | |
| 89 | + sql<Row[]>`select id, level, component, message, created_at from system_alerts where resolved_at is null order by created_at desc limit 20`, | |
| 90 | + ]); | |
| 91 | + const c = counts[0] ?? {}; | |
| 92 | + const bySeverity: Record<string, number> = {}; | |
| 93 | + for (const r of sev) bySeverity[reqStr(r.severity)] = int(r.n); | |
| 94 | + return { | |
| 95 | + openFlags: int(c.open_flags), | |
| 96 | + bySeverity, | |
| 97 | + byCode: codes.map((r) => ({ code: reqStr(r.code), count: int(r.n), label: flagLabel(reqStr(r.code)) })), | |
| 98 | + possibleDuplicates: int(c.possible_duplicates), | |
| 99 | + suspiciousMw: int(c.suspicious_mw), | |
| 100 | + suspiciousInvestment: int(c.suspicious_inv), | |
| 101 | + projectFalsePositiveCandidates: int(c.project_fp), | |
| 102 | + unscopedClaims: int(c.unscoped_claims), | |
| 103 | + claimsInReview: int(c.claims_review), | |
| 104 | + unverifiedLargeProjects: int(c.unverified_large), | |
| 105 | + facilitiesMissingCountry: int(c.fac_no_country), | |
| 106 | + projectsWithoutFacility: int(c.prj_no_facility), | |
| 107 | + orphanOperators: int(c.orphan_operators), | |
| 108 | + largest: { | |
| 109 | + operationalFacilityMw: largeOps.map((r) => ({ ...named(r), mw: num(r.mw) ?? 0, scope: str(r.scope), semantics: str(r.semantics) })), | |
| 110 | + constructionFacilityMw: largeCon.map((r) => ({ ...named(r), mw: num(r.mw) ?? 0 })), | |
| 111 | + projectMw: largePrj.map((r) => ({ ...named(r), mw: num(r.mw) ?? 0, scope: str(r.scope) })), | |
| 112 | + investment: largeInv.map((r) => ({ ...named(r), investmentUsd: num(r.inv) ?? 0, scope: str(r.scope) })), | |
| 113 | + operatorPipelineMw: opPipe.filter((r) => (num(r.mw) ?? 0) > 0).map((r) => ({ ...named(r), mw: Math.round((num(r.mw) ?? 0) * 100) / 100 })), | |
| 114 | + metroPipelineMw: metPipe.filter((r) => (num(r.mw) ?? 0) > 0).map((r) => ({ ...named(r), mw: Math.round((num(r.mw) ?? 0) * 100) / 100 })), | |
| 115 | + }, | |
| 116 | + reviewQueue: await flagsWithRefs(queue), | |
| 117 | + alerts: alerts.map((a) => ({ id: reqStr(a.id), level: reqStr(a.level), component: reqStr(a.component), message: reqStr(a.message), createdAt: iso(a.created_at) ?? "" })), | |
| 118 | + }; | |
| 119 | +} | |
| 120 | + | |
| 121 | +export interface FlagFilters { status?: string; code?: string; entityType?: string; minPriority?: number; page?: number; perPage?: number } | |
| 122 | + | |
| 123 | +export async function listFlags(f: FlagFilters): Promise<{ items: QualityFlagDTO[]; total: number; page: number; perPage: number }> { | |
| 124 | + const sql = pg(); | |
| 125 | + const pg_ = page(f.page, f.perPage, 200, 50); | |
| 126 | + const c: Fragment[] = []; | |
| 127 | + if (f.status && f.status !== "all") c.push(sql`q.status = ${f.status}`); | |
| 128 | + if (f.code) c.push(sql`q.code = ${f.code}`); | |
| 129 | + if (f.entityType) c.push(sql`q.entity_type = ${f.entityType}`); | |
| 130 | + if (f.minPriority != null) c.push(sql`q.priority >= ${f.minPriority}`); | |
| 131 | + const rows = await sql<Row[]>`select q.*, count(*) over() as total from quality_flags q where ${andAll(sql, c)} order by q.priority desc, q.created_at desc limit ${pg_.perPage} offset ${pg_.offset}`; | |
| 132 | + return { items: await flagsWithRefs(rows), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }; | |
| 133 | +} | |
| 134 | + | |
| 135 | +export async function setFlagStatus(id: string, status: "resolved" | "dismissed", resolution: string | null): Promise<QualityFlagDTO | null> { | |
| 136 | + const sql = pg(); | |
| 137 | + const rows = await sql<Row[]>`update quality_flags set status = ${status}, resolution = ${resolution}, resolved_by = 'admin', resolved_at = now(), updated_at = now() where id = ${id} returning *`; | |
| 138 | + if (!rows[0]) return null; | |
| 139 | + return (await flagsWithRefs(rows))[0] ?? null; | |
| 140 | +} | |
added
apps/api/src/repositories/admin/runs.ts
+70 −0
@@ -0,0 +1,70 @@ | ||
| 1 | +/** | |
| 2 | + * Per-run change inspection and rollback. A rollback never deletes: claims → rejected (reason rollback), provenance rows | |
| 3 | + * → is_current = false (winner restored to the latest remaining current row per field), events → review_status rejected. | |
| 4 | + * Entity columns are left as they are; the worker's reconciliation re-derives them from the remaining claims. | |
| 5 | + */ | |
| 6 | +import type { ClaimDTO, EventDTO, ProvenanceDTO } from "@dci/core"; | |
| 7 | +import { pg, claimCols, eventCols, eventJoins } from "../../lib/sql.js"; | |
| 8 | +import { bool, int, iso, reqStr, str, type Row } from "../../lib/rows.js"; | |
| 9 | +import { claimDto, eventDto, provenanceDto } from "../../lib/dto.js"; | |
| 10 | +import { entityKey, resolveEntityRefs } from "../../lib/resolve.js"; | |
| 11 | + | |
| 12 | +const CAP = 2000; | |
| 13 | + | |
| 14 | +export interface RunChanges { | |
| 15 | + runId: string; | |
| 16 | + provenance: Array<ProvenanceDTO & { entityType: string; entityId: string; isCurrent: boolean }>; | |
| 17 | + claims: ClaimDTO[]; | |
| 18 | + events: EventDTO[]; | |
| 19 | + documentVersions: Array<{ id: string; documentId: string; url: string | null; fetchedAt: string | null; significance: number; changes: number }>; | |
| 20 | + counts: { provenance: number; claims: number; events: number; documentVersions: number }; | |
| 21 | +} | |
| 22 | + | |
| 23 | +export async function runExists(id: string): Promise<boolean> { | |
| 24 | + const sql = pg(); | |
| 25 | + return (await sql`select 1 from connector_runs where id = ${id}`).length > 0; | |
| 26 | +} | |
| 27 | + | |
| 28 | +export async function runChanges(id: string): Promise<RunChanges | null> { | |
| 29 | + const sql = pg(); | |
| 30 | + if (!(await runExists(id))) return null; | |
| 31 | + const [prov, claims, events, versions, counts] = await Promise.all([ | |
| 32 | + sql<Row[]>`select p.*, s.name as source_name, s.kind as source_kind from provenance p left join sources s on s.id = p.source_id where p.run_id = ${id} order by p.entity_type, p.entity_id, p.field limit ${CAP}`, | |
| 33 | + sql<Row[]>`select ${claimCols(sql)} from claims k left join sources s on s.id = k.source_id where k.run_id = ${id} order by k.subject_type, k.subject_id, k.predicate limit ${CAP}`, | |
| 34 | + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.run_id = ${id} order by e.detected_at desc limit ${CAP}`, | |
| 35 | + sql<Row[]>`select v.id, v.document_id, d.url, v.fetched_at, v.significance, jsonb_array_length(coalesce(v.detected_changes, '[]'::jsonb)) as changes from document_versions v left join documents d on d.id = v.document_id where v.run_id = ${id} order by v.fetched_at desc limit ${CAP}`, | |
| 36 | + sql<Row[]>`select (select count(*) from provenance where run_id = ${id})::int as p, (select count(*) from claims where run_id = ${id})::int as c, (select count(*) from events where run_id = ${id})::int as e, (select count(*) from document_versions where run_id = ${id})::int as v`, | |
| 37 | + ]); | |
| 38 | + const refs = await resolveEntityRefs(events.map((r) => ({ type: reqStr(r.entity_type), id: str(r.entity_id) }))); | |
| 39 | + const c = counts[0] ?? {}; | |
| 40 | + return { | |
| 41 | + runId: id, | |
| 42 | + provenance: prov.map((p) => ({ ...provenanceDto(p), entityType: reqStr(p.entity_type), entityId: reqStr(p.entity_id), isCurrent: bool(p.is_current) })), | |
| 43 | + claims: claims.map((k) => claimDto(k, false)), | |
| 44 | + events: events.map((e) => eventDto(e, refs.get(entityKey(e.entity_type, e.entity_id)) ?? null)), | |
| 45 | + documentVersions: versions.map((v) => ({ id: reqStr(v.id), documentId: reqStr(v.document_id), url: str(v.url), fetchedAt: iso(v.fetched_at), significance: int(v.significance), changes: int(v.changes) })), | |
| 46 | + counts: { provenance: int(c.p), claims: int(c.c), events: int(c.e), documentVersions: int(c.v) }, | |
| 47 | + }; | |
| 48 | +} | |
| 49 | + | |
| 50 | +export interface RollbackResult { runId: string; claimsRejected: number; provenanceRetired: number; winnersRestored: number; eventsRejected: number } | |
| 51 | + | |
| 52 | +export async function rollbackRun(id: string): Promise<RollbackResult | null> { | |
| 53 | + const sql = pg(); | |
| 54 | + if (!(await runExists(id))) return null; | |
| 55 | + return sql.begin(async (tx) => { | |
| 56 | + const claims = await tx`update claims set status = 'rejected', rejection_reason = 'rollback' where run_id = ${id} and status <> 'rejected'`; | |
| 57 | + const touched = await tx<Row[]>`select distinct entity_type, entity_id, field from provenance where run_id = ${id} and is_current`; | |
| 58 | + const prov = await tx`update provenance set is_current = false, is_winner = false where run_id = ${id} and is_current`; | |
| 59 | + let restored = 0; | |
| 60 | + for (const t of touched) { | |
| 61 | + const r = await tx`update provenance set is_winner = true where id = ( | |
| 62 | + select id from provenance where entity_type = ${reqStr(t.entity_type)} and entity_id = ${reqStr(t.entity_id)} and field = ${reqStr(t.field)} and is_current order by last_observed desc, retrieved_at desc limit 1) | |
| 63 | + and not exists (select 1 from provenance w where w.entity_type = ${reqStr(t.entity_type)} and w.entity_id = ${reqStr(t.entity_id)} and w.field = ${reqStr(t.field)} and w.is_current and w.is_winner)`; | |
| 64 | + restored += r.count; | |
| 65 | + } | |
| 66 | + const events = await tx`update events set review_status = 'rejected' where run_id = ${id} and review_status <> 'rejected'`; | |
| 67 | + await tx`update connector_runs set log = coalesce(log, '[]'::jsonb) || ${JSON.stringify([{ t: new Date().toISOString(), level: "warn", msg: `rolled back by admin: ${claims.count} claims rejected, ${prov.count} provenance rows retired, ${events.count} events rejected` }])}::jsonb where id = ${id}`; | |
| 68 | + return { runId: id, claimsRejected: claims.count, provenanceRetired: prov.count, winnersRestored: restored, eventsRejected: events.count }; | |
| 69 | + }); | |
| 70 | +} | |
added
apps/api/src/repositories/ai.ts
+76 −0
@@ -0,0 +1,76 @@ | ||
| 1 | +/** | |
| 2 | + * /ai-infrastructure — AI / HPC index. Counts use ai_evidence confirmed | likely (or the legacy is_ai flag); | |
| 3 | + * "associated" facilities are reported separately and never counted as AI infrastructure. | |
| 4 | + */ | |
| 5 | +import type { AiIndex } from "@dci/core"; | |
| 6 | +import { pg, facilityJoins, facilitySummaryCols, facilityView, knownMwAgg, pipelineMwAgg, countedAgg, hasMwAgg, mwExpr, projectJoins, projectSummaryCols, projectLive, CONSTRUCTION_SET, type Sql } from "../lib/sql.js"; | |
| 7 | +import { int, num, reqStr, type Row } from "../lib/rows.js"; | |
| 8 | +import { asStatus, facilitySummary, projectSummary, round2, share } from "../lib/dto.js"; | |
| 9 | +import { eventsWhere } from "./events.js"; | |
| 10 | + | |
| 11 | +export const AI_EVIDENCE_NOTE = "AI evidence levels: confirmed = a source explicitly describes the site as AI / HPC / GPU / accelerated-computing infrastructure; likely = AI-ready, high-density, liquid- or direct-to-chip cooling or GPU signals in the source; associated = only an AI tenant, customer or company mention (reported separately, NOT counted as AI infrastructure); unknown = no signal. A single keyword never confirms a site. Counts and MW cover confirmed + likely only; MW figures are published site-scoped figures (containment-aware), never extrapolated."; | |
| 12 | + | |
| 13 | +const AI_F = (sql: Sql) => sql`(f.ai_evidence in ('confirmed', 'likely') or f.is_ai or f.facility_type = 'ai')`; | |
| 14 | +const AI_P = (sql: Sql) => sql`(p.ai_evidence in ('confirmed', 'likely') or p.is_ai)`; | |
| 15 | + | |
| 16 | +export async function aiIndex(): Promise<AiIndex> { | |
| 17 | + const sql = pg(); | |
| 18 | + const known = knownMwAgg(sql), pipe = pipelineMwAgg(sql), counted = countedAgg(sql), hasMw = hasMwAgg(sql); | |
| 19 | + const [fs, ps, topOps, topMetros, topCountries, stages, recentProjects, facilities, recentAnnouncements] = await Promise.all([ | |
| 20 | + sql<Row[]>`select count(*) filter (where ${counted} and ${AI_F(sql)})::int as facilities, | |
| 21 | + count(*) filter (where ${counted} and f.ai_evidence = 'confirmed')::int as confirmed, | |
| 22 | + count(*) filter (where ${counted} and f.ai_evidence = 'likely')::int as likely, | |
| 23 | + count(*) filter (where ${counted} and f.ai_evidence = 'associated')::int as associated, | |
| 24 | + sum(${pipe}) filter (where ${AI_F(sql)} and f.status = any(${CONSTRUCTION_SET}))::float as construction_mw, | |
| 25 | + sum(${known}) filter (where ${AI_F(sql)} and f.status in ('operational', 'partially_operational', 'expansion'))::float as known_mw, | |
| 26 | + count(*) filter (where ${counted} and ${AI_F(sql)} and ${hasMw})::int as with_mw, | |
| 27 | + count(distinct f.country_iso2) filter (where ${AI_F(sql)})::int as f_countries, | |
| 28 | + count(distinct f.operator_id) filter (where ${AI_F(sql)})::int as f_operators | |
| 29 | + from ${facilityView(sql)} f`, | |
| 30 | + sql<Row[]>`select count(*)::int as projects, sum(p.planned_mw)::float as planned_mw, count(distinct p.country_iso2)::int as countries, count(distinct p.operator_id)::int as operators from projects p where ${projectLive(sql)} and ${AI_P(sql)}`, | |
| 31 | + sql<Row[]>`select * from (select o.id, o.slug, o.name, | |
| 32 | + (select count(*) from ${facilityView(sql)} f where f.operator_id = o.id and ${counted} and ${AI_F(sql)})::int as facilities, | |
| 33 | + (select count(*) from projects p where p.operator_id = o.id and ${projectLive(sql)} and ${AI_P(sql)})::int as projects, | |
| 34 | + (select sum(p.planned_mw) from projects p where p.operator_id = o.id and ${projectLive(sql)} and ${AI_P(sql)})::float as planned_mw | |
| 35 | + from operators o where exists (select 1 from facilities f where f.operator_id = o.id and f.merged_into is null and ${AI_F(sql)}) or exists (select 1 from projects p where p.operator_id = o.id and ${projectLive(sql)} and ${AI_P(sql)})) t order by t.facilities + t.projects desc, t.planned_mw desc nulls last, t.name limit 10`, | |
| 36 | + sql<Row[]>`select * from (select m.id, m.slug, m.name, m.country_iso2, | |
| 37 | + (select count(*) from ${facilityView(sql)} f where f.metro_id = m.id and ${counted} and ${AI_F(sql)})::int as facilities, | |
| 38 | + (select count(*) from projects p where p.metro_id = m.id and ${projectLive(sql)} and ${AI_P(sql)})::int as projects, | |
| 39 | + (select sum(p.planned_mw) from projects p where p.metro_id = m.id and ${projectLive(sql)} and ${AI_P(sql)})::float as planned_mw | |
| 40 | + from metros m where exists (select 1 from facilities f where f.metro_id = m.id and f.merged_into is null and ${AI_F(sql)}) or exists (select 1 from projects p where p.metro_id = m.id and ${projectLive(sql)} and ${AI_P(sql)})) t order by t.facilities + t.projects desc, t.planned_mw desc nulls last, t.name limit 10`, | |
| 41 | + sql<Row[]>`select * from (select c.iso2, c.slug, c.name, | |
| 42 | + (select count(*) from ${facilityView(sql)} f where f.country_iso2 = c.iso2 and ${counted} and ${AI_F(sql)})::int as facilities, | |
| 43 | + (select count(*) from projects p where p.country_iso2 = c.iso2 and ${projectLive(sql)} and ${AI_P(sql)})::int as projects, | |
| 44 | + (select sum(p.planned_mw) from projects p where p.country_iso2 = c.iso2 and ${projectLive(sql)} and ${AI_P(sql)})::float as planned_mw | |
| 45 | + from countries c where exists (select 1 from facilities f where f.country_iso2 = c.iso2 and f.merged_into is null and ${AI_F(sql)}) or exists (select 1 from projects p where p.country_iso2 = c.iso2 and ${projectLive(sql)} and ${AI_P(sql)})) t order by t.facilities + t.projects desc, t.planned_mw desc nulls last, t.name limit 10`, | |
| 46 | + sql<Row[]>`select p.status, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${projectLive(sql)} and ${AI_P(sql)} group by 1 order by n desc`, | |
| 47 | + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and ${AI_P(sql)} order by p.last_update desc limit 10`, | |
| 48 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and ${AI_F(sql)} order by ${mwExpr(sql)} desc nulls last, f.completeness desc limit 30`, | |
| 49 | + eventsWhere(sql`e.is_ai = true`, 15), | |
| 50 | + ]); | |
| 51 | + const f = fs[0] ?? {}; | |
| 52 | + const p = ps[0] ?? {}; | |
| 53 | + const facilitiesN = int(f.facilities); | |
| 54 | + return { | |
| 55 | + stats: { | |
| 56 | + facilities: facilitiesN, | |
| 57 | + confirmed: int(f.confirmed), | |
| 58 | + likely: int(f.likely), | |
| 59 | + associated: int(f.associated), | |
| 60 | + projects: int(p.projects), | |
| 61 | + plannedMw: round2(num(p.planned_mw)), | |
| 62 | + constructionMw: round2(num(f.construction_mw)), | |
| 63 | + countries: Math.max(int(f.f_countries), int(p.countries)), | |
| 64 | + operators: Math.max(int(f.f_operators), int(p.operators)), | |
| 65 | + mwCoverage: share(int(f.with_mw), facilitiesN), | |
| 66 | + }, | |
| 67 | + topOperators: topOps.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), facilities: int(r.facilities), projects: int(r.projects), plannedMw: round2(num(r.planned_mw)) })), | |
| 68 | + topMetros: topMetros.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), countryIso2: reqStr(r.country_iso2), facilities: int(r.facilities), projects: int(r.projects), plannedMw: round2(num(r.planned_mw)) })), | |
| 69 | + topCountries: topCountries.map((r) => ({ iso2: reqStr(r.iso2), slug: reqStr(r.slug), name: reqStr(r.name), facilities: int(r.facilities), projects: int(r.projects), plannedMw: round2(num(r.planned_mw)) })), | |
| 70 | + pipelineByStage: stages.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 71 | + recentAnnouncements, | |
| 72 | + recentProjects: recentProjects.map(projectSummary), | |
| 73 | + facilities: facilities.map(facilitySummary), | |
| 74 | + evidenceNote: AI_EVIDENCE_NOTE, | |
| 75 | + }; | |
| 76 | +} | |
modified
apps/api/src/repositories/cloud-regions.ts
+33 −15
@@ -1,8 +1,10 @@ | ||
| 1 | −import type { CloudRegionSummary, FacilitySummary } from "@dci/core"; | |
| 2 | −import { pg, andAll, cloudRegionCols, facilityJoins, facilitySummaryCols, type Fragment } from "../lib/sql.js"; | |
| 3 | −import { str, type Row } from "../lib/rows.js"; | |
| 1 | +import type { CloudRegionDetail as CloudRegionDetailContract, CloudRegionSummary, FacilitySummary } from "@dci/core"; | |
| 2 | +import { pg, andAll, cloudRegionCols, facilityJoins, facilitySummaryCols, mwExpr, type Fragment } from "../lib/sql.js"; | |
| 3 | +import { record, reqIso, str, type Row } from "../lib/rows.js"; | |
| 4 | 4 | import { cloudRegionSummary, facilitySummary } from "../lib/dto.js"; |
| 5 | 5 | import { findBySlugOrId } from "../lib/resolve.js"; |
| 6 | +import { eventsForEntity } from "./events.js"; | |
| 7 | +import { provenanceFor } from "./facilities.js"; | |
| 6 | 8 | |
| 7 | 9 | export async function listCloudRegions(f: { provider?: string; country?: string; metroId?: string; status?: string }): Promise<CloudRegionSummary[]> { |
| 8 | 10 | const sql = pg(); |
@@ -15,40 +17,56 @@ export async function listCloudRegions(f: { provider?: string; country?: string; | ||
| 15 | 17 | return rows.map(cloudRegionSummary); |
| 16 | 18 | } |
| 17 | 19 | |
| 18 | −export interface CloudRegionDetail extends CloudRegionSummary { | |
| 20 | +/** Contract CloudRegionDetail + the extra fields the web already reads. */ | |
| 21 | +export interface CloudRegionDetail extends CloudRegionDetailContract { | |
| 19 | 22 | regionName: string | null; |
| 20 | − metro: { id: string; slug: string; name: string } | null; | |
| 21 | 23 | countryName: string | null; |
| 22 | − siblingRegions: CloudRegionSummary[]; | |
| 23 | 24 | facilitiesInMetro: FacilitySummary[]; |
| 24 | − externalIds: Record<string, string | number>; | |
| 25 | 25 | updatedAt: string; |
| 26 | 26 | } |
| 27 | 27 | |
| 28 | +export const MARKET_ASSOCIATION_NOTE = "Region associated with market, not hosted at a specific facility"; | |
| 29 | + | |
| 28 | 30 | export async function getCloudRegion(idOrSlug: string): Promise<CloudRegionDetail | null> { |
| 29 | 31 | const sql = pg(); |
| 30 | 32 | const row = await findBySlugOrId("cloud_regions", idOrSlug); |
| 31 | 33 | if (!row) return null; |
| 32 | 34 | const id = String(row.id); |
| 33 | 35 | const metroId = str(row.metro_id); |
| 34 | − const [main, siblings, facs, metroRows, countryRows] = await Promise.all([ | |
| 36 | + const providerId = String(row.provider_id); | |
| 37 | + const [main, siblings, facs, hosts, metroRows, countryRows, events, provenance] = await Promise.all([ | |
| 35 | 38 | sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.id = ${id}`, |
| 36 | − sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.provider_id = ${String(row.provider_id)} and r.id <> ${id} and r.country_iso2 is not distinct from ${str(row.country_iso2)} order by r.code limit 12`, | |
| 39 | + sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.provider_id = ${providerId} and r.id <> ${id} and r.country_iso2 is not distinct from ${str(row.country_iso2)} order by r.code limit 12`, | |
| 37 | 40 | metroId |
| 38 | − ? sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.metro_id = ${metroId} order by coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) desc nulls last, f.name limit 12` | |
| 41 | + ? sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.metro_id = ${metroId} order by ${mwExpr(sql)} desc nulls last, f.name limit 12` | |
| 39 | 42 | : Promise.resolve([] as Row[]), |
| 43 | + // explicit hosting link only: a cloud tenancy of this provider at a facility in the region's metro / country | |
| 44 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} join facility_tenants t on t.facility_id = f.id and t.role = 'cloud' and t.operator_id = ${providerId} | |
| 45 | + where f.merged_into is null and ${metroId ? sql`f.metro_id = ${metroId}` : row.country_iso2 ? sql`f.country_iso2 = ${String(row.country_iso2)}` : sql`false`} order by f.name limit 50`, | |
| 40 | 46 | metroId ? sql<Row[]>`select id, slug, name from metros where id = ${metroId}` : Promise.resolve([] as Row[]), |
| 41 | 47 | row.country_iso2 ? sql<Row[]>`select name from countries where iso2 = ${String(row.country_iso2)}` : Promise.resolve([] as Row[]), |
| 48 | + eventsForEntity("cloud_region", id, 50), | |
| 49 | + provenanceFor("cloud_region", id), | |
| 42 | 50 | ]); |
| 43 | 51 | const base = cloudRegionSummary(main[0] ?? row); |
| 52 | + const hostFacilities = hosts.map(facilitySummary); | |
| 53 | + const marketFacilities = facs.map(facilitySummary); | |
| 54 | + const announcedProv = provenance.find((p) => p.field === "announcedOn" && typeof p.value === "string"); | |
| 55 | + const announcedOn = announcedProv ? String(announcedProv.value) : base.status === "announced" && base.launchedOn ? base.launchedOn : null; | |
| 44 | 56 | return { |
| 45 | 57 | ...base, |
| 46 | − regionName: str(row.region_name), | |
| 47 | 58 | metro: metroRows[0] ? { id: String(metroRows[0].id), slug: String(metroRows[0].slug), name: String(metroRows[0].name) } : null, |
| 48 | − countryName: countryRows[0] ? String(countryRows[0].name) : null, | |
| 59 | + hostFacilities, | |
| 60 | + marketFacilities, | |
| 49 | 61 | siblingRegions: siblings.map(cloudRegionSummary), |
| 50 | − facilitiesInMetro: facs.map(facilitySummary), | |
| 51 | − externalIds: (row.external_ids as Record<string, string | number> | null) ?? {}, | |
| 52 | − updatedAt: String(row.updated_at ?? ""), | |
| 62 | + announcedOn, | |
| 63 | + events, | |
| 64 | + provenance, | |
| 65 | + externalIds: record(row.external_ids), | |
| 66 | + note: hostFacilities.length ? `Hosting publicly verified for ${hostFacilities.length} facilit${hostFacilities.length > 1 ? "ies" : "y"} (cloud tenancy recorded at the facility); marketFacilities lists the other facilities of the market.` : MARKET_ASSOCIATION_NOTE, | |
| 67 | + regionName: str(row.region_name), | |
| 68 | + countryName: countryRows[0] ? String(countryRows[0].name) : null, | |
| 69 | + facilitiesInMetro: marketFacilities, | |
| 70 | + updatedAt: reqIso(row.updated_at), | |
| 53 | 71 | }; |
| 54 | 72 | } |
added
apps/api/src/repositories/compare.ts
+34 −0
@@ -0,0 +1,34 @@ | ||
| 1 | +/** Side-by-side operator comparison (2–5 operators): summary, pipeline, velocity, new markets, recent projects. */ | |
| 2 | +import type { OperatorComparison } from "@dci/core"; | |
| 3 | +import { pg, projectLive } from "../lib/sql.js"; | |
| 4 | +import { reqStr, type Row } from "../lib/rows.js"; | |
| 5 | +import { notFound } from "../lib/http.js"; | |
| 6 | +import { expansionVelocity, pipelineBreakdown, scopeFor } from "../lib/pipeline.js"; | |
| 7 | +import { newMarkets, operatorSummaryById } from "./operators.js"; | |
| 8 | +import { projectsWhere } from "./projects.js"; | |
| 9 | + | |
| 10 | +export const COMPARE_METHODOLOGY = "Each column is computed the same way as the operator page: containment-aware facility counts, known MW = published operational figures only (coverage given per operator), pipeline from facility statuses and live project records (hidden / merged projects excluded), velocity windows from opened_on (else first_seen) and announced_on (else first indexed)."; | |
| 11 | + | |
| 12 | +export async function compareOperators(slugs: string[]): Promise<OperatorComparison> { | |
| 13 | + const sql = pg(); | |
| 14 | + const ids: string[] = []; | |
| 15 | + for (const slug of slugs) { | |
| 16 | + const r = (await sql<Row[]>`select id from operators where slug = ${slug} or id = ${slug} limit 1`)[0]; | |
| 17 | + if (!r) throw notFound(`operator ${slug}`); | |
| 18 | + ids.push(reqStr(r.id)); | |
| 19 | + } | |
| 20 | + const operators = await Promise.all(ids.map(async (id) => { | |
| 21 | + const scope = scopeFor(sql, { operatorId: id }); | |
| 22 | + const [summary, pipeline, velocity, markets, recentProjects, countries] = await Promise.all([ | |
| 23 | + operatorSummaryById(id), | |
| 24 | + pipelineBreakdown(scope), | |
| 25 | + expansionVelocity(scope), | |
| 26 | + newMarkets(id, 12), | |
| 27 | + projectsWhere(sql`${projectLive(sql)} and p.operator_id = ${id}`, 5), | |
| 28 | + sql<Row[]>`select distinct country_iso2 from facilities where operator_id = ${id} and merged_into is null and country_iso2 is not null order by 1`, | |
| 29 | + ]); | |
| 30 | + if (!summary) throw notFound(`operator ${id}`); | |
| 31 | + return { ...summary, pipeline, velocity, aiCount: summary.aiCount ?? 0, newMarkets12m: markets, recentProjects, countriesList: countries.map((c) => reqStr(c.country_iso2)), coverage: summary.mwCoverage ?? 0 }; | |
| 32 | + })); | |
| 33 | + return { operators, generatedAt: new Date().toISOString() }; | |
| 34 | +} | |
added
apps/api/src/repositories/connectivity.ts
+39 −0
@@ -0,0 +1,39 @@ | ||
| 1 | +/** | |
| 2 | + * /connectivity — IXPs, cloud regions, carrier hotels and per-metro interconnection density. Cable landing stations | |
| 3 | + * have no connector yet and are returned empty with a note. | |
| 4 | + */ | |
| 5 | +import type { ConnectivityOverview } from "@dci/core"; | |
| 6 | +import { pg, cloudRegionCols, facilityJoins, facilitySummaryCols } from "../lib/sql.js"; | |
| 7 | +import { int, reqStr, type Row } from "../lib/rows.js"; | |
| 8 | +import { cloudRegionSummary, facilitySummary, ixpSummary } from "../lib/dto.js"; | |
| 9 | + | |
| 10 | +export const CONNECTIVITY_NOTE = "IXP facility / operator counts come from published IXP↔facility links (facility_ixps); IXPs have no coordinates of their own and are placed at their metro reference point on maps. Carrier hotels = facility_type carrier_hotel or ≥ 20 carriers published on the facility. Submarine cable landing stations are empty until a licensed connector exists — never inferred. Cloud regions are associated with a market, not hosted at a specific facility unless a source says so."; | |
| 11 | + | |
| 12 | +export async function connectivityOverview(): Promise<ConnectivityOverview> { | |
| 13 | + const sql = pg(); | |
| 14 | + const [ixps, cloudRegions, carrierHotels, byMetro] = await Promise.all([ | |
| 15 | + sql<Row[]>`select x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, | |
| 16 | + (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count, | |
| 17 | + (select count(distinct f.operator_id)::int from facility_ixps fx join facilities f on f.id = fx.facility_id where fx.ixp_id = x.id and f.operator_id is not null) as operator_count, | |
| 18 | + m.id as met_id, m.slug as met_slug, m.name as met_name, m.lat, m.lng | |
| 19 | + from ixps x left join metros m on m.id = x.metro_id | |
| 20 | + order by x.network_count desc nulls last, facility_count desc, x.name limit 100`, | |
| 21 | + sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.status <> 'retired' order by pr.name, r.country_iso2, r.code limit 2000`, | |
| 22 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and (f.facility_type = 'carrier_hotel' or coalesce(f.carriers_count, 0) >= 20) order by f.carriers_count desc nulls last, f.completeness desc, f.name limit 50`, | |
| 23 | + sql<Row[]>`select * from (select m.id, m.slug, m.name, m.country_iso2, | |
| 24 | + (select count(*)::int from ixps x where x.metro_id = m.id) as ixps, | |
| 25 | + (select count(*)::int from cloud_regions r where r.metro_id = m.id and r.status <> 'retired') as cloud_regions, | |
| 26 | + (select count(*)::int from facilities f where f.metro_id = m.id and f.merged_into is null and (coalesce(f.carriers_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id and t.role in ('carrier', 'network')))) as facilities_with_carriers, | |
| 27 | + (select count(*)::int from facilities f where f.metro_id = m.id and f.merged_into is null) as facilities | |
| 28 | + from metros m | |
| 29 | + where exists (select 1 from ixps x where x.metro_id = m.id) or exists (select 1 from cloud_regions r where r.metro_id = m.id) or exists (select 1 from facilities f where f.metro_id = m.id and f.merged_into is null and (coalesce(f.carriers_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id)))) t order by t.ixps + t.cloud_regions + t.facilities_with_carriers desc, t.facilities desc limit 50`, | |
| 30 | + ]); | |
| 31 | + return { | |
| 32 | + ixps: ixps.map((r) => ({ ...ixpSummary(r), operatorCount: int(r.operator_count) })), | |
| 33 | + cloudRegions: cloudRegions.map(cloudRegionSummary), | |
| 34 | + carrierHotels: carrierHotels.map(facilitySummary), | |
| 35 | + landingStations: [], | |
| 36 | + byMetro: byMetro.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), countryIso2: reqStr(r.country_iso2), ixps: int(r.ixps), cloudRegions: int(r.cloud_regions), facilitiesWithCarriers: int(r.facilities_with_carriers), facilities: int(r.facilities) })), | |
| 37 | + note: CONNECTIVITY_NOTE, | |
| 38 | + }; | |
| 39 | +} | |
modified
apps/api/src/repositories/countries.ts
+65 −40
@@ -1,15 +1,18 @@ | ||
| 1 | −import type { CountryDetail, CountrySummary } from "@dci/core"; | |
| 2 | −import { pg, mwExpr, plannedMwExpr, OPERATIONAL_SET, CONSTRUCTION_SET, PLANNED_SET, yearExpr, type Fragment, type Sql } from "../lib/sql.js"; | |
| 1 | +import type { CountryDetail, CountrySummary, EnergyContext } from "@dci/core"; | |
| 2 | +import { pg, mwExpr, yearExpr, facilityView, knownMwAgg, countedAgg, projectLive, facilityJoins, facilitySummaryCols, projectJoins, projectSummaryCols, cloudRegionCols, AI_LEVELS, type Sql } from "../lib/sql.js"; | |
| 3 | 3 | import { int, num, reqStr, str, type Row } from "../lib/rows.js"; |
| 4 | −import { growthSeries, statusBreakdown, typeBreakdown, cloudRegionSummary } from "../lib/dto.js"; | |
| 4 | +import { growthSeries, statusBreakdown, typeBreakdown, cloudRegionSummary, facilitySummary, projectSummary, ixpSummary, round2, share } from "../lib/dto.js"; | |
| 5 | 5 | import { findCountry } from "../lib/resolve.js"; |
| 6 | 6 | import { listFacilities } from "./facilities.js"; |
| 7 | 7 | import { operatorsForScope } from "./operators.js"; |
| 8 | −import { listMetros } from "./metros.js"; | |
| 8 | +import { gridConstraintsFor, listMetros } from "./metros.js"; | |
| 9 | 9 | import { projectsWhere } from "./projects.js"; |
| 10 | 10 | import { eventsForCountry } from "./events.js"; |
| 11 | 11 | import { rankingPositions } from "./rankings.js"; |
| 12 | −import { cloudRegionCols } from "../lib/sql.js"; | |
| 12 | +import { coverageRow, dimAggCols, pipelineBreakdown, scopeFor } from "../lib/pipeline.js"; | |
| 13 | +import { claimsFor } from "../lib/quality.js"; | |
| 14 | + | |
| 15 | +export const ENERGY_NOTE = "National grid averages (renewable share, generation) describe the country's electricity mix, not the electricity a facility contracts; utility/grid MW on facilities are supply figures, not IT load."; | |
| 13 | 16 | |
| 14 | 17 | function countrySummary(r: Row): CountrySummary { |
| 15 | 18 | const n = int(r.facility_count); |
@@ -27,79 +30,101 @@ function countrySummary(r: Row): CountrySummary { | ||
| 27 | 30 | operationalCount: int(r.operational), |
| 28 | 31 | constructionCount: int(r.construction), |
| 29 | 32 | plannedCount: int(r.planned), |
| 30 | − knownMw: num(r.known_mw), | |
| 31 | − constructionMw: num(r.construction_mw), | |
| 32 | − plannedMw: num(r.planned_mw), | |
| 33 | + knownMw: round2(num(r.known_mw)), | |
| 34 | + constructionMw: round2(num(r.construction_mw)), | |
| 35 | + plannedMw: round2(num(r.planned_mw)), | |
| 33 | 36 | operatorCount: int(r.operator_count), |
| 34 | 37 | cloudRegionCount: int(r.cloud_region_count), |
| 35 | 38 | hyperscaleCount: int(r.hyperscale), |
| 36 | 39 | aiCount: int(r.ai), |
| 37 | 40 | projectCount: int(r.project_count), |
| 38 | − mwCoverage: n ? Math.round((withMw / n) * 1000) / 1000 : 0, | |
| 41 | + projectPlannedMw: round2(num(r.project_planned_mw)), | |
| 42 | + projectConstructionMw: round2(num(r.project_construction_mw)), | |
| 43 | + ixpCount: int(r.ixp_count), | |
| 44 | + mwCoverage: n ? share(withMw, n) : 0, | |
| 39 | 45 | lat: num(r.lat), |
| 40 | 46 | lng: num(r.lng), |
| 41 | 47 | }; |
| 42 | 48 | } |
| 43 | 49 | |
| 44 | −function facilityAgg(sql: Sql): Fragment { | |
| 45 | − const mw = mwExpr(sql); | |
| 46 | − return sql`( | |
| 47 | − select f.country_iso2, count(*)::int as facility_count, | |
| 48 | − count(*) filter (where f.status = any(${OPERATIONAL_SET}))::int as operational, | |
| 49 | − count(*) filter (where f.status = any(${CONSTRUCTION_SET}))::int as construction, | |
| 50 | − count(*) filter (where f.status = any(${PLANNED_SET}))::int as planned, | |
| 51 | − sum(${mw}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw, | |
| 52 | − sum(${mw}) filter (where f.status = any(${CONSTRUCTION_SET}))::float as construction_mw, | |
| 53 | − sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PLANNED_SET}))::float as planned_mw, | |
| 54 | − count(distinct f.operator_id)::int as operator_count, | |
| 55 | − count(*) filter (where f.is_hyperscale or f.facility_type = 'hyperscale')::int as hyperscale, | |
| 56 | − count(*) filter (where f.is_ai or f.facility_type = 'ai')::int as ai, | |
| 57 | − count(*) filter (where ${mw} is not null)::int as with_mw | |
| 58 | − from facilities f where f.merged_into is null and f.country_iso2 is not null group by f.country_iso2 | |
| 59 | − )`; | |
| 60 | −} | |
| 61 | − | |
| 62 | 50 | const SELECT = (sql: Sql) => sql` |
| 63 | − select c.iso2, c.iso3, c.slug, c.name, c.region, c.subregion, c.population, c.gdp_usd, c.lat, c.lng, c.electricity_twh, c.renewable_share, | |
| 64 | − coalesce(fa.facility_count, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned, | |
| 65 | − fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operator_count, 0) as operator_count, coalesce(fa.hyperscale, 0) as hyperscale, coalesce(fa.ai, 0) as ai, coalesce(fa.with_mw, 0) as with_mw, | |
| 66 | − coalesce(cr.n, 0) as cloud_region_count, coalesce(pj.n, 0) as project_count | |
| 51 | + select c.iso2, c.iso3, c.slug, c.name, c.region, c.subregion, c.population, c.gdp_usd, c.lat, c.lng, c.electricity_twh, c.renewable_share, c.stats_year, c.stats, | |
| 52 | + coalesce(fa.facilities, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned, | |
| 53 | + fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operators, 0) as operator_count, coalesce(fa.hyperscale, 0) as hyperscale, coalesce(fa.ai, 0) as ai, coalesce(fa.with_mw, 0) as with_mw, | |
| 54 | + coalesce(cr.n, 0) as cloud_region_count, coalesce(pj.n, 0) as project_count, pj.planned_mw as project_planned_mw, pj.construction_mw as project_construction_mw, coalesce(ix.n, 0) as ixp_count | |
| 67 | 55 | from countries c |
| 68 | − left join ${facilityAgg(sql)} fa on fa.country_iso2 = c.iso2 | |
| 69 | − left join (select country_iso2, count(*)::int as n from cloud_regions group by 1) cr on cr.country_iso2 = c.iso2 | |
| 70 | − left join (select country_iso2, count(*)::int as n from projects group by 1) pj on pj.country_iso2 = c.iso2`; | |
| 56 | + left join (select f.country_iso2, ${dimAggCols(sql)} from ${facilityView(sql)} f where f.country_iso2 is not null group by f.country_iso2) fa on fa.country_iso2 = c.iso2 | |
| 57 | + left join (select country_iso2, count(*)::int as n from cloud_regions where status <> 'retired' group by 1) cr on cr.country_iso2 = c.iso2 | |
| 58 | + left join (select country_iso2, count(*)::int as n from ixps group by 1) ix on ix.country_iso2 = c.iso2 | |
| 59 | + left join (select p.country_iso2, count(*)::int as n, | |
| 60 | + sum(p.planned_mw) filter (where p.status in ('rumored','proposed','announced','permitting','approved','delayed'))::float as planned_mw, | |
| 61 | + sum(p.planned_mw) filter (where p.status = 'under_construction')::float as construction_mw | |
| 62 | + from projects p where ${projectLive(sql)} group by 1) pj on pj.country_iso2 = c.iso2`; | |
| 71 | 63 | |
| 72 | 64 | export async function listCountries(opts: { all?: boolean; region?: string } = {}): Promise<CountrySummary[]> { |
| 73 | 65 | const sql = pg(); |
| 74 | 66 | const rows = await sql<Row[]>`${SELECT(sql)} |
| 75 | − where ${opts.all ? sql`true` : sql`(coalesce(fa.facility_count, 0) > 0 or coalesce(cr.n, 0) > 0)`} | |
| 67 | + where ${opts.all ? sql`true` : sql`(coalesce(fa.facilities, 0) > 0 or coalesce(cr.n, 0) > 0)`} | |
| 76 | 68 | and ${opts.region ? sql`(c.region ilike ${opts.region} or c.subregion ilike ${opts.region})` : sql`true`} |
| 77 | − order by coalesce(fa.facility_count, 0) desc, coalesce(cr.n, 0) desc, c.name`; | |
| 69 | + order by coalesce(fa.facilities, 0) desc, coalesce(cr.n, 0) desc, c.name`; | |
| 78 | 70 | return rows.map(countrySummary); |
| 79 | 71 | } |
| 80 | 72 | |
| 73 | +async function energyContext(base: Row): Promise<EnergyContext> { | |
| 74 | + const sql = pg(); | |
| 75 | + const stats = (base.stats as Record<string, unknown> | null) ?? {}; | |
| 76 | + let sourceName: string | null = null; | |
| 77 | + let sourceUrl: string | null = null; | |
| 78 | + const es = stats.energySource; | |
| 79 | + if (es && typeof es === "object") { sourceName = str((es as Record<string, unknown>).name); sourceUrl = str((es as Record<string, unknown>).url); } | |
| 80 | + else if (typeof es === "string") sourceName = es; | |
| 81 | + if (!sourceName) { | |
| 82 | + const rows = await sql<Row[]>`select s.name, p.url from provenance p left join sources s on s.id = p.source_id where p.entity_type = 'country' and p.entity_id = ${String(base.iso2)} and p.is_current and p.field in ('renewableShare', 'electricityTwh', 'renewable_share', 'electricity_twh') order by p.last_observed desc limit 1`; | |
| 83 | + if (rows[0]) { sourceName = str(rows[0].name); sourceUrl = str(rows[0].url); } | |
| 84 | + } | |
| 85 | + return { renewableShare: num(base.renewable_share), electricityTwh: num(base.electricity_twh), statsYear: num(base.stats_year), gridCarbonIntensity: null, sourceName, sourceUrl, note: ENERGY_NOTE }; | |
| 86 | +} | |
| 87 | + | |
| 81 | 88 | export async function getCountryDetail(slugOrIso2: string, fPage = 1): Promise<CountryDetail | null> { |
| 82 | 89 | const sql = pg(); |
| 83 | 90 | const base = await findCountry(slugOrIso2); |
| 84 | 91 | if (!base) return null; |
| 85 | 92 | const iso2 = String(base.iso2); |
| 86 | − const [sumRows, topOperators, metros, cloudRegionRows, recentProjects, recentEvents, growthRows, statusRows, typeRows, invRows, rankings, facilities] = await Promise.all([ | |
| 93 | + const scope = scopeFor(sql, { countryIso2: iso2 }); | |
| 94 | + const [sumRows, topOperators, metros, cloudRegionRows, recentProjects, recentEvents, growthRows, statusRows, typeRows, invRows, rankings, facilities, ixpRows, gridConstraints, energy, aiFacRows, aiPrjRows, pipeline, coverage, claims] = await Promise.all([ | |
| 87 | 95 | sql<Row[]>`${SELECT(sql)} where c.iso2 = ${iso2}`, |
| 88 | 96 | operatorsForScope({ countryIso2: iso2 }, 10), |
| 89 | 97 | listMetros({ country: iso2 }), |
| 90 | 98 | sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.country_iso2 = ${iso2} order by pr.name, r.code`, |
| 91 | − projectsWhere(sql`p.country_iso2 = ${iso2}`, 10), | |
| 99 | + projectsWhere(sql`${projectLive(sql)} and p.country_iso2 = ${iso2}`, 10), | |
| 92 | 100 | eventsForCountry(iso2, 20), |
| 93 | − sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*)::int as n, sum(${mwExpr(sql)})::float as mw from facilities f where f.country_iso2 = ${iso2} and f.merged_into is null and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 101 | + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${knownMwAgg(sql)})::float as mw from ${facilityView(sql)} f where f.country_iso2 = ${iso2} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 94 | 102 | sql<Row[]>`select status, count(*)::int as n from facilities where country_iso2 = ${iso2} and merged_into is null group by status`, |
| 95 | 103 | sql<Row[]>`select facility_type, count(*)::int as n from facilities where country_iso2 = ${iso2} and merged_into is null group by facility_type`, |
| 96 | − sql<Row[]>`select sum(investment_usd)::float as inv from projects where country_iso2 = ${iso2} and status not in ('cancelled')`, | |
| 104 | + sql<Row[]>`select sum(p.investment_usd)::float as inv from projects p where ${projectLive(sql)} and p.country_iso2 = ${iso2} and p.status not in ('cancelled') and (p.investment_scope is null or p.investment_scope in ('facility', 'campus', 'building'))`, | |
| 97 | 105 | rankingPositions("countries", [iso2, String(base.slug)]), |
| 98 | 106 | listFacilities({ countryIso2: iso2, page: fPage, per_page: 50, sort: "mw", order: "desc" }), |
| 107 | + sql<Row[]>`select x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count, m.id as met_id, m.slug as met_slug, m.name as met_name | |
| 108 | + from ixps x left join metros m on m.id = x.metro_id where x.country_iso2 = ${iso2} order by x.network_count desc nulls last, x.name limit 500`, | |
| 109 | + gridConstraintsFor({ countryIso2: iso2 }), | |
| 110 | + energyContext(base), | |
| 111 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.country_iso2 = ${iso2} and f.ai_evidence = any(${AI_LEVELS}) order by ${mwExpr(sql)} desc nulls last, f.name limit 20`, | |
| 112 | + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and p.country_iso2 = ${iso2} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai) order by p.planned_mw desc nulls last, p.last_update desc limit 20`, | |
| 113 | + pipelineBreakdown(scope), | |
| 114 | + coverageRow(iso2, String(base.name), String(base.slug), scope), | |
| 115 | + claimsFor("country", iso2), | |
| 99 | 116 | ]); |
| 100 | 117 | const summary = countrySummary(sumRows[0] ?? { ...base, facility_count: 0 }); |
| 101 | 118 | return { |
| 102 | 119 | ...summary, |
| 120 | + ixps: ixpRows.map(ixpSummary), | |
| 121 | + gridConstraints, | |
| 122 | + energy, | |
| 123 | + aiFacilities: aiFacRows.map(facilitySummary), | |
| 124 | + aiProjects: aiPrjRows.map(projectSummary), | |
| 125 | + pipeline, | |
| 126 | + coverage, | |
| 127 | + claims, | |
| 103 | 128 | topOperators, |
| 104 | 129 | metros, |
| 105 | 130 | cloudRegions: cloudRegionRows.map(cloudRegionSummary), |
added
apps/api/src/repositories/coverage.ts
+124 −0
@@ -0,0 +1,124 @@ | ||
| 1 | +/** | |
| 2 | + * Coverage report: how much of the index is actually known, per country / metro / operator / field / source kind. | |
| 3 | + * Containment-aware (campus rows with buildings are not counted). Shares are 0..1. | |
| 4 | + */ | |
| 5 | +import type { CoverageReport, CoverageRow, SourceCoverage } from "@dci/core"; | |
| 6 | +import { authorityTier } from "@dci/core"; | |
| 7 | +import { pg, facilityView, countedAgg, hasMwAgg, projectLive, type Fragment, type Sql } from "../lib/sql.js"; | |
| 8 | +import { int, iso, reqStr, str, type Row } from "../lib/rows.js"; | |
| 9 | +import { asSourceKind, share } from "../lib/dto.js"; | |
| 10 | +import { coverageRow, scopeFor } from "../lib/pipeline.js"; | |
| 11 | + | |
| 12 | +const PRIMARY_KINDS = ["operator", "government", "filing", "utility", "cloud_provider", "registry"]; | |
| 13 | + | |
| 14 | +/** Grouped version of lib/pipeline coverageRow: one row per dimension value. */ | |
| 15 | +function coverageCols(sql: Sql): Fragment { | |
| 16 | + const counted = countedAgg(sql), hasMw = hasMwAgg(sql); | |
| 17 | + return sql` | |
| 18 | + count(*) filter (where ${counted})::int as n, | |
| 19 | + count(*) filter (where ${counted} and ${hasMw})::int as with_mw, | |
| 20 | + count(*) filter (where ${counted} and f.operator_id is not null)::int as with_op, | |
| 21 | + count(*) filter (where ${counted} and f.lat is not null and f.geo_precision in ('exact', 'parcel', 'street'))::int as precise, | |
| 22 | + count(*) filter (where ${counted} and f.lat is not null)::int as any_loc, | |
| 23 | + count(*) filter (where ${counted} and f.status <> 'unknown')::int as with_status, | |
| 24 | + count(*) filter (where ${counted} and f.opened_on ~ '^\\d{4}')::int as with_open, | |
| 25 | + count(*) filter (where ${counted} and f.source_count >= 2)::int as multi, | |
| 26 | + count(*) filter (where ${counted} and (coalesce(f.carriers_count, 0) > 0 or coalesce(f.ixp_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id) or exists (select 1 from facility_ixps x where x.facility_id = f.id)))::int as conn, | |
| 27 | + count(*) filter (where ${counted} and exists (select 1 from provenance p join sources s on s.id = p.source_id where p.entity_type = 'facility' and p.entity_id = f.id and p.is_current and s.kind = any(${PRIMARY_KINDS})))::int as primary_src`; | |
| 28 | +} | |
| 29 | + | |
| 30 | +function toRow(r: Row, projects: { n: number; loc: number }): CoverageRow { | |
| 31 | + const n = int(r.n); | |
| 32 | + return { | |
| 33 | + key: reqStr(r.key), name: reqStr(r.name), slug: reqStr(r.slug), | |
| 34 | + facilities: n, | |
| 35 | + capacityCoverage: share(int(r.with_mw), n), | |
| 36 | + operatorCoverage: share(int(r.with_op), n), | |
| 37 | + preciseLocationCoverage: share(int(r.precise), n), | |
| 38 | + anyLocationCoverage: share(int(r.any_loc), n), | |
| 39 | + statusCoverage: share(int(r.with_status), n), | |
| 40 | + openingDateCoverage: share(int(r.with_open), n), | |
| 41 | + multiSourceCoverage: share(int(r.multi), n), | |
| 42 | + projects: projects.n, | |
| 43 | + projectsWithLocation: projects.loc, | |
| 44 | + connectivityCoverage: share(int(r.conn), n), | |
| 45 | + primarySourceShare: share(int(r.primary_src), n), | |
| 46 | + }; | |
| 47 | +} | |
| 48 | + | |
| 49 | +async function projectCounts(sql: Sql, col: "country_iso2" | "metro_id" | "operator_id"): Promise<Map<string, { n: number; loc: number }>> { | |
| 50 | + const rows = await sql<Row[]>`select p.${sql(col)} as key, count(*)::int as n, count(*) filter (where p.lat is not null)::int as loc from projects p where ${projectLive(sql)} and p.${sql(col)} is not null group by 1`; | |
| 51 | + return new Map(rows.map((r) => [reqStr(r.key), { n: int(r.n), loc: int(r.loc) }])); | |
| 52 | +} | |
| 53 | + | |
| 54 | +export const COVERAGE_METHODOLOGY = "Shares of counted facilities (campus rows with building rows excluded; merged duplicates excluded) with each field known. capacityCoverage counts a building covered by its campus figure as covered. Precise location = exact / parcel / street. Primary sources = operator, government, filing, utility, cloud provider, registry. Nothing is extrapolated: a low share means the index does not know, not that the value is zero."; | |
| 55 | + | |
| 56 | +export async function coverageReport(): Promise<CoverageReport> { | |
| 57 | + const sql = pg(); | |
| 58 | + const [global, countries, metros, operators, pc, pm, po, fields, bySourceKind] = await Promise.all([ | |
| 59 | + coverageRow("global", "Global", "global", scopeFor(sql, {})), | |
| 60 | + sql<Row[]>`select c.iso2 as key, c.name, c.slug, ${coverageCols(sql)} from ${facilityView(sql)} f join countries c on c.iso2 = f.country_iso2 group by c.iso2, c.name, c.slug having count(*) filter (where ${countedAgg(sql)}) >= 1 order by n desc, c.name`, | |
| 61 | + sql<Row[]>`select m.id as key, m.name, m.slug, ${coverageCols(sql)} from ${facilityView(sql)} f join metros m on m.id = f.metro_id group by m.id, m.name, m.slug having count(*) filter (where ${countedAgg(sql)}) >= 1 order by n desc, m.name`, | |
| 62 | + sql<Row[]>`select o.id as key, o.name, o.slug, ${coverageCols(sql)} from ${facilityView(sql)} f join operators o on o.id = f.operator_id group by o.id, o.name, o.slug having count(*) filter (where ${countedAgg(sql)}) >= 5 order by n desc, o.name`, | |
| 63 | + projectCounts(sql, "country_iso2"), | |
| 64 | + projectCounts(sql, "metro_id"), | |
| 65 | + projectCounts(sql, "operator_id"), | |
| 66 | + sql<Row[]>`select count(*)::int as n, | |
| 67 | + count(*) filter (where f.operator_id is not null)::int as operator, | |
| 68 | + count(*) filter (where f.lat is not null)::int as coordinates, | |
| 69 | + count(*) filter (where f.lat is not null and f.geo_precision in ('exact', 'parcel', 'street'))::int as precise_coordinates, | |
| 70 | + count(*) filter (where f.status <> 'unknown')::int as status, | |
| 71 | + count(*) filter (where f.facility_type <> 'unknown')::int as facility_type, | |
| 72 | + count(*) filter (where f.it_capacity_mw is not null)::int as it_capacity_mw, | |
| 73 | + count(*) filter (where f.total_power_mw is not null)::int as total_power_mw, | |
| 74 | + count(*) filter (where f.planned_power_mw is not null)::int as planned_power_mw, | |
| 75 | + count(*) filter (where f.opened_on ~ '^\\d{4}')::int as opened_on, | |
| 76 | + count(*) filter (where f.address is not null)::int as address, | |
| 77 | + count(*) filter (where coalesce(f.carriers_count, 0) > 0 or coalesce(f.ixp_count, 0) > 0 or exists (select 1 from facility_tenants t where t.facility_id = f.id) or exists (select 1 from facility_ixps x where x.facility_id = f.id))::int as connectivity, | |
| 78 | + count(*) filter (where f.website is not null)::int as website, | |
| 79 | + count(*) filter (where f.description is not null)::int as description, | |
| 80 | + count(*) filter (where f.tier is not null)::int as tier | |
| 81 | + from ${facilityView(sql)} f where ${countedAgg(sql)}`, | |
| 82 | + sql<Row[]>`select s.kind, count(distinct p.entity_id) filter (where p.entity_type = 'facility')::int as facilities, count(*)::int as fields from provenance p join sources s on s.id = p.source_id where p.is_current group by s.kind order by facilities desc`, | |
| 83 | + ]); | |
| 84 | + const fr = fields[0] ?? {}; | |
| 85 | + const total = int(fr.n); | |
| 86 | + const FIELDS: Array<[string, string]> = [["operator", "Operator"], ["coordinates", "Coordinates (any precision)"], ["precise_coordinates", "Precise coordinates (exact / parcel / street)"], ["status", "Lifecycle status"], ["facility_type", "Facility type"], ["it_capacity_mw", "IT capacity (MW)"], ["total_power_mw", "Total power (MW)"], ["planned_power_mw", "Planned power (MW)"], ["opened_on", "Opening date"], ["address", "Street address"], ["connectivity", "Carriers / IXPs"], ["website", "Website"], ["description", "Description"], ["tier", "Tier / certification"]]; | |
| 87 | + return { | |
| 88 | + global, | |
| 89 | + countries: countries.map((r) => toRow(r, pc.get(reqStr(r.key)) ?? { n: 0, loc: 0 })), | |
| 90 | + metros: metros.map((r) => toRow(r, pm.get(reqStr(r.key)) ?? { n: 0, loc: 0 })), | |
| 91 | + operators: operators.map((r) => toRow(r, po.get(reqStr(r.key)) ?? { n: 0, loc: 0 })), | |
| 92 | + fields: FIELDS.map(([field, label]) => ({ field, label, coverage: share(int(fr[field]), total), count: int(fr[field]) })), | |
| 93 | + bySourceKind: bySourceKind.map((r) => ({ kind: asSourceKind(r.kind), facilities: int(r.facilities), fields: int(r.fields) })), | |
| 94 | + generatedAt: new Date().toISOString(), | |
| 95 | + }; | |
| 96 | +} | |
| 97 | + | |
| 98 | +export async function sourceCoverage(): Promise<SourceCoverage[]> { | |
| 99 | + const sql = pg(); | |
| 100 | + const rows = await sql<Row[]>` | |
| 101 | + with cur as (select source_id, entity_type, entity_id from provenance where is_current), | |
| 102 | + per_entity as (select entity_type, entity_id, count(distinct source_id) as sources from cur group by 1, 2) | |
| 103 | + select s.id, s.name, s.kind, s.license, s.redistribution, s.connector_id, | |
| 104 | + (select count(distinct (c.entity_type, c.entity_id)) from cur c where c.source_id = s.id)::int as records, | |
| 105 | + (select count(*) from cur c where c.source_id = s.id)::int as fields, | |
| 106 | + (select count(distinct (c.entity_type, c.entity_id)) from cur c join per_entity pe on pe.entity_type = c.entity_type and pe.entity_id = c.entity_id where c.source_id = s.id and pe.sources = 1)::int as unique_records, | |
| 107 | + (select max(r.finished_at) from connector_runs r where r.connector_id = s.connector_id and r.status in ('ok', 'partial')) as last_ok, | |
| 108 | + (select count(*) from connector_runs r where r.connector_id = s.connector_id and r.started_at >= now() - interval '30 days' and r.status <> 'running')::int as runs30, | |
| 109 | + (select count(*) from connector_runs r where r.connector_id = s.connector_id and r.started_at >= now() - interval '30 days' and r.status in ('failed', 'aborted'))::int as failed30 | |
| 110 | + from sources s order by records desc, s.name`; | |
| 111 | + return rows.map((r) => { | |
| 112 | + const lastOk = iso(r.last_ok); | |
| 113 | + const runs = int(r.runs30); | |
| 114 | + return { | |
| 115 | + id: reqStr(r.id), name: reqStr(r.name), kind: asSourceKind(r.kind), | |
| 116 | + recordsContributed: int(r.records), fieldsContributed: int(r.fields), uniqueRecords: int(r.unique_records), | |
| 117 | + lastSuccessfulCrawl: lastOk, | |
| 118 | + freshnessDays: lastOk ? Math.max(0, Math.round((Date.now() - new Date(lastOk).getTime()) / 86_400_000)) : null, | |
| 119 | + authority: authorityTier({ field: "identity", sourceKind: str(r.kind) }), | |
| 120 | + failureRate: runs ? Math.round((int(r.failed30) / runs) * 1000) / 1000 : null, | |
| 121 | + license: str(r.license), redistribution: str(r.redistribution), | |
| 122 | + }; | |
| 123 | + }); | |
| 124 | +} | |
modified
apps/api/src/repositories/dashboard.ts
+98 −55
@@ -1,74 +1,108 @@ | ||
| 1 | −import type { Dashboard, DashboardStats, ProjectSummary, RankingRow } from "@dci/core"; | |
| 2 | −import { pg, mwExpr, plannedMwExpr, OPERATIONAL_SET, CONSTRUCTION_SET, PLANNED_SET, yearExpr, projectJoins, projectSummaryCols } from "../lib/sql.js"; | |
| 1 | +import type { Dashboard, DashboardStats, GridConstraintDTO, MarketMomentum, ProjectSummary, RankingRow } from "@dci/core"; | |
| 2 | +import { pg, yearExpr, projectJoins, projectSummaryCols, projectLive, facilityView, knownMwAgg, countedAgg, eventCols, eventJoins, gridConstraintCols, gridConstraintJoins, AI_LEVELS, POWER_EVENT_TYPES, GRID_EVENT_TYPES } from "../lib/sql.js"; | |
| 3 | 3 | import { int, iso, num, reqStr, str, type Row } from "../lib/rows.js"; |
| 4 | −import { asStatus, asType, projectSummary } from "../lib/dto.js"; | |
| 4 | +import { asStatus, asType, gridConstraintDto, gridConstraintFromEvent, projectSummary, round2, share } from "../lib/dto.js"; | |
| 5 | 5 | import { hrefFor } from "../lib/resolve.js"; |
| 6 | −import { latestEvents } from "./events.js"; | |
| 6 | +import { latestEvents, toEventDtos } from "./events.js"; | |
| 7 | 7 | import { recentlyVerifiedFacilities } from "./facilities.js"; |
| 8 | 8 | import { topRows } from "./rankings.js"; |
| 9 | +import { pulse } from "./pulse.js"; | |
| 10 | +import { coverageRow, dimAggCols, momentum, scopeFor } from "../lib/pipeline.js"; | |
| 9 | 11 | |
| 10 | −async function liveTopCountries(limit: number): Promise<RankingRow[]> { | |
| 12 | +async function liveTop(dim: "country" | "metro" | "operator", limit: number): Promise<RankingRow[]> { | |
| 11 | 13 | const sql = pg(); |
| 12 | − const rows = await sql<Row[]>` | |
| 13 | − select c.iso2 as id, c.slug, c.name, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw, | |
| 14 | − (count(*) filter (where ${mwExpr(sql)} is not null))::float / count(*) as coverage | |
| 15 | − from facilities f join countries c on c.iso2 = f.country_iso2 where f.merged_into is null group by 1, 2, 3 order by n desc limit ${limit}`; | |
| 16 | − return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor("country", reqStr(r.slug)), value: int(r.n), secondary: num(r.mw), countryIso2: reqStr(r.id), coverage: num(r.coverage) })); | |
| 14 | + const known = knownMwAgg(sql), counted = countedAgg(sql), hasMw = sql`(coalesce(f.it_capacity_mw, f.total_power_mw, f.planned_power_mw) is not null or f.covered_by_parent)`; | |
| 15 | + const rows = dim === "country" | |
| 16 | + ? await sql<Row[]>`select c.iso2 as id, c.slug, c.name, c.iso2 as country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational','partially_operational','expansion'))::float as mw, count(*) filter (where ${counted} and ${hasMw})::int as with_mw from ${facilityView(sql)} f join countries c on c.iso2 = f.country_iso2 group by 1, 2, 3, 4 order by n desc limit ${limit}` | |
| 17 | + : dim === "metro" | |
| 18 | + ? await sql<Row[]>`select m.id, m.slug, m.name, m.country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational','partially_operational','expansion'))::float as mw, count(*) filter (where ${counted} and ${hasMw})::int as with_mw from ${facilityView(sql)} f join metros m on m.id = f.metro_id group by 1, 2, 3, 4 order by n desc limit ${limit}` | |
| 19 | + : await sql<Row[]>`select o.id, o.slug, o.name, o.hq_country_iso2 as country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational','partially_operational','expansion'))::float as mw, count(*) filter (where ${counted} and ${hasMw})::int as with_mw from ${facilityView(sql)} f join operators o on o.id = f.operator_id group by 1, 2, 3, 4 order by n desc limit ${limit}`; | |
| 20 | + return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor(dim, reqStr(r.slug)), value: int(r.n), secondary: round2(num(r.mw)), countryIso2: str(r.country_iso2), coverage: share(int(r.with_mw), int(r.n)) })); | |
| 17 | 21 | } |
| 18 | 22 | |
| 19 | −async function liveTopMetros(limit: number): Promise<RankingRow[]> { | |
| 23 | +async function fastestGrowingMetros(limit = 8): Promise<Dashboard["fastestGrowingMetros"]> { | |
| 20 | 24 | const sql = pg(); |
| 25 | + const since = sql`(current_date - interval '12 months')`; | |
| 26 | + const pd = (col: string) => sql`(case when p.${sql(col)} ~ '^\\d{4}-\\d{2}-\\d{2}' then p.${sql(col)}::date when p.${sql(col)} ~ '^\\d{4}-\\d{2}$' then (p.${sql(col)} || '-01')::date when p.${sql(col)} ~ '^\\d{4}$' then (p.${sql(col)} || '-01-01')::date else null end)`; | |
| 21 | 27 | const rows = await sql<Row[]>` |
| 22 | − select m.id, m.slug, m.name, m.country_iso2, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw, | |
| 23 | − (count(*) filter (where ${mwExpr(sql)} is not null))::float / count(*) as coverage | |
| 24 | − from facilities f join metros m on m.id = f.metro_id where f.merged_into is null group by 1, 2, 3, 4 order by n desc limit ${limit}`; | |
| 25 | − return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor("metro", reqStr(r.slug)), value: int(r.n), secondary: num(r.mw), countryIso2: str(r.country_iso2), coverage: num(r.coverage) })); | |
| 28 | + with pa as (select p.metro_id, count(*) filter (where coalesce(${pd("announced_on")}, p.created_at::date) >= ${since})::int as announced, count(*) filter (where ${pd("construction_started_on")} >= ${since})::int as construction from projects p where ${projectLive(sql)} and p.metro_id is not null group by 1), | |
| 29 | + fo as (select f.metro_id, count(*)::int as opened from facilities f where f.merged_into is null and f.metro_id is not null and f.status in ('operational','partially_operational','expansion') and (case when f.opened_on ~ '^\\d{4}-\\d{2}-\\d{2}' then f.opened_on::date when f.opened_on ~ '^\\d{4}-\\d{2}$' then (f.opened_on || '-01')::date when f.opened_on ~ '^\\d{4}$' then (f.opened_on || '-01-01')::date else null end) >= ${since} group by 1) | |
| 30 | + select m.id, m.slug, m.name, m.country_iso2, coalesce(pa.announced, 0) + coalesce(pa.construction, 0) + coalesce(fo.opened, 0) as score | |
| 31 | + from metros m left join pa on pa.metro_id = m.id left join fo on fo.metro_id = m.id | |
| 32 | + where coalesce(pa.announced, 0) + coalesce(pa.construction, 0) + coalesce(fo.opened, 0) > 0 | |
| 33 | + order by score desc, m.name limit ${limit}`; | |
| 34 | + return Promise.all(rows.map(async (r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), countryIso2: reqStr(r.country_iso2), momentum: (await momentum(scopeFor(sql, { metroId: reqStr(r.id) }))) as MarketMomentum }))); | |
| 26 | 35 | } |
| 27 | 36 | |
| 28 | −async function liveTopOperators(limit: number): Promise<RankingRow[]> { | |
| 37 | +async function operatorExpansion(limit = 10): Promise<Dashboard["operatorExpansion"]> { | |
| 29 | 38 | const sql = pg(); |
| 39 | + const since = sql`(current_date - interval '12 months')`; | |
| 30 | 40 | const rows = await sql<Row[]>` |
| 31 | − select o.id, o.slug, o.name, o.hq_country_iso2, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw, | |
| 32 | − (count(*) filter (where ${mwExpr(sql)} is not null))::float / count(*) as coverage | |
| 33 | − from facilities f join operators o on o.id = f.operator_id where f.merged_into is null group by 1, 2, 3, 4 order by n desc limit ${limit}`; | |
| 34 | − return rows.map((r, i) => ({ rank: i + 1, id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), href: hrefFor("operator", reqStr(r.slug)), value: int(r.n), secondary: num(r.mw), countryIso2: str(r.hq_country_iso2), coverage: num(r.coverage) })); | |
| 41 | + with entries as ( | |
| 42 | + select f.operator_id, f.metro_id, f.country_iso2, coalesce(case when f.opened_on ~ '^\\d{4}-\\d{2}-\\d{2}' then f.opened_on::date when f.opened_on ~ '^\\d{4}-\\d{2}$' then (f.opened_on || '-01')::date when f.opened_on ~ '^\\d{4}$' then (f.opened_on || '-01-01')::date else null end, f.first_seen::date) as d | |
| 43 | + from facilities f where f.merged_into is null and f.operator_id is not null | |
| 44 | + union all | |
| 45 | + select p.operator_id, p.metro_id, p.country_iso2, case when p.announced_on ~ '^\\d{4}-\\d{2}-\\d{2}' then p.announced_on::date when p.announced_on ~ '^\\d{4}-\\d{2}$' then (p.announced_on || '-01')::date when p.announced_on ~ '^\\d{4}$' then (p.announced_on || '-01-01')::date else p.created_at::date end | |
| 46 | + from projects p where ${projectLive(sql)} and p.operator_id is not null | |
| 47 | + ), | |
| 48 | + fc as (select operator_id, country_iso2, min(d) as first_d from entries where country_iso2 is not null group by 1, 2), | |
| 49 | + fm as (select operator_id, metro_id, min(d) as first_d from entries where metro_id is not null group by 1, 2), | |
| 50 | + agg as ( | |
| 51 | + select o.id, o.slug, o.name, | |
| 52 | + (select array_agg(fc.country_iso2 order by fc.first_d desc) from fc where fc.operator_id = o.id and fc.first_d >= ${since}) as new_countries, | |
| 53 | + (select array_agg(m.name order by fm.first_d desc) from fm join metros m on m.id = fm.metro_id where fm.operator_id = o.id and fm.first_d >= ${since}) as new_metros, | |
| 54 | + (select count(*)::int from projects p where ${projectLive(sql)} and p.operator_id = o.id and coalesce(case when p.announced_on ~ '^\\d{4}' then (left(p.announced_on, 4) || '-01-01')::date else null end, p.created_at::date) >= ${since}) as projects12m, | |
| 55 | + (select count(*) from fc where fc.operator_id = o.id and fc.first_d < ${since}) as had_before | |
| 56 | + from operators o | |
| 57 | + ) | |
| 58 | + select * from agg where (coalesce(array_length(new_countries, 1), 0) > 0 or coalesce(array_length(new_metros, 1), 0) > 0) and had_before > 0 | |
| 59 | + order by coalesce(array_length(new_countries, 1), 0) + coalesce(array_length(new_metros, 1), 0) desc, projects12m desc, name limit ${limit}`; | |
| 60 | + return rows.map((r) => ({ operator: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, newCountries: ((r.new_countries as string[] | null) ?? []).map(String), newMetros: ((r.new_metros as string[] | null) ?? []).map(String), projects12m: int(r.projects12m) })); | |
| 61 | +} | |
| 62 | + | |
| 63 | +async function globalGridConstraints(limit = 10): Promise<GridConstraintDTO[]> { | |
| 64 | + const sql = pg(); | |
| 65 | + const [rows, ev] = await Promise.all([ | |
| 66 | + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} order by g.created_at desc limit ${limit}`, | |
| 67 | + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.event_type = any(${GRID_EVENT_TYPES}) and e.review_status <> 'rejected' order by e.detected_at desc limit ${limit}`, | |
| 68 | + ]); | |
| 69 | + const out = rows.map(gridConstraintDto); | |
| 70 | + const seen = new Set(out.map((g) => g.eventId).filter(Boolean)); | |
| 71 | + for (const e of await toEventDtos(ev)) { if (seen.has(e.id)) continue; seen.add(e.id); out.push(gridConstraintFromEvent(e)); } | |
| 72 | + return out.slice(0, limit); | |
| 35 | 73 | } |
| 36 | 74 | |
| 37 | 75 | export async function dashboard(): Promise<Dashboard> { |
| 38 | 76 | const sql = pg(); |
| 39 | − const mw = mwExpr(sql); | |
| 40 | − const [facStats, countsRow, evRow, crawlRow, capSeries, capFallback, projStatus, typeRows, pipeline, aiRow, aiRecent, latest, newProjects, recentlyVerified, rkCountries, rkMetros, rkOperators] = await Promise.all([ | |
| 41 | − sql<Row[]>` | |
| 42 | − select count(*)::int as facilities, | |
| 43 | − count(*) filter (where f.status = any(${OPERATIONAL_SET}))::int as operational, | |
| 44 | − count(*) filter (where f.status = any(${CONSTRUCTION_SET}))::int as under_construction, | |
| 45 | − count(*) filter (where f.status = any(${PLANNED_SET}))::int as planned, | |
| 46 | − coalesce(sum(${mw}) filter (where f.status = any(${OPERATIONAL_SET})), 0)::float as known_operational_mw, | |
| 47 | − coalesce(sum(${mw}) filter (where f.status = any(${CONSTRUCTION_SET})), 0)::float as construction_mw, | |
| 48 | − coalesce(sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PLANNED_SET})), 0)::float as planned_mw, | |
| 49 | − count(*) filter (where ${mw} is not null)::int as with_mw, | |
| 50 | − count(distinct f.country_iso2)::int as countries, | |
| 51 | − count(distinct f.metro_id)::int as metros, | |
| 52 | − count(distinct f.operator_id)::int as operators | |
| 53 | − from facilities f where f.merged_into is null`, | |
| 77 | + const known = knownMwAgg(sql); | |
| 78 | + const [facStats, countsRow, evRow, crawlRow, capSeries, capFallback, projStatus, typeRows, pipeline, aiRow, aiRecent, latest, newProjects, recentlyVerified, rkCountries, rkMetros, rkOperators, pulseData, growing, major, expansion, powerRows, gridConstraints, coverage, ingRow] = await Promise.all([ | |
| 79 | + sql<Row[]>`select ${dimAggCols(sql)}, count(distinct f.country_iso2)::int as countries, count(distinct f.metro_id)::int as metros from ${facilityView(sql)} f`, | |
| 54 | 80 | sql<Row[]>` |
| 55 | − select (select count(*)::int from cloud_regions) as cloud_regions, (select count(*)::int from ixps) as ixps, (select count(*)::int from projects) as projects, | |
| 81 | + select (select count(*)::int from cloud_regions where status <> 'retired') as cloud_regions, (select count(*)::int from ixps) as ixps, (select count(*)::int from projects p where ${projectLive(sql)}) as projects, | |
| 56 | 82 | (select count(*)::int from sources) as sources, (select count(*)::int from documents) as documents, (select count(*)::int from operators) as all_operators`, |
| 57 | 83 | sql<Row[]>`select count(*) filter (where detected_at >= now() - interval '24 hours')::int as e24, count(*) filter (where detected_at >= now() - interval '7 days')::int as e7 from events where review_status <> 'rejected'`, |
| 58 | − sql<Row[]>`select max(finished_at) as last_finished, max(started_at) as last_started from connector_runs`, | |
| 84 | + sql<Row[]>`select max(finished_at) as last_finished, max(started_at) as last_started, count(*) filter (where started_at >= now() - interval '24 hours')::int as runs24 from connector_runs`, | |
| 59 | 85 | sql<Row[]>`select day, metric, value from daily_metrics where dim = 'global' and metric in ('facilities_total', 'known_mw') order by day`, |
| 60 | − sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*)::int as n, sum(${mw})::float as mw from facilities f where f.merged_into is null and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 61 | − sql<Row[]>`select status, count(*)::int as n, sum(planned_mw)::float as mw from projects group by status order by n desc`, | |
| 86 | + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${known})::float as mw from ${facilityView(sql)} f where f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 87 | + sql<Row[]>`select p.status, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${projectLive(sql)} group by p.status order by n desc`, | |
| 62 | 88 | sql<Row[]>`select facility_type, count(*)::int as n from facilities where merged_into is null group by facility_type order by n desc`, |
| 63 | − sql<Row[]>`select left(expected_opening, 4) as year, count(*)::int as n, sum(planned_mw)::float as mw from projects where expected_opening ~ '^\\d{4}' and status not in ('cancelled', 'closed') group by 1 order by 1`, | |
| 64 | − sql<Row[]>`select (select count(*)::int from facilities where merged_into is null and (is_ai or facility_type = 'ai')) as facilities, (select count(*)::int from projects where is_ai) as projects, (select sum(planned_mw)::float from projects where is_ai and status not in ('cancelled')) as planned_mw`, | |
| 65 | − sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where p.is_ai order by p.last_update desc limit 6`, | |
| 89 | + sql<Row[]>`select left(p.expected_opening, 4) as year, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${projectLive(sql)} and p.expected_opening ~ '^\\d{4}' and p.status not in ('cancelled', 'closed') group by 1 order by 1`, | |
| 90 | + sql<Row[]>`select (select count(*)::int from ${facilityView(sql)} f where ${countedAgg(sql)} and (f.ai_evidence = any(${AI_LEVELS}) or f.is_ai or f.facility_type = 'ai')) as facilities, (select count(*)::int from projects p where ${projectLive(sql)} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai)) as projects, (select sum(p.planned_mw)::float from projects p where ${projectLive(sql)} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai) and p.status not in ('cancelled')) as planned_mw`, | |
| 91 | + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and (p.ai_evidence = any(${AI_LEVELS}) or p.is_ai) order by p.last_update desc limit 6`, | |
| 66 | 92 | latestEvents(20), |
| 67 | − sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} order by p.created_at desc limit 10`, | |
| 93 | + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} order by p.created_at desc limit 10`, | |
| 68 | 94 | recentlyVerifiedFacilities(10), |
| 69 | 95 | topRows(["countries_by_facilities", "countries_by_known_mw", "countries_facilities", "top_countries"], "countries", 10), |
| 70 | 96 | topRows(["metros_by_facilities", "metros_by_known_mw", "metros_facilities", "top_metros"], "metros", 10), |
| 71 | 97 | topRows(["operators_by_facilities", "operators_by_known_mw", "operators_facilities", "top_operators"], "operators", 10), |
| 98 | + pulse("24h"), | |
| 99 | + fastestGrowingMetros(8), | |
| 100 | + sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and p.planned_mw >= 100 and p.status not in ('cancelled', 'closed') order by p.planned_mw desc, p.last_update desc limit 12`, | |
| 101 | + operatorExpansion(10), | |
| 102 | + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.event_type = any(${POWER_EVENT_TYPES}) and e.review_status <> 'rejected' order by e.detected_at desc limit 10`, | |
| 103 | + globalGridConstraints(10), | |
| 104 | + coverageRow("global", "Global", "global", scopeFor(sql, {})), | |
| 105 | + sql<Row[]>`select (select count(*)::int from documents where last_fetched >= now() - interval '24 hours') as docs24, (select count(*)::int from connectors where enabled and not paused) as total, (select count(*)::int from connectors where enabled and not paused and health = 'ok') as healthy`, | |
| 72 | 106 | ]); |
| 73 | 107 | const fs = facStats[0] ?? {}; |
| 74 | 108 | const cn = countsRow[0] ?? {}; |
@@ -76,12 +110,12 @@ export async function dashboard(): Promise<Dashboard> { | ||
| 76 | 110 | const stats: DashboardStats = { |
| 77 | 111 | facilities, |
| 78 | 112 | operational: int(fs.operational), |
| 79 | − underConstruction: int(fs.under_construction), | |
| 113 | + underConstruction: int(fs.construction), | |
| 80 | 114 | planned: int(fs.planned), |
| 81 | − knownOperationalMw: Math.round((num(fs.known_operational_mw) ?? 0) * 100) / 100, | |
| 82 | − constructionMw: Math.round((num(fs.construction_mw) ?? 0) * 100) / 100, | |
| 83 | − plannedMw: Math.round((num(fs.planned_mw) ?? 0) * 100) / 100, | |
| 84 | − mwCoverage: facilities ? Math.round((int(fs.with_mw) / facilities) * 1000) / 1000 : 0, | |
| 115 | + knownOperationalMw: round2(num(fs.known_mw)) ?? 0, | |
| 116 | + constructionMw: round2(num(fs.construction_mw)) ?? 0, | |
| 117 | + plannedMw: round2(num(fs.planned_mw)) ?? 0, | |
| 118 | + mwCoverage: share(int(fs.with_mw), facilities), | |
| 85 | 119 | countries: int(fs.countries), |
| 86 | 120 | metros: int(fs.metros), |
| 87 | 121 | operators: int(fs.operators) || int(cn.all_operators), |
@@ -112,22 +146,31 @@ export async function dashboard(): Promise<Dashboard> { | ||
| 112 | 146 | } |
| 113 | 147 | if (!capacityOverTime.length) { |
| 114 | 148 | let cumF = 0, cumMw = 0, saw = false; |
| 115 | − capacityOverTime = capFallback.map((r) => { cumF += int(r.n); const m = num(r.mw); if (m != null) { cumMw += m; saw = true; } return { year: int(r.year), facilities: cumF, knownMw: m, cumulativeMw: saw ? Math.round(cumMw * 100) / 100 : null }; }).filter((x) => x.year > 0); | |
| 149 | + capacityOverTime = capFallback.map((r) => { cumF += int(r.n); const m = num(r.mw); if (m != null) { cumMw += m; saw = true; } return { year: int(r.year), facilities: cumF, knownMw: round2(m), cumulativeMw: saw ? Math.round(cumMw * 100) / 100 : null }; }).filter((x) => x.year > 0); | |
| 116 | 150 | } |
| 117 | 151 | |
| 118 | 152 | const ai = aiRow[0] ?? {}; |
| 153 | + const ing = ingRow[0] ?? {}; | |
| 119 | 154 | return { |
| 120 | 155 | stats, |
| 121 | 156 | capacityOverTime, |
| 122 | − projectsByStatus: projStatus.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: num(r.mw) })), | |
| 123 | − topCountries: rkCountries ?? (await liveTopCountries(10)), | |
| 124 | − topMetros: rkMetros ?? (await liveTopMetros(10)), | |
| 125 | − topOperators: rkOperators ?? (await liveTopOperators(10)), | |
| 157 | + projectsByStatus: projStatus.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 158 | + topCountries: rkCountries ?? (await liveTop("country", 10)), | |
| 159 | + topMetros: rkMetros ?? (await liveTop("metro", 10)), | |
| 160 | + topOperators: rkOperators ?? (await liveTop("operator", 10)), | |
| 126 | 161 | typeBreakdown: typeRows.map((r) => ({ type: asType(r.facility_type), count: int(r.n) })), |
| 127 | − pipelineByYear: pipeline.map((r) => ({ year: reqStr(r.year), count: int(r.n), mw: num(r.mw) })), | |
| 128 | − aiExpansion: { facilities: int(ai.facilities), projects: int(ai.projects), plannedMw: num(ai.planned_mw), recent: aiRecent.map(projectSummary) as ProjectSummary[] }, | |
| 162 | + pipelineByYear: pipeline.map((r) => ({ year: reqStr(r.year), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 163 | + aiExpansion: { facilities: int(ai.facilities), projects: int(ai.projects), plannedMw: round2(num(ai.planned_mw)), recent: aiRecent.map(projectSummary) as ProjectSummary[] }, | |
| 129 | 164 | latestEvents: latest, |
| 130 | 165 | newProjects: newProjects.map(projectSummary), |
| 131 | 166 | recentlyVerified, |
| 167 | + pulse: pulseData, | |
| 168 | + fastestGrowingMetros: growing, | |
| 169 | + majorProjects: major.map(projectSummary), | |
| 170 | + operatorExpansion: expansion, | |
| 171 | + powerEvents: await toEventDtos(powerRows), | |
| 172 | + gridConstraints, | |
| 173 | + coverage, | |
| 174 | + ingestion: { lastRunAt: iso(crawlRow[0]?.last_started), runs24h: int(crawlRow[0]?.runs24), documents24h: int(ing.docs24), events24h: int(evRow[0]?.e24), connectorsHealthy: int(ing.healthy), connectorsTotal: int(ing.total) }, | |
| 132 | 175 | }; |
| 133 | 176 | } |
added
apps/api/src/repositories/download.ts
+196 −0
@@ -0,0 +1,196 @@ | ||
| 1 | +/** | |
| 2 | + * Dataset downloads (CSV / JSON / GeoJSON), streamed in 1 000-row batches, ≤ 50 000 rows, with license gating: | |
| 3 | + * a row whose ONLY current provenance sources forbid redistribution (sources.redistribution = 'restricted') is | |
| 4 | + * excluded and the excluded sources are listed. Nothing is ever blanked silently. | |
| 5 | + */ | |
| 6 | +import type { DownloadDataset } from "@dci/core"; | |
| 7 | +import { pg, andAll, facilityJoins, projectLive, yearExpr, type Fragment, type Sql } from "../lib/sql.js"; | |
| 8 | +import { int, num, reqStr, str, type Row } from "../lib/rows.js"; | |
| 9 | +import { csvLine } from "../lib/csv.js"; | |
| 10 | +import { csv } from "../lib/http.js"; | |
| 11 | +import { facilityConds, toFacilityQuery } from "./facilities.js"; | |
| 12 | + | |
| 13 | +export const DOWNLOAD_KEYS = ["facilities", "projects", "operators", "events", "cloud-regions", "ixps", "countries", "markets"] as const; | |
| 14 | +export type DownloadKey = (typeof DownloadKeys)[number]; | |
| 15 | +const DownloadKeys = DOWNLOAD_KEYS; | |
| 16 | +export type DownloadFormat = "csv" | "json" | "geojson"; | |
| 17 | +export const MAX_ROWS = 50_000; | |
| 18 | +export const DOWNLOAD_LICENSE = "Each row carries the names of the sources it was built from; re-users must keep that attribution and respect each source's licence (see /api/v1/sources). Rows whose only sources forbid redistribution are excluded from every download and listed in excludedSources. Figures are published values only (no estimates unless flagged), with the coverage caveats of the API."; | |
| 19 | + | |
| 20 | +export interface DownloadFilters { country?: string; status?: string; type?: string; operator?: string; metro?: string; min_mw?: number; max_mw?: number; ai?: boolean; hyperscale?: boolean; has_mw?: boolean; project_status?: string; expected_from?: number; expected_to?: number; event_type?: string; since?: string; until?: string; min_significance?: number } | |
| 21 | + | |
| 22 | +interface Spec { key: DownloadKey; label: string; description: string; formats: DownloadFormat[]; filters: string[]; columns: string[]; entityType: string | null; gated: boolean } | |
| 23 | + | |
| 24 | +export const SPECS: Spec[] = [ | |
| 25 | + { key: "facilities", label: "Facilities", description: "Every indexed data center / campus / building record (merged duplicates excluded) with location, status, capacity ontology, AI evidence and sources.", formats: ["csv", "json", "geojson"], filters: ["country", "status", "type", "operator", "metro", "min_mw", "max_mw", "ai", "hyperscale", "has_mw"], columns: ["id", "slug", "name", "operator", "country_iso2", "city", "metro", "lat", "lng", "geo_precision", "status", "facility_type", "it_capacity_mw", "total_power_mw", "planned_power_mw", "utility_capacity_mw", "grid_connection_mw", "mw_is_estimate", "record_scope", "parent_facility_id", "ai_evidence", "opened_on", "confidence", "completeness", "source_count", "sources", "updated_at"], entityType: "facility", gated: true }, | |
| 26 | + { key: "projects", label: "Projects", description: "Live infrastructure projects (false positives hidden by review and merged duplicates excluded) with lifecycle dates, planned MW and investment scopes.", formats: ["csv", "json", "geojson"], filters: ["country", "project_status", "operator", "metro", "min_mw", "max_mw", "ai", "expected_from", "expected_to"], columns: ["id", "slug", "name", "operator", "country_iso2", "city", "metro", "lat", "lng", "geo_precision", "status", "project_class", "evidence_level", "announced_on", "permit_filed_on", "approved_on", "construction_started_on", "expected_opening", "opened_on", "planned_mw", "capacity_scope", "investment_usd", "investment_scope", "ai_evidence", "confidence", "sources", "last_update"], entityType: "project", gated: true }, | |
| 27 | + { key: "operators", label: "Operators", description: "Operators with containment-aware facility counts and known MW (from the worker-refreshed stats).", formats: ["csv", "json"], filters: ["country"], columns: ["id", "slug", "name", "kind", "hq_country_iso2", "website", "is_cloud_provider", "is_carrier", "facility_count", "country_count", "known_mw", "planned_mw", "project_count", "mw_coverage", "sources", "updated_at"], entityType: "operator", gated: true }, | |
| 28 | + { key: "events", label: "Events", description: "Change feed: detected facility / project / market events with source and significance (rejected events excluded).", formats: ["csv", "json"], filters: ["event_type", "country", "operator", "since", "until", "min_significance"], columns: ["id", "entity_type", "entity_id", "event_type", "detected_at", "effective_date", "title", "url", "source", "source_kind", "significance", "significance_band", "confidence", "country_iso2", "operator", "old_value", "new_value"], entityType: null, gated: true }, | |
| 29 | + { key: "cloud-regions", label: "Cloud regions", description: "Public cloud regions by provider with city-level coordinates.", formats: ["csv", "json", "geojson"], filters: ["country", "operator"], columns: ["id", "slug", "provider", "code", "name", "city", "country_iso2", "lat", "lng", "geo_precision", "availability_zones", "launched_on", "status", "is_sovereign", "source_url", "sources", "updated_at"], entityType: "cloud_region", gated: true }, | |
| 30 | + { key: "ixps", label: "Internet exchanges", description: "IXPs with network counts and facility links; coordinates are the metro reference point when known.", formats: ["csv", "json", "geojson"], filters: ["country"], columns: ["id", "slug", "name", "name_long", "city", "country_iso2", "website", "network_count", "facility_count", "metro", "lat", "lng", "sources", "updated_at"], entityType: "ixp", gated: true }, | |
| 31 | + { key: "countries", label: "Countries", description: "Countries with containment-aware facility totals (worker-refreshed stats) and public energy indicators.", formats: ["csv", "json", "geojson"], filters: [], columns: ["iso2", "iso3", "slug", "name", "region", "subregion", "population", "gdp_usd", "electricity_twh", "renewable_share", "stats_year", "facility_count", "known_mw", "mw_coverage", "lat", "lng", "updated_at"], entityType: null, gated: false }, | |
| 32 | + { key: "markets", label: "Markets (metros)", description: "Data center markets / metros with reference point, radius and containment-aware totals.", formats: ["csv", "json", "geojson"], filters: ["country"], columns: ["id", "slug", "name", "country_iso2", "region_name", "lat", "lng", "radius_km", "facility_count", "known_mw", "mw_coverage", "operator_count", "updated_at"], entityType: null, gated: false }, | |
| 33 | +]; | |
| 34 | + | |
| 35 | +export function specFor(key: string): Spec | null { | |
| 36 | + return SPECS.find((s) => s.key === key) ?? null; | |
| 37 | +} | |
| 38 | + | |
| 39 | +/** Sources that forbid redistribution (their rows are gated out). */ | |
| 40 | +export async function restrictedSources(): Promise<Array<{ id: string; name: string; reason: string }>> { | |
| 41 | + const sql = pg(); | |
| 42 | + const rows = await sql<Row[]>`select id, name from sources where redistribution = 'restricted' order by name`; | |
| 43 | + return rows.map((r) => ({ id: reqStr(r.id), name: reqStr(r.name), reason: "redistribution = restricted" })); | |
| 44 | +} | |
| 45 | + | |
| 46 | +/** A row is gated out when it has provenance and every current source is restricted. */ | |
| 47 | +function gate(sql: Sql, entityType: string, idCol: Fragment): Fragment { | |
| 48 | + return sql`not (exists (select 1 from provenance gp where gp.entity_type = ${entityType} and gp.entity_id = ${idCol} and gp.is_current) | |
| 49 | + and not exists (select 1 from provenance gp join sources gs on gs.id = gp.source_id where gp.entity_type = ${entityType} and gp.entity_id = ${idCol} and gp.is_current and coalesce(gs.redistribution, 'unknown') <> 'restricted'))`; | |
| 50 | +} | |
| 51 | + | |
| 52 | +function sourcesCol(sql: Sql, entityType: string, idCol: Fragment): Fragment { | |
| 53 | + return sql`(select string_agg(distinct s.name, ';' order by s.name) from provenance sp join sources s on s.id = sp.source_id where sp.entity_type = ${entityType} and sp.entity_id = ${idCol} and sp.is_current)`; | |
| 54 | +} | |
| 55 | + | |
| 56 | +/** SELECT for a dataset, columns aliased exactly as in the spec (extra `lat`/`lng` are used for GeoJSON). */ | |
| 57 | +function query(sql: Sql, key: DownloadKey, f: DownloadFilters): Fragment { | |
| 58 | + switch (key) { | |
| 59 | + case "facilities": { | |
| 60 | + const conds = facilityConds(sql, toFacilityQuery({ country: f.country, status: f.status, type: f.type, operator: f.operator, metro: f.metro, min_mw: f.min_mw, max_mw: f.max_mw, ai: f.ai, hyperscale: f.hyperscale, has_mw: f.has_mw })); | |
| 61 | + conds.push(gate(sql, "facility", sql`f.id`)); | |
| 62 | + return sql`select f.id, f.slug, f.name, o.name as operator, f.country_iso2, f.city, m.name as metro, f.lat, f.lng, f.geo_precision, f.status, f.facility_type, f.it_capacity_mw, f.total_power_mw, f.planned_power_mw, f.utility_capacity_mw, f.grid_connection_mw, f.mw_is_estimate, f.record_scope, f.parent_facility_id, f.ai_evidence, f.opened_on, f.confidence, f.completeness, f.source_count, ${sourcesCol(sql, "facility", sql`f.id`)} as sources, f.updated_at | |
| 63 | + from facilities f ${facilityJoins(sql)} where ${andAll(sql, conds)} order by f.country_iso2, f.name limit ${MAX_ROWS}`; | |
| 64 | + } | |
| 65 | + case "projects": { | |
| 66 | + const c: Fragment[] = [projectLive(sql), gate(sql, "project", sql`p.id`)]; | |
| 67 | + const st = csv(f.project_status ?? f.status); | |
| 68 | + if (st.length) c.push(sql`p.status = any(${st})`); | |
| 69 | + const co = csv(f.country).map((x) => x.toUpperCase()); | |
| 70 | + if (co.length) c.push(sql`p.country_iso2 = any(${co})`); | |
| 71 | + if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`); | |
| 72 | + if (f.metro) c.push(sql`(m.slug = ${f.metro} or m.id = ${f.metro})`); | |
| 73 | + if (f.min_mw != null) c.push(sql`p.planned_mw >= ${f.min_mw}`); | |
| 74 | + if (f.max_mw != null) c.push(sql`p.planned_mw <= ${f.max_mw}`); | |
| 75 | + if (f.ai === true) c.push(sql`(p.is_ai or p.ai_evidence in ('confirmed', 'likely'))`); | |
| 76 | + if (f.expected_from != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} >= ${f.expected_from}`); | |
| 77 | + if (f.expected_to != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} <= ${f.expected_to}`); | |
| 78 | + return sql`select p.id, p.slug, p.name, o.name as operator, p.country_iso2, p.city, m.name as metro, p.lat, p.lng, p.geo_precision, p.status, p.project_class, p.evidence_level, p.announced_on, p.permit_filed_on, p.approved_on, p.construction_started_on, p.expected_opening, p.opened_on, p.planned_mw, p.capacity_scope, p.investment_usd, p.investment_scope, p.ai_evidence, p.confidence, ${sourcesCol(sql, "project", sql`p.id`)} as sources, p.last_update | |
| 79 | + from projects p left join operators o on o.id = p.operator_id left join metros m on m.id = p.metro_id where ${andAll(sql, c)} order by p.planned_mw desc nulls last, p.name limit ${MAX_ROWS}`; | |
| 80 | + } | |
| 81 | + case "operators": { | |
| 82 | + const c: Fragment[] = [gate(sql, "operator", sql`o.id`)]; | |
| 83 | + if (f.country) c.push(sql`(o.hq_country_iso2 = ${f.country.toUpperCase()} or exists (select 1 from facilities x where x.operator_id = o.id and x.country_iso2 = ${f.country.toUpperCase()} and x.merged_into is null))`); | |
| 84 | + return sql`select o.id, o.slug, o.name, o.kind, o.hq_country_iso2, o.website, o.is_cloud_provider, o.is_carrier, (o.stats->>'facilityCount')::int as facility_count, (o.stats->>'countryCount')::int as country_count, (o.stats->>'knownMw')::float as known_mw, (o.stats->>'plannedMw')::float as planned_mw, (o.stats->>'projectCount')::int as project_count, (o.stats->>'mwCoverage')::float as mw_coverage, ${sourcesCol(sql, "operator", sql`o.id`)} as sources, o.updated_at | |
| 85 | + from operators o where ${andAll(sql, c)} order by (o.stats->>'facilityCount')::int desc nulls last, o.name limit ${MAX_ROWS}`; | |
| 86 | + } | |
| 87 | + case "events": { | |
| 88 | + const c: Fragment[] = [sql`e.review_status <> 'rejected'`, sql`coalesce(s.redistribution, 'unknown') <> 'restricted'`]; | |
| 89 | + const types = csv(f.event_type); | |
| 90 | + if (types.length) c.push(sql`e.event_type = any(${types})`); | |
| 91 | + if (f.country) c.push(sql`e.country_iso2 = ${f.country.toUpperCase()}`); | |
| 92 | + if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`); | |
| 93 | + if (f.since) c.push(sql`e.detected_at >= ${f.since}::timestamptz`); | |
| 94 | + if (f.until) c.push(sql`e.detected_at <= ${f.until}::timestamptz`); | |
| 95 | + if (f.min_significance != null) c.push(sql`e.significance >= ${f.min_significance}`); | |
| 96 | + return sql`select e.id, e.entity_type, e.entity_id, e.event_type, e.detected_at, e.effective_date, e.title, e.url, s.name as source, coalesce(e.source_kind, s.kind) as source_kind, e.significance, case when e.significance >= 75 then 'major' when e.significance >= 45 then 'medium' else 'minor' end as significance_band, e.confidence, e.country_iso2, o.name as operator, e.old_value, e.new_value | |
| 97 | + from events e left join sources s on s.id = e.source_id left join operators o on o.id = e.operator_id where ${andAll(sql, c)} order by e.detected_at desc limit ${MAX_ROWS}`; | |
| 98 | + } | |
| 99 | + case "cloud-regions": { | |
| 100 | + const c: Fragment[] = [gate(sql, "cloud_region", sql`r.id`)]; | |
| 101 | + if (f.country) c.push(sql`r.country_iso2 = ${f.country.toUpperCase()}`); | |
| 102 | + if (f.operator) c.push(sql`(pr.slug = ${f.operator} or pr.id = ${f.operator})`); | |
| 103 | + return sql`select r.id, r.slug, pr.name as provider, r.code, r.name, r.city, r.country_iso2, r.lat, r.lng, r.geo_precision, r.availability_zones, r.launched_on, r.status, r.is_sovereign, r.source_url, ${sourcesCol(sql, "cloud_region", sql`r.id`)} as sources, r.updated_at | |
| 104 | + from cloud_regions r join operators pr on pr.id = r.provider_id where ${andAll(sql, c)} order by pr.name, r.code limit ${MAX_ROWS}`; | |
| 105 | + } | |
| 106 | + case "ixps": { | |
| 107 | + const c: Fragment[] = [gate(sql, "ixp", sql`x.id`)]; | |
| 108 | + if (f.country) c.push(sql`x.country_iso2 = ${f.country.toUpperCase()}`); | |
| 109 | + return sql`select x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count, m.name as metro, m.lat, m.lng, ${sourcesCol(sql, "ixp", sql`x.id`)} as sources, x.updated_at | |
| 110 | + from ixps x left join metros m on m.id = x.metro_id where ${andAll(sql, c)} order by x.network_count desc nulls last, x.name limit ${MAX_ROWS}`; | |
| 111 | + } | |
| 112 | + case "countries": | |
| 113 | + return sql`select c.iso2, c.iso3, c.slug, c.name, c.region, c.subregion, c.population, c.gdp_usd, c.electricity_twh, c.renewable_share, c.stats_year, (c.stats->>'facilityCount')::int as facility_count, (c.stats->>'knownMw')::float as known_mw, (c.stats->>'mwCoverage')::float as mw_coverage, c.lat, c.lng, c.updated_at | |
| 114 | + from countries c where coalesce((c.stats->>'facilityCount')::int, 0) > 0 or exists (select 1 from cloud_regions r where r.country_iso2 = c.iso2) order by (c.stats->>'facilityCount')::int desc nulls last, c.name limit ${MAX_ROWS}`; | |
| 115 | + case "markets": { | |
| 116 | + const c: Fragment[] = [sql`true`]; | |
| 117 | + if (f.country) c.push(sql`m.country_iso2 = ${f.country.toUpperCase()}`); | |
| 118 | + return sql`select m.id, m.slug, m.name, m.country_iso2, m.region_name, m.lat, m.lng, m.radius_km, (m.stats->>'facilityCount')::int as facility_count, (m.stats->>'knownMw')::float as known_mw, (m.stats->>'mwCoverage')::float as mw_coverage, (m.stats->>'operatorCount')::int as operator_count, m.updated_at | |
| 119 | + from metros m where ${andAll(sql, c)} order by (m.stats->>'facilityCount')::int desc nulls last, m.name limit ${MAX_ROWS}`; | |
| 120 | + } | |
| 121 | + } | |
| 122 | +} | |
| 123 | + | |
| 124 | +/** Dataset catalogue with live row counts. */ | |
| 125 | +export async function listDatasets(): Promise<DownloadDataset[]> { | |
| 126 | + const sql = pg(); | |
| 127 | + const [counts, excluded, attributions] = await Promise.all([ | |
| 128 | + sql<Row[]>`select | |
| 129 | + (select count(*) from facilities where merged_into is null)::int as facilities, | |
| 130 | + (select count(*) from projects p where ${projectLive(sql)})::int as projects, | |
| 131 | + (select count(*) from operators)::int as operators, | |
| 132 | + (select count(*) from events where review_status <> 'rejected')::int as events, | |
| 133 | + (select count(*) from cloud_regions)::int as "cloud-regions", | |
| 134 | + (select count(*) from ixps)::int as ixps, | |
| 135 | + (select count(*) from countries c where coalesce((c.stats->>'facilityCount')::int, 0) > 0 or exists (select 1 from cloud_regions r where r.country_iso2 = c.iso2))::int as countries, | |
| 136 | + (select count(*) from metros)::int as markets`, | |
| 137 | + restrictedSources(), | |
| 138 | + sql<Row[]>`select p.entity_type, array_agg(distinct s.attribution) filter (where s.attribution is not null) as attributions from provenance p join sources s on s.id = p.source_id where p.is_current group by p.entity_type`, | |
| 139 | + ]); | |
| 140 | + const attr = new Map<string, string[]>(attributions.map((r) => [reqStr(r.entity_type), ((r.attributions as string[] | null) ?? []).slice(0, 30)])); | |
| 141 | + const eventAttr = (await sql<Row[]>`select distinct s.attribution from events e join sources s on s.id = e.source_id where s.attribution is not null limit 30`).map((r) => reqStr(r.attribution)); | |
| 142 | + const c = counts[0] ?? {}; | |
| 143 | + return SPECS.map((s) => ({ | |
| 144 | + key: s.key, | |
| 145 | + label: s.label, | |
| 146 | + description: s.description, | |
| 147 | + formats: s.formats, | |
| 148 | + filters: s.filters, | |
| 149 | + rows: Math.min(MAX_ROWS, int(c[s.key])), | |
| 150 | + license: DOWNLOAD_LICENSE, | |
| 151 | + attribution: s.entityType ? attr.get(s.entityType) ?? [] : s.key === "events" ? eventAttr : [], | |
| 152 | + excludedSources: s.gated ? excluded : [], | |
| 153 | + })); | |
| 154 | +} | |
| 155 | + | |
| 156 | +function cell(v: unknown): unknown { | |
| 157 | + if (v == null) return null; | |
| 158 | + if (typeof v === "object" && !(v instanceof Date)) return JSON.stringify(v); | |
| 159 | + return v; | |
| 160 | +} | |
| 161 | + | |
| 162 | +/** Async generator of chunks for a dataset in the requested format (header first). */ | |
| 163 | +export async function* streamDataset(key: DownloadKey, format: DownloadFormat, f: DownloadFilters, excluded: Array<{ id: string; name: string; reason: string }>): AsyncGenerator<string> { | |
| 164 | + const sql = pg(); | |
| 165 | + const spec = specFor(key)!; | |
| 166 | + const cols = spec.columns; | |
| 167 | + const q = query(sql, key, f); | |
| 168 | + let total = 0, skipped = 0, first = true; | |
| 169 | + if (format === "csv") yield csvLine(cols); | |
| 170 | + else if (format === "json") yield `{"data":[`; | |
| 171 | + else yield `{"type":"FeatureCollection","features":[`; | |
| 172 | + const cursor = q.cursor(1000); | |
| 173 | + for await (const rows of cursor) { | |
| 174 | + let chunk = ""; | |
| 175 | + for (const r of rows as Row[]) { | |
| 176 | + total++; | |
| 177 | + if (format === "csv") { chunk += csvLine(cols.map((c) => cell(r[c]))); continue; } | |
| 178 | + const obj: Record<string, unknown> = {}; | |
| 179 | + for (const c of cols) obj[c] = r[c] ?? null; | |
| 180 | + if (format === "json") { chunk += `${first ? "" : ","}${JSON.stringify(obj)}`; first = false; continue; } | |
| 181 | + const lat = num(r.lat), lng = num(r.lng); | |
| 182 | + if (lat == null || lng == null) { skipped++; continue; } | |
| 183 | + const { lat: _a, lng: _b, ...props } = obj; | |
| 184 | + chunk += `${first ? "" : ","}${JSON.stringify({ type: "Feature", id: str(r.id ?? r.iso2), geometry: { type: "Point", coordinates: [lng, lat] }, properties: props })}`; | |
| 185 | + first = false; | |
| 186 | + } | |
| 187 | + if (chunk) yield chunk; | |
| 188 | + } | |
| 189 | + const meta = { total: format === "geojson" ? total - skipped : total, rowsWithoutCoordinates: format === "geojson" ? skipped : undefined, excludedSources: excluded, generatedAt: new Date().toISOString(), license: DOWNLOAD_LICENSE, maxRows: MAX_ROWS }; | |
| 190 | + if (format === "json") yield `],"meta":${JSON.stringify(meta)}}`; | |
| 191 | + else if (format === "geojson") yield `],"meta":${JSON.stringify(meta)}}`; | |
| 192 | +} | |
| 193 | + | |
| 194 | +export function contentType(format: DownloadFormat): string { | |
| 195 | + return format === "csv" ? "text/csv; charset=utf-8" : format === "json" ? "application/json; charset=utf-8" : "application/geo+json; charset=utf-8"; | |
| 196 | +} | |
modified
apps/api/src/repositories/events.ts
+121 −52
@@ -1,7 +1,7 @@ | ||
| 1 | −import type { EventDTO } from "@dci/core"; | |
| 2 | −import { pg, andAll, eventCols, eventJoins, page, type Fragment } from "../lib/sql.js"; | |
| 3 | −import { int, type Row } from "../lib/rows.js"; | |
| 4 | −import { eventDto } from "../lib/dto.js"; | |
| 1 | +import type { EventDTO, SourceKind } from "@dci/core"; | |
| 2 | +import { pg, andAll, eventCols, eventJoins, page, likePattern, type Fragment } from "../lib/sql.js"; | |
| 3 | +import { int, reqStr, type Row } from "../lib/rows.js"; | |
| 4 | +import { asSourceKind, eventDto } from "../lib/dto.js"; | |
| 5 | 5 | import { entityKey, resolveEntityRefs } from "../lib/resolve.js"; |
| 6 | 6 | |
| 7 | 7 | export interface EventFilters { |
@@ -13,102 +13,171 @@ export interface EventFilters { | ||
| 13 | 13 | entityType?: string; |
| 14 | 14 | entityId?: string; |
| 15 | 15 | minSignificance?: number; |
| 16 | + significance?: "major" | "medium" | "minor"; | |
| 17 | + confidence?: string[]; | |
| 18 | + sourceKind?: string[]; | |
| 19 | + ai?: boolean; | |
| 16 | 20 | since?: string; |
| 21 | + until?: string; | |
| 22 | + q?: string; | |
| 23 | + /** collapse events sharing a cluster id into one row (default true) */ | |
| 24 | + dedupe?: boolean; | |
| 17 | 25 | reviewStatus?: string; |
| 18 | 26 | page?: number; |
| 19 | 27 | perPage?: number; |
| 20 | 28 | } |
| 21 | 29 | |
| 30 | +/** Primary-source order used to pick the representative member of a cluster. */ | |
| 31 | +const SOURCE_RANK = ["operator", "government", "filing", "utility", "cloud_provider", "registry", "secondary", "news", "dataset", "community"]; | |
| 32 | + | |
| 33 | +export const EVENTS_DEDUPE_METHODOLOGY = "dedupe=true (default) collapses events that share a cluster_id (documents describing the same announcement) into one row: the member from the most authoritative source kind (operator > government > filing > utility > cloud provider > registry > secondary > news > dataset > community), then the highest significance, then the earliest detection. evidenceCount = number of documents in the cluster; otherSources lists the other members (or same-day events of the same operator and type when no cluster id exists). Significance bands: major ≥ 75, medium 45–74, minor < 45."; | |
| 34 | + | |
| 35 | +function sourceRankExpr(sql: ReturnType<typeof pg>): Fragment { | |
| 36 | + return sql`(case coalesce(e.source_kind, s.kind) ${sql.unsafe(SOURCE_RANK.map((k, i) => `when '${k}' then ${i}`).join(" "))} else 99 end)`; | |
| 37 | +} | |
| 38 | + | |
| 22 | 39 | /** Map rows to EventDTOs with {slug,name} resolved per entity type. */ |
| 23 | 40 | export async function toEventDtos(rows: Row[]): Promise<EventDTO[]> { |
| 24 | 41 | const refs = await resolveEntityRefs(rows.map((r) => ({ type: String(r.entity_type ?? ""), id: r.entity_id == null ? null : String(r.entity_id) }))); |
| 25 | 42 | return rows.map((r) => eventDto(r, refs.get(entityKey(r.entity_type, r.entity_id)) ?? null)); |
| 26 | 43 | } |
| 27 | 44 | |
| 45 | +/** | |
| 46 | + * Fill evidenceCount / otherSources for a page of events in one query: other members of the same cluster, or — when | |
| 47 | + * the event has no cluster id — same-day events with the same operator and event type. | |
| 48 | + */ | |
| 49 | +export async function attachOtherSources(dtos: EventDTO[]): Promise<EventDTO[]> { | |
| 50 | + if (!dtos.length) return dtos; | |
| 51 | + const sql = pg(); | |
| 52 | + const ids = dtos.map((d) => d.id); | |
| 53 | + const rows = await sql<Row[]>` | |
| 54 | + select x.id as for_id, e.id, s.name as source_name, coalesce(e.source_kind, s.kind) as source_kind, e.url | |
| 55 | + from events x | |
| 56 | + join events e on e.id <> x.id and e.review_status <> 'rejected' and ( | |
| 57 | + (x.cluster_id is not null and e.cluster_id = x.cluster_id) | |
| 58 | + or (x.cluster_id is null and x.operator_id is not null and e.operator_id = x.operator_id and e.event_type = x.event_type and e.detected_at::date = x.detected_at::date) | |
| 59 | + ) | |
| 60 | + left join sources s on s.id = e.source_id | |
| 61 | + where x.id = any(${ids}) | |
| 62 | + order by x.id, e.detected_at asc`; | |
| 63 | + const by = new Map<string, Array<{ sourceName: string; sourceKind: SourceKind; url: string }>>(); | |
| 64 | + const seen = new Set<string>(); | |
| 65 | + for (const r of rows) { | |
| 66 | + const forId = reqStr(r.for_id); | |
| 67 | + const url = reqStr(r.url); | |
| 68 | + const k = `${forId}|${url}`; | |
| 69 | + if (seen.has(k)) continue; | |
| 70 | + seen.add(k); | |
| 71 | + const list = by.get(forId) ?? []; | |
| 72 | + if (list.length < 10) list.push({ sourceName: reqStr(r.source_name, "unknown source"), sourceKind: asSourceKind(r.source_kind), url }); | |
| 73 | + by.set(forId, list); | |
| 74 | + } | |
| 75 | + for (const d of dtos) { | |
| 76 | + const others = by.get(d.id) ?? []; | |
| 77 | + d.otherSources = others; | |
| 78 | + d.evidenceCount = Math.max(d.evidenceCount ?? 1, others.length + 1); | |
| 79 | + } | |
| 80 | + return dtos; | |
| 81 | +} | |
| 82 | + | |
| 83 | +function conds(f: EventFilters): Fragment[] { | |
| 84 | + const sql = pg(); | |
| 85 | + const c: Fragment[] = []; | |
| 86 | + if (f.type?.length) c.push(sql`e.event_type = any(${f.type})`); | |
| 87 | + if (f.country) c.push(sql`e.country_iso2 = ${f.country.toUpperCase()}`); | |
| 88 | + if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`); | |
| 89 | + if (f.metro) c.push(sql`e.metro_id in (select id from metros where slug = ${f.metro} or id = ${f.metro})`); | |
| 90 | + if (f.project) c.push(sql`(e.project_id in (select id from projects where slug = ${f.project} or id = ${f.project}) or (e.entity_type = 'project' and e.entity_id in (select id from projects where slug = ${f.project} or id = ${f.project})))`); | |
| 91 | + if (f.entityType) c.push(sql`e.entity_type = ${f.entityType}`); | |
| 92 | + if (f.entityId) c.push(sql`e.entity_id = ${f.entityId}`); | |
| 93 | + if (f.minSignificance != null) c.push(sql`e.significance >= ${f.minSignificance}`); | |
| 94 | + if (f.significance === "major") c.push(sql`e.significance >= 75`); | |
| 95 | + if (f.significance === "medium") c.push(sql`e.significance >= 45 and e.significance < 75`); | |
| 96 | + if (f.significance === "minor") c.push(sql`e.significance < 45`); | |
| 97 | + if (f.confidence?.length) c.push(sql`e.confidence = any(${f.confidence})`); | |
| 98 | + if (f.sourceKind?.length) c.push(sql`coalesce(e.source_kind, s.kind) = any(${f.sourceKind})`); | |
| 99 | + if (f.ai === true) c.push(sql`e.is_ai`); | |
| 100 | + if (f.ai === false) c.push(sql`not e.is_ai`); | |
| 101 | + if (f.since) c.push(sql`e.detected_at >= ${f.since}::timestamptz`); | |
| 102 | + if (f.until) c.push(sql`e.detected_at <= ${f.until}::timestamptz`); | |
| 103 | + if (f.q) { const t = f.q.trim(); if (t) c.push(sql`(e.title ilike ${likePattern(t)} or e.summary ilike ${likePattern(t)})`); } | |
| 104 | + if (f.reviewStatus) c.push(sql`e.review_status = ${f.reviewStatus}`); | |
| 105 | + else c.push(sql`e.review_status <> 'rejected'`); | |
| 106 | + return c; | |
| 107 | +} | |
| 108 | + | |
| 28 | 109 | export async function listEvents(f: EventFilters): Promise<{ items: EventDTO[]; total: number; page: number; perPage: number }> { |
| 29 | 110 | const sql = pg(); |
| 30 | 111 | const pg_ = page(f.page, f.perPage, 100, 50); |
| 31 | − const conds: Fragment[] = []; | |
| 32 | − if (f.type?.length) conds.push(sql`e.event_type = any(${f.type})`); | |
| 33 | − if (f.country) conds.push(sql`e.country_iso2 = ${f.country.toUpperCase()}`); | |
| 34 | − if (f.operator) conds.push(sql`o.slug = ${f.operator}`); | |
| 35 | − if (f.metro) conds.push(sql`e.metro_id in (select id from metros where slug = ${f.metro} or id = ${f.metro})`); | |
| 36 | − if (f.project) conds.push(sql`(e.project_id in (select id from projects where slug = ${f.project} or id = ${f.project}) or (e.entity_type = 'project' and e.entity_id in (select id from projects where slug = ${f.project} or id = ${f.project})))`); | |
| 37 | − if (f.entityType) conds.push(sql`e.entity_type = ${f.entityType}`); | |
| 38 | − if (f.entityId) conds.push(sql`e.entity_id = ${f.entityId}`); | |
| 39 | − if (f.minSignificance != null) conds.push(sql`e.significance >= ${f.minSignificance}`); | |
| 40 | − if (f.since) conds.push(sql`e.detected_at >= ${f.since}::timestamptz`); | |
| 41 | − if (f.reviewStatus) conds.push(sql`e.review_status = ${f.reviewStatus}`); | |
| 42 | − else conds.push(sql`e.review_status <> 'rejected'`); | |
| 43 | − const rows = await sql<Row[]>` | |
| 44 | − select ${eventCols(sql)}, count(*) over() as total | |
| 45 | − from events e ${eventJoins(sql)} | |
| 46 | − where ${andAll(sql, conds)} | |
| 47 | − order by e.detected_at desc, e.id desc | |
| 48 | − limit ${pg_.perPage} offset ${pg_.offset}`; | |
| 112 | + const dedupe = f.dedupe !== false; | |
| 113 | + const where = andAll(sql, conds(f)); | |
| 114 | + const rows = dedupe | |
| 115 | + ? await sql<Row[]>` | |
| 116 | + with ranked as ( | |
| 117 | + select ${eventCols(sql)}, row_number() over (partition by coalesce(e.cluster_id, e.id) order by ${sourceRankExpr(sql)} asc, e.significance desc, e.detected_at asc, e.id asc) as rn | |
| 118 | + from events e ${eventJoins(sql)} | |
| 119 | + where ${where} | |
| 120 | + ) | |
| 121 | + select *, count(*) over() as total from ranked where rn = 1 | |
| 122 | + order by detected_at desc, id desc | |
| 123 | + limit ${pg_.perPage} offset ${pg_.offset}` | |
| 124 | + : await sql<Row[]>` | |
| 125 | + select ${eventCols(sql)}, count(*) over() as total | |
| 126 | + from events e ${eventJoins(sql)} | |
| 127 | + where ${where} | |
| 128 | + order by e.detected_at desc, e.id desc | |
| 129 | + limit ${pg_.perPage} offset ${pg_.offset}`; | |
| 49 | 130 | const total = rows.length ? int(rows[0]!.total) : 0; |
| 50 | − return { items: await toEventDtos(rows), total, page: pg_.page, perPage: pg_.perPage }; | |
| 131 | + const items = await attachOtherSources(await toEventDtos(rows)); | |
| 132 | + return { items, total, page: pg_.page, perPage: pg_.perPage }; | |
| 51 | 133 | } |
| 52 | 134 | |
| 53 | 135 | export async function getEvent(id: string): Promise<EventDTO | null> { |
| 54 | 136 | const sql = pg(); |
| 55 | 137 | const rows = await sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.id = ${id} limit 1`; |
| 56 | 138 | if (!rows.length) return null; |
| 57 | − return (await toEventDtos(rows))[0] ?? null; | |
| 139 | + const dtos = await attachOtherSources(await toEventDtos(rows)); | |
| 140 | + return dtos[0] ?? null; | |
| 58 | 141 | } |
| 59 | 142 | |
| 60 | −/** Events attached to an entity (entity_type + entity_id), newest first. */ | |
| 61 | −export async function eventsForEntity(entityType: string, entityId: string, limit = 50): Promise<EventDTO[]> { | |
| 143 | +/** Arbitrary condition over `events e` (joined with sources s, operators o, metros em, projects ep), newest first. */ | |
| 144 | +export async function eventsWhere(cond: Fragment, limit = 20): Promise<EventDTO[]> { | |
| 62 | 145 | const sql = pg(); |
| 63 | 146 | const rows = await sql<Row[]>` |
| 64 | 147 | select ${eventCols(sql)} from events e ${eventJoins(sql)} |
| 65 | − where e.entity_type = ${entityType} and e.entity_id = ${entityId} and e.review_status <> 'rejected' | |
| 148 | + where (${cond}) and e.review_status <> 'rejected' | |
| 66 | 149 | order by e.detected_at desc limit ${limit}`; |
| 67 | 150 | return toEventDtos(rows); |
| 68 | 151 | } |
| 69 | 152 | |
| 153 | +/** Events attached to an entity (entity_type + entity_id), newest first. */ | |
| 154 | +export async function eventsForEntity(entityType: string, entityId: string, limit = 50): Promise<EventDTO[]> { | |
| 155 | + const sql = pg(); | |
| 156 | + return eventsWhere(sql`e.entity_type = ${entityType} and e.entity_id = ${entityId}`, limit); | |
| 157 | +} | |
| 158 | + | |
| 70 | 159 | /** Recent events for an operator (by operator_id or entity = operator). */ |
| 71 | 160 | export async function eventsForOperator(operatorId: string, limit = 20): Promise<EventDTO[]> { |
| 72 | 161 | const sql = pg(); |
| 73 | − const rows = await sql<Row[]>` | |
| 74 | − select ${eventCols(sql)} from events e ${eventJoins(sql)} | |
| 75 | − where (e.operator_id = ${operatorId} or (e.entity_type = 'operator' and e.entity_id = ${operatorId})) and e.review_status <> 'rejected' | |
| 76 | − order by e.detected_at desc limit ${limit}`; | |
| 77 | − return toEventDtos(rows); | |
| 162 | + return eventsWhere(sql`e.operator_id = ${operatorId} or (e.entity_type = 'operator' and e.entity_id = ${operatorId})`, limit); | |
| 78 | 163 | } |
| 79 | 164 | |
| 80 | 165 | export async function eventsForCountry(iso2: string, limit = 20): Promise<EventDTO[]> { |
| 81 | 166 | const sql = pg(); |
| 82 | − const rows = await sql<Row[]>` | |
| 83 | − select ${eventCols(sql)} from events e ${eventJoins(sql)} | |
| 84 | − where e.country_iso2 = ${iso2} and e.review_status <> 'rejected' | |
| 85 | − order by e.detected_at desc limit ${limit}`; | |
| 86 | − return toEventDtos(rows); | |
| 167 | + return eventsWhere(sql`e.country_iso2 = ${iso2}`, limit); | |
| 87 | 168 | } |
| 88 | 169 | |
| 89 | 170 | export async function eventsForMetro(metroId: string, limit = 20): Promise<EventDTO[]> { |
| 90 | 171 | const sql = pg(); |
| 91 | − const rows = await sql<Row[]>` | |
| 92 | − select ${eventCols(sql)} from events e ${eventJoins(sql)} | |
| 93 | − where (e.metro_id = ${metroId} or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = ${metroId}))) and e.review_status <> 'rejected' | |
| 94 | − order by e.detected_at desc limit ${limit}`; | |
| 95 | − return toEventDtos(rows); | |
| 172 | + return eventsWhere(sql`e.metro_id = ${metroId} or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = ${metroId}))`, limit); | |
| 96 | 173 | } |
| 97 | 174 | |
| 98 | 175 | export async function eventsForProject(projectId: string, limit = 50): Promise<EventDTO[]> { |
| 99 | 176 | const sql = pg(); |
| 100 | − const rows = await sql<Row[]>` | |
| 101 | − select ${eventCols(sql)} from events e ${eventJoins(sql)} | |
| 102 | − where (e.project_id = ${projectId} or (e.entity_type = 'project' and e.entity_id = ${projectId})) and e.review_status <> 'rejected' | |
| 103 | − order by e.detected_at desc limit ${limit}`; | |
| 104 | − return toEventDtos(rows); | |
| 177 | + return eventsWhere(sql`e.project_id = ${projectId} or (e.entity_type = 'project' and e.entity_id = ${projectId})`, limit); | |
| 105 | 178 | } |
| 106 | 179 | |
| 107 | 180 | export async function latestEvents(limit = 20, minSignificance = 0): Promise<EventDTO[]> { |
| 108 | 181 | const sql = pg(); |
| 109 | − const rows = await sql<Row[]>` | |
| 110 | − select ${eventCols(sql)} from events e ${eventJoins(sql)} | |
| 111 | − where e.review_status <> 'rejected' and e.significance >= ${minSignificance} | |
| 112 | − order by e.detected_at desc limit ${limit}`; | |
| 113 | − return toEventDtos(rows); | |
| 182 | + return eventsWhere(sql`e.significance >= ${minSignificance}`, limit); | |
| 114 | 183 | } |
added
apps/api/src/repositories/explore.ts
+198 −0
@@ -0,0 +1,198 @@ | ||
| 1 | +/** | |
| 2 | + * /explore — structured query over facilities or live projects with facets, charts (containment-aware MW) and | |
| 3 | + * map points. Facets and charts are computed on the FULL filtered set; `items` is one page. | |
| 4 | + */ | |
| 5 | +import type { ExploreQuery, ExploreResponse, MapPoint } from "@dci/core"; | |
| 6 | +import { pg, andAll, facilityJoins, facilitySummaryCols, facilityView, knownMwAgg, pipelineMwAgg, countedAgg, hasMwAgg, mwExpr, projectJoins, projectSummaryCols, projectLive, yearExpr, likePattern, page, PIPELINE_SET, OPERATIONAL_SET, type Fragment, type Sql } from "../lib/sql.js"; | |
| 7 | +import { int, num, reqStr, str, type Row } from "../lib/rows.js"; | |
| 8 | +import { asPrecision, asStatus, asType, facilitySummary, projectSummary, round2, share } from "../lib/dto.js"; | |
| 9 | +import { csv } from "../lib/http.js"; | |
| 10 | +import { facilityConds, toFacilityQuery } from "./facilities.js"; | |
| 11 | +import { MAX_POINTS } from "./map.js"; | |
| 12 | + | |
| 13 | +export const EXPLORE_METHODOLOGY = "Facets and charts cover the whole filtered set (items are one page). Facility MW in charts is containment-aware (a campus and its buildings are never both summed) and uses published figures only: IT capacity, else total power, else planned power. Project MW is the planned MW published on live project records (false positives hidden by review and merged duplicates excluded). mwCoverage is the share of rows with any MW figure. Map points are capped at 5 000 (degraded = true when the cap was hit)."; | |
| 14 | + | |
| 15 | +const AI_LEVELS_FOR: Record<string, string[]> = { confirmed: ["confirmed"], likely: ["confirmed", "likely"], associated: ["confirmed", "likely", "associated"] }; | |
| 16 | + | |
| 17 | +function aiCond(sql: Sql, alias: Fragment, ai: ExploreQuery["ai"], legacyFlag: Fragment): Fragment | null { | |
| 18 | + if (!ai) return null; | |
| 19 | + if (ai === "any") return sql`(${alias}.ai_evidence <> 'unknown' or ${legacyFlag})`; | |
| 20 | + const levels = AI_LEVELS_FOR[ai] ?? ["confirmed", "likely"]; | |
| 21 | + return ai === "confirmed" ? sql`${alias}.ai_evidence = any(${levels})` : sql`(${alias}.ai_evidence = any(${levels}) or ${legacyFlag})`; | |
| 22 | +} | |
| 23 | + | |
| 24 | +function facilityExploreConds(sql: Sql, q: ExploreQuery): Fragment[] { | |
| 25 | + const conds = facilityConds(sql, toFacilityQuery({ q: q.q, country: q.country, metro: q.metro, operator: q.operator, status: q.status, type: q.type, min_mw: q.min_mw, max_mw: q.max_mw, hyperscale: q.hyperscale, has_mw: q.has_mw, confidence: q.confidence, opened_from: q.opened_from, opened_to: q.opened_to })); | |
| 26 | + const ai = aiCond(sql, sql`f`, q.ai, sql`f.is_ai`); | |
| 27 | + if (ai) conds.push(ai); | |
| 28 | + const prec = csv(q.location_precision); | |
| 29 | + if (prec.length) conds.push(sql`f.geo_precision = any(${prec})`); | |
| 30 | + const expected = yearExpr(sql, sql`coalesce(f.opened_on, f.construction_started_on, f.announced_on)`); | |
| 31 | + if (q.expected_before != null) conds.push(sql`f.status = any(${PIPELINE_SET}) and ${expected} <= ${q.expected_before}`); | |
| 32 | + if (q.expected_after != null) conds.push(sql`f.status = any(${PIPELINE_SET}) and ${expected} >= ${q.expected_after}`); | |
| 33 | + if (q.announced_since) conds.push(sql`f.announced_on >= ${q.announced_since}`); | |
| 34 | + return conds; | |
| 35 | +} | |
| 36 | + | |
| 37 | +function projectExploreConds(sql: Sql, q: ExploreQuery): Fragment[] { | |
| 38 | + const conds: Fragment[] = [projectLive(sql)]; | |
| 39 | + const statuses = csv(q.project_status ?? q.status); | |
| 40 | + if (statuses.length) conds.push(sql`p.status = any(${statuses})`); | |
| 41 | + const countries = csv(q.country).map((c) => c.toUpperCase()); | |
| 42 | + if (countries.length) conds.push(sql`p.country_iso2 = any(${countries})`); | |
| 43 | + if (q.metro) conds.push(sql`(m.slug = ${q.metro} or m.id = ${q.metro})`); | |
| 44 | + if (q.operator) conds.push(sql`(o.slug = ${q.operator} or o.id = ${q.operator})`); | |
| 45 | + if (q.min_mw != null) conds.push(sql`p.planned_mw >= ${q.min_mw}`); | |
| 46 | + if (q.max_mw != null) conds.push(sql`p.planned_mw <= ${q.max_mw}`); | |
| 47 | + const ai = aiCond(sql, sql`p`, q.ai, sql`p.is_ai`); | |
| 48 | + if (ai) conds.push(ai); | |
| 49 | + if (q.hyperscale === true) conds.push(sql`exists (select 1 from operators ho where ho.id = p.operator_id and ho.kind = 'hyperscaler')`); | |
| 50 | + const expected = yearExpr(sql, sql`p.expected_opening`); | |
| 51 | + if (q.expected_before != null) conds.push(sql`${expected} <= ${q.expected_before}`); | |
| 52 | + if (q.expected_after != null) conds.push(sql`${expected} >= ${q.expected_after}`); | |
| 53 | + if (q.announced_since) conds.push(sql`p.announced_on >= ${q.announced_since}`); | |
| 54 | + const conf = csv(q.confidence); | |
| 55 | + if (conf.length) conds.push(sql`p.confidence = any(${conf})`); | |
| 56 | + const cls = csv(q.project_class); | |
| 57 | + if (cls.length) conds.push(sql`p.project_class = any(${cls})`); | |
| 58 | + const prec = csv(q.location_precision); | |
| 59 | + if (prec.length) conds.push(sql`p.geo_precision = any(${prec})`); | |
| 60 | + if (q.has_mw === true) conds.push(sql`p.planned_mw is not null`); | |
| 61 | + if (q.has_mw === false) conds.push(sql`p.planned_mw is null`); | |
| 62 | + if (q.q) conds.push(sql`(p.name ilike ${likePattern(q.q)} or o.name ilike ${likePattern(q.q)} or p.city ilike ${likePattern(q.q)})`); | |
| 63 | + return conds; | |
| 64 | +} | |
| 65 | + | |
| 66 | +function facilityOrder(sql: Sql, sort: string | undefined, order: "asc" | "desc" | undefined): Fragment { | |
| 67 | + const asc = (order ?? (sort === "name" ? "asc" : "desc")) === "asc"; | |
| 68 | + const dir = asc ? sql`asc` : sql`desc`; | |
| 69 | + switch (sort) { | |
| 70 | + case "name": return sql`f.name ${dir}, f.id`; | |
| 71 | + case "mw": return sql`${mwExpr(sql)} ${dir} nulls last, f.name`; | |
| 72 | + case "opened": case "opening": return sql`f.opened_on ${dir} nulls last, f.name`; | |
| 73 | + case "announced": return sql`f.announced_on ${dir} nulls last, f.name`; | |
| 74 | + case "completeness": return sql`f.completeness ${dir}, f.name`; | |
| 75 | + default: return sql`f.updated_at ${dir}, f.id`; | |
| 76 | + } | |
| 77 | +} | |
| 78 | + | |
| 79 | +function projectOrder(sql: Sql, sort: string | undefined, order: "asc" | "desc" | undefined): Fragment { | |
| 80 | + const asc = (order ?? (sort === "name" ? "asc" : "desc")) === "asc"; | |
| 81 | + const dir = asc ? sql`asc` : sql`desc`; | |
| 82 | + switch (sort) { | |
| 83 | + case "name": return sql`p.name ${dir}, p.id`; | |
| 84 | + case "mw": return sql`p.planned_mw ${dir} nulls last, p.name`; | |
| 85 | + case "announced": return sql`p.announced_on ${dir} nulls last, p.name`; | |
| 86 | + case "opening": case "opened": return sql`p.expected_opening ${dir} nulls last, p.name`; | |
| 87 | + default: return sql`p.last_update ${dir}, p.id`; | |
| 88 | + } | |
| 89 | +} | |
| 90 | + | |
| 91 | +export async function explore(q: ExploreQuery): Promise<ExploreResponse> { | |
| 92 | + const sql = pg(); | |
| 93 | + const pg_ = page(q.page, q.per_page, 100, 24); | |
| 94 | + const entity = q.entity ?? "facilities"; | |
| 95 | + if (entity === "projects") { | |
| 96 | + const where = andAll(sql, projectExploreConds(sql, q)); | |
| 97 | + const [items, facets, charts, pts, cov] = await Promise.all([ | |
| 98 | + sql<Row[]>`select ${projectSummaryCols(sql)}, count(*) over() as total from projects p ${projectJoins(sql)} where ${where} order by ${projectOrder(sql, q.sort, q.order)} limit ${pg_.perPage} offset ${pg_.offset}`, | |
| 99 | + Promise.all([ | |
| 100 | + sql<Row[]>`select p.status as key, count(*)::int as n from projects p ${projectJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 101 | + sql<Row[]>`select p.country_iso2 as key, c.name, count(*)::int as n from projects p ${projectJoins(sql)} left join countries c on c.iso2 = p.country_iso2 where ${where} and p.country_iso2 is not null group by 1, 2 order by n desc limit 30`, | |
| 102 | + sql<Row[]>`select o.slug as key, o.name, count(*)::int as n from projects p ${projectJoins(sql)} where ${where} and o.id is not null group by 1, 2 order by n desc limit 15`, | |
| 103 | + sql<Row[]>`select p.ai_evidence as key, count(*)::int as n from projects p ${projectJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 104 | + ]), | |
| 105 | + Promise.all([ | |
| 106 | + sql<Row[]>`select p.status as key, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p ${projectJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 107 | + sql<Row[]>`select p.country_iso2 as key, c.name, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p ${projectJoins(sql)} left join countries c on c.iso2 = p.country_iso2 where ${where} and p.country_iso2 is not null group by 1, 2 order by n desc limit 15`, | |
| 108 | + sql<Row[]>`select ${yearExpr(sql, sql`p.expected_opening`)} as year, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p ${projectJoins(sql)} where ${where} and p.expected_opening ~ '^\\d{4}' group by 1 order by 1`, | |
| 109 | + ]), | |
| 110 | + sql<Row[]>`select p.id, p.slug, p.name, o.name as op_name, p.lat, p.lng, p.status, p.planned_mw, p.geo_precision, p.is_ai, p.ai_evidence, p.country_iso2, ${yearExpr(sql, sql`p.expected_opening`)} as y from projects p ${projectJoins(sql)} where ${where} and p.lat is not null and p.lng is not null order by p.planned_mw desc nulls last limit ${MAX_POINTS + 1}`, | |
| 111 | + sql<Row[]>`select count(*)::int as n, count(*) filter (where p.planned_mw is not null)::int as with_mw from projects p ${projectJoins(sql)} where ${where}`, | |
| 112 | + ]); | |
| 113 | + const [fStatus, fCountry, fOperator, fAi] = facets; | |
| 114 | + const [cStatus, cCountry, cYear] = charts; | |
| 115 | + const degraded = pts.length > MAX_POINTS; | |
| 116 | + const points: MapPoint[] = (degraded ? [] : pts).map((r) => { | |
| 117 | + const p: MapPoint = { id: reqStr(r.id), slug: reqStr(r.slug), n: reqStr(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: "unknown", mw: num(r.planned_mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2), k: "project" }; | |
| 118 | + const ai = str(r.ai_evidence); | |
| 119 | + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1; | |
| 120 | + const y = num(r.y); | |
| 121 | + if (y != null) p.y = y; | |
| 122 | + return p; | |
| 123 | + }); | |
| 124 | + return { | |
| 125 | + query: q, | |
| 126 | + total: items.length ? int(items[0]!.total) : int(cov[0]?.n), | |
| 127 | + items: items.map(projectSummary), | |
| 128 | + facets: { | |
| 129 | + status: fStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n) })), | |
| 130 | + country: fCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n) })), | |
| 131 | + operator: fOperator.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name), count: int(r.n) })), | |
| 132 | + ai: fAi.map((r) => ({ key: reqStr(r.key, "unknown"), count: int(r.n) })), | |
| 133 | + }, | |
| 134 | + charts: { | |
| 135 | + byStatus: cStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 136 | + byCountry: cCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 137 | + byYear: cYear.map((r) => ({ year: int(r.year), count: int(r.n), mw: round2(num(r.mw)) })).filter((x) => x.year > 0), | |
| 138 | + }, | |
| 139 | + map: { points, total: degraded ? int(cov[0]?.n) : points.length, degraded }, | |
| 140 | + mwCoverage: share(int(cov[0]?.with_mw), int(cov[0]?.n)), | |
| 141 | + }; | |
| 142 | + } | |
| 143 | + | |
| 144 | + const conds = facilityExploreConds(sql, q); | |
| 145 | + const where = andAll(sql, conds); | |
| 146 | + const known = knownMwAgg(sql), pipe = pipelineMwAgg(sql), counted = countedAgg(sql), hasMw = hasMwAgg(sql); | |
| 147 | + const chartMw = sql`sum(case when f.status = any(${OPERATIONAL_SET}) then ${known} else ${pipe} end)::float`; | |
| 148 | + const [items, facets, charts, pts, cov] = await Promise.all([ | |
| 149 | + sql<Row[]>`select ${facilitySummaryCols(sql)}, count(*) over() as total from facilities f ${facilityJoins(sql)} where ${where} order by ${facilityOrder(sql, q.sort, q.order)} limit ${pg_.perPage} offset ${pg_.offset}`, | |
| 150 | + Promise.all([ | |
| 151 | + sql<Row[]>`select f.status as key, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 152 | + sql<Row[]>`select f.country_iso2 as key, c.name, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} and f.country_iso2 is not null group by 1, 2 order by n desc limit 30`, | |
| 153 | + sql<Row[]>`select o.slug as key, o.name, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} and o.id is not null group by 1, 2 order by n desc limit 15`, | |
| 154 | + sql<Row[]>`select f.facility_type as key, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 155 | + sql<Row[]>`select f.ai_evidence as key, count(*)::int as n from facilities f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 156 | + ]), | |
| 157 | + Promise.all([ | |
| 158 | + sql<Row[]>`select f.status as key, count(*) filter (where ${counted})::int as n, ${chartMw} as mw from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where} group by 1 order by n desc`, | |
| 159 | + sql<Row[]>`select f.country_iso2 as key, c.name, count(*) filter (where ${counted})::int as n, ${chartMw} as mw from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where} and f.country_iso2 is not null group by 1, 2 order by n desc limit 15`, | |
| 160 | + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${counted})::int as n, sum(${known})::float as mw from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 161 | + ]), | |
| 162 | + sql<Row[]>`select f.id, f.slug, f.name, o.name as op_name, f.lat, f.lng, f.status, f.facility_type, ${mwExpr(sql)} as mw, f.geo_precision, f.is_ai, f.ai_evidence, f.is_hyperscale, f.country_iso2, f.record_scope, ${yearExpr(sql, sql`f.opened_on`)} as y from facilities f ${facilityJoins(sql)} where ${where} and f.lat is not null and f.lng is not null order by ${mwExpr(sql)} desc nulls last, f.id limit ${MAX_POINTS + 1}`, | |
| 163 | + sql<Row[]>`select count(*) filter (where ${counted})::int as n, count(*) filter (where ${counted} and ${hasMw})::int as with_mw, count(*)::int as rows from ${facilityView(sql)} f ${facilityJoins(sql)} where ${where}`, | |
| 164 | + ]); | |
| 165 | + const [fStatus, fCountry, fOperator, fType, fAi] = facets; | |
| 166 | + const [cStatus, cCountry, cYear] = charts; | |
| 167 | + const degraded = pts.length > MAX_POINTS; | |
| 168 | + const points: MapPoint[] = (degraded ? [] : pts).map((r) => { | |
| 169 | + const p: MapPoint = { id: reqStr(r.id), slug: reqStr(r.slug), n: reqStr(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: asType(r.facility_type), mw: num(r.mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2) }; | |
| 170 | + const ai = str(r.ai_evidence); | |
| 171 | + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1; | |
| 172 | + if (r.is_hyperscale === true) p.hs = 1; | |
| 173 | + const y = num(r.y); | |
| 174 | + if (y != null) p.y = y; | |
| 175 | + const rs = str(r.record_scope); | |
| 176 | + if (rs === "building" || rs === "campus") p.rs = rs; | |
| 177 | + return p; | |
| 178 | + }); | |
| 179 | + return { | |
| 180 | + query: q, | |
| 181 | + total: items.length ? int(items[0]!.total) : int(cov[0]?.rows), | |
| 182 | + items: items.map(facilitySummary), | |
| 183 | + facets: { | |
| 184 | + status: fStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n) })), | |
| 185 | + country: fCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n) })), | |
| 186 | + operator: fOperator.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name), count: int(r.n) })), | |
| 187 | + type: fType.map((r) => ({ key: reqStr(r.key), count: int(r.n) })), | |
| 188 | + ai: fAi.map((r) => ({ key: reqStr(r.key, "unknown"), count: int(r.n) })), | |
| 189 | + }, | |
| 190 | + charts: { | |
| 191 | + byStatus: cStatus.map((r) => ({ key: reqStr(r.key), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 192 | + byCountry: cCountry.map((r) => ({ key: reqStr(r.key), name: reqStr(r.name, reqStr(r.key)), count: int(r.n), mw: round2(num(r.mw)) })), | |
| 193 | + byYear: cYear.map((r) => ({ year: int(r.year), count: int(r.n), mw: round2(num(r.mw)) })).filter((x) => x.year > 0), | |
| 194 | + }, | |
| 195 | + map: { points, total: degraded ? int(cov[0]?.rows) : points.length, degraded }, | |
| 196 | + mwCoverage: share(int(cov[0]?.with_mw), int(cov[0]?.n)), | |
| 197 | + }; | |
| 198 | +} | |
modified
apps/api/src/repositories/facilities.ts
+122 −19
@@ -1,10 +1,12 @@ | ||
| 1 | −import type { CloudRegionSummary, FacilityDetail, FacilityFilters, FacilitySummary, ProvenanceDTO, SourceRef } from "@dci/core"; | |
| 2 | −import { pg, andAll, facilityJoins, facilitySummaryCols, mwExpr, page, bboxAround, haversineExpr, yearExpr, likePattern, PIPELINE_SET, cloudRegionCols, type Fragment, type Sql } from "../lib/sql.js"; | |
| 3 | −import { int, num, record, reqIso, str, strArray, type Row } from "../lib/rows.js"; | |
| 4 | −import { cloudRegionSummary, facilitySummary, provenanceDto } from "../lib/dto.js"; | |
| 1 | +import type { ClaimDTO, CloudRegionSummary, EntityHistory, EventDTO, FacilityDetail, FacilityFilters, FacilitySummary, GridConstraintDTO, ProvenanceDTO, SourceRef } from "@dci/core"; | |
| 2 | +import { pg, andAll, facilityJoins, facilitySummaryCols, mwExpr, page, bboxAround, haversineExpr, yearExpr, likePattern, PIPELINE_SET, cloudRegionCols, gridConstraintCols, gridConstraintJoins, POWER_EVENT_TYPES, GRID_EVENT_TYPES, AI_LEVELS, type Fragment, type Sql } from "../lib/sql.js"; | |
| 3 | +import { int, num, record, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js"; | |
| 4 | +import { cloudRegionSummary, facilitySummary, gridConstraintDto, gridConstraintFromEvent, provenanceDto } from "../lib/dto.js"; | |
| 5 | 5 | import { csv } from "../lib/http.js"; |
| 6 | 6 | import { findBySlugOrId } from "../lib/resolve.js"; |
| 7 | −import { eventsForEntity } from "./events.js"; | |
| 7 | +import { nearby } from "../lib/nearby.js"; | |
| 8 | +import { capacityHistory, claimsFor, dataQualityFor, entityHistory, eventsForSubject, provenanceAll } from "../lib/quality.js"; | |
| 9 | +import { eventsForEntity, eventsWhere } from "./events.js"; | |
| 8 | 10 | import { projectsForFacility } from "./projects.js"; |
| 9 | 11 | import { buildSourceHistory, documentVersionsFor, sourceIdsOf, sourceRefsFor } from "../lib/source-history.js"; |
| 10 | 12 | |
@@ -21,6 +23,12 @@ export interface FacilityQuery extends Omit<FacilityFilters, "country" | "status | ||
| 21 | 23 | metroId?: string; |
| 22 | 24 | countryIso2?: string; |
| 23 | 25 | operatorId?: string; |
| 26 | + /** extra: AI evidence levels (confirmed, likely, associated, unknown) */ | |
| 27 | + aiEvidence?: string[]; | |
| 28 | + /** extra: geo precision values */ | |
| 29 | + locationPrecision?: string[]; | |
| 30 | + /** extra: record scope (building | facility | campus) */ | |
| 31 | + recordScope?: string[]; | |
| 24 | 32 | } |
| 25 | 33 | |
| 26 | 34 | /** Translate raw FacilityFilters (csv strings) into a FacilityQuery. */ |
@@ -63,8 +71,11 @@ export function facilityConds(sql: Sql, q: FacilityQuery): Fragment[] { | ||
| 63 | 71 | if (q.hyperscale === true) conds.push(sql`(f.is_hyperscale or f.facility_type = 'hyperscale')`); |
| 64 | 72 | if (q.hyperscale === false) conds.push(sql`(not f.is_hyperscale and f.facility_type <> 'hyperscale')`); |
| 65 | 73 | if (q.colocation === true) conds.push(sql`f.facility_type in ('colocation', 'carrier_hotel', 'wholesale')`); |
| 66 | − if (q.ai === true) conds.push(sql`(f.is_ai or f.facility_type = 'ai')`); | |
| 67 | − if (q.ai === false) conds.push(sql`(not f.is_ai and f.facility_type <> 'ai')`); | |
| 74 | + if (q.ai === true) conds.push(sql`(f.is_ai or f.facility_type = 'ai' or f.ai_evidence = any(${AI_LEVELS}))`); | |
| 75 | + if (q.ai === false) conds.push(sql`(not f.is_ai and f.facility_type <> 'ai' and f.ai_evidence <> all(${AI_LEVELS}))`); | |
| 76 | + if (q.aiEvidence?.length) conds.push(sql`f.ai_evidence = any(${q.aiEvidence})`); | |
| 77 | + if (q.locationPrecision?.length) conds.push(sql`f.geo_precision = any(${q.locationPrecision})`); | |
| 78 | + if (q.recordScope?.length) conds.push(sql`f.record_scope = any(${q.recordScope})`); | |
| 68 | 79 | if (q.renewable === true) conds.push(sql`f.renewable_claim is not null`); |
| 69 | 80 | if (q.has_mw === true) conds.push(sql`${mw} is not null`); |
| 70 | 81 | if (q.has_mw === false) conds.push(sql`${mw} is null`); |
@@ -113,6 +124,13 @@ export async function facilitiesByIds(ids: string[]): Promise<FacilitySummary[]> | ||
| 113 | 124 | return ids.map((id) => by.get(id)).filter((x): x is FacilitySummary => Boolean(x)); |
| 114 | 125 | } |
| 115 | 126 | |
| 127 | +/** Facilities matching an arbitrary condition over `facilities f` (+ facilityJoins aliases), ordered by MW. */ | |
| 128 | +export async function facilitiesWhere(cond: Fragment, limit = 20, orderBy?: Fragment): Promise<FacilitySummary[]> { | |
| 129 | + const sql = pg(); | |
| 130 | + const rows = await sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and (${cond}) order by ${orderBy ?? sql`${mwExpr(sql)} desc nulls last, f.name`} limit ${limit}`; | |
| 131 | + return rows.map(facilitySummary); | |
| 132 | +} | |
| 133 | + | |
| 116 | 134 | export async function recentlyVerifiedFacilities(limit = 10): Promise<FacilitySummary[]> { |
| 117 | 135 | const sql = pg(); |
| 118 | 136 | const rows = await sql<Row[]>` |
@@ -144,42 +162,99 @@ async function cloudRegionsInMetro(metroId: string | null, limit = 12): Promise< | ||
| 144 | 162 | export async function provenanceFor(entityType: string, entityId: string, limit = 400): Promise<ProvenanceDTO[]> { |
| 145 | 163 | const sql = pg(); |
| 146 | 164 | const rows = await sql<Row[]>` |
| 147 | − select p.field, p.value, p.source_id, s.name as source_name, s.kind as source_kind, p.url, p.first_observed, p.last_observed, p.retrieved_at, p.confidence, p.is_estimate, p.method | |
| 165 | + select p.field, p.value, p.source_id, s.name as source_name, s.kind as source_kind, p.url, p.first_observed, p.last_observed, p.retrieved_at, p.confidence, p.is_estimate, p.method, p.is_winner, p.scope, p.run_id, p.document_id | |
| 148 | 166 | from provenance p left join sources s on s.id = p.source_id |
| 149 | 167 | where p.entity_type = ${entityType} and p.entity_id = ${entityId} and p.is_current |
| 150 | − order by p.field, p.last_observed desc limit ${limit}`; | |
| 168 | + order by p.field, p.is_winner desc, p.last_observed desc limit ${limit}`; | |
| 151 | 169 | return rows.map(provenanceDto); |
| 152 | 170 | } |
| 153 | 171 | |
| 154 | −export async function getFacilityDetail(idOrSlug: string): Promise<{ detail: FacilityDetail; sources: SourceRef[] } | null> { | |
| 172 | +/** Grid constraints (grid_constraints rows + grid / utility / power events) for a metro and/or country. */ | |
| 173 | +export async function gridConstraintsFor(scope: { metroId?: string | null; countryIso2?: string | null }, limit = 20): Promise<GridConstraintDTO[]> { | |
| 174 | + const sql = pg(); | |
| 175 | + const conds: Fragment[] = []; | |
| 176 | + if (scope.metroId) conds.push(sql`g.metro_id = ${scope.metroId}`); | |
| 177 | + if (scope.countryIso2 && !scope.metroId) conds.push(sql`g.country_iso2 = ${scope.countryIso2}`); | |
| 178 | + if (!conds.length) return []; | |
| 179 | + const [rows, events] = await Promise.all([ | |
| 180 | + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} where ${conds.reduce<Fragment>((a, c) => sql`${a} or ${c}`, sql`false`)} order by g.effective_date desc nulls last, g.created_at desc limit ${limit}`, | |
| 181 | + scope.metroId ? eventsWhere(sql`e.metro_id = ${scope.metroId} and e.event_type = any(${GRID_EVENT_TYPES})`, limit) : Promise.resolve([] as EventDTO[]), | |
| 182 | + ]); | |
| 183 | + const out = rows.map(gridConstraintDto); | |
| 184 | + const seen = new Set(out.map((g) => g.eventId).filter(Boolean)); | |
| 185 | + for (const e of events) if (!seen.has(e.id)) out.push(gridConstraintFromEvent(e)); | |
| 186 | + return out.slice(0, limit); | |
| 187 | +} | |
| 188 | + | |
| 189 | +export const POWER_CONTEXT_NOTE = "utilityCapacityMw / gridConnectionMw are utility-side figures (power available or contracted from the grid), not IT load — never compare them with itCapacityMw. countryEnergy shows national grid averages (renewable share, generation), which do not describe this facility's contracted electricity or power purchase agreements. Grid constraints are public reports attached to the facility's market or country, not to the site itself."; | |
| 190 | + | |
| 191 | +/** Resolve a facility by slug or id, following merged_into to the survivor. */ | |
| 192 | +export async function resolveFacilityRow(idOrSlug: string): Promise<Row | null> { | |
| 155 | 193 | const sql = pg(); |
| 156 | 194 | const base = await findBySlugOrId("facilities", idOrSlug); |
| 157 | 195 | if (!base) return null; |
| 158 | − // follow merges: a merged facility redirects to its survivor | |
| 159 | 196 | const mergedInto = str(base.merged_into); |
| 160 | − const row = mergedInto ? ((await sql<Row[]>`select * from facilities where id = ${mergedInto} limit 1`)[0] ?? base) : base; | |
| 197 | + return mergedInto ? ((await sql<Row[]>`select * from facilities where id = ${mergedInto} limit 1`)[0] ?? base) : base; | |
| 198 | +} | |
| 199 | + | |
| 200 | +export async function getFacilityDetail(idOrSlug: string, opts: { radiusKm?: number } = {}): Promise<{ detail: FacilityDetail; sources: SourceRef[] } | null> { | |
| 201 | + const sql = pg(); | |
| 202 | + const row = await resolveFacilityRow(idOrSlug); | |
| 203 | + if (!row) return null; | |
| 161 | 204 | const id = String(row.id); |
| 162 | − const [sumRows, aliasRows, tenantRows, ixpRows, cloudRegions, nearby, projects, provenance, events, versions, ownerRows, campusRows] = await Promise.all([ | |
| 205 | + const lat = num(row.lat), lng = num(row.lng); | |
| 206 | + const metroId = str(row.metro_id); | |
| 207 | + const countryIso2 = str(row.country_iso2); | |
| 208 | + const operatorId = str(row.operator_id); | |
| 209 | + const [sumRows, aliasRows, tenantRows, ixpRows, cloudRegions, nearbyRows, projects, provenance, events, versions, ownerRows, campusRows, buildingRows, roleRows, claims, nearbyInfra, powerEvents, gridConstraints, energyRows, dataQuality] = await Promise.all([ | |
| 163 | 210 | sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.id = ${id}`, |
| 164 | 211 | sql<Row[]>`select alias from facility_aliases where facility_id = ${id} order by alias`, |
| 165 | 212 | sql<Row[]>`select t.role, t.asn, o.id, o.slug, o.name from facility_tenants t join operators o on o.id = t.operator_id where t.facility_id = ${id} order by o.name`, |
| 166 | 213 | sql<Row[]>`select x.id, x.slug, x.name from facility_ixps fx join ixps x on x.id = fx.ixp_id where fx.facility_id = ${id} order by x.name`, |
| 167 | − cloudRegionsInMetro(str(row.metro_id)), | |
| 168 | − num(row.lat) != null && num(row.lng) != null ? nearbyFacilities(id, num(row.lat)!, num(row.lng)!) : Promise.resolve([] as FacilitySummary[]), | |
| 169 | − projectsForFacility(id, str(row.operator_id), str(row.metro_id)), | |
| 214 | + cloudRegionsInMetro(metroId), | |
| 215 | + lat != null && lng != null ? nearbyFacilities(id, lat, lng) : Promise.resolve([] as FacilitySummary[]), | |
| 216 | + projectsForFacility(id, operatorId, metroId), | |
| 170 | 217 | provenanceFor("facility", id), |
| 171 | 218 | eventsForEntity("facility", id, 50), |
| 172 | 219 | documentVersionsFor("facility", id), |
| 173 | 220 | row.owner_id ? sql<Row[]>`select id, slug, name from operators where id = ${String(row.owner_id)}` : Promise.resolve([] as Row[]), |
| 174 | 221 | row.campus_id ? sql<Row[]>`select id, slug, name from campuses where id = ${String(row.campus_id)}` : Promise.resolve([] as Row[]), |
| 222 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.parent_facility_id = ${id} and f.merged_into is null order by f.name`, | |
| 223 | + sql<Row[]>`select o.id, o.slug, o.name, (o.id = ${str(row.developer_id)}) as is_developer, (o.id = ${str(row.landowner_id)}) as is_landowner from operators o where o.id in (${str(row.developer_id) ?? ""}, ${str(row.landowner_id) ?? ""})`, | |
| 224 | + claimsFor("facility", id), | |
| 225 | + lat != null && lng != null ? nearby({ lat, lng, radiusKm: opts.radiusKm ?? 25, excludeFacilityId: id, limitPerType: 50 }) : Promise.resolve(null), | |
| 226 | + eventsWhere(sql`e.event_type = any(${POWER_EVENT_TYPES}) and ((e.entity_type = 'facility' and e.entity_id = ${id}) ${operatorId && metroId ? sql`or (e.operator_id = ${operatorId} and e.metro_id = ${metroId})` : sql``})`, 20), | |
| 227 | + gridConstraintsFor({ metroId, countryIso2 }), | |
| 228 | + countryIso2 ? sql<Row[]>`select renewable_share, electricity_twh, stats_year from countries where iso2 = ${countryIso2}` : Promise.resolve([] as Row[]), | |
| 229 | + dataQualityFor("facility", id, int(row.completeness), str(row.last_verified)), | |
| 175 | 230 | ]); |
| 176 | 231 | const summary = facilitySummary(sumRows[0] ?? row); |
| 177 | 232 | const carriers = tenantRows.filter((t) => t.role !== "cloud").map((t) => ({ id: String(t.id), slug: String(t.slug), name: String(t.name), asn: num(t.asn) })); |
| 178 | 233 | const cloudProviders = tenantRows.filter((t) => t.role === "cloud").map((t) => ({ id: String(t.id), slug: String(t.slug), name: String(t.name) })); |
| 179 | 234 | const owner = ownerRows[0] ? { id: String(ownerRows[0].id), slug: String(ownerRows[0].slug), name: String(ownerRows[0].name) } : null; |
| 180 | 235 | const campus = campusRows[0] ? { id: String(campusRows[0].id), slug: String(campusRows[0].slug), name: String(campusRows[0].name) } : null; |
| 236 | + const refOf = (r: Row | undefined) => (r ? { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) } : null); | |
| 237 | + const developer = refOf(roleRows.find((r) => r.is_developer === true)); | |
| 238 | + const landowner = refOf(roleRows.find((r) => r.is_landowner === true)); | |
| 239 | + const energy = energyRows[0]; | |
| 181 | 240 | const detail: FacilityDetail = { |
| 182 | 241 | ...summary, |
| 242 | + buildings: buildingRows.map(facilitySummary), | |
| 243 | + developer, | |
| 244 | + landowner, | |
| 245 | + tenants: tenantRows.map((t) => ({ id: String(t.id), slug: String(t.slug), name: String(t.name), role: reqStr(t.role, "tenant") })), | |
| 246 | + claims, | |
| 247 | + capacityHistory: capacityHistory(provenance, claims, events), | |
| 248 | + nearbyInfrastructure: nearbyInfra, | |
| 249 | + powerContext: { | |
| 250 | + utilityCapacityMw: num(row.utility_capacity_mw), | |
| 251 | + gridConnectionMw: num(row.grid_connection_mw), | |
| 252 | + powerEvents, | |
| 253 | + gridConstraints, | |
| 254 | + countryEnergy: energy ? { renewableShare: num(energy.renewable_share), electricityTwh: num(energy.electricity_twh), statsYear: num(energy.stats_year) } : null, | |
| 255 | + note: POWER_CONTEXT_NOTE, | |
| 256 | + }, | |
| 257 | + dataQuality, | |
| 183 | 258 | aliases: aliasRows.map((a) => String(a.alias)), |
| 184 | 259 | owner, |
| 185 | 260 | campus, |
@@ -202,7 +277,7 @@ export async function getFacilityDetail(idOrSlug: string): Promise<{ detail: Fac | ||
| 202 | 277 | cloudProviders, |
| 203 | 278 | ixps: ixpRows.map((x) => ({ id: String(x.id), slug: String(x.slug), name: String(x.name) })), |
| 204 | 279 | cloudRegions, |
| 205 | − nearby, | |
| 280 | + nearby: nearbyRows, | |
| 206 | 281 | projects, |
| 207 | 282 | provenance, |
| 208 | 283 | events, |
@@ -211,10 +286,39 @@ export async function getFacilityDetail(idOrSlug: string): Promise<{ detail: Fac | ||
| 211 | 286 | firstSeen: reqIso(row.first_seen), |
| 212 | 287 | updatedAt: reqIso(row.updated_at), |
| 213 | 288 | }; |
| 214 | − const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions)); | |
| 289 | + const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions, claims)); | |
| 215 | 290 | return { detail, sources }; |
| 216 | 291 | } |
| 217 | 292 | |
| 293 | +/** /datacenters/:id/history */ | |
| 294 | +export async function getFacilityHistory(idOrSlug: string): Promise<{ history: EntityHistory; sources: SourceRef[] } | null> { | |
| 295 | + const row = await resolveFacilityRow(idOrSlug); | |
| 296 | + if (!row) return null; | |
| 297 | + const id = String(row.id); | |
| 298 | + const [prov, claims, events] = await Promise.all([provenanceAll("facility", id), claimsFor("facility", id), eventsForSubject("facility", id)]); | |
| 299 | + return { history: entityHistory("facility", id, prov, claims, events), sources: await sourceRefsFor(sourceIdsOf(prov, claims, events)) }; | |
| 300 | +} | |
| 301 | + | |
| 302 | +/** /datacenters/:id/claims */ | |
| 303 | +export async function getFacilityClaims(idOrSlug: string, opts: { status?: string[]; predicate?: string } = {}): Promise<{ id: string; claims: ClaimDTO[]; sources: SourceRef[] } | null> { | |
| 304 | + const row = await resolveFacilityRow(idOrSlug); | |
| 305 | + if (!row) return null; | |
| 306 | + const id = String(row.id); | |
| 307 | + const claims = await claimsFor("facility", id, opts); | |
| 308 | + return { id, claims, sources: await sourceRefsFor(sourceIdsOf(claims)) }; | |
| 309 | +} | |
| 310 | + | |
| 311 | +/** /datacenters/:id/provenance — every observation, current and superseded. */ | |
| 312 | +export async function getFacilityProvenance(idOrSlug: string): Promise<{ id: string; provenance: ProvenanceDTO[]; current: number; sources: SourceRef[] } | null> { | |
| 313 | + const row = await resolveFacilityRow(idOrSlug); | |
| 314 | + if (!row) return null; | |
| 315 | + const id = String(row.id); | |
| 316 | + const provenance = await provenanceAll("facility", id); | |
| 317 | + const sql = pg(); | |
| 318 | + const cur = await sql<Row[]>`select count(*)::int as n from provenance where entity_type = 'facility' and entity_id = ${id} and is_current`; | |
| 319 | + return { id, provenance, current: int(cur[0]?.n), sources: await sourceRefsFor(sourceIdsOf(provenance)) }; | |
| 320 | +} | |
| 321 | + | |
| 218 | 322 | /** Distinct facility cities matching a text (for search). */ |
| 219 | 323 | export async function searchCities(text: string, limit = 5): Promise<Array<{ city: string; countryIso2: string | null; count: number }>> { |
| 220 | 324 | const sql = pg(); |
@@ -226,4 +330,3 @@ export async function searchCities(text: string, limit = 5): Promise<Array<{ cit | ||
| 226 | 330 | order by (lower(f.city) = lower(${text})) desc, n desc limit ${limit}`; |
| 227 | 331 | return rows.map((r) => ({ city: String(r.city), countryIso2: str(r.country_iso2), count: int(r.n) })); |
| 228 | 332 | } |
| 229 | − | |
modified
apps/api/src/repositories/ixps.ts
+43 −16
@@ -1,51 +1,78 @@ | ||
| 1 | −import type { FacilitySummary, IxpSummary } from "@dci/core"; | |
| 2 | −import { pg, facilityJoins, facilitySummaryCols } from "../lib/sql.js"; | |
| 3 | −import { record, reqIso, str, type Row } from "../lib/rows.js"; | |
| 1 | +import type { FacilitySummary, IxpDetail as IxpDetailContract, IxpSummary } from "@dci/core"; | |
| 2 | +import { pg, facilityJoins, facilitySummaryCols, haversineExpr, withinBbox, likePattern } from "../lib/sql.js"; | |
| 3 | +import { num, record, reqIso, reqStr, round, str, type Row } from "../lib/rows.js"; | |
| 4 | 4 | import { facilitySummary, ixpSummary } from "../lib/dto.js"; |
| 5 | 5 | import { findBySlugOrId } from "../lib/resolve.js"; |
| 6 | +import { eventsForEntity } from "./events.js"; | |
| 7 | +import { provenanceFor } from "./facilities.js"; | |
| 8 | +import { buildSourceHistory, documentVersionsFor } from "../lib/source-history.js"; | |
| 6 | 9 | |
| 7 | −const COLS = "x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count"; | |
| 10 | +/** IXP columns joined with the metro (ixps have no coordinates of their own — the metro reference point stands in). */ | |
| 11 | +const COLS = "x.id, x.slug, x.name, x.name_long, x.city, x.country_iso2, x.website, x.network_count, xm.id as met_id, xm.slug as met_slug, xm.name as met_name, xm.lat as lat, xm.lng as lng"; | |
| 8 | 12 | |
| 9 | 13 | export async function listIxps(f: { country?: string; metroId?: string; q?: string }): Promise<IxpSummary[]> { |
| 10 | 14 | const sql = pg(); |
| 11 | 15 | const rows = await sql<Row[]>` |
| 12 | 16 | select ${sql.unsafe(COLS)}, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count |
| 13 | − from ixps x | |
| 17 | + from ixps x left join metros xm on xm.id = x.metro_id | |
| 14 | 18 | where ${f.country ? sql`x.country_iso2 = ${f.country.toUpperCase()}` : sql`true`} |
| 15 | 19 | and ${f.metroId ? sql`x.metro_id = ${f.metroId}` : sql`true`} |
| 16 | − and ${f.q ? sql`(x.name ilike ${"%" + f.q + "%"} or x.name_long ilike ${"%" + f.q + "%"})` : sql`true`} | |
| 20 | + and ${f.q ? sql`(x.name ilike ${likePattern(f.q)} or x.name_long ilike ${likePattern(f.q)})` : sql`true`} | |
| 17 | 21 | order by x.network_count desc nulls last, x.name limit 2000`; |
| 18 | 22 | return rows.map(ixpSummary); |
| 19 | 23 | } |
| 20 | 24 | |
| 21 | −export interface IxpDetail extends IxpSummary { | |
| 25 | +/** Contract IxpDetail + the extra fields the web already reads. */ | |
| 26 | +export interface IxpDetail extends IxpDetailContract { | |
| 22 | 27 | countryName: string | null; |
| 23 | − metro: { id: string; slug: string; name: string } | null; | |
| 24 | 28 | regionContinent: string | null; |
| 25 | − facilities: FacilitySummary[]; | |
| 26 | − externalIds: Record<string, string | number>; | |
| 27 | 29 | updatedAt: string; |
| 28 | 30 | } |
| 29 | 31 | |
| 32 | +export const IXP_COORDS_NOTE = "This IXP has no published coordinates in the index; lat/lng and the nearby facilities use its metro reference point. Traffic figures are not indexed (no licensed source)."; | |
| 33 | + | |
| 30 | 34 | export async function getIxp(idOrSlug: string): Promise<IxpDetail | null> { |
| 31 | 35 | const sql = pg(); |
| 32 | 36 | const row = await findBySlugOrId("ixps", idOrSlug); |
| 33 | 37 | if (!row) return null; |
| 34 | 38 | const id = String(row.id); |
| 35 | 39 | const metroId = str(row.metro_id); |
| 36 | − const [main, facs, metroRows, countryRows] = await Promise.all([ | |
| 37 | − sql<Row[]>`select ${sql.unsafe(COLS)}, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count from ixps x where x.id = ${id}`, | |
| 40 | + const [main, facs, metroRows, countryRows, provenance, events, versions] = await Promise.all([ | |
| 41 | + sql<Row[]>`select ${sql.unsafe(COLS)}, (select count(*)::int from facility_ixps fx where fx.ixp_id = x.id) as facility_count from ixps x left join metros xm on xm.id = x.metro_id where x.id = ${id}`, | |
| 38 | 42 | sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} join facility_ixps fx on fx.facility_id = f.id where fx.ixp_id = ${id} and f.merged_into is null order by f.name limit 200`, |
| 39 | − metroId ? sql<Row[]>`select id, slug, name from metros where id = ${metroId}` : Promise.resolve([] as Row[]), | |
| 43 | + metroId ? sql<Row[]>`select id, slug, name, lat, lng from metros where id = ${metroId}` : Promise.resolve([] as Row[]), | |
| 40 | 44 | row.country_iso2 ? sql<Row[]>`select name from countries where iso2 = ${String(row.country_iso2)}` : Promise.resolve([] as Row[]), |
| 45 | + provenanceFor("ixp", id), | |
| 46 | + eventsForEntity("ixp", id, 50), | |
| 47 | + documentVersionsFor("ixp", id), | |
| 41 | 48 | ]); |
| 49 | + const metro = metroRows[0] ?? null; | |
| 50 | + const lat = metro ? num(metro.lat) : null, lng = metro ? num(metro.lng) : null; | |
| 51 | + const facilities = facs.map(facilitySummary); | |
| 52 | + const linkedIds = facilities.map((f) => f.id); | |
| 53 | + let nearbyFacilities: Array<FacilitySummary & { distanceKm: number }> = []; | |
| 54 | + if (lat != null && lng != null) { | |
| 55 | + const dist = haversineExpr(sql, lat, lng); | |
| 56 | + const rows = await sql<Row[]>`select ${facilitySummaryCols(sql)}, ${dist} as distance_km from facilities f ${facilityJoins(sql)} | |
| 57 | + where f.merged_into is null and f.lat is not null and ${withinBbox(sql, lat, lng, 25, sql`f.lat`, sql`f.lng`)} and ${dist} <= 25 ${linkedIds.length ? sql`and f.id <> all(${linkedIds})` : sql``} | |
| 58 | + order by ${dist} asc limit 20`; | |
| 59 | + nearbyFacilities = rows.map((r) => ({ ...facilitySummary(r), distanceKm: round(num(r.distance_km), 2) ?? 0 })); | |
| 60 | + } | |
| 61 | + const opMap = new Map<string, { id: string; slug: string; name: string; facilityCount: number }>(); | |
| 62 | + for (const f of facilities) if (f.operator) { const cur = opMap.get(f.operator.id) ?? { ...f.operator, facilityCount: 0 }; cur.facilityCount++; opMap.set(f.operator.id, cur); } | |
| 42 | 63 | return { |
| 43 | 64 | ...ixpSummary(main[0] ?? { ...row, facility_count: facs.length }), |
| 65 | + metro: metro ? { id: reqStr(metro.id), slug: reqStr(metro.slug), name: reqStr(metro.name) } : null, | |
| 66 | + lat, | |
| 67 | + lng, | |
| 68 | + facilities, | |
| 69 | + operators: [...opMap.values()].sort((a, b) => b.facilityCount - a.facilityCount || a.name.localeCompare(b.name)), | |
| 70 | + nearbyFacilities, | |
| 71 | + trafficNote: metro ? IXP_COORDS_NOTE : null, | |
| 72 | + externalIds: record(row.external_ids), | |
| 73 | + sourceHistory: buildSourceHistory(provenance, events, versions), | |
| 44 | 74 | countryName: countryRows[0] ? String(countryRows[0].name) : null, |
| 45 | − metro: metroRows[0] ? { id: String(metroRows[0].id), slug: String(metroRows[0].slug), name: String(metroRows[0].name) } : null, | |
| 46 | 75 | regionContinent: str(row.region_continent), |
| 47 | − facilities: facs.map(facilitySummary), | |
| 48 | − externalIds: record(row.external_ids), | |
| 49 | 76 | updatedAt: reqIso(row.updated_at), |
| 50 | 77 | }; |
| 51 | 78 | } |
modified
apps/api/src/repositories/map.ts
+241 −34
@@ -1,5 +1,12 @@ | ||
| 1 | −import type { FacilityStatus, MapCluster, MapPoint, MapResponse } from "@dci/core"; | |
| 2 | −import { pg, andAll, facilityJoins, mwExpr, type Fragment, type Sql } from "../lib/sql.js"; | |
| 1 | +/** | |
| 2 | + * /map data: zoom-tiered clusters / points for facilities, thematic layers (capacity, projects, AI, cloud, IXPs, | |
| 3 | + * connectivity, power, pipeline), density grids and a "time machine" year filter. Only entities with coordinates are | |
| 4 | + * drawn; IXPs without a published location fall back to their metro reference point (p = "metro"). Layers without a | |
| 5 | + * connector (landing stations, substations, power plants) are never drawn — power events are exposed as metro-level | |
| 6 | + * clusters, not as invented plant locations. | |
| 7 | + */ | |
| 8 | +import type { DensityCell, DensityView, FacilityStatus, MapCluster, MapLayer, MapPoint, MapResponse } from "@dci/core"; | |
| 9 | +import { pg, andAll, facilityJoins, facilityView, knownMwAgg, pipelineMwAgg, countedAgg, mwExpr, projectLive, yearExpr, PIPELINE_SET, POWER_EVENT_TYPES, GRID_EVENT_TYPES, type Fragment, type Sql } from "../lib/sql.js"; | |
| 3 | 10 | import { facilityConds, type FacilityQuery } from "./facilities.js"; |
| 4 | 11 | import { int, num, str, type Row } from "../lib/rows.js"; |
| 5 | 12 | import { asPrecision, asStatus, asType } from "../lib/dto.js"; |
@@ -12,15 +19,30 @@ export interface MapQuery { | ||
| 12 | 19 | status?: string[]; |
| 13 | 20 | type?: string[]; |
| 14 | 21 | operator?: string; |
| 22 | + metro?: string; | |
| 15 | 23 | country?: string[]; |
| 24 | + q?: string; | |
| 16 | 25 | min_mw?: number; |
| 17 | 26 | max_mw?: number; |
| 18 | 27 | ai?: boolean; |
| 19 | 28 | hyperscale?: boolean; |
| 29 | + has_mw?: boolean; | |
| 30 | + confidence?: string[]; | |
| 31 | + opened_from?: number; | |
| 32 | + opened_to?: number; | |
| 20 | 33 | cloud_regions?: boolean; |
| 34 | + layer?: MapLayer; | |
| 35 | + density?: DensityView; | |
| 36 | + year?: number; | |
| 37 | + project_status?: string[]; | |
| 38 | + expected_from?: number; | |
| 39 | + expected_to?: number; | |
| 40 | + location_precision?: string[]; | |
| 41 | + ai_evidence?: string[]; | |
| 21 | 42 | } |
| 22 | 43 | |
| 23 | 44 | export const MAX_POINTS = 5000; |
| 45 | +export const MAP_METHODOLOGY = "Only entities with coordinates are drawn; `p` is the geo precision (city/metro-level points are approximate). Facility `mw` = best known figure (IT, else total, else planned) — on the power layer it is utility / grid supply capacity, NOT IT load. `k` marks overlay kinds (project, ixp, cloud_region); IXPs without a published location sit at their metro reference point. Density weights are containment-aware (a campus and its buildings are never both summed). `year` keeps facilities whose opening date is ≤ year; `yearCoverage` is the share of matching facilities that have an opening date at all — undated facilities are absent from every year frame."; | |
| 24 | 46 | |
| 25 | 47 | /** Map mode from zoom: <5 countries, 5–8 grid, ≥9 points. Exported for tests. */ |
| 26 | 48 | export function mapMode(zoom: number): "country" | "grid" | "points" { |
@@ -39,9 +61,46 @@ function bboxCond(sql: Sql, b: Bbox | undefined, latCol: Fragment, lngCol: Fragm | ||
| 39 | 61 | return sql`(${latCol} between ${b.s} and ${b.n} and ${lng})`; |
| 40 | 62 | } |
| 41 | 63 | |
| 64 | +const AI_COND = (sql: Sql) => sql`(f.ai_evidence in ('confirmed', 'likely') or f.is_ai)`; | |
| 65 | +const PIPELINE_PROJECT_STATUSES = ["rumored", "proposed", "announced", "permitting", "approved", "under_construction", "delayed"]; | |
| 66 | + | |
| 67 | +/** Layer-specific facility conditions (on top of the generic filters). */ | |
| 68 | +function layerConds(sql: Sql, q: MapQuery): Fragment[] { | |
| 69 | + const out: Fragment[] = []; | |
| 70 | + switch (q.layer) { | |
| 71 | + case "capacity": out.push(sql`${mwExpr(sql)} is not null`); break; | |
| 72 | + case "ai": out.push(AI_COND(sql)); break; | |
| 73 | + case "power": out.push(sql`(f.utility_capacity_mw is not null or f.grid_connection_mw is not null)`); break; | |
| 74 | + case "pipeline": out.push(sql`f.status = any(${PIPELINE_SET})`); break; | |
| 75 | + case "connectivity": out.push(sql`(f.facility_type = 'carrier_hotel' or coalesce(f.carriers_count, 0) >= 20)`); break; | |
| 76 | + default: break; | |
| 77 | + } | |
| 78 | + if (q.location_precision?.length) out.push(sql`f.geo_precision = any(${q.location_precision})`); | |
| 79 | + if (q.ai_evidence?.length) out.push(sql`f.ai_evidence = any(${q.ai_evidence})`); | |
| 80 | + return out; | |
| 81 | +} | |
| 82 | + | |
| 83 | +/** Time-machine condition: opened_on year ≤ year; pipeline rows only when the caller asked for pipeline statuses. */ | |
| 84 | +function yearCond(sql: Sql, q: MapQuery): Fragment { | |
| 85 | + if (q.year == null) return sql`true`; | |
| 86 | + const opened = sql`${yearExpr(sql, sql`f.opened_on`)} <= ${q.year}`; | |
| 87 | + const pipelineAsked = (q.status ?? []).some((s) => (PIPELINE_SET as string[]).includes(s)); | |
| 88 | + if (!pipelineAsked) return opened; | |
| 89 | + return sql`(${opened} or (f.status = any(${PIPELINE_SET}) and ${yearExpr(sql, sql`coalesce(f.announced_on, f.construction_started_on, f.opened_on)`)} <= ${q.year}))`; | |
| 90 | +} | |
| 91 | + | |
| 92 | +function baseConds(sql: Sql, q: MapQuery): Fragment[] { | |
| 93 | + const fq: FacilityQuery = { status: q.status, type: q.type, operator: q.operator, metro: q.metro, country: q.country, q: q.q, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, confidence: q.confidence, opened_from: q.opened_from, opened_to: q.opened_to }; | |
| 94 | + return [...facilityConds(sql, fq), ...layerConds(sql, q), sql`f.lat is not null and f.lng is not null`, bboxCond(sql, q.bbox, sql`f.lat`, sql`f.lng`)]; | |
| 95 | +} | |
| 96 | + | |
| 42 | 97 | function conds(sql: Sql, q: MapQuery): Fragment { |
| 43 | − const fq: FacilityQuery = { status: q.status, type: q.type, operator: q.operator, country: q.country, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale }; | |
| 44 | − return andAll(sql, [...facilityConds(sql, fq), sql`f.lat is not null and f.lng is not null`, bboxCond(sql, q.bbox, sql`f.lat`, sql`f.lng`)]); | |
| 98 | + return andAll(sql, [...baseConds(sql, q), yearCond(sql, q)]); | |
| 99 | +} | |
| 100 | + | |
| 101 | +/** Layers whose facility points are hidden (overlay-only layers). */ | |
| 102 | +function facilitiesDrawn(q: MapQuery): boolean { | |
| 103 | + return q.layer !== "cloud" && q.layer !== "ixps" && q.layer !== "projects"; | |
| 45 | 104 | } |
| 46 | 105 | |
| 47 | 106 | function addStatus(target: Partial<Record<FacilityStatus, number>>, status: unknown, n: number): void { |
@@ -49,9 +108,15 @@ function addStatus(target: Partial<Record<FacilityStatus, number>>, status: unkn | ||
| 49 | 108 | target[s] = (target[s] ?? 0) + n; |
| 50 | 109 | } |
| 51 | 110 | |
| 111 | +/** MW column drawn for the layer: power → utility/grid supply, else the best known figure. */ | |
| 112 | +function pointMw(sql: Sql, q: MapQuery): Fragment { | |
| 113 | + return q.layer === "power" ? sql`coalesce(f.utility_capacity_mw, f.grid_connection_mw)` : mwExpr(sql); | |
| 114 | +} | |
| 115 | + | |
| 52 | 116 | async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> { |
| 117 | + const mw = pointMw(sql, q); | |
| 53 | 118 | const rows = await sql<Row[]>` |
| 54 | − select f.country_iso2, f.status, count(*)::int as n, sum(${mwExpr(sql)})::float as mw, sum(f.lat)::float as slat, sum(f.lng)::float as slng, c.lat as clat, c.lng as clng, c.name as cname | |
| 119 | + select f.country_iso2, f.status, count(*)::int as n, sum(${mw})::float as mw, sum(f.lat)::float as slat, sum(f.lng)::float as slng, c.lat as clat, c.lng as clng, c.name as cname | |
| 55 | 120 | from facilities f ${facilityJoins(sql)} |
| 56 | 121 | where ${conds(sql, q)} |
| 57 | 122 | group by f.country_iso2, f.status, c.lat, c.lng, c.name`; |
@@ -64,8 +129,8 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> { | ||
| 64 | 129 | c.count += n; |
| 65 | 130 | c._slat += num(r.slat) ?? 0; |
| 66 | 131 | c._slng += num(r.slng) ?? 0; |
| 67 | − const mw = num(r.mw); | |
| 68 | − if (mw != null) { c.mw = (c.mw ?? 0) + mw; c._hasMw = true; } | |
| 132 | + const m = num(r.mw); | |
| 133 | + if (m != null) { c.mw = (c.mw ?? 0) + m; c._hasMw = true; } | |
| 69 | 134 | addStatus(c.statuses, r.status, n); |
| 70 | 135 | const clat = num(r.clat), clng = num(r.clng); |
| 71 | 136 | if (clat != null && clng != null) { c.lat = clat; c.lng = clng; } |
@@ -77,7 +142,7 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> { | ||
| 77 | 142 | by.delete("unknown"); |
| 78 | 143 | const size = cellSize(4); |
| 79 | 144 | const cells = await sql<Row[]>` |
| 80 | − select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mwExpr(sql)})::float as mw, avg(f.lat)::float as alat, avg(f.lng)::float as alng | |
| 145 | + select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mw})::float as mw, avg(f.lat)::float as alat, avg(f.lng)::float as alng | |
| 81 | 146 | from facilities f ${facilityJoins(sql)} |
| 82 | 147 | where ${conds(sql, q)} and f.country_iso2 is null |
| 83 | 148 | group by 1, 2, 3`; |
@@ -87,8 +152,8 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> { | ||
| 87 | 152 | if (!c) { c = { key, lat: num(r.alat) ?? 0, lng: num(r.alng) ?? 0, count: 0, mw: null, statuses: {}, label: null, _slat: 0, _slng: 0, _hasMw: false }; by.set(key, c); } |
| 88 | 153 | const n = int(r.n); |
| 89 | 154 | c.count += n; |
| 90 | − const mw = num(r.mw); | |
| 91 | − if (mw != null) { c.mw = (c.mw ?? 0) + mw; c._hasMw = true; } | |
| 155 | + const m = num(r.mw); | |
| 156 | + if (m != null) { c.mw = (c.mw ?? 0) + m; c._hasMw = true; } | |
| 92 | 157 | addStatus(c.statuses, r.status, n); |
| 93 | 158 | } |
| 94 | 159 | } |
@@ -103,8 +168,9 @@ async function countryClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> { | ||
| 103 | 168 | |
| 104 | 169 | async function gridClusters(sql: Sql, q: MapQuery, zoom: number): Promise<MapCluster[]> { |
| 105 | 170 | const size = cellSize(zoom); |
| 171 | + const mw = pointMw(sql, q); | |
| 106 | 172 | const rows = await sql<Row[]>` |
| 107 | − select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mwExpr(sql)})::float as mw | |
| 173 | + select floor((f.lng + 180) / ${size})::int as i, floor((f.lat + 90) / ${size})::int as j, f.status, count(*)::int as n, sum(${mw})::float as mw | |
| 108 | 174 | from facilities f ${facilityJoins(sql)} |
| 109 | 175 | where ${conds(sql, q)} |
| 110 | 176 | group by 1, 2, 3`; |
@@ -116,25 +182,31 @@ async function gridClusters(sql: Sql, q: MapQuery, zoom: number): Promise<MapClu | ||
| 116 | 182 | if (!c) { c = { key, lat: Math.round((j * size - 90 + size / 2) * 1e4) / 1e4, lng: Math.round((i * size - 180 + size / 2) * 1e4) / 1e4, count: 0, mw: null, statuses: {}, label: null }; by.set(key, c); } |
| 117 | 183 | const n = int(r.n); |
| 118 | 184 | c.count += n; |
| 119 | − const mw = num(r.mw); | |
| 120 | − if (mw != null) c.mw = Math.round(((c.mw ?? 0) + mw) * 100) / 100; | |
| 185 | + const m = num(r.mw); | |
| 186 | + if (m != null) c.mw = Math.round(((c.mw ?? 0) + m) * 100) / 100; | |
| 121 | 187 | addStatus(c.statuses, r.status, n); |
| 122 | 188 | } |
| 123 | 189 | return [...by.values()].sort((a, b) => b.count - a.count); |
| 124 | 190 | } |
| 125 | 191 | |
| 126 | 192 | async function points(sql: Sql, q: MapQuery): Promise<MapPoint[] | null> { |
| 193 | + const mw = pointMw(sql, q); | |
| 127 | 194 | const rows = await sql<Row[]>` |
| 128 | − select f.id, f.slug, f.name, o.name as op_name, f.lat, f.lng, f.status, f.facility_type, ${mwExpr(sql)} as mw, f.geo_precision, f.is_ai, f.is_hyperscale, f.country_iso2 | |
| 195 | + select f.id, f.slug, f.name, o.name as op_name, f.lat, f.lng, f.status, f.facility_type, ${mw} as mw, f.geo_precision, f.is_ai, f.ai_evidence, f.is_hyperscale, f.country_iso2, f.record_scope, ${yearExpr(sql, sql`f.opened_on`)} as y | |
| 129 | 196 | from facilities f ${facilityJoins(sql)} |
| 130 | 197 | where ${conds(sql, q)} |
| 131 | − order by ${mwExpr(sql)} desc nulls last, f.id | |
| 198 | + order by ${mw} desc nulls last, f.id | |
| 132 | 199 | limit ${MAX_POINTS + 1}`; |
| 133 | 200 | if (rows.length > MAX_POINTS) return null; |
| 134 | 201 | return rows.map((r) => { |
| 135 | 202 | const p: MapPoint = { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: asType(r.facility_type), mw: num(r.mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2) }; |
| 136 | − if (r.is_ai === true) p.ai = 1; | |
| 203 | + const ai = str(r.ai_evidence); | |
| 204 | + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1; | |
| 137 | 205 | if (r.is_hyperscale === true) p.hs = 1; |
| 206 | + const y = num(r.y); | |
| 207 | + if (y != null) p.y = y; | |
| 208 | + const rs = str(r.record_scope); | |
| 209 | + if (rs === "building" || rs === "campus") p.rs = rs; | |
| 138 | 210 | return p; |
| 139 | 211 | }); |
| 140 | 212 | } |
@@ -143,36 +215,171 @@ async function cloudRegionPoints(sql: Sql, q: MapQuery): Promise<MapPoint[]> { | ||
| 143 | 215 | const c: Fragment[] = [sql`r.lat is not null and r.lng is not null`, bboxCond(sql, q.bbox, sql`r.lat`, sql`r.lng`)]; |
| 144 | 216 | if (q.country?.length) c.push(sql`r.country_iso2 = any(${q.country})`); |
| 145 | 217 | if (q.operator) c.push(sql`(pr.slug = ${q.operator} or pr.id = ${q.operator})`); |
| 146 | − const rows = await sql<Row[]>`select r.id, r.slug, r.name, pr.name as pr_name, r.lat, r.lng, r.status, r.geo_precision, r.country_iso2 from cloud_regions r join operators pr on pr.id = r.provider_id where ${andAll(sql, c)} limit 2000`; | |
| 218 | + if (q.metro) c.push(sql`r.metro_id in (select id from metros where slug = ${q.metro} or id = ${q.metro})`); | |
| 219 | + const rows = await sql<Row[]>`select r.id, r.slug, r.name, pr.name as pr_name, r.lat, r.lng, r.status, r.geo_precision, r.country_iso2, ${yearExpr(sql, sql`r.launched_on`)} as y from cloud_regions r join operators pr on pr.id = r.provider_id where ${andAll(sql, c)} limit 2000`; | |
| 147 | 220 | return rows.map((r) => { |
| 148 | 221 | const st = str(r.status); |
| 149 | − return { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.pr_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: st === "announced" ? "announced" : st === "retired" ? "closed" : "operational", t: "cloud_region", p: asPrecision(r.geo_precision ?? "city"), c: str(r.country_iso2) }; | |
| 222 | + const p: MapPoint = { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.pr_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: st === "announced" ? "announced" : st === "retired" ? "closed" : "operational", t: "cloud_region", p: asPrecision(r.geo_precision ?? "city"), c: str(r.country_iso2), k: "cloud_region" }; | |
| 223 | + const y = num(r.y); | |
| 224 | + if (y != null) p.y = y; | |
| 225 | + return p; | |
| 150 | 226 | }); |
| 151 | 227 | } |
| 152 | 228 | |
| 229 | +/** Live projects with coordinates (city-level geocodes → p = city). */ | |
| 230 | +async function projectPoints(sql: Sql, q: MapQuery, onlyAi = false): Promise<MapPoint[]> { | |
| 231 | + const c: Fragment[] = [projectLive(sql), sql`p.lat is not null and p.lng is not null`, bboxCond(sql, q.bbox, sql`p.lat`, sql`p.lng`)]; | |
| 232 | + if (q.country?.length) c.push(sql`p.country_iso2 = any(${q.country})`); | |
| 233 | + if (q.operator) c.push(sql`p.operator_id in (select id from operators where slug = ${q.operator} or id = ${q.operator})`); | |
| 234 | + if (q.metro) c.push(sql`p.metro_id in (select id from metros where slug = ${q.metro} or id = ${q.metro})`); | |
| 235 | + if (q.project_status?.length) c.push(sql`p.status = any(${q.project_status})`); | |
| 236 | + else if (q.layer === "pipeline") c.push(sql`p.status = any(${PIPELINE_PROJECT_STATUSES})`); | |
| 237 | + if (q.expected_from != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} >= ${q.expected_from}`); | |
| 238 | + if (q.expected_to != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} <= ${q.expected_to}`); | |
| 239 | + if (q.min_mw != null) c.push(sql`p.planned_mw >= ${q.min_mw}`); | |
| 240 | + if (q.max_mw != null) c.push(sql`p.planned_mw <= ${q.max_mw}`); | |
| 241 | + if (onlyAi || q.ai === true) c.push(sql`(p.is_ai or p.ai_evidence in ('confirmed', 'likely'))`); | |
| 242 | + if (q.ai_evidence?.length) c.push(sql`p.ai_evidence = any(${q.ai_evidence})`); | |
| 243 | + if (q.year != null) c.push(sql`${yearExpr(sql, sql`coalesce(p.announced_on, p.expected_opening)`)} <= ${q.year}`); | |
| 244 | + const rows = await sql<Row[]>` | |
| 245 | + select p.id, p.slug, p.name, o.name as op_name, p.lat, p.lng, p.status, p.planned_mw, p.geo_precision, p.is_ai, p.ai_evidence, p.country_iso2, pf.facility_type, ${yearExpr(sql, sql`p.expected_opening`)} as y | |
| 246 | + from projects p left join operators o on o.id = p.operator_id left join facilities pf on pf.id = p.facility_id | |
| 247 | + where ${andAll(sql, c)} order by p.planned_mw desc nulls last, p.id limit ${MAX_POINTS}`; | |
| 248 | + return rows.map((r) => { | |
| 249 | + const p: MapPoint = { id: String(r.id), slug: String(r.slug), n: String(r.name), o: str(r.op_name), lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: asStatus(r.status), t: r.facility_type ? asType(r.facility_type) : "unknown", mw: num(r.planned_mw), p: asPrecision(r.geo_precision), c: str(r.country_iso2), k: "project" }; | |
| 250 | + const ai = str(r.ai_evidence); | |
| 251 | + if (r.is_ai === true || ai === "confirmed" || ai === "likely") p.ai = 1; | |
| 252 | + const y = num(r.y); | |
| 253 | + if (y != null) p.y = y; | |
| 254 | + return p; | |
| 255 | + }); | |
| 256 | +} | |
| 257 | + | |
| 258 | +/** IXPs at their metro's reference point (ixps carry no coordinates); metro resolved by id, else by city name. */ | |
| 259 | +function ixpMetroJoin(sql: Sql): Fragment { | |
| 260 | + return sql` | |
| 261 | + left join metros xm on xm.id = x.metro_id | |
| 262 | + left join lateral (select m2.id, m2.slug, m2.name, m2.lat, m2.lng from metros m2 where x.metro_id is null and x.country_iso2 = m2.country_iso2 and x.city is not null and (lower(m2.name) = lower(x.city) or exists (select 1 from unnest(m2.aliases) a where lower(a) = lower(x.city))) limit 1) xm2 on true`; | |
| 263 | +} | |
| 264 | + | |
| 265 | +async function ixpPoints(sql: Sql, q: MapQuery): Promise<MapPoint[]> { | |
| 266 | + const lat = sql`coalesce(xm.lat, xm2.lat)`, lng = sql`coalesce(xm.lng, xm2.lng)`; | |
| 267 | + const c: Fragment[] = [sql`${lat} is not null`, bboxCond(sql, q.bbox, lat, lng)]; | |
| 268 | + if (q.country?.length) c.push(sql`x.country_iso2 = any(${q.country})`); | |
| 269 | + if (q.metro) c.push(sql`coalesce(xm.id, xm2.id) in (select id from metros where slug = ${q.metro} or id = ${q.metro})`); | |
| 270 | + const rows = await sql<Row[]>`select x.id, x.slug, x.name, x.country_iso2, x.network_count, ${lat} as lat, ${lng} as lng from ixps x ${ixpMetroJoin(sql)} where ${andAll(sql, c)} order by x.network_count desc nulls last limit 2000`; | |
| 271 | + return rows.map((r) => ({ id: String(r.id), slug: String(r.slug), n: String(r.name), o: null, lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, s: "operational", t: "internet_exchange", mw: null, p: "metro", c: str(r.country_iso2), k: "ixp" })); | |
| 272 | +} | |
| 273 | + | |
| 274 | +/** Power / grid events by metro, as clusters at the metro reference point (no invented plant locations). */ | |
| 275 | +async function powerEventClusters(sql: Sql, q: MapQuery): Promise<MapCluster[]> { | |
| 276 | + const types = [...new Set([...POWER_EVENT_TYPES, ...GRID_EVENT_TYPES])]; | |
| 277 | + const c: Fragment[] = [sql`e.event_type = any(${types})`, sql`e.review_status <> 'rejected'`, bboxCond(sql, q.bbox, sql`m.lat`, sql`m.lng`)]; | |
| 278 | + if (q.country?.length) c.push(sql`m.country_iso2 = any(${q.country})`); | |
| 279 | + if (q.metro) c.push(sql`(m.slug = ${q.metro} or m.id = ${q.metro})`); | |
| 280 | + const rows = await sql<Row[]>`select m.id, m.name, m.lat, m.lng, count(*)::int as n from events e join metros m on m.id = e.metro_id where ${andAll(sql, c)} group by m.id, m.name, m.lat, m.lng order by n desc limit 500`; | |
| 281 | + return rows.map((r) => ({ key: `power:${String(r.id)}`, lat: num(r.lat) ?? 0, lng: num(r.lng) ?? 0, count: int(r.n), mw: null, statuses: {}, label: `${String(r.name)} — ${int(r.n)} power / grid event${int(r.n) > 1 ? "s" : ""}` })); | |
| 282 | +} | |
| 283 | + | |
| 284 | +/** Share of filter-matching facilities that carry an opening year (time machine honesty figure). */ | |
| 285 | +async function yearCoverage(sql: Sql, q: MapQuery): Promise<number> { | |
| 286 | + const rows = await sql<Row[]>`select count(*)::int as n, count(*) filter (where f.opened_on ~ '^\\d{4}')::int as dated from facilities f ${facilityJoins(sql)} where ${andAll(sql, baseConds(sql, q))}`; | |
| 287 | + const n = int(rows[0]?.n); | |
| 288 | + return n ? Math.round((int(rows[0]?.dated) / n) * 1000) / 1000 : 0; | |
| 289 | +} | |
| 290 | + | |
| 291 | +async function densityCells(sql: Sql, q: MapQuery, zoom: number): Promise<{ view: DensityView; cells: DensityCell[]; max: number; total: number }> { | |
| 292 | + const view = q.density ?? "facilities"; | |
| 293 | + const size = cellSize(zoom); | |
| 294 | + const cell = (latCol: Fragment, lngCol: Fragment) => sql`floor((${lngCol} + 180) / ${size})::int as i, floor((${latCol} + 90) / ${size})::int as j`; | |
| 295 | + let rows: Row[]; | |
| 296 | + if (view === "cloud") { | |
| 297 | + const c: Fragment[] = [sql`r.lat is not null and r.lng is not null and r.status <> 'retired'`, bboxCond(sql, q.bbox, sql`r.lat`, sql`r.lng`)]; | |
| 298 | + if (q.country?.length) c.push(sql`r.country_iso2 = any(${q.country})`); | |
| 299 | + if (q.operator) c.push(sql`r.provider_id in (select id from operators where slug = ${q.operator} or id = ${q.operator})`); | |
| 300 | + rows = await sql<Row[]>`select ${cell(sql`r.lat`, sql`r.lng`)}, count(*)::int as n, count(*)::float as w from cloud_regions r where ${andAll(sql, c)} group by 1, 2`; | |
| 301 | + } else if (view === "ixps") { | |
| 302 | + const lat = sql`coalesce(xm.lat, xm2.lat)`, lng = sql`coalesce(xm.lng, xm2.lng)`; | |
| 303 | + const c: Fragment[] = [sql`${lat} is not null`, bboxCond(sql, q.bbox, lat, lng)]; | |
| 304 | + if (q.country?.length) c.push(sql`x.country_iso2 = any(${q.country})`); | |
| 305 | + rows = await sql<Row[]>`select ${cell(lat, lng)}, count(*)::int as n, count(*)::float as w from ixps x ${ixpMetroJoin(sql)} where ${andAll(sql, c)} group by 1, 2`; | |
| 306 | + } else { | |
| 307 | + const weight: Fragment = | |
| 308 | + view === "known_mw" ? sql`coalesce(sum(${knownMwAgg(sql)}) filter (where f.status in ('operational', 'partially_operational', 'expansion')), 0)::float` | |
| 309 | + : view === "pipeline_mw" ? sql`coalesce(sum(${pipelineMwAgg(sql)}) filter (where f.status = any(${PIPELINE_SET})), 0)::float` | |
| 310 | + : view === "ai" ? sql`count(*) filter (where ${countedAgg(sql)} and ${AI_COND(sql)})::float` | |
| 311 | + : view === "operators" ? sql`count(distinct f.operator_id)::float` | |
| 312 | + : sql`count(*) filter (where ${countedAgg(sql)})::float`; | |
| 313 | + rows = await sql<Row[]>`select ${cell(sql`f.lat`, sql`f.lng`)}, count(*)::int as n, ${weight} as w | |
| 314 | + from ${facilityView(sql)} f ${facilityJoins(sql)} where ${conds(sql, q)} group by 1, 2`; | |
| 315 | + } | |
| 316 | + const cells: DensityCell[] = []; | |
| 317 | + let max = 0, total = 0; | |
| 318 | + for (const r of rows) { | |
| 319 | + const w = Math.round((num(r.w) ?? 0) * 100) / 100; | |
| 320 | + const n = int(r.n); | |
| 321 | + if (w <= 0 && n <= 0) continue; | |
| 322 | + total += n; | |
| 323 | + if (w > max) max = w; | |
| 324 | + cells.push({ lat: Math.round((int(r.j) * size - 90 + size / 2) * 1e4) / 1e4, lng: Math.round((int(r.i) * size - 180 + size / 2) * 1e4) / 1e4, w, n }); | |
| 325 | + } | |
| 326 | + cells.sort((a, b) => b.w - a.w); | |
| 327 | + return { view, cells, max, total }; | |
| 328 | +} | |
| 329 | + | |
| 330 | +/** Overlay points for the requested layer (non-facility entities). */ | |
| 331 | +async function overlayFor(sql: Sql, q: MapQuery): Promise<{ overlay: MapPoint[]; clusters: MapCluster[] }> { | |
| 332 | + const layer = q.layer ?? "facilities"; | |
| 333 | + const tasks: Array<Promise<MapPoint[]>> = []; | |
| 334 | + let clusters: Promise<MapCluster[]> = Promise.resolve([]); | |
| 335 | + if (layer === "projects" || layer === "pipeline") tasks.push(projectPoints(sql, q)); | |
| 336 | + if (layer === "ai") tasks.push(projectPoints(sql, q, true)); | |
| 337 | + if (layer === "cloud" || layer === "connectivity" || q.cloud_regions) tasks.push(cloudRegionPoints(sql, q)); | |
| 338 | + if (layer === "ixps" || layer === "connectivity") tasks.push(ixpPoints(sql, q)); | |
| 339 | + if (layer === "power") clusters = powerEventClusters(sql, q); | |
| 340 | + const [parts, cl] = await Promise.all([Promise.all(tasks), clusters]); | |
| 341 | + return { overlay: parts.flat(), clusters: cl }; | |
| 342 | +} | |
| 343 | + | |
| 153 | 344 | export async function mapData(q: MapQuery): Promise<MapResponse> { |
| 154 | 345 | const sql = pg(); |
| 155 | 346 | const zoom = Math.max(0, Math.min(22, Math.round(q.zoom))); |
| 156 | − const mode = mapMode(zoom); | |
| 157 | − const cloud = q.cloud_regions ? cloudRegionPoints(sql, q) : Promise.resolve([] as MapPoint[]); | |
| 158 | − if (mode === "country") { | |
| 159 | − const [clusters, cr] = await Promise.all([countryClusters(sql, q), cloud]); | |
| 160 | − const res: MapResponse = { zoom, mode: "clusters", clusters, total: clusters.reduce((a, c) => a + c.count, 0) }; | |
| 161 | − if (cr.length) res.points = cr; | |
| 347 | + const layer: MapLayer = q.layer ?? "facilities"; | |
| 348 | + const withOverlay = (res: MapResponse, o: { overlay: MapPoint[]; clusters: MapCluster[] }): MapResponse => { | |
| 349 | + if (o.overlay.length) res.overlay = o.overlay; | |
| 350 | + // API 1.x compatibility: the legacy `cloud_regions=1` flag (no `layer`) returned cloud regions inside `points` | |
| 351 | + if (o.overlay.length && q.cloud_regions && !q.layer) res.points = [...(res.points ?? []), ...o.overlay.filter((p) => p.k === "cloud_region")]; | |
| 352 | + if (o.clusters.length) res.clusters = [...(res.clusters ?? []), ...o.clusters]; | |
| 162 | 353 | return res; |
| 354 | + }; | |
| 355 | + const year = q.year ?? null; | |
| 356 | + const cov = year != null ? yearCoverage(sql, q) : Promise.resolve(null); | |
| 357 | + const overlayP = overlayFor(sql, q); | |
| 358 | + | |
| 359 | + if (q.density) { | |
| 360 | + const [d, o, yc] = await Promise.all([densityCells(sql, q, zoom), overlayP, cov]); | |
| 361 | + const res: MapResponse = { zoom, mode: "density", layer, density: { view: d.view, cells: d.cells, max: d.max }, total: d.total, year, yearCoverage: yc }; | |
| 362 | + return withOverlay(res, o); | |
| 163 | 363 | } |
| 164 | − if (mode === "grid") { | |
| 165 | − const [clusters, cr] = await Promise.all([gridClusters(sql, q, zoom), cloud]); | |
| 166 | − const res: MapResponse = { zoom, mode: "clusters", clusters, total: clusters.reduce((a, c) => a + c.count, 0) }; | |
| 167 | − if (cr.length) res.points = cr; | |
| 168 | − return res; | |
| 364 | + | |
| 365 | + if (!facilitiesDrawn(q)) { | |
| 366 | + const [o, yc] = await Promise.all([overlayP, cov]); | |
| 367 | + const res: MapResponse = { zoom, mode: "points", layer, points: [], total: o.overlay.length, year, yearCoverage: yc }; | |
| 368 | + return withOverlay(res, o); | |
| 169 | 369 | } |
| 170 | − const [pts, cr] = await Promise.all([points(sql, q), cloud]); | |
| 370 | + | |
| 371 | + const mode = mapMode(zoom); | |
| 372 | + if (mode === "country" || mode === "grid") { | |
| 373 | + const [clusters, o, yc] = await Promise.all([mode === "country" ? countryClusters(sql, q) : gridClusters(sql, q, zoom), overlayP, cov]); | |
| 374 | + const res: MapResponse = { zoom, mode: "clusters", layer, clusters, total: clusters.reduce((a, c) => a + c.count, 0), year, yearCoverage: yc }; | |
| 375 | + return withOverlay(res, o); | |
| 376 | + } | |
| 377 | + const [pts, o, yc] = await Promise.all([points(sql, q), overlayP, cov]); | |
| 171 | 378 | if (pts === null) { |
| 172 | 379 | const clusters = await gridClusters(sql, q, 8); |
| 173 | − const res: MapResponse = { zoom, mode: "clusters", clusters, total: clusters.reduce((a, c) => a + c.count, 0) }; | |
| 174 | − if (cr.length) res.points = cr; | |
| 175 | − return res; | |
| 380 | + const res: MapResponse = { zoom, mode: "clusters", layer, clusters, total: clusters.reduce((a, c) => a + c.count, 0), degraded: true, year, yearCoverage: yc }; | |
| 381 | + return withOverlay(res, o); | |
| 176 | 382 | } |
| 177 | − return { zoom, mode: "points", points: [...pts, ...cr], total: pts.length }; | |
| 383 | + const res: MapResponse = { zoom, mode: "points", layer, points: pts, total: pts.length, year, yearCoverage: yc }; | |
| 384 | + return withOverlay(res, o); | |
| 178 | 385 | } |
modified
apps/api/src/repositories/metros.ts
+58 −33
@@ -1,15 +1,18 @@ | ||
| 1 | −import type { MetroDetail, MetroSummary } from "@dci/core"; | |
| 2 | −import { pg, mwExpr, plannedMwExpr, OPERATIONAL_SET, CONSTRUCTION_SET, PLANNED_SET, yearExpr, cloudRegionCols, likePattern, type Fragment, type Sql } from "../lib/sql.js"; | |
| 1 | +import type { GridConstraintDTO, MetroDetail, MetroSummary } from "@dci/core"; | |
| 2 | +import { pg, mwExpr, yearExpr, cloudRegionCols, likePattern, facilityView, knownMwAgg, countedAgg, projectLive, facilityJoins, facilitySummaryCols, eventCols, eventJoins, gridConstraintCols, gridConstraintJoins, AI_LEVELS, GRID_EVENT_TYPES, OPERATIONAL_SET, type Fragment, type Sql } from "../lib/sql.js"; | |
| 3 | 3 | import { int, iso, num, reqStr, str, strArray, type Row } from "../lib/rows.js"; |
| 4 | −import { cloudRegionSummary, growthSeries } from "../lib/dto.js"; | |
| 4 | +import { cloudRegionSummary, facilitySummary, gridConstraintDto, gridConstraintFromEvent, growthSeries, round2, share } from "../lib/dto.js"; | |
| 5 | 5 | import { findBySlugOrId } from "../lib/resolve.js"; |
| 6 | 6 | import { listFacilities } from "./facilities.js"; |
| 7 | 7 | import { operatorsForScope } from "./operators.js"; |
| 8 | 8 | import { projectsWhere } from "./projects.js"; |
| 9 | −import { eventsForMetro } from "./events.js"; | |
| 9 | +import { eventsForMetro, toEventDtos } from "./events.js"; | |
| 10 | 10 | import { rankingPositions } from "./rankings.js"; |
| 11 | +import { concentration, coverageRow, dimAggCols, momentum, pipelineBreakdown, scopeFor } from "../lib/pipeline.js"; | |
| 12 | +import { claimsFor } from "../lib/quality.js"; | |
| 11 | 13 | |
| 12 | 14 | function metroSummary(r: Row): MetroSummary { |
| 15 | + const n = int(r.facility_count); | |
| 13 | 16 | return { |
| 14 | 17 | id: reqStr(r.id), |
| 15 | 18 | slug: reqStr(r.slug), |
@@ -19,44 +22,34 @@ function metroSummary(r: Row): MetroSummary { | ||
| 19 | 22 | regionName: str(r.region_name), |
| 20 | 23 | lat: num(r.lat) ?? 0, |
| 21 | 24 | lng: num(r.lng) ?? 0, |
| 22 | − facilityCount: int(r.facility_count), | |
| 25 | + facilityCount: n, | |
| 23 | 26 | operationalCount: int(r.operational), |
| 24 | 27 | constructionCount: int(r.construction), |
| 25 | 28 | plannedCount: int(r.planned), |
| 26 | − knownMw: num(r.known_mw), | |
| 27 | − constructionMw: num(r.construction_mw), | |
| 28 | − plannedMw: num(r.planned_mw), | |
| 29 | + knownMw: round2(num(r.known_mw)), | |
| 30 | + constructionMw: round2(num(r.construction_mw)), | |
| 31 | + plannedMw: round2(num(r.planned_mw)), | |
| 29 | 32 | operatorCount: int(r.operator_count), |
| 30 | 33 | cloudRegionCount: int(r.cloud_region_count), |
| 31 | 34 | ixpCount: int(r.ixp_count), |
| 32 | 35 | projectCount: int(r.project_count), |
| 36 | + projectPlannedMw: round2(num(r.project_planned_mw)), | |
| 37 | + aiCount: int(r.ai), | |
| 38 | + mwCoverage: n ? share(int(r.with_mw), n) : 0, | |
| 33 | 39 | }; |
| 34 | 40 | } |
| 35 | 41 | |
| 36 | −const SELECT = (sql: Sql) => { | |
| 37 | − const mw = mwExpr(sql); | |
| 38 | − return sql` | |
| 42 | +const SELECT = (sql: Sql) => sql` | |
| 39 | 43 | select m.id, m.slug, m.name, m.country_iso2, c.name as country_name, m.region_name, m.lat, m.lng, m.aliases, m.description, m.updated_at, |
| 40 | − coalesce(fa.facility_count, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned, | |
| 41 | − fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operator_count, 0) as operator_count, | |
| 42 | − coalesce(cr.n, 0) as cloud_region_count, coalesce(ix.n, 0) as ixp_count, coalesce(pj.n, 0) as project_count | |
| 44 | + coalesce(fa.facilities, 0) as facility_count, coalesce(fa.operational, 0) as operational, coalesce(fa.construction, 0) as construction, coalesce(fa.planned, 0) as planned, | |
| 45 | + fa.known_mw, fa.construction_mw, fa.planned_mw, coalesce(fa.operators, 0) as operator_count, coalesce(fa.with_mw, 0) as with_mw, coalesce(fa.ai, 0) as ai, | |
| 46 | + coalesce(cr.n, 0) as cloud_region_count, coalesce(ix.n, 0) as ixp_count, coalesce(pj.n, 0) as project_count, pj.mw as project_planned_mw | |
| 43 | 47 | from metros m |
| 44 | 48 | left join countries c on c.iso2 = m.country_iso2 |
| 45 | − left join ( | |
| 46 | − select f.metro_id, count(*)::int as facility_count, | |
| 47 | − count(*) filter (where f.status = any(${OPERATIONAL_SET}))::int as operational, | |
| 48 | − count(*) filter (where f.status = any(${CONSTRUCTION_SET}))::int as construction, | |
| 49 | − count(*) filter (where f.status = any(${PLANNED_SET}))::int as planned, | |
| 50 | − sum(${mw}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw, | |
| 51 | − sum(${mw}) filter (where f.status = any(${CONSTRUCTION_SET}))::float as construction_mw, | |
| 52 | − sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PLANNED_SET}))::float as planned_mw, | |
| 53 | − count(distinct f.operator_id)::int as operator_count | |
| 54 | − from facilities f where f.merged_into is null and f.metro_id is not null group by f.metro_id | |
| 55 | − ) fa on fa.metro_id = m.id | |
| 56 | − left join (select metro_id, count(*)::int as n from cloud_regions where metro_id is not null group by 1) cr on cr.metro_id = m.id | |
| 49 | + left join (select f.metro_id, ${dimAggCols(sql)} from ${facilityView(sql)} f where f.metro_id is not null group by f.metro_id) fa on fa.metro_id = m.id | |
| 50 | + left join (select metro_id, count(*)::int as n from cloud_regions where metro_id is not null and status <> 'retired' group by 1) cr on cr.metro_id = m.id | |
| 57 | 51 | left join (select metro_id, count(*)::int as n from ixps where metro_id is not null group by 1) ix on ix.metro_id = m.id |
| 58 | − left join (select metro_id, count(*)::int as n from projects where metro_id is not null group by 1) pj on pj.metro_id = m.id`; | |
| 59 | −}; | |
| 52 | + left join (select p.metro_id, count(*)::int as n, sum(p.planned_mw) filter (where p.status in ('rumored','proposed','announced','permitting','approved','delayed','under_construction'))::float as mw from projects p where p.metro_id is not null and ${projectLive(sql)} group by p.metro_id) pj on pj.metro_id = m.id`; | |
| 60 | 53 | |
| 61 | 54 | export async function listMetros(f: { country?: string; q?: string; limit?: number } = {}): Promise<MetroSummary[]> { |
| 62 | 55 | const sql = pg(); |
@@ -64,7 +57,7 @@ export async function listMetros(f: { country?: string; q?: string; limit?: numb | ||
| 64 | 57 | if (f.country) conds.push(sql`m.country_iso2 = ${f.country.toUpperCase()}`); |
| 65 | 58 | if (f.q) conds.push(sql`(m.name ilike ${likePattern(f.q)} or similarity(m.name, ${f.q}) > 0.35 or exists (select 1 from unnest(m.aliases) a where a ilike ${likePattern(f.q)}))`); |
| 66 | 59 | const rows = await sql<Row[]>`${SELECT(sql)} where ${conds.reduce<Fragment>((a, c) => sql`${a} and ${c}`, sql`true`)} |
| 67 | − order by coalesce(fa.facility_count, 0) desc, coalesce(cr.n, 0) desc, m.name limit ${f.limit ?? 1000}`; | |
| 60 | + order by coalesce(fa.facilities, 0) desc, coalesce(cr.n, 0) desc, m.name limit ${f.limit ?? 1000}`; | |
| 68 | 61 | return rows.map(metroSummary); |
| 69 | 62 | } |
| 70 | 63 | |
@@ -83,7 +76,7 @@ async function metroConstraints(metroId: string, name: string, aliases: string[] | ||
| 83 | 76 | const rows = await sql<Row[]>` |
| 84 | 77 | select n.title, n.summary, n.url, n.published_at, n.created_at, s.name as source_name |
| 85 | 78 | from news_items n left join sources s on s.id = n.source_id |
| 86 | − where (n.mentions->>'metroId' = ${metroId} or n.mentions->'metros' ? ${metroId} or n.mentions->'cities' ?| ${terms}::text[] or (${nameCond})) | |
| 79 | + where (n.metro_id = ${metroId} or n.mentions->>'metroId' = ${metroId} or n.mentions->'metros' ? ${metroId} or n.mentions->'cities' ?| ${terms}::text[] or (${nameCond})) | |
| 87 | 80 | and (n.title ~* '(moratorium|grid|power|electricity|substation|water|zoning|land|transmission|interconnection)' or n.summary ~* '(moratorium|grid (constraint|capacity)|power (shortage|constraint|crunch)|water (usage|shortage|restriction)|zoning|land (scarcity|shortage|constraint))') |
| 88 | 81 | order by n.published_at desc nulls last, n.created_at desc limit 40`; |
| 89 | 82 | const out: MetroDetail["constraints"] = []; |
@@ -97,28 +90,60 @@ async function metroConstraints(metroId: string, name: string, aliases: string[] | ||
| 97 | 90 | return out; |
| 98 | 91 | } |
| 99 | 92 | |
| 93 | +/** grid_constraints rows + grid / utility / power events for a metro or a country, deduplicated by event id. */ | |
| 94 | +export async function gridConstraintsFor(scope: { metroId?: string; countryIso2?: string }, limit = 30): Promise<GridConstraintDTO[]> { | |
| 95 | + const sql = pg(); | |
| 96 | + const gCond = scope.metroId ? sql`g.metro_id = ${scope.metroId}` : scope.countryIso2 ? sql`g.country_iso2 = ${scope.countryIso2}` : sql`true`; | |
| 97 | + const eCond = scope.metroId ? sql`(e.metro_id = ${scope.metroId} or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = ${scope.metroId})))` : scope.countryIso2 ? sql`e.country_iso2 = ${scope.countryIso2}` : sql`true`; | |
| 98 | + const [rows, evRows] = await Promise.all([ | |
| 99 | + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} where ${gCond} order by g.effective_date desc nulls last, g.created_at desc limit ${limit}`, | |
| 100 | + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where ${eCond} and e.event_type = any(${GRID_EVENT_TYPES}) and e.review_status <> 'rejected' order by e.detected_at desc limit ${limit}`, | |
| 101 | + ]); | |
| 102 | + const out = rows.map(gridConstraintDto); | |
| 103 | + const seen = new Set(out.map((g) => g.eventId).filter(Boolean)); | |
| 104 | + for (const e of await toEventDtos(evRows)) { if (seen.has(e.id)) continue; seen.add(e.id); out.push(gridConstraintFromEvent(e)); } | |
| 105 | + return out.slice(0, limit); | |
| 106 | +} | |
| 107 | + | |
| 100 | 108 | export async function getMetroDetail(idOrSlug: string, fPage = 1): Promise<MetroDetail | null> { |
| 101 | 109 | const sql = pg(); |
| 102 | 110 | const base = await findBySlugOrId("metros", idOrSlug); |
| 103 | 111 | if (!base) return null; |
| 104 | 112 | const id = String(base.id); |
| 105 | 113 | const aliases = strArray(base.aliases); |
| 106 | − const [sumRows, operators, cloudProviderRows, cloudRegionRows, ixpRows, facilities, projects, recentEvents, constraints, growthRows, rankings] = await Promise.all([ | |
| 114 | + const scope = scopeFor(sql, { metroId: id }); | |
| 115 | + const [sumRows, operators, cloudProviderRows, cloudRegionRows, ixpRows, facilities, projects, recentEvents, constraints, growthRows, rankings, conc, mom, pipeline, gridConstraints, aiRows, openingRows, coverage, claims] = await Promise.all([ | |
| 107 | 116 | sql<Row[]>`${SELECT(sql)} where m.id = ${id}`, |
| 108 | 117 | operatorsForScope({ metroId: id }, 15), |
| 109 | 118 | sql<Row[]>`select pr.id, pr.slug, pr.name, count(*)::int as n from cloud_regions r join operators pr on pr.id = r.provider_id where r.metro_id = ${id} group by 1, 2, 3 order by n desc, pr.name`, |
| 110 | 119 | sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.metro_id = ${id} order by pr.name, r.code`, |
| 111 | 120 | sql<Row[]>`select id, slug, name, network_count from ixps where metro_id = ${id} order by network_count desc nulls last, name`, |
| 112 | 121 | listFacilities({ metroId: id, page: fPage, per_page: 50, sort: "mw", order: "desc" }), |
| 113 | − projectsWhere(sql`p.metro_id = ${id}`, 20), | |
| 122 | + projectsWhere(sql`${projectLive(sql)} and p.metro_id = ${id}`, 20), | |
| 114 | 123 | eventsForMetro(id, 20), |
| 115 | 124 | metroConstraints(id, String(base.name), aliases), |
| 116 | − sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*)::int as n, sum(${mwExpr(sql)})::float as mw from facilities f where f.metro_id = ${id} and f.merged_into is null and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 125 | + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${knownMwAgg(sql)})::float as mw from ${facilityView(sql)} f where f.metro_id = ${id} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 117 | 126 | rankingPositions("metros", [id, String(base.slug)]), |
| 127 | + concentration(scope), | |
| 128 | + momentum(scope), | |
| 129 | + pipelineBreakdown(scope), | |
| 130 | + gridConstraintsFor({ metroId: id }), | |
| 131 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and f.metro_id = ${id} and f.ai_evidence = any(${AI_LEVELS}) order by ${mwExpr(sql)} desc nulls last, f.name limit 20`, | |
| 132 | + sql<Row[]>`select ${yearExpr(sql, sql`f.opened_on`)} as year, count(*) filter (where ${countedAgg(sql)})::int as n, sum(${knownMwAgg(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as mw from ${facilityView(sql)} f where f.metro_id = ${id} and f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 133 | + coverageRow(id, String(base.name), String(base.slug), scope), | |
| 134 | + Promise.all([claimsFor("metro", id), claimsFor("market", id)]).then(([a, b]) => [...a, ...b]), | |
| 118 | 135 | ]); |
| 119 | 136 | const summary = metroSummary(sumRows[0] ?? { ...base, facility_count: 0 }); |
| 120 | 137 | return { |
| 121 | 138 | ...summary, |
| 139 | + concentration: conc, | |
| 140 | + momentum: mom, | |
| 141 | + pipeline, | |
| 142 | + gridConstraints, | |
| 143 | + aiFacilities: aiRows.map(facilitySummary), | |
| 144 | + openingTimeline: openingRows.filter((r) => int(r.year) > 0).map((r) => ({ year: int(r.year), opened: int(r.n), openedMw: round2(num(r.mw)) })), | |
| 145 | + coverage, | |
| 146 | + claims, | |
| 122 | 147 | aliases, |
| 123 | 148 | description: str(base.description), |
| 124 | 149 | operators, |
modified
apps/api/src/repositories/misc.ts
+4 −3
@@ -1,6 +1,6 @@ | ||
| 1 | 1 | /** Smaller public repositories: time series, sources, news, sitemap feeds. */ |
| 2 | 2 | import type { SourceRef } from "@dci/core"; |
| 3 | −import { pg, page, type Fragment } from "../lib/sql.js"; | |
| 3 | +import { pg, page, likePattern, type Fragment } from "../lib/sql.js"; | |
| 4 | 4 | import { int, iso, json, num, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js"; |
| 5 | 5 | import { sourceRef } from "../lib/dto.js"; |
| 6 | 6 | |
@@ -57,7 +57,7 @@ export async function listNews(f: { country?: string; operator?: string; since?: | ||
| 57 | 57 | if (f.country) c.push(sql`${f.country.toUpperCase()} = any(n.country_iso2s)`); |
| 58 | 58 | if (f.operator) c.push(sql`exists (select 1 from operators o where o.id = any(n.operator_ids) and (o.slug = ${f.operator} or o.id = ${f.operator}))`); |
| 59 | 59 | if (f.since) c.push(sql`coalesce(n.published_at, n.created_at) >= ${f.since}::timestamptz`); |
| 60 | − if (f.q) c.push(sql`(n.title ilike ${"%" + f.q + "%"} or n.summary ilike ${"%" + f.q + "%"})`); | |
| 60 | + if (f.q) c.push(sql`(n.title ilike ${likePattern(f.q)} or n.summary ilike ${likePattern(f.q)})`); | |
| 61 | 61 | const rows = await sql<Row[]>` |
| 62 | 62 | select n.*, s.name as source_name, s.kind as source_kind, |
| 63 | 63 | (select coalesce(jsonb_agg(jsonb_build_object('id', o.id, 'slug', o.slug, 'name', o.name)), '[]'::jsonb) from operators o where o.id = any(n.operator_ids)) as ops, |
@@ -81,7 +81,8 @@ export async function sitemap(kind: SitemapKind, pageNo: number): Promise<{ item | ||
| 81 | 81 | const sql = pg(); |
| 82 | 82 | const offset = Math.max(0, pageNo - 1) * SITEMAP_PAGE; |
| 83 | 83 | const table = kind === "cloud-regions" ? "cloud_regions" : kind; |
| 84 | − const where = kind === "facilities" ? sql`where merged_into is null` : sql``; | |
| 84 | + // facilities: merged duplicates hidden; projects: false positives (hidden) and merged duplicates never listed | |
| 85 | + const where = kind === "facilities" ? sql`where merged_into is null` : kind === "projects" ? sql`where hidden = false and merged_into is null` : sql``; | |
| 85 | 86 | const rows = await sql<Row[]>`select slug, updated_at, count(*) over() as total from ${sql(table)} ${where} order by slug limit ${SITEMAP_PAGE} offset ${offset}`; |
| 86 | 87 | const total = rows.length ? int(rows[0]!.total) : int((await sql<Row[]>`select count(*)::int as n from ${sql(table)} ${where}`)[0]?.n); |
| 87 | 88 | return { items: rows.map((r) => ({ slug: reqStr(r.slug), updatedAt: reqIso(r.updated_at) })), total, page: pageNo, perPage: SITEMAP_PAGE }; |
modified
apps/api/src/repositories/operators.ts
+115 −43
@@ -1,18 +1,27 @@ | ||
| 1 | −import type { OperatorDetail, OperatorSummary, SourceRef } from "@dci/core"; | |
| 2 | −import { pg, andAll, page, mwExpr, plannedMwExpr, OPERATIONAL_SET, PIPELINE_SET, likePattern, type Fragment, type Sql } from "../lib/sql.js"; | |
| 3 | −import { int, num, record, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js"; | |
| 4 | −import { statusBreakdown } from "../lib/dto.js"; | |
| 1 | +import type { EventDTO, OperatorDetail, OperatorSummary, SourceRef } from "@dci/core"; | |
| 2 | +import { pg, andAll, page, likePattern, facilityView, knownMwAgg, countedAgg, facilityJoins, facilitySummaryCols, mwExpr, cloudRegionCols, eventCols, eventJoins, projectLive, OPERATIONAL_SET, AI_LEVELS, CORPORATE_EVENT_TYPES, type Fragment, type Sql } from "../lib/sql.js"; | |
| 3 | +import { bool, int, num, record, reqIso, reqStr, str, strArray, type Row } from "../lib/rows.js"; | |
| 4 | +import { cloudRegionSummary, facilitySummary, round2, share, statusBreakdown } from "../lib/dto.js"; | |
| 5 | 5 | import { findBySlugOrId } from "../lib/resolve.js"; |
| 6 | 6 | import { listFacilities, provenanceFor } from "./facilities.js"; |
| 7 | 7 | import { projectsWhere } from "./projects.js"; |
| 8 | −import { eventsForOperator } from "./events.js"; | |
| 8 | +import { eventsForOperator, toEventDtos } from "./events.js"; | |
| 9 | 9 | import { buildSourceHistory, documentVersionsFor, sourceIdsOf, sourceRefsFor } from "../lib/source-history.js"; |
| 10 | +import { dimAggCols, expansionVelocity, pipelineBreakdown, scopeFor } from "../lib/pipeline.js"; | |
| 11 | +import { claimsFor, dataQualityFor } from "../lib/quality.js"; | |
| 10 | 12 | |
| 11 | 13 | export interface OperatorFilters { q?: string; kind?: string; country?: string; sort?: "facilities" | "name" | "mw"; order?: "asc" | "desc"; page?: number; per_page?: number } |
| 12 | 14 | |
| 13 | −function operatorSummary(r: Row): OperatorSummary { | |
| 15 | +/** | |
| 16 | + * OperatorSummary from a row carrying `stats` (worker-refreshed, containment-aware) and optional live aggregate | |
| 17 | + * columns (facility_count, known_mw, …). Live columns win when present (they reflect the current database); the | |
| 18 | + * stats jsonb fills what the live query did not compute. | |
| 19 | + */ | |
| 20 | +export function operatorSummary(r: Row): OperatorSummary { | |
| 14 | 21 | const stats = (r.stats as Record<string, unknown> | null) ?? {}; |
| 15 | − const fc = int(r.facility_count); | |
| 22 | + const live = r.facility_count != null; | |
| 23 | + const fc = live ? int(r.facility_count) : int(stats.facilityCount); | |
| 24 | + const withMw = live ? int(r.with_mw) : null; | |
| 16 | 25 | return { |
| 17 | 26 | id: reqStr(r.id), |
| 18 | 27 | slug: reqStr(r.slug), |
@@ -20,28 +29,44 @@ function operatorSummary(r: Row): OperatorSummary { | ||
| 20 | 29 | kind: (str(r.kind) as OperatorSummary["kind"]) ?? null, |
| 21 | 30 | website: str(r.website), |
| 22 | 31 | hqCountryIso2: str(r.hq_country_iso2), |
| 23 | − facilityCount: fc || int(stats.facilityCount), | |
| 24 | − countryCount: int(r.country_count) || int(stats.countryCount), | |
| 25 | − metroCount: int(r.metro_count) || int(stats.metroCount), | |
| 26 | − knownMw: num(r.known_mw) ?? num(stats.knownMw), | |
| 27 | − plannedMw: num(r.planned_mw) ?? num(stats.plannedMw), | |
| 28 | − projectCount: int(r.project_count) || int(stats.projectCount), | |
| 32 | + facilityCount: fc, | |
| 33 | + countryCount: live ? int(r.country_count) : int(stats.countryCount), | |
| 34 | + metroCount: live ? int(r.metro_count) : int(stats.metroCount), | |
| 35 | + knownMw: live ? round2(num(r.known_mw)) : round2(num(stats.knownMw)), | |
| 36 | + plannedMw: live ? round2(num(r.planned_mw)) : round2(num(stats.plannedMw)), | |
| 37 | + constructionMw: live ? round2(num(r.construction_mw)) : round2(num(stats.constructionMw)), | |
| 38 | + projectCount: r.project_count != null ? int(r.project_count) : int(stats.projectCount), | |
| 39 | + projectPlannedMw: r.project_planned_mw != null ? round2(num(r.project_planned_mw)) : round2(num(stats.projectPlannedMw)), | |
| 40 | + aiCount: live ? int(r.ai) : int(stats.aiCount), | |
| 41 | + cloudRegionCount: r.cloud_region_count != null ? int(r.cloud_region_count) : int(stats.cloudRegionCount), | |
| 42 | + mwCoverage: withMw != null ? (fc ? share(withMw, fc) : 0) : num(stats.mwCoverage), | |
| 43 | + isCloudProvider: bool(r.is_cloud_provider), | |
| 44 | + isCarrier: bool(r.is_carrier), | |
| 29 | 45 | }; |
| 30 | 46 | } |
| 31 | 47 | |
| 32 | −/** Aggregate over live facilities per operator, optionally restricted (country / metro). */ | |
| 48 | +/** Containment-aware aggregate per operator over the facility view, optionally restricted to a country / metro. */ | |
| 33 | 49 | function aggFragment(sql: Sql, scope: { countryIso2?: string; metroId?: string } = {}): Fragment { |
| 34 | − const conds: Fragment[] = [sql`f.merged_into is null`, sql`f.operator_id is not null`]; | |
| 50 | + const conds: Fragment[] = [sql`f.operator_id is not null`]; | |
| 35 | 51 | if (scope.countryIso2) conds.push(sql`f.country_iso2 = ${scope.countryIso2}`); |
| 36 | 52 | if (scope.metroId) conds.push(sql`f.metro_id = ${scope.metroId}`); |
| 37 | 53 | return sql`( |
| 38 | − select f.operator_id, count(*)::int as facility_count, count(distinct f.country_iso2)::int as country_count, count(distinct f.metro_id)::int as metro_count, | |
| 39 | − sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw, | |
| 40 | − sum(${plannedMwExpr(sql)}) filter (where f.status = any(${PIPELINE_SET}))::float as planned_mw | |
| 41 | − from facilities f where ${andAll(sql, conds)} group by f.operator_id | |
| 54 | + select f.operator_id, count(distinct f.country_iso2)::int as country_count, count(distinct f.metro_id)::int as metro_count, ${dimAggCols(sql)} | |
| 55 | + from ${facilityView(sql)} f where ${andAll(sql, conds)} group by f.operator_id | |
| 42 | 56 | )`; |
| 43 | 57 | } |
| 44 | 58 | |
| 59 | +function projectAgg(sql: Sql): Fragment { | |
| 60 | + return sql`(select p.operator_id, count(*)::int as n, sum(p.planned_mw) filter (where p.status in ('rumored','proposed','announced','permitting','approved','delayed','under_construction'))::float as mw from projects p where ${projectLive(sql)} group by p.operator_id)`; | |
| 61 | +} | |
| 62 | + | |
| 63 | +const OPERATOR_COLS = (sql: Sql) => sql` | |
| 64 | + o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, o.stats, o.is_cloud_provider, o.is_carrier, | |
| 65 | + coalesce(a.facilities, 0) as facility_count, coalesce(a.country_count, 0) as country_count, coalesce(a.metro_count, 0) as metro_count, | |
| 66 | + a.known_mw, a.planned_mw, a.construction_mw, coalesce(a.with_mw, 0) as with_mw, coalesce(a.ai, 0) as ai, | |
| 67 | + coalesce(pc.n, 0) as project_count, pc.mw as project_planned_mw, | |
| 68 | + (select count(*)::int from cloud_regions cr where cr.provider_id = o.id and cr.status <> 'retired') as cloud_region_count`; | |
| 69 | + | |
| 45 | 70 | export async function listOperators(f: OperatorFilters): Promise<{ items: OperatorSummary[]; total: number; page: number; perPage: number }> { |
| 46 | 71 | const sql = pg(); |
| 47 | 72 | const pg_ = page(f.page, f.per_page, 100, 50); |
@@ -50,66 +75,113 @@ export async function listOperators(f: OperatorFilters): Promise<{ items: Operat | ||
| 50 | 75 | if (f.kind) conds.push(sql`o.kind = ${f.kind}`); |
| 51 | 76 | if (f.country) { const iso = f.country.toUpperCase(); conds.push(sql`(o.hq_country_iso2 = ${iso} or exists (select 1 from facilities x where x.operator_id = o.id and x.country_iso2 = ${iso} and x.merged_into is null))`); } |
| 52 | 77 | const asc = f.order === "asc"; |
| 53 | − const order = f.sort === "name" ? (asc ? sql`o.name asc` : sql`o.name desc`) : f.sort === "mw" ? (asc ? sql`a.known_mw asc nulls last, o.name` : sql`a.known_mw desc nulls last, o.name`) : asc ? sql`coalesce(a.facility_count, 0) asc, o.name` : sql`coalesce(a.facility_count, 0) desc, o.name`; | |
| 78 | + const order = f.sort === "name" ? (asc ? sql`o.name asc` : sql`o.name desc`) : f.sort === "mw" ? (asc ? sql`a.known_mw asc nulls last, o.name` : sql`a.known_mw desc nulls last, o.name`) : asc ? sql`coalesce(a.facilities, 0) asc, o.name` : sql`coalesce(a.facilities, 0) desc, o.name`; | |
| 54 | 79 | const rows = await sql<Row[]>` |
| 55 | − select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, o.stats, | |
| 56 | − coalesce(a.facility_count, 0) as facility_count, coalesce(a.country_count, 0) as country_count, coalesce(a.metro_count, 0) as metro_count, a.known_mw, a.planned_mw, | |
| 57 | − coalesce(pc.n, 0) as project_count, count(*) over() as total | |
| 80 | + select ${OPERATOR_COLS(sql)}, count(*) over() as total | |
| 58 | 81 | from operators o |
| 59 | 82 | left join ${aggFragment(sql)} a on a.operator_id = o.id |
| 60 | − left join (select operator_id, count(*)::int as n from projects group by operator_id) pc on pc.operator_id = o.id | |
| 83 | + left join ${projectAgg(sql)} pc on pc.operator_id = o.id | |
| 61 | 84 | where ${andAll(sql, conds)} |
| 62 | 85 | order by ${order} |
| 63 | 86 | limit ${pg_.perPage} offset ${pg_.offset}`; |
| 64 | 87 | return { items: rows.map(operatorSummary), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }; |
| 65 | 88 | } |
| 66 | 89 | |
| 67 | −/** Operators ranked by facilities inside a scope (country or metro). */ | |
| 90 | +/** Operators ranked by facilities inside a scope (country or metro) — containment-aware counts. */ | |
| 68 | 91 | export async function operatorsForScope(scope: { countryIso2?: string; metroId?: string }, limit = 10): Promise<OperatorSummary[]> { |
| 69 | 92 | const sql = pg(); |
| 70 | 93 | const rows = await sql<Row[]>` |
| 71 | − select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, '{}'::jsonb as stats, | |
| 72 | − a.facility_count, a.country_count, a.metro_count, a.known_mw, a.planned_mw, 0 as project_count | |
| 94 | + select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, '{}'::jsonb as stats, o.is_cloud_provider, o.is_carrier, | |
| 95 | + a.facilities as facility_count, a.country_count, a.metro_count, a.known_mw, a.planned_mw, a.construction_mw, a.with_mw, a.ai, 0 as project_count, null::float as project_planned_mw, 0 as cloud_region_count | |
| 73 | 96 | from ${aggFragment(sql, scope)} a join operators o on o.id = a.operator_id |
| 74 | − order by a.facility_count desc, a.known_mw desc nulls last, o.name limit ${limit}`; | |
| 97 | + where a.facilities > 0 | |
| 98 | + order by a.facilities desc, a.known_mw desc nulls last, o.name limit ${limit}`; | |
| 75 | 99 | return rows.map(operatorSummary); |
| 76 | 100 | } |
| 77 | 101 | |
| 102 | +/** OperatorSummary for one id (live aggregates). null when unknown. */ | |
| 103 | +export async function operatorSummaryById(id: string): Promise<OperatorSummary | null> { | |
| 104 | + const sql = pg(); | |
| 105 | + const rows = await sql<Row[]>`select ${OPERATOR_COLS(sql)} from operators o left join ${aggFragment(sql)} a on a.operator_id = o.id left join ${projectAgg(sql)} pc on pc.operator_id = o.id where o.id = ${id}`; | |
| 106 | + return rows[0] ? operatorSummary(rows[0]) : null; | |
| 107 | +} | |
| 108 | + | |
| 109 | +/** Corporate events (acquisitions, financing, partnerships, executive changes…) linked to an operator. */ | |
| 110 | +export async function corporateEventsForOperator(operatorId: string, limit = 30): Promise<EventDTO[]> { | |
| 111 | + const sql = pg(); | |
| 112 | + const rows = await sql<Row[]>` | |
| 113 | + select ${eventCols(sql)} from events e ${eventJoins(sql)} | |
| 114 | + where (e.operator_id = ${operatorId} or (e.entity_type = 'operator' and e.entity_id = ${operatorId})) and e.event_type = any(${CORPORATE_EVENT_TYPES}) and e.review_status <> 'rejected' | |
| 115 | + order by e.detected_at desc limit ${limit}`; | |
| 116 | + return toEventDtos(rows); | |
| 117 | +} | |
| 118 | + | |
| 119 | +/** Metros first entered by the operator within the last `months` months (by earliest facility opened_on / first_seen or project announcement). */ | |
| 120 | +export async function newMarkets(operatorId: string, months = 12): Promise<Array<{ id: string; slug: string; name: string }>> { | |
| 121 | + const sql = pg(); | |
| 122 | + const rows = await sql<Row[]>` | |
| 123 | + with fm as ( | |
| 124 | + select f.metro_id, min(case when f.opened_on ~ '^\\d{4}-\\d{2}-\\d{2}' then f.opened_on::date when f.opened_on ~ '^\\d{4}-\\d{2}$' then (f.opened_on || '-01')::date when f.opened_on ~ '^\\d{4}$' then (f.opened_on || '-01-01')::date else f.first_seen::date end) as first_d | |
| 125 | + from facilities f where f.merged_into is null and f.operator_id = ${operatorId} and f.metro_id is not null group by 1 | |
| 126 | + union all | |
| 127 | + select p.metro_id, min(case when p.announced_on ~ '^\\d{4}-\\d{2}-\\d{2}' then p.announced_on::date when p.announced_on ~ '^\\d{4}-\\d{2}$' then (p.announced_on || '-01')::date when p.announced_on ~ '^\\d{4}$' then (p.announced_on || '-01-01')::date else p.created_at::date end) | |
| 128 | + from projects p where ${projectLive(sql)} and p.operator_id = ${operatorId} and p.metro_id is not null group by 1 | |
| 129 | + ), first as (select metro_id, min(first_d) as first_d from fm group by 1) | |
| 130 | + select m.id, m.slug, m.name from first join metros m on m.id = first.metro_id where first.first_d >= current_date - make_interval(months => ${months}) order by first.first_d desc limit 25`; | |
| 131 | + return rows.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) })); | |
| 132 | +} | |
| 133 | + | |
| 78 | 134 | export async function getOperatorDetail(idOrSlug: string, fPage = 1): Promise<{ detail: OperatorDetail; sources: SourceRef[] } | null> { |
| 79 | 135 | const sql = pg(); |
| 80 | 136 | const row = await findBySlugOrId("operators", idOrSlug); |
| 81 | 137 | if (!row) return null; |
| 82 | 138 | const id = String(row.id); |
| 83 | − const [sumRows, countryRows, metroRows, statusRows, facilities, projects, recentEvents, provenance, versions, parentRows] = await Promise.all([ | |
| 84 | − sql<Row[]>` | |
| 85 | − select o.id, o.slug, o.name, o.kind, o.website, o.hq_country_iso2, o.stats, | |
| 86 | − coalesce(a.facility_count, 0) as facility_count, coalesce(a.country_count, 0) as country_count, coalesce(a.metro_count, 0) as metro_count, a.known_mw, a.planned_mw, | |
| 87 | − (select count(*)::int from projects p where p.operator_id = o.id) as project_count | |
| 88 | − from operators o left join ${aggFragment(sql)} a on a.operator_id = o.id where o.id = ${id}`, | |
| 139 | + const scope = scopeFor(sql, { operatorId: id }); | |
| 140 | + const known = knownMwAgg(sql), counted = countedAgg(sql); | |
| 141 | + const [sumRows, countryRows, metroRows, statusRows, facilities, projects, recentEvents, provenance, versions, parentRows, pipeline, velocity, aiRows, cloudRows, corporateEvents, claims] = await Promise.all([ | |
| 142 | + sql<Row[]>`select ${OPERATOR_COLS(sql)} from operators o left join ${aggFragment(sql)} a on a.operator_id = o.id left join ${projectAgg(sql)} pc on pc.operator_id = o.id where o.id = ${id}`, | |
| 89 | 143 | sql<Row[]>` |
| 90 | − select f.country_iso2 as iso2, c.name, c.slug, count(*)::int as n, sum(${mwExpr(sql)}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw | |
| 91 | − from facilities f left join countries c on c.iso2 = f.country_iso2 | |
| 92 | − where f.operator_id = ${id} and f.merged_into is null and f.country_iso2 is not null | |
| 144 | + select f.country_iso2 as iso2, c.name, c.slug, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw | |
| 145 | + from ${facilityView(sql)} f left join countries c on c.iso2 = f.country_iso2 | |
| 146 | + where f.operator_id = ${id} and f.country_iso2 is not null | |
| 93 | 147 | group by 1, 2, 3 order by n desc, c.name`, |
| 94 | 148 | sql<Row[]>` |
| 95 | − select m.id, m.slug, m.name, m.country_iso2, count(*)::int as n | |
| 96 | − from facilities f join metros m on m.id = f.metro_id | |
| 97 | − where f.operator_id = ${id} and f.merged_into is null group by 1, 2, 3, 4 order by n desc, m.name limit 100`, | |
| 149 | + select m.id, m.slug, m.name, m.country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw | |
| 150 | + from ${facilityView(sql)} f join metros m on m.id = f.metro_id | |
| 151 | + where f.operator_id = ${id} group by 1, 2, 3, 4 order by n desc, m.name limit 100`, | |
| 98 | 152 | sql<Row[]>`select status, count(*)::int as n from facilities where operator_id = ${id} and merged_into is null group by status`, |
| 99 | 153 | listFacilities({ operatorId: id, page: fPage, per_page: 50, sort: "mw", order: "desc" }), |
| 100 | − projectsWhere(sql`p.operator_id = ${id}`, 20), | |
| 154 | + projectsWhere(sql`${projectLive(sql)} and p.operator_id = ${id}`, 20), | |
| 101 | 155 | eventsForOperator(id, 20), |
| 102 | 156 | provenanceFor("operator", id), |
| 103 | 157 | documentVersionsFor("operator", id), |
| 104 | 158 | row.parent_id ? sql<Row[]>`select id, slug, name from operators where id = ${String(row.parent_id)}` : Promise.resolve([] as Row[]), |
| 159 | + pipelineBreakdown(scope), | |
| 160 | + expansionVelocity(scope), | |
| 161 | + sql<Row[]>`select ${facilitySummaryCols(sql)} from facilities f ${facilityJoins(sql)} where f.merged_into is null and (f.operator_id = ${id} or f.owner_id = ${id}) and f.ai_evidence = any(${AI_LEVELS}) order by ${mwExpr(sql)} desc nulls last, f.name limit 20`, | |
| 162 | + sql<Row[]>`select ${cloudRegionCols(sql)} from cloud_regions r join operators pr on pr.id = r.provider_id where r.provider_id = ${id} order by r.country_iso2, r.code limit 500`, | |
| 163 | + corporateEventsForOperator(id, 30), | |
| 164 | + claimsFor("operator", id), | |
| 105 | 165 | ]); |
| 106 | 166 | const summary = operatorSummary(sumRows[0] ?? row); |
| 167 | + const completeness = Math.round(([row.website, row.hq_country_iso2, row.kind, row.description].filter((v) => v != null && v !== "").length / 4) * 100); | |
| 168 | + const dataQuality = await dataQualityFor("operator", id, completeness, str(row.updated_at)); | |
| 169 | + const counted_total = summary.facilityCount || countryRows.reduce((a, c) => a + int(c.n), 0); | |
| 107 | 170 | const detail: OperatorDetail = { |
| 108 | 171 | ...summary, |
| 172 | + pipeline, | |
| 173 | + velocity, | |
| 174 | + topCountries: countryRows.slice(0, 10).map((c) => ({ iso2: reqStr(c.iso2), name: reqStr(c.name, reqStr(c.iso2)), slug: reqStr(c.slug, reqStr(c.iso2).toLowerCase()), facilityCount: int(c.n), knownMw: round2(num(c.known_mw)), share: share(int(c.n), counted_total) })), | |
| 175 | + topMetros: metroRows.slice(0, 10).map((m) => ({ id: reqStr(m.id), slug: reqStr(m.slug), name: reqStr(m.name), countryIso2: reqStr(m.country_iso2), facilityCount: int(m.n), knownMw: round2(num(m.known_mw)), share: share(int(m.n), counted_total) })), | |
| 176 | + aiFacilities: aiRows.map(facilitySummary), | |
| 177 | + cloudRegions: cloudRows.map(cloudRegionSummary), | |
| 178 | + corporateEvents, | |
| 179 | + claims, | |
| 180 | + dataQuality, | |
| 109 | 181 | aliases: strArray(row.aliases), |
| 110 | 182 | description: str(row.description), |
| 111 | 183 | parent: parentRows[0] ? { id: String(parentRows[0].id), slug: String(parentRows[0].slug), name: String(parentRows[0].name) } : null, |
| 112 | − countries: countryRows.map((c) => ({ iso2: reqStr(c.iso2), name: reqStr(c.name, reqStr(c.iso2)), slug: reqStr(c.slug, reqStr(c.iso2).toLowerCase()), facilityCount: int(c.n), knownMw: num(c.known_mw) })), | |
| 184 | + countries: countryRows.map((c) => ({ iso2: reqStr(c.iso2), name: reqStr(c.name, reqStr(c.iso2)), slug: reqStr(c.slug, reqStr(c.iso2).toLowerCase()), facilityCount: int(c.n), knownMw: round2(num(c.known_mw)) })), | |
| 113 | 185 | metros: metroRows.map((m) => ({ id: reqStr(m.id), slug: reqStr(m.slug), name: reqStr(m.name), countryIso2: reqStr(m.country_iso2), facilityCount: int(m.n) })), |
| 114 | 186 | statusBreakdown: statusBreakdown(statusRows), |
| 115 | 187 | facilities: facilities.items, |
@@ -119,7 +191,7 @@ export async function getOperatorDetail(idOrSlug: string, fPage = 1): Promise<{ | ||
| 119 | 191 | externalIds: record(row.external_ids), |
| 120 | 192 | updatedAt: reqIso(row.updated_at), |
| 121 | 193 | }; |
| 122 | − const sources = await sourceRefsFor(sourceIdsOf(provenance, recentEvents, versions)); | |
| 194 | + const sources = await sourceRefsFor(sourceIdsOf(provenance, recentEvents, versions, claims, corporateEvents)); | |
| 123 | 195 | return { detail, sources }; |
| 124 | 196 | } |
| 125 | 197 | |
added
apps/api/src/repositories/power.ts
+74 −0
@@ -0,0 +1,74 @@ | ||
| 1 | +/** | |
| 2 | + * /power — grid constraints, power / grid events, large loads (utility / grid MW published on facilities, current | |
| 3 | + * grid claims, ≥ 200 MW planned projects) and national energy context. Utility / grid MW are supply-side figures, | |
| 4 | + * never IT load; national grid averages never describe a facility's contracted electricity. | |
| 5 | + */ | |
| 6 | +import type { EnergyContext, PowerOverview } from "@dci/core"; | |
| 7 | +import { pg, facilityView, knownMwAgg, countedAgg, gridConstraintCols, gridConstraintJoins, projectLive, POWER_EVENT_TYPES, GRID_EVENT_TYPES, OPERATIONAL_SET } from "../lib/sql.js"; | |
| 8 | +import { int, num, reqStr, str, type Row } from "../lib/rows.js"; | |
| 9 | +import { gridConstraintDto, gridConstraintFromEvent, round2 } from "../lib/dto.js"; | |
| 10 | +import { eventsWhere } from "./events.js"; | |
| 11 | + | |
| 12 | +export const ENERGY_NOTE = "National grid averages (renewable share, total generation) describe the country's electricity system, not the electricity a given facility contracts (PPAs, on-site generation, utility tariffs). Grid carbon intensity is null until a reliable public source is connected."; | |
| 13 | +export const POWER_NOTE = "utilityCapacityMw / gridConnectionMw are supply-side figures (power available or contracted from the utility / grid) — never IT load and never comparable with itCapacityMw. Large loads list published site-scoped figures only (utility capacity, grid connection, planned load ≥ 200 MW on live projects). Grid constraints combine curated public reports (grid_constraints) with detected grid / utility / power-agreement events. Substations, power plants and transmission lines have no connector yet and are not shown; `utilities` lists operators whose name says energy / power / electric AND that appear as tenants or in events — empty otherwise."; | |
| 14 | + | |
| 15 | +export function energyContext(r: Row): EnergyContext { | |
| 16 | + return { renewableShare: num(r.renewable_share), electricityTwh: num(r.electricity_twh), statsYear: num(r.stats_year), gridCarbonIntensity: null, sourceName: str(r.energy_source_name), sourceUrl: str(r.energy_source_url), note: ENERGY_NOTE }; | |
| 17 | +} | |
| 18 | + | |
| 19 | +export async function powerOverview(): Promise<PowerOverview> { | |
| 20 | + const sql = pg(); | |
| 21 | + const known = knownMwAgg(sql), counted = countedAgg(sql); | |
| 22 | + const [gcRows, gridEvents, powerEvents, loadsF, loadsC, loadsP, energy, utilities] = await Promise.all([ | |
| 23 | + sql<Row[]>`select ${gridConstraintCols(sql)} from grid_constraints g ${gridConstraintJoins(sql)} order by g.effective_date desc nulls last, g.created_at desc limit 20`, | |
| 24 | + eventsWhere(sql`e.event_type = any(${GRID_EVENT_TYPES})`, 30), | |
| 25 | + eventsWhere(sql`e.event_type = any(${POWER_EVENT_TYPES})`, 30), | |
| 26 | + sql<Row[]>`select f.id, f.slug, f.name, f.country_iso2, f.utility_capacity_mw, f.grid_connection_mw, m.id as met_id, m.slug as met_slug, m.name as met_name, | |
| 27 | + (select s.name from provenance p join sources s on s.id = p.source_id where p.entity_type = 'facility' and p.entity_id = f.id and p.is_current and p.field in ('utilityCapacityMw', 'gridConnectionMw') order by p.is_winner desc, p.last_observed desc limit 1) as source_name, | |
| 28 | + (select p.url from provenance p where p.entity_type = 'facility' and p.entity_id = f.id and p.is_current and p.field in ('utilityCapacityMw', 'gridConnectionMw') order by p.is_winner desc, p.last_observed desc limit 1) as url | |
| 29 | + from facilities f left join metros m on m.id = f.metro_id | |
| 30 | + where f.merged_into is null and (f.utility_capacity_mw is not null or f.grid_connection_mw is not null) | |
| 31 | + order by greatest(coalesce(f.utility_capacity_mw, 0), coalesce(f.grid_connection_mw, 0)) desc limit 50`, | |
| 32 | + sql<Row[]>`select k.predicate, k.value, k.url, s.name as source_name, f.id, f.slug, f.name, f.country_iso2, m.id as met_id, m.slug as met_slug, m.name as met_name | |
| 33 | + from claims k join facilities f on f.id = k.subject_id and k.subject_type = 'facility' left join metros m on m.id = f.metro_id left join sources s on s.id = k.source_id | |
| 34 | + where k.status = 'current' and k.predicate in ('grid_connection_mw', 'utility_capacity_mw') and k.value is not null and f.merged_into is null | |
| 35 | + and ((k.predicate = 'grid_connection_mw' and f.grid_connection_mw is null) or (k.predicate = 'utility_capacity_mw' and f.utility_capacity_mw is null)) | |
| 36 | + order by k.value desc limit 50`, | |
| 37 | + sql<Row[]>`select p.id, p.slug, p.name, p.country_iso2, p.planned_mw, p.source_url, m.id as met_id, m.slug as met_slug, m.name as met_name, | |
| 38 | + (select s.name from provenance pr join sources s on s.id = pr.source_id where pr.entity_type = 'project' and pr.entity_id = p.id and pr.is_current and pr.field = 'plannedMw' order by pr.is_winner desc, pr.last_observed desc limit 1) as source_name | |
| 39 | + from projects p left join metros m on m.id = p.metro_id where ${projectLive(sql)} and p.planned_mw >= 200 order by p.planned_mw desc limit 50`, | |
| 40 | + sql<Row[]>`select c.iso2, c.name, c.slug, c.renewable_share, c.electricity_twh, c.stats_year, | |
| 41 | + (select s.name from provenance p join sources s on s.id = p.source_id where p.entity_type = 'country' and p.entity_id = c.iso2 and p.is_current and p.field in ('renewableShare', 'electricityTwh') limit 1) as energy_source_name, | |
| 42 | + (select p.url from provenance p where p.entity_type = 'country' and p.entity_id = c.iso2 and p.is_current and p.field in ('renewableShare', 'electricityTwh') limit 1) as energy_source_url, | |
| 43 | + fa.n as facilities, fa.known_mw | |
| 44 | + from countries c join (select f.country_iso2, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status = any(${OPERATIONAL_SET}))::float as known_mw from ${facilityView(sql)} f where f.country_iso2 is not null group by 1) fa on fa.country_iso2 = c.iso2 | |
| 45 | + where c.renewable_share is not null or c.electricity_twh is not null | |
| 46 | + order by fa.n desc limit 60`, | |
| 47 | + sql<Row[]>`select o.id, o.slug, o.name, o.kind, (select count(*)::int from facility_tenants t where t.operator_id = o.id) as tenancies, (select count(*)::int from facilities f where f.operator_id = o.id and f.merged_into is null) as facility_count | |
| 48 | + from operators o where o.name ~* '(energy|power|electric|utility|utilities)' | |
| 49 | + and (exists (select 1 from facility_tenants t where t.operator_id = o.id) or exists (select 1 from events e where e.operator_id = o.id or (e.entity_type = 'operator' and e.entity_id = o.id))) | |
| 50 | + order by tenancies desc, o.name limit 50`, | |
| 51 | + ]); | |
| 52 | + const gridConstraints = [...gcRows.map(gridConstraintDto), ...gridEvents.map(gridConstraintFromEvent)]; | |
| 53 | + const seen = new Set<string>(); | |
| 54 | + const deduped = gridConstraints.filter((g) => { const k = g.eventId ?? g.id; if (seen.has(k)) return false; seen.add(k); return true; }).slice(0, 30); | |
| 55 | + const metro = (r: Row) => (r.met_id ? { id: reqStr(r.met_id), slug: reqStr(r.met_slug), name: reqStr(r.met_name) } : null); | |
| 56 | + const largeLoads: PowerOverview["largeLoads"] = []; | |
| 57 | + for (const r of loadsF) { | |
| 58 | + const fac = { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }; | |
| 59 | + const u = num(r.utility_capacity_mw), g = num(r.grid_connection_mw); | |
| 60 | + if (u != null) largeLoads.push({ facility: fac, project: null, kind: "utility_capacity", mw: u, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.url) }); | |
| 61 | + if (g != null) largeLoads.push({ facility: fac, project: null, kind: "grid_connection", mw: g, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.url) }); | |
| 62 | + } | |
| 63 | + for (const r of loadsC) largeLoads.push({ facility: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, project: null, kind: reqStr(r.predicate) === "grid_connection_mw" ? "grid_connection" : "utility_capacity", mw: num(r.value) ?? 0, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.url) }); | |
| 64 | + for (const r of loadsP) largeLoads.push({ facility: null, project: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, kind: "planned_load", mw: num(r.planned_mw) ?? 0, countryIso2: str(r.country_iso2), metro: metro(r), sourceName: str(r.source_name), url: str(r.source_url) }); | |
| 65 | + largeLoads.sort((a, b) => b.mw - a.mw); | |
| 66 | + return { | |
| 67 | + gridConstraints: deduped, | |
| 68 | + powerEvents, | |
| 69 | + largeLoads: largeLoads.slice(0, 50), | |
| 70 | + countryEnergy: energy.map((r) => ({ iso2: reqStr(r.iso2), name: reqStr(r.name), slug: reqStr(r.slug), energy: energyContext(r), knownMw: round2(num(r.known_mw)), facilities: int(r.facilities) })), | |
| 71 | + utilities: utilities.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), facilityCount: int(r.facility_count), kind: str(r.kind) })), | |
| 72 | + note: POWER_NOTE, | |
| 73 | + }; | |
| 74 | +} | |
modified
apps/api/src/repositories/projects.ts
+162 −20
@@ -1,9 +1,11 @@ | ||
| 1 | −import type { FacilityStatus, ProjectDetail, ProjectSummary, SourceRef } from "@dci/core"; | |
| 2 | −import { partialDateSortKey } from "@dci/core"; | |
| 3 | −import { pg, andAll, projectJoins, projectSummaryCols, page, type Fragment } from "../lib/sql.js"; | |
| 1 | +import type { ClaimDTO, EntityHistory, EventDTO, FacilityStatus, HistoryPoint, ProjectDetail, ProjectStageDTO, ProjectSummary, SourceRef } from "@dci/core"; | |
| 2 | +import { PROJECT_STAGES, parsePartialDate, partialDateSortKey, projectTransition } from "@dci/core"; | |
| 3 | +import { pg, andAll, projectJoins, projectSummaryCols, projectLive, page, likePattern, yearExpr, AI_LEVELS, type Fragment } from "../lib/sql.js"; | |
| 4 | 4 | import { int, num, reqStr, str, type Row } from "../lib/rows.js"; |
| 5 | 5 | import { asStatus, projectSummary } from "../lib/dto.js"; |
| 6 | 6 | import { findBySlugOrId } from "../lib/resolve.js"; |
| 7 | +import { nearby } from "../lib/nearby.js"; | |
| 8 | +import { capacityHistory, claimsFor, dataQualityFor, entityHistory, eventsForSubject, provenanceAll } from "../lib/quality.js"; | |
| 7 | 9 | import { eventsForProject } from "./events.js"; |
| 8 | 10 | import { provenanceFor } from "./facilities.js"; |
| 9 | 11 | import { buildSourceHistory, documentVersionsFor, sourceIdsOf, sourceRefsFor } from "../lib/source-history.js"; |
@@ -15,7 +17,13 @@ export interface ProjectFilters { | ||
| 15 | 17 | metro?: string; // slug or id |
| 16 | 18 | facilityId?: string; |
| 17 | 19 | min_mw?: number; |
| 20 | + max_mw?: number; | |
| 18 | 21 | ai?: boolean; |
| 22 | + project_class?: string[]; | |
| 23 | + evidence_level?: string[]; | |
| 24 | + expected_from?: number; | |
| 25 | + expected_to?: number; | |
| 26 | + announced_since?: string; | |
| 19 | 27 | q?: string; |
| 20 | 28 | sort?: "updated" | "mw" | "announced" | "opening"; |
| 21 | 29 | order?: "asc" | "desc"; |
@@ -23,17 +31,24 @@ export interface ProjectFilters { | ||
| 23 | 31 | per_page?: number; |
| 24 | 32 | } |
| 25 | 33 | |
| 26 | −function conds(f: ProjectFilters): Fragment[] { | |
| 34 | +export function projectConds(f: ProjectFilters): Fragment[] { | |
| 27 | 35 | const sql = pg(); |
| 28 | − const c: Fragment[] = [sql`true`]; | |
| 36 | + const c: Fragment[] = [projectLive(sql)]; | |
| 29 | 37 | if (f.status?.length) c.push(sql`p.status = any(${f.status})`); |
| 30 | 38 | if (f.country) c.push(sql`p.country_iso2 = ${f.country.toUpperCase()}`); |
| 31 | 39 | if (f.operator) c.push(sql`(o.slug = ${f.operator} or o.id = ${f.operator})`); |
| 32 | 40 | if (f.metro) c.push(sql`(m.slug = ${f.metro} or m.id = ${f.metro})`); |
| 33 | 41 | if (f.facilityId) c.push(sql`p.facility_id = ${f.facilityId}`); |
| 34 | 42 | if (f.min_mw != null) c.push(sql`p.planned_mw >= ${f.min_mw}`); |
| 35 | − if (f.ai === true) c.push(sql`p.is_ai`); | |
| 36 | − if (f.q) c.push(sql`(p.name ilike ${"%" + f.q + "%"} or similarity(p.name, ${f.q}) > 0.3)`); | |
| 43 | + if (f.max_mw != null) c.push(sql`p.planned_mw <= ${f.max_mw}`); | |
| 44 | + if (f.ai === true) c.push(sql`(p.is_ai or p.ai_evidence = any(${AI_LEVELS}))`); | |
| 45 | + if (f.ai === false) c.push(sql`(not p.is_ai and p.ai_evidence <> all(${AI_LEVELS}))`); | |
| 46 | + if (f.project_class?.length) c.push(sql`p.project_class = any(${f.project_class})`); | |
| 47 | + if (f.evidence_level?.length) c.push(sql`p.evidence_level = any(${f.evidence_level})`); | |
| 48 | + if (f.expected_from != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} >= ${f.expected_from}`); | |
| 49 | + if (f.expected_to != null) c.push(sql`${yearExpr(sql, sql`p.expected_opening`)} <= ${f.expected_to}`); | |
| 50 | + if (f.announced_since) c.push(sql`(p.announced_on >= ${f.announced_since} or (p.announced_on is null and p.created_at >= ${f.announced_since}::timestamptz))`); | |
| 51 | + if (f.q) { const t = f.q.trim(); if (t) c.push(sql`(p.name ilike ${likePattern(t)} or similarity(p.name, ${t}) > 0.3 or o.name ilike ${likePattern(t)})`); } | |
| 37 | 52 | return c; |
| 38 | 53 | } |
| 39 | 54 | |
@@ -54,7 +69,7 @@ export async function listProjects(f: ProjectFilters): Promise<{ items: ProjectS | ||
| 54 | 69 | const rows = await sql<Row[]>` |
| 55 | 70 | select ${projectSummaryCols(sql)}, count(*) over() as total |
| 56 | 71 | from projects p ${projectJoins(sql)} |
| 57 | − where ${andAll(sql, conds(f))} | |
| 72 | + where ${andAll(sql, projectConds(f))} | |
| 58 | 73 | order by ${order(f.sort, f.order)} |
| 59 | 74 | limit ${pg_.perPage} offset ${pg_.offset}`; |
| 60 | 75 | return { items: rows.map(projectSummary), total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }; |
@@ -65,29 +80,126 @@ export async function projectsForFacility(facilityId: string, operatorId: string | ||
| 65 | 80 | const sql = pg(); |
| 66 | 81 | const rows = await sql<Row[]>` |
| 67 | 82 | select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} |
| 68 | − where p.facility_id = ${facilityId} | |
| 69 | − ${operatorId && metroId ? sql`or (p.operator_id = ${operatorId} and p.metro_id = ${metroId})` : sql``} | |
| 83 | + where ${projectLive(sql)} and (p.facility_id = ${facilityId} | |
| 84 | + ${operatorId && metroId ? sql`or (p.operator_id = ${operatorId} and p.metro_id = ${metroId})` : sql``}) | |
| 70 | 85 | order by (p.facility_id = ${facilityId}) desc, p.last_update desc limit ${limit}`; |
| 71 | 86 | return rows.map(projectSummary); |
| 72 | 87 | } |
| 73 | 88 | |
| 89 | +/** Live projects matching a condition over `projects p` (+ projectJoins aliases o / pf / m). */ | |
| 74 | 90 | export async function projectsWhere(where: Fragment, limit = 10, orderBy?: Fragment): Promise<ProjectSummary[]> { |
| 75 | 91 | const sql = pg(); |
| 76 | − const rows = await sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${where} order by ${orderBy ?? sql`p.last_update desc`} limit ${limit}`; | |
| 92 | + const rows = await sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where ${projectLive(sql)} and (${where}) order by ${orderBy ?? sql`p.last_update desc`} limit ${limit}`; | |
| 77 | 93 | return rows.map(projectSummary); |
| 78 | 94 | } |
| 79 | 95 | |
| 80 | −export async function getProjectDetail(idOrSlug: string): Promise<{ detail: ProjectDetail; sources: SourceRef[] } | null> { | |
| 96 | +/** Resolve a project by slug or id; merged projects redirect to their survivor, hidden projects are never served. */ | |
| 97 | +export async function resolveProjectRow(idOrSlug: string): Promise<Row | null> { | |
| 81 | 98 | const sql = pg(); |
| 82 | − const row = await findBySlugOrId("projects", idOrSlug); | |
| 99 | + const base = await findBySlugOrId("projects", idOrSlug); | |
| 100 | + if (!base) return null; | |
| 101 | + let row = base; | |
| 102 | + for (let hops = 0; hops < 5 && str(row.merged_into); hops++) { | |
| 103 | + const next = (await sql<Row[]>`select * from projects where id = ${String(row.merged_into)} limit 1`)[0]; | |
| 104 | + if (!next) break; | |
| 105 | + row = next; | |
| 106 | + } | |
| 107 | + if (row.hidden === true) return null; | |
| 108 | + return row; | |
| 109 | +} | |
| 110 | + | |
| 111 | +// ─── lifecycle stages ────────────────────────────────────────────────────────────────────────────── | |
| 112 | + | |
| 113 | +type Stage = ProjectStageDTO["stage"]; | |
| 114 | +const STAGE_COLUMN: Partial<Record<Stage, string>> = { announced: "announced_on", permitting: "permit_filed_on", approved: "approved_on", under_construction: "construction_started_on", operational: "opened_on" }; | |
| 115 | +const TIMELINE_STAGE: Array<[RegExp, Stage]> = [ | |
| 116 | + [/cancel/i, "cancelled"], [/delay|on hold|paused/i, "delayed"], [/open|launch|live|commission|energi[sz]ed/i, "operational"], [/partial/i, "partially_operational"], | |
| 117 | + [/construction|ground|topped/i, "under_construction"], [/approv|consent|permit(ted)?\b|green/i, "approved"], [/planning|filed|permit|zoning|application/i, "permitting"], | |
| 118 | + [/announce|unveil|reveal|plan/i, "announced"], [/propos/i, "proposed"], [/rumo/i, "rumored"], | |
| 119 | +]; | |
| 120 | +const EVENT_STAGE: Record<string, Stage> = { construction_started: "under_construction", planning_filed: "permitting", planning_approved: "approved", project_delayed: "delayed", project_cancelled: "cancelled", facility_opened: "operational", project_announced: "announced" }; | |
| 121 | + | |
| 122 | +interface StageEvidence { date: string; url: string | null; sourceName: string | null; eventId: string | null } | |
| 123 | + | |
| 124 | +function isStage(s: unknown): s is Stage { return typeof s === "string" && (PROJECT_STAGES as readonly string[]).includes(s); } | |
| 125 | + | |
| 126 | +export function buildStages(row: Row, timeline: Array<{ date: string; type: string; url: string | null; sourceName: string | null }>, events: EventDTO[]): ProjectStageDTO[] { | |
| 127 | + const current = asStatus(row.status); | |
| 128 | + const ev = new Map<Stage, StageEvidence>(); | |
| 129 | + const put = (stage: Stage, e: StageEvidence) => { | |
| 130 | + if (!e.date || !/^\d{4}/.test(e.date)) return; | |
| 131 | + const prev = ev.get(stage); | |
| 132 | + // earliest dated evidence wins; a dated column beats a detected_at fallback | |
| 133 | + if (!prev || partialDateSortKey(e.date) < partialDateSortKey(prev.date)) ev.set(stage, e); | |
| 134 | + }; | |
| 135 | + for (const [stage, col] of Object.entries(STAGE_COLUMN) as Array<[Stage, string]>) { const d = str(row[col]); if (d) put(stage, { date: d, url: str(row.source_url), sourceName: null, eventId: null }); } | |
| 136 | + if (current === "operational" && !ev.has("operational") && str(row.expected_opening) && partialDateSortKey(str(row.expected_opening)) <= Date.now()) put("operational", { date: String(row.expected_opening), url: str(row.source_url), sourceName: null, eventId: null }); | |
| 137 | + for (const t of timeline) { const stage = TIMELINE_STAGE.find(([re]) => re.test(t.type))?.[1]; if (stage) put(stage, { date: t.date, url: t.url, sourceName: t.sourceName, eventId: null }); } | |
| 138 | + for (const e of events) { | |
| 139 | + let stage: Stage | undefined = EVENT_STAGE[e.eventType]; | |
| 140 | + if (e.eventType === "project_status_changed" || e.eventType === "status_changed") { const to = typeof e.newValue === "object" && e.newValue ? (e.newValue as Record<string, unknown>).status ?? (e.newValue as Record<string, unknown>).to : e.newValue; stage = isStage(to) ? to : undefined; } | |
| 141 | + if (stage) put(stage, { date: e.effectiveDate ?? e.detectedAt.slice(0, 10), url: e.url || null, sourceName: e.sourceName, eventId: e.id }); | |
| 142 | + } | |
| 143 | + // `expansion` is an operational site growing — it sits at the operational stage of the lifecycle view | |
| 144 | + const currentStage: Stage | null = current === "expansion" ? "operational" : isStage(current) ? current : null; | |
| 145 | + return PROJECT_STAGES.map((stage) => { | |
| 146 | + const isCurrent = currentStage === stage; | |
| 147 | + const side = stage === "delayed" || stage === "cancelled"; | |
| 148 | + // a stage is reached when it is the current one, when dated evidence placed the project there, or when the pipeline | |
| 149 | + // moved forward past it (a project under construction was announced); side branches only when current or dated | |
| 150 | + const forward = !side && currentStage != null && (currentStage === "delayed" || currentStage === "cancelled" ? ["rumored", "proposed", "announced"].includes(stage) : projectTransition(stage, currentStage) === "forward"); | |
| 151 | + const reached = isCurrent || ev.has(stage) || forward; | |
| 152 | + const e = ev.get(stage); | |
| 153 | + return { stage, date: e?.date ?? null, reached, current: isCurrent, url: e?.url ?? null, sourceName: e?.sourceName ?? null, eventId: e?.eventId ?? null }; | |
| 154 | + }); | |
| 155 | +} | |
| 156 | + | |
| 157 | +function daysBetween(a: string | null, b: string | null): number | null { | |
| 158 | + if (!a || !b) return null; | |
| 159 | + const pa = parsePartialDate(a), pb = parsePartialDate(b); | |
| 160 | + if (!pa || !pb) return null; | |
| 161 | + const da = new Date(pa.length === 4 ? `${pa}-01-01` : pa.length === 7 ? `${pa}-01` : pa.slice(0, 10)); | |
| 162 | + const db = new Date(pb.length === 4 ? `${pb}-01-01` : pb.length === 7 ? `${pb}-01` : pb.slice(0, 10)); | |
| 163 | + if (Number.isNaN(da.getTime()) || Number.isNaN(db.getTime())) return null; | |
| 164 | + return Math.round((db.getTime() - da.getTime()) / 86_400_000); | |
| 165 | +} | |
| 166 | + | |
| 167 | +export function velocityDays(stages: ProjectStageDTO[]): ProjectDetail["velocityDays"] { | |
| 168 | + const d = (s: Stage) => stages.find((x) => x.stage === s)?.date ?? null; | |
| 169 | + return { | |
| 170 | + announcedToPermitting: daysBetween(d("announced"), d("permitting")), | |
| 171 | + permittingToApproval: daysBetween(d("permitting"), d("approved")), | |
| 172 | + approvalToConstruction: daysBetween(d("approved"), d("under_construction")), | |
| 173 | + constructionToOpening: daysBetween(d("under_construction"), d("operational")), | |
| 174 | + announcedToConstruction: daysBetween(d("announced"), d("under_construction")), | |
| 175 | + }; | |
| 176 | +} | |
| 177 | + | |
| 178 | +const COMPLETENESS_COLS = ["operator_id", "country_iso2", "lat", "planned_mw", "expected_opening", "announced_on", "description"]; | |
| 179 | +export function projectCompleteness(row: Row): number { | |
| 180 | + return Math.round((COMPLETENESS_COLS.filter((c) => row[c] != null && row[c] !== "").length / COMPLETENESS_COLS.length) * 100); | |
| 181 | +} | |
| 182 | + | |
| 183 | +function investmentPoints(claims: ClaimDTO[]): HistoryPoint[] { | |
| 184 | + return claims.filter((c) => /usd$/.test(c.predicate) && c.status !== "rejected").map((c) => ({ date: c.publishedAt && /^\d{4}/.test(c.publishedAt) ? c.publishedAt : c.firstObserved, field: "investmentUsd", predicate: c.predicate, value: c.value ?? c.valueText, sourceId: c.sourceId, sourceName: c.sourceName, sourceKind: c.sourceKind, url: c.url, claimId: c.id, kind: "claim" as const })); | |
| 185 | +} | |
| 186 | + | |
| 187 | +export async function getProjectDetail(idOrSlug: string, opts: { radiusKm?: number } = {}): Promise<{ detail: ProjectDetail; sources: SourceRef[] } | null> { | |
| 188 | + const sql = pg(); | |
| 189 | + const row = await resolveProjectRow(idOrSlug); | |
| 83 | 190 | if (!row) return null; |
| 84 | 191 | const id = String(row.id); |
| 85 | − const [sumRows, timelineRows, provenance, events, versions] = await Promise.all([ | |
| 192 | + const lat = num(row.lat), lng = num(row.lng); | |
| 193 | + const [sumRows, timelineRows, provenance, events, versions, campusRows, claims, nearbyInfra, dataQuality] = await Promise.all([ | |
| 86 | 194 | sql<Row[]>`select ${projectSummaryCols(sql)} from projects p ${projectJoins(sql)} where p.id = ${id}`, |
| 87 | 195 | sql<Row[]>`select t.event_date, t.event_type, t.description, t.url, s.name as source_name from project_timeline t left join sources s on s.id = t.source_id where t.project_id = ${id}`, |
| 88 | 196 | provenanceFor("project", id), |
| 89 | 197 | eventsForProject(id, 50), |
| 90 | 198 | documentVersionsFor("project", id), |
| 199 | + row.campus_id ? sql<Row[]>`select id, slug, name from campuses where id = ${String(row.campus_id)}` : Promise.resolve([] as Row[]), | |
| 200 | + claimsFor("project", id), | |
| 201 | + lat != null && lng != null ? nearby({ lat, lng, radiusKm: opts.radiusKm ?? 25, excludeProjectId: id, limitPerType: 50 }) : Promise.resolve(null), | |
| 202 | + dataQualityFor("project", id, projectCompleteness(row), null), | |
| 91 | 203 | ]); |
| 92 | 204 | const summary = projectSummary(sumRows[0] ?? row); |
| 93 | 205 | const timeline = timelineRows |
@@ -95,22 +207,51 @@ export async function getProjectDetail(idOrSlug: string): Promise<{ detail: Proj | ||
| 95 | 207 | .sort((a, b) => partialDateSortKey(a.date) - partialDateSortKey(b.date)); |
| 96 | 208 | const statusHistory = events |
| 97 | 209 | .filter((e) => e.eventType === "project_status_changed" || e.eventType === "status_changed") |
| 98 | − .map((e) => ({ date: e.effectiveDate ?? e.detectedAt, from: e.oldValue != null ? asStatus(e.oldValue) : null, to: asStatus(e.newValue), url: e.url || null })) | |
| 210 | + .map((e) => ({ date: e.effectiveDate ?? e.detectedAt, from: e.oldValue != null ? asStatus(typeof e.oldValue === "object" ? (e.oldValue as Record<string, unknown>).status : e.oldValue) : null, to: asStatus(typeof e.newValue === "object" && e.newValue ? (e.newValue as Record<string, unknown>).status : e.newValue), url: e.url || null })) | |
| 99 | 211 | .sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0)); |
| 212 | + const stages = buildStages(row, timeline, events); | |
| 213 | + const relatedEvents = events.filter((e) => e.project && e.project.id === id && !(e.entityType === "project" && e.entityId === id)); | |
| 214 | + const history = [...capacityHistory(provenance, claims, events), ...investmentPoints(claims)].sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0)); | |
| 100 | 215 | const detail: ProjectDetail = { |
| 101 | 216 | ...summary, |
| 102 | 217 | description: str(row.description), |
| 103 | 218 | sourceUrl: str(row.source_url), |
| 219 | + campus: campusRows[0] ? { id: reqStr(campusRows[0].id), slug: reqStr(campusRows[0].slug), name: reqStr(campusRows[0].name) } : null, | |
| 220 | + stages, | |
| 221 | + velocityDays: velocityDays(stages), | |
| 104 | 222 | timeline, |
| 105 | 223 | statusHistory, |
| 224 | + claims, | |
| 225 | + history, | |
| 226 | + nearbyInfrastructure: nearbyInfra, | |
| 227 | + relatedEvents, | |
| 228 | + dataQuality, | |
| 106 | 229 | provenance, |
| 107 | 230 | events, |
| 108 | 231 | sourceHistory: buildSourceHistory(provenance, events, versions), |
| 109 | 232 | }; |
| 110 | − const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions)); | |
| 233 | + const sources = await sourceRefsFor(sourceIdsOf(provenance, events, versions, claims)); | |
| 111 | 234 | return { detail, sources }; |
| 112 | 235 | } |
| 113 | 236 | |
| 237 | +/** /projects/:slug/history */ | |
| 238 | +export async function getProjectHistory(idOrSlug: string): Promise<{ history: EntityHistory; sources: SourceRef[] } | null> { | |
| 239 | + const row = await resolveProjectRow(idOrSlug); | |
| 240 | + if (!row) return null; | |
| 241 | + const id = String(row.id); | |
| 242 | + const [prov, claims, events] = await Promise.all([provenanceAll("project", id), claimsFor("project", id), eventsForSubject("project", id)]); | |
| 243 | + return { history: entityHistory("project", id, prov, claims, events), sources: await sourceRefsFor(sourceIdsOf(prov, claims, events)) }; | |
| 244 | +} | |
| 245 | + | |
| 246 | +/** /projects/:slug/claims */ | |
| 247 | +export async function getProjectClaims(idOrSlug: string, opts: { status?: string[]; predicate?: string } = {}): Promise<{ id: string; claims: ClaimDTO[]; sources: SourceRef[] } | null> { | |
| 248 | + const row = await resolveProjectRow(idOrSlug); | |
| 249 | + if (!row) return null; | |
| 250 | + const id = String(row.id); | |
| 251 | + const claims = await claimsFor("project", id, opts); | |
| 252 | + return { id, claims, sources: await sourceRefsFor(sourceIdsOf(claims)) }; | |
| 253 | +} | |
| 254 | + | |
| 114 | 255 | export interface PipelineAggregates { |
| 115 | 256 | byStatus: Array<{ status: FacilityStatus; count: number; mw: number | null; investmentUsd: number | null }>; |
| 116 | 257 | byYear: Array<{ year: string; count: number; mw: number | null }>; |
@@ -120,11 +261,12 @@ export interface PipelineAggregates { | ||
| 120 | 261 | |
| 121 | 262 | export async function projectPipeline(): Promise<PipelineAggregates> { |
| 122 | 263 | const sql = pg(); |
| 264 | + const live = projectLive(sql); | |
| 123 | 265 | const [st, yr, co, tot] = await Promise.all([ |
| 124 | − sql<Row[]>`select status, count(*)::int as n, sum(planned_mw)::float as mw, sum(investment_usd)::float as inv from projects group by status order by n desc`, | |
| 125 | − sql<Row[]>`select left(expected_opening, 4) as year, count(*)::int as n, sum(planned_mw)::float as mw from projects where expected_opening ~ '^\\d{4}' and status not in ('cancelled','closed') group by 1 order by 1`, | |
| 126 | − sql<Row[]>`select p.country_iso2, c.name, c.slug, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p left join countries c on c.iso2 = p.country_iso2 where p.country_iso2 is not null group by 1,2,3 order by n desc, mw desc nulls last limit 15`, | |
| 127 | − sql<Row[]>`select count(*)::int as n, sum(planned_mw)::float as mw, sum(investment_usd)::float as inv, count(*) filter (where is_ai)::int as ai from projects`, | |
| 266 | + sql<Row[]>`select p.status, count(*)::int as n, sum(p.planned_mw)::float as mw, sum(p.investment_usd)::float as inv from projects p where ${live} group by p.status order by n desc`, | |
| 267 | + sql<Row[]>`select left(p.expected_opening, 4) as year, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p where ${live} and p.expected_opening ~ '^\\d{4}' and p.status not in ('cancelled','closed') group by 1 order by 1`, | |
| 268 | + sql<Row[]>`select p.country_iso2, c.name, c.slug, count(*)::int as n, sum(p.planned_mw)::float as mw from projects p left join countries c on c.iso2 = p.country_iso2 where ${live} and p.country_iso2 is not null group by 1,2,3 order by n desc, mw desc nulls last limit 15`, | |
| 269 | + sql<Row[]>`select count(*)::int as n, sum(p.planned_mw)::float as mw, sum(p.investment_usd)::float as inv, count(*) filter (where p.is_ai or p.ai_evidence = any(${AI_LEVELS}))::int as ai from projects p where ${live}`, | |
| 128 | 270 | ]); |
| 129 | 271 | return { |
| 130 | 272 | byStatus: st.map((r) => ({ status: asStatus(r.status), count: int(r.n), mw: num(r.mw), investmentUsd: num(r.inv) })), |
added
apps/api/src/repositories/pulse.ts
+93 −0
@@ -0,0 +1,93 @@ | ||
| 1 | +/** | |
| 2 | + * Global infrastructure pulse — deterministic counts over a time window, computed from events and entity dates. | |
| 3 | + * Nothing here is scored or weighted: every figure is a count or a sum of published MW figures. | |
| 4 | + */ | |
| 5 | +import type { Pulse } from "@dci/core"; | |
| 6 | +import { pg, eventCols, eventJoins, projectLive, type Fragment, type Sql } from "../lib/sql.js"; | |
| 7 | +import { int, num, reqStr, str, type Row } from "../lib/rows.js"; | |
| 8 | +import { round2 } from "../lib/dto.js"; | |
| 9 | +import { toEventDtos } from "./events.js"; | |
| 10 | + | |
| 11 | +export type PulseWindow = Pulse["window"]; | |
| 12 | +const WINDOW_INTERVAL: Record<PulseWindow, string> = { "24h": "24 hours", "7d": "7 days", "30d": "30 days" }; | |
| 13 | + | |
| 14 | +export const PULSE_METHODOLOGY = "Deterministic counts over the window: events with review_status ≠ rejected; projects by created_at / construction_started_on; facilities by opened_on / first_seen; MW figures are sums of published site-scoped values on the records (no estimates). operatorsNewMarkets = operators whose first facility or project in a metro (or country when no metro) falls inside the window and who had none there before. Hidden (false-positive) and merged projects are excluded."; | |
| 15 | + | |
| 16 | +function partialDate(sql: Sql, col: Fragment, fallback: Fragment): Fragment { | |
| 17 | + return sql`(case when ${col} ~ '^\\d{4}-\\d{2}-\\d{2}' then ${col}::date when ${col} ~ '^\\d{4}-\\d{2}$' then (${col} || '-01')::date when ${col} ~ '^\\d{4}$' then (${col} || '-01-01')::date else ${fallback} end)`; | |
| 18 | +} | |
| 19 | + | |
| 20 | +export async function pulse(window: PulseWindow = "24h"): Promise<Pulse> { | |
| 21 | + const sql = pg(); | |
| 22 | + const since = sql`(now() - ${WINDOW_INTERVAL[window]}::interval)`; | |
| 23 | + const sinceDate = sql`(now() - ${WINDOW_INTERVAL[window]}::interval)::date`; | |
| 24 | + const conDate = partialDate(sql, sql`p.construction_started_on`, sql`null::date`); | |
| 25 | + const openedDate = partialDate(sql, sql`f.opened_on`, sql`null::date`); | |
| 26 | + const fFirst = sql`coalesce(${partialDate(sql, sql`f.opened_on`, sql`null::date`)}, f.first_seen::date)`; | |
| 27 | + const pFirst = partialDate(sql, sql`p.announced_on`, sql`p.created_at::date`); | |
| 28 | + const [ev, pr, fa, markets, major, byCountry, byOperator, sinceRow] = await Promise.all([ | |
| 29 | + sql<Row[]>`select count(*)::int as total, | |
| 30 | + count(*) filter (where e.event_type = 'facility_opened')::int as opened_ev, | |
| 31 | + count(*) filter (where e.event_type = 'cloud_region_announced')::int as cloud_ann, | |
| 32 | + count(*) filter (where e.event_type = 'power_agreement')::int as power, | |
| 33 | + count(*) filter (where e.event_type in ('grid_constraint', 'utility_event'))::int as grid, | |
| 34 | + count(*) filter (where e.event_type = 'acquisition')::int as acq, | |
| 35 | + count(*) filter (where e.event_type = 'investment_announced')::int as fin, | |
| 36 | + count(*) filter (where e.event_type = 'facility_discovered')::int as discovered, | |
| 37 | + count(*) filter (where e.event_type in ('capacity_changed', 'planned_capacity_changed'))::int as cap, | |
| 38 | + count(distinct e.project_id) filter (where e.project_id is not null and (e.event_type = 'construction_started' or (e.event_type = 'project_status_changed' and e.new_value::text ilike '%under_construction%')))::int as con_ev | |
| 39 | + from events e where e.review_status <> 'rejected' and e.detected_at >= ${since}`, | |
| 40 | + sql<Row[]>`select | |
| 41 | + count(*) filter (where p.created_at >= ${since})::int as new_n, sum(p.planned_mw) filter (where p.created_at >= ${since})::float as new_mw, | |
| 42 | + count(*) filter (where ${conDate} >= ${sinceDate})::int as con_n, sum(p.planned_mw) filter (where ${conDate} >= ${sinceDate})::float as con_mw | |
| 43 | + from projects p where ${projectLive(sql)}`, | |
| 44 | + sql<Row[]>`select count(*) filter (where ${openedDate} >= ${sinceDate} and f.status in ('operational', 'partially_operational', 'expansion'))::int as opened, count(*) filter (where f.first_seen >= ${since})::int as indexed from facilities f where f.merged_into is null`, | |
| 45 | + sql<Row[]>` | |
| 46 | + with entries as ( | |
| 47 | + select f.operator_id, f.metro_id, f.country_iso2, ${fFirst} as d from facilities f where f.merged_into is null and f.operator_id is not null and (f.metro_id is not null or f.country_iso2 is not null) | |
| 48 | + union all | |
| 49 | + select p.operator_id, p.metro_id, p.country_iso2, ${pFirst} from projects p where ${projectLive(sql)} and p.operator_id is not null and (p.metro_id is not null or p.country_iso2 is not null) | |
| 50 | + ), firsts as ( | |
| 51 | + select operator_id, metro_id, country_iso2, min(d) as first_d from entries group by 1, 2, 3 | |
| 52 | + ) | |
| 53 | + select o.id, o.slug, o.name, m.id as met_id, m.slug as met_slug, m.name as met_name, fs.country_iso2, fs.first_d | |
| 54 | + from firsts fs join operators o on o.id = fs.operator_id left join metros m on m.id = fs.metro_id | |
| 55 | + where fs.first_d >= ${sinceDate} | |
| 56 | + and not exists (select 1 from firsts f2 where f2.operator_id = fs.operator_id and f2.first_d < ${sinceDate} and (f2.metro_id is not distinct from fs.metro_id) and (fs.metro_id is not null or f2.country_iso2 is not distinct from fs.country_iso2)) | |
| 57 | + order by fs.first_d desc, o.name limit 25`, | |
| 58 | + sql<Row[]>`select ${eventCols(sql)} from events e ${eventJoins(sql)} where e.review_status <> 'rejected' and e.detected_at >= ${since} and e.significance >= 75 order by e.significance desc, e.detected_at desc limit 15`, | |
| 59 | + sql<Row[]>`select c.iso2, c.name, c.slug, coalesce(ev.n, 0) as events, coalesce(pj.n, 0) as new_projects | |
| 60 | + from countries c | |
| 61 | + left join (select country_iso2, count(*)::int as n from events where review_status <> 'rejected' and detected_at >= ${since} and country_iso2 is not null group by 1) ev on ev.country_iso2 = c.iso2 | |
| 62 | + left join (select p.country_iso2, count(*)::int as n from projects p where ${projectLive(sql)} and p.created_at >= ${since} and p.country_iso2 is not null group by 1) pj on pj.country_iso2 = c.iso2 | |
| 63 | + where coalesce(ev.n, 0) > 0 or coalesce(pj.n, 0) > 0 order by events desc, new_projects desc, c.name limit 15`, | |
| 64 | + sql<Row[]>`select o.id, o.slug, o.name, coalesce(ev.n, 0) as events, coalesce(pj.n, 0) as new_projects | |
| 65 | + from operators o | |
| 66 | + left join (select operator_id, count(*)::int as n from events where review_status <> 'rejected' and detected_at >= ${since} and operator_id is not null group by 1) ev on ev.operator_id = o.id | |
| 67 | + left join (select p.operator_id, count(*)::int as n from projects p where ${projectLive(sql)} and p.created_at >= ${since} and p.operator_id is not null group by 1) pj on pj.operator_id = o.id | |
| 68 | + where coalesce(ev.n, 0) > 0 or coalesce(pj.n, 0) > 0 order by events desc, new_projects desc, o.name limit 15`, | |
| 69 | + sql<Row[]>`select ${since} as since`, | |
| 70 | + ]); | |
| 71 | + const e = ev[0] ?? {}, p = pr[0] ?? {}, f = fa[0] ?? {}; | |
| 72 | + return { | |
| 73 | + window, | |
| 74 | + since: new Date(String(sinceRow[0]?.since ?? Date.now())).toISOString(), | |
| 75 | + newProjects: int(p.new_n), | |
| 76 | + projectsEnteredConstruction: Math.max(int(p.con_n), int(e.con_ev)), | |
| 77 | + facilitiesOpened: Math.max(int(f.opened), int(e.opened_ev)), | |
| 78 | + newlyAnnouncedMw: round2(num(p.new_mw)), | |
| 79 | + constructionStartedMw: round2(num(p.con_mw)), | |
| 80 | + operatorsNewMarkets: markets.map((r) => ({ operator: { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name) }, market: r.met_id ? { id: reqStr(r.met_id), slug: reqStr(r.met_slug), name: reqStr(r.met_name) } : null, countryIso2: str(r.country_iso2) })), | |
| 81 | + cloudRegionsAnnounced: int(e.cloud_ann), | |
| 82 | + powerAgreements: int(e.power), | |
| 83 | + gridConstraintEvents: int(e.grid), | |
| 84 | + acquisitions: int(e.acq), | |
| 85 | + financingEvents: int(e.fin), | |
| 86 | + newFacilitiesIndexed: Math.max(int(f.indexed), int(e.discovered)), | |
| 87 | + capacityChanges: int(e.cap), | |
| 88 | + eventsTotal: int(e.total), | |
| 89 | + majorEvents: await toEventDtos(major), | |
| 90 | + byCountry: byCountry.map((r) => ({ iso2: reqStr(r.iso2), name: reqStr(r.name), slug: reqStr(r.slug), events: int(r.events), newProjects: int(r.new_projects) })), | |
| 91 | + byOperator: byOperator.map((r) => ({ id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), events: int(r.events), newProjects: int(r.new_projects) })), | |
| 92 | + }; | |
| 93 | +} | |
modified
apps/api/src/repositories/search.ts
+41 −7
@@ -1,10 +1,11 @@ | ||
| 1 | 1 | import type { SearchHit, SearchResponse } from "@dci/core"; |
| 2 | −import { pg, likePattern } from "../lib/sql.js"; | |
| 2 | +import { pg, likePattern, projectLive } from "../lib/sql.js"; | |
| 3 | 3 | import { int, num, reqStr, str, type Row } from "../lib/rows.js"; |
| 4 | 4 | import { hrefFor } from "../lib/resolve.js"; |
| 5 | −import { hasFilters, interpretQuery } from "../search/interpret.js"; | |
| 5 | +import { hasFilters, interpretQuery, type InterpretOptions } from "../search/interpret.js"; | |
| 6 | 6 | import { operatorCandidates } from "./operators.js"; |
| 7 | 7 | import { listFacilities, searchCities } from "./facilities.js"; |
| 8 | +import { listProjects } from "./projects.js"; | |
| 8 | 9 | import { searchCountries } from "./countries.js"; |
| 9 | 10 | |
| 10 | 11 | const MAX_HITS = 30; |
@@ -57,7 +58,7 @@ async function projectHits(text: string, limit = 5): Promise<SearchHit[]> { | ||
| 57 | 58 | const rows = await sql<Row[]>` |
| 58 | 59 | select p.id, p.slug, p.name, p.status, p.country_iso2, p.planned_mw, o.name as op_name, greatest(similarity(p.name, ${text}), case when p.name ilike ${likePattern(text)} then 0.7 else 0 end) as score |
| 59 | 60 | from projects p left join operators o on o.id = p.operator_id |
| 60 | − where p.name % ${text} or p.name ilike ${likePattern(text)} | |
| 61 | + where ${projectLive(sql)} and (p.name % ${text} or p.name ilike ${likePattern(text)}) | |
| 61 | 62 | order by score desc, p.last_update desc limit ${limit}`; |
| 62 | 63 | return rows.map((r) => ({ type: "project", id: reqStr(r.id), slug: reqStr(r.slug), title: reqStr(r.name), subtitle: [str(r.op_name), str(r.status)?.replace(/_/g, " "), num(r.planned_mw) != null ? `${num(r.planned_mw)} MW` : null, str(r.country_iso2)].filter(Boolean).join(" · ") || null, href: hrefFor("project", reqStr(r.slug)), score: Math.min(1, 0.45 + 0.4 * (num(r.score) ?? 0)) })); |
| 63 | 64 | } |
@@ -71,11 +72,30 @@ async function ixpHits(text: string, limit = 3): Promise<SearchHit[]> { | ||
| 71 | 72 | return rows.map((r) => ({ type: "ixp", id: reqStr(r.id), slug: reqStr(r.slug), title: reqStr(r.name), subtitle: [str(r.name_long), str(r.city), str(r.country_iso2)].filter(Boolean).join(" · ") || null, href: hrefFor("ixp", reqStr(r.slug)), score: Math.min(1, 0.4 + 0.4 * (num(r.score) ?? 0)) })); |
| 72 | 73 | } |
| 73 | 74 | |
| 75 | +/** Metro candidates for query interpretation (trigram on name, substring on aliases). */ | |
| 76 | +async function metroCandidates(q: string, limit = 5): Promise<NonNullable<InterpretOptions["metros"]>> { | |
| 77 | + const sql = pg(); | |
| 78 | + const rows = await sql<Row[]>` | |
| 79 | + select m.slug, m.name, m.aliases from metros m | |
| 80 | + where position(lower(m.name) in lower(${q})) > 0 or (m.name % ${q} and similarity(m.name, ${q}) >= 0.5) | |
| 81 | + or exists (select 1 from unnest(m.aliases) a where length(a) >= 3 and position(lower(a) in lower(${q})) > 0) | |
| 82 | + order by similarity(m.name, ${q}) desc limit ${limit}`; | |
| 83 | + return rows.map((r) => ({ slug: reqStr(r.slug), name: reqStr(r.name), aliases: Array.isArray(r.aliases) ? (r.aliases as string[]) : [] })); | |
| 84 | +} | |
| 85 | + | |
| 86 | +/** Facility ids behind a search response (hits + list), for the envelope `sources`. */ | |
| 87 | +export function facilityIdsOf(res: SearchResponse): string[] { | |
| 88 | + const ids = new Set<string>(); | |
| 89 | + for (const h of res.hits) if (h.type === "facility") ids.add(h.id); | |
| 90 | + for (const f of res.facilities?.items ?? []) ids.add(f.id); | |
| 91 | + return [...ids]; | |
| 92 | +} | |
| 93 | + | |
| 74 | 94 | export async function search(q: string): Promise<SearchResponse> { |
| 75 | 95 | const query = q.replace(/\s+/g, " ").trim().slice(0, 200); |
| 76 | 96 | if (!query) return { query: "", hits: [] }; |
| 77 | − const ops = await operatorCandidates(query, 5).catch(() => []); | |
| 78 | − const interpreted = interpretQuery(query, { operators: ops }); | |
| 97 | + const [ops, metros] = await Promise.all([operatorCandidates(query, 5).catch(() => []), metroCandidates(query, 5).catch(() => [])]); | |
| 98 | + const interpreted = interpretQuery(query, { operators: ops, metros }); | |
| 79 | 99 | const text = interpreted.text ?? ""; |
| 80 | 100 | const hits: SearchHit[] = []; |
| 81 | 101 | const tasks: Promise<SearchHit[]>[] = []; |
@@ -87,15 +107,29 @@ export async function search(q: string): Promise<SearchResponse> { | ||
| 87 | 107 | // structured hits for the interpreted filters (country / operator) |
| 88 | 108 | if (interpreted.countryIso2) tasks.push(searchCountries(interpreted.countryIso2, 1).then((cs) => cs.map((c) => ({ type: "country" as const, id: c.iso2, slug: c.slug, title: c.name, subtitle: `${c.facilityCount} facilities`, href: hrefFor("country", c.slug), score: 0.99, meta: { facilityCount: c.facilityCount } })))); |
| 89 | 109 | if (interpreted.operator) { const op = ops.find((o) => o.slug === interpreted.operator); if (op) hits.push({ type: "operator", id: op.id, slug: op.slug, title: op.name, subtitle: "Operator", href: hrefFor("operator", op.slug), score: 0.97 }); } |
| 110 | + if (interpreted.metro) { const m = metros.find((x) => x.slug === interpreted.metro); if (m) hits.push({ type: "metro", id: `metro:${m.slug}`, slug: m.slug, title: m.name, subtitle: "Market", href: hrefFor("metro", m.slug), score: 0.96 }); } | |
| 90 | 111 | const results = await Promise.allSettled(tasks); |
| 91 | 112 | for (const r of results) if (r.status === "fulfilled") hits.push(...r.value); |
| 92 | 113 | // dedupe by type+id |
| 93 | 114 | const seen = new Set<string>(); |
| 94 | 115 | const merged = hits.filter((h) => { const k = `${h.type}:${h.id}`; if (seen.has(k)) return false; seen.add(k); return true; }).sort((a, b) => b.score - a.score).slice(0, MAX_HITS); |
| 95 | 116 | const response: SearchResponse = { query, interpreted, hits: merged.map((h) => ({ ...h, score: Math.round(h.score * 1000) / 1000 })) }; |
| 117 | + const yearFrom = interpreted.year != null && interpreted.yearOp !== "before" ? interpreted.year : undefined; | |
| 118 | + const yearTo = interpreted.year != null && interpreted.yearOp !== "after" ? interpreted.year : undefined; | |
| 119 | + if (interpreted.entity === "project") { | |
| 120 | + // project intent: rank project rows instead of facilities (planned MW, status, country, AI, expected year) | |
| 121 | + const pl = await listProjects({ q: text || undefined, country: interpreted.countryIso2, operator: interpreted.operator, metro: interpreted.metro, status: interpreted.status ? [interpreted.status] : undefined, min_mw: interpreted.minMw, ai: interpreted.ai ? true : undefined, per_page: 20, page: 1, sort: text ? "updated" : "mw", order: "desc" }).catch(() => ({ items: [], total: 0 })); | |
| 122 | + const filtered = pl.items.filter((p) => (interpreted.maxMw == null || p.plannedMw == null || p.plannedMw <= interpreted.maxMw) && (yearFrom == null || !p.expectedOpening || Number(p.expectedOpening.slice(0, 4)) >= yearFrom) && (yearTo == null || !p.expectedOpening || Number(p.expectedOpening.slice(0, 4)) <= yearTo)); | |
| 123 | + const seenP = new Set(response.hits.filter((h) => h.type === "project").map((h) => h.id)); | |
| 124 | + for (const p of filtered) { | |
| 125 | + if (seenP.has(p.id)) continue; | |
| 126 | + response.hits.push({ type: "project", id: p.id, slug: p.slug, title: p.name, subtitle: [p.operator?.name, p.status.replace(/_/g, " "), p.plannedMw != null ? `${p.plannedMw} MW` : null, p.expectedOpening ? `opening ${p.expectedOpening}` : null, p.countryIso2].filter(Boolean).join(" · ") || null, href: hrefFor("project", p.slug), score: 0.9, meta: { status: p.status, plannedMw: p.plannedMw, countryIso2: p.countryIso2 } }); | |
| 127 | + } | |
| 128 | + response.hits = response.hits.sort((a, b) => b.score - a.score).slice(0, MAX_HITS); | |
| 129 | + return response; | |
| 130 | + } | |
| 96 | 131 | if (hasFilters(interpreted) || text) { |
| 97 | − const fl = await listFacilities({ q: text || undefined, country: interpreted.countryIso2 ? [interpreted.countryIso2] : undefined, operator: interpreted.operator, status: interpreted.status ? [interpreted.status] : undefined, type: interpreted.facilityType && interpreted.facilityType !== "cloud_region" ? [interpreted.facilityType] : undefined, ai: interpreted.facilityType === "ai" ? true : undefined, min_mw: interpreted.minMw, per_page: 20, page: 1, sort: text ? "completeness" : "mw", order: "desc" }); | |
| 98 | − // when the type is "ai" we already matched is_ai OR type = ai above; drop the strict type filter result if it is empty and retry without it | |
| 132 | + const fl = await listFacilities({ q: text || undefined, country: interpreted.countryIso2 ? [interpreted.countryIso2] : undefined, operator: interpreted.operator, metro: interpreted.metro, status: interpreted.status ? [interpreted.status] : undefined, type: interpreted.facilityType && interpreted.facilityType !== "cloud_region" && interpreted.facilityType !== "ai" ? [interpreted.facilityType] : undefined, ai: interpreted.ai || interpreted.facilityType === "ai" ? true : undefined, min_mw: interpreted.minMw, max_mw: interpreted.maxMw, opened_from: yearFrom, opened_to: yearTo, per_page: 20, page: 1, sort: text ? "completeness" : "mw", order: "desc" }); | |
| 99 | 133 | response.facilities = { total: fl.total, items: fl.items }; |
| 100 | 134 | } |
| 101 | 135 | return response; |
added
apps/api/src/repositories/time-machine.ts
+43 −0
@@ -0,0 +1,43 @@ | ||
| 1 | +/** | |
| 2 | + * /time-machine — yearly frames of the index as it would have looked from published opening dates. Frames only | |
| 3 | + * include facilities WITH an opening date, so every year is a lower bound; coverage is stated. | |
| 4 | + */ | |
| 5 | +import type { TimeMachine, TimeMachineFrame } from "@dci/core"; | |
| 6 | +import { pg, facilityView, knownMwAgg, countedAgg, projectLive, yearExpr } from "../lib/sql.js"; | |
| 7 | +import { int, num, type Row } from "../lib/rows.js"; | |
| 8 | +import { round2, share } from "../lib/dto.js"; | |
| 9 | + | |
| 10 | +export const TIME_MACHINE_NOTE = "Frames only include facilities with a published opening date (coverage given); facilities without a date are absent from every frame, so early years are lower bounds. knownMw is the cumulative known operational MW of dated facilities (containment-aware, published figures only). announced / construction count live projects by announcement / construction-start year."; | |
| 11 | +export const RELIABLE_MIN_FACILITIES = 20; | |
| 12 | + | |
| 13 | +export async function timeMachine(): Promise<TimeMachine> { | |
| 14 | + const sql = pg(); | |
| 15 | + const known = knownMwAgg(sql), counted = countedAgg(sql); | |
| 16 | + const openedYear = yearExpr(sql, sql`f.opened_on`); | |
| 17 | + const [byYear, cov, ann, con] = await Promise.all([ | |
| 18 | + sql<Row[]>`select ${openedYear} as year, count(*) filter (where ${counted})::int as n, sum(${known}) filter (where f.status in ('operational', 'partially_operational', 'expansion'))::float as mw from ${facilityView(sql)} f where f.opened_on ~ '^\\d{4}' group by 1 order by 1`, | |
| 19 | + sql<Row[]>`select count(*) filter (where ${counted})::int as n, count(*) filter (where ${counted} and f.opened_on ~ '^\\d{4}')::int as dated from ${facilityView(sql)} f`, | |
| 20 | + sql<Row[]>`select ${yearExpr(sql, sql`p.announced_on`)} as year, count(*)::int as n from projects p where ${projectLive(sql)} and p.announced_on ~ '^\\d{4}' group by 1`, | |
| 21 | + sql<Row[]>`select ${yearExpr(sql, sql`p.construction_started_on`)} as year, count(*)::int as n from projects p where ${projectLive(sql)} and p.construction_started_on ~ '^\\d{4}' group by 1`, | |
| 22 | + ]); | |
| 23 | + const opened = new Map<number, { n: number; mw: number | null }>(); | |
| 24 | + for (const r of byYear) { const y = int(r.year); if (y) opened.set(y, { n: int(r.n), mw: num(r.mw) }); } | |
| 25 | + const annMap = new Map<number, number>(ann.map((r) => [int(r.year), int(r.n)])); | |
| 26 | + const conMap = new Map<number, number>(con.map((r) => [int(r.year), int(r.n)])); | |
| 27 | + const nowYear = new Date().getUTCFullYear(); | |
| 28 | + const years = [...opened.keys(), ...annMap.keys(), ...conMap.keys()].filter((y) => y >= 1950 && y <= nowYear + 10); | |
| 29 | + const frames: TimeMachineFrame[] = []; | |
| 30 | + let earliestReliableYear = 0; | |
| 31 | + if (years.length) { | |
| 32 | + const from = Math.min(...years); | |
| 33 | + const to = Math.max(nowYear, ...years.filter((y) => y <= nowYear)); | |
| 34 | + let cumN = 0, cumMw = 0, sawMw = false; | |
| 35 | + for (let y = from; y <= to; y++) { | |
| 36 | + const o = opened.get(y); | |
| 37 | + if (o) { cumN += o.n; if (o.mw != null) { cumMw += o.mw; sawMw = true; } } | |
| 38 | + if (!earliestReliableYear && cumN >= RELIABLE_MIN_FACILITIES) earliestReliableYear = y; | |
| 39 | + frames.push({ year: y, facilities: cumN, knownMw: sawMw ? round2(cumMw) : null, announced: annMap.get(y) ?? 0, construction: conMap.get(y) ?? 0, opened: o?.n ?? 0 }); | |
| 40 | + } | |
| 41 | + } | |
| 42 | + return { frames, earliestReliableYear: earliestReliableYear || (frames.at(-1)?.year ?? nowYear), openingDateCoverage: share(int(cov[0]?.dated), int(cov[0]?.n)), note: TIME_MACHINE_NOTE }; | |
| 43 | +} | |
added
apps/api/src/repositories/watchlist.ts
+99 −0
@@ -0,0 +1,99 @@ | ||
| 1 | +/** | |
| 2 | + * Private watchlists — no accounts. The owner is an opaque random token stored in the httpOnly `dci_watch` cookie; | |
| 3 | + * rows are only ever read / deleted through that token. | |
| 4 | + */ | |
| 5 | +import { randomBytes } from "node:crypto"; | |
| 6 | +import type { EventDTO, WatchlistItem } from "@dci/core"; | |
| 7 | +import { newId } from "@dci/core"; | |
| 8 | +import { pg, eventCols, eventJoins, page } from "../lib/sql.js"; | |
| 9 | +import { reqIso, reqStr, str, type Row } from "../lib/rows.js"; | |
| 10 | +import { findBySlugOrId, findCountry, resolveEntityRefs } from "../lib/resolve.js"; | |
| 11 | +import { toEventDtos } from "./events.js"; | |
| 12 | + | |
| 13 | +export const WATCH_COOKIE = "dci_watch"; | |
| 14 | +export const WATCH_MAX_ITEMS = 200; | |
| 15 | +export const WATCH_ENTITY_TYPES = ["operator", "metro", "country", "project", "facility"] as const; | |
| 16 | +export type WatchEntityType = (typeof WATCH_ENTITY_TYPES)[number]; | |
| 17 | + | |
| 18 | +export function newWatchToken(): string { | |
| 19 | + return randomBytes(24).toString("base64url"); | |
| 20 | +} | |
| 21 | + | |
| 22 | +export function parseCookies(header: string | undefined): Record<string, string> { | |
| 23 | + const out: Record<string, string> = {}; | |
| 24 | + if (!header) return out; | |
| 25 | + for (const part of header.split(";")) { | |
| 26 | + const i = part.indexOf("="); | |
| 27 | + if (i < 0) continue; | |
| 28 | + const k = part.slice(0, i).trim(); | |
| 29 | + const v = part.slice(i + 1).trim(); | |
| 30 | + if (k && /^[A-Za-z0-9_-]{8,128}$/.test(v)) out[k] = v; | |
| 31 | + } | |
| 32 | + return out; | |
| 33 | +} | |
| 34 | + | |
| 35 | +const TABLE: Record<WatchEntityType, "operators" | "metros" | "projects" | "facilities" | null> = { operator: "operators", metro: "metros", project: "projects", facility: "facilities", country: null }; | |
| 36 | + | |
| 37 | +/** Resolve a slug or id to the canonical entity id (null when unknown). */ | |
| 38 | +export async function resolveWatchEntity(type: WatchEntityType, idOrSlug: string): Promise<string | null> { | |
| 39 | + if (type === "country") { const c = await findCountry(idOrSlug); return c ? reqStr(c.iso2) : null; } | |
| 40 | + const row = await findBySlugOrId(TABLE[type]!, idOrSlug); | |
| 41 | + if (!row) return null; | |
| 42 | + if (type === "project" && (row.hidden === true || row.merged_into)) return str(row.merged_into) ?? null; | |
| 43 | + return reqStr(row.id); | |
| 44 | +} | |
| 45 | + | |
| 46 | +async function toItems(rows: Row[]): Promise<WatchlistItem[]> { | |
| 47 | + const refs = await resolveEntityRefs(rows.map((r) => ({ type: reqStr(r.entity_type), id: reqStr(r.entity_id) }))); | |
| 48 | + return rows.map((r) => { | |
| 49 | + const ref = refs.get(`${reqStr(r.entity_type)}:${reqStr(r.entity_id)}`); | |
| 50 | + return { id: reqStr(r.id), entityType: reqStr(r.entity_type) as WatchEntityType, entityId: reqStr(r.entity_id), slug: ref?.slug ?? reqStr(r.entity_id), name: ref?.name ?? reqStr(r.entity_id), createdAt: reqIso(r.created_at) }; | |
| 51 | + }); | |
| 52 | +} | |
| 53 | + | |
| 54 | +export async function listWatchlist(token: string): Promise<WatchlistItem[]> { | |
| 55 | + const sql = pg(); | |
| 56 | + const rows = await sql<Row[]>`select * from watchlists where owner_token = ${token} order by created_at desc limit ${WATCH_MAX_ITEMS}`; | |
| 57 | + return toItems(rows); | |
| 58 | +} | |
| 59 | + | |
| 60 | +export async function addWatch(token: string, type: WatchEntityType, entityId: string): Promise<{ item: WatchlistItem; created: boolean } | { error: "full" }> { | |
| 61 | + const sql = pg(); | |
| 62 | + const n = await sql<Row[]>`select count(*)::int as n from watchlists where owner_token = ${token}`; | |
| 63 | + const existing = await sql<Row[]>`select * from watchlists where owner_token = ${token} and entity_type = ${type} and entity_id = ${entityId}`; | |
| 64 | + if (existing[0]) return { item: (await toItems(existing))[0]!, created: false }; | |
| 65 | + if (Number(n[0]?.n ?? 0) >= WATCH_MAX_ITEMS) return { error: "full" }; | |
| 66 | + const rows = await sql<Row[]>`insert into watchlists (id, owner_token, entity_type, entity_id) values (${newId("watch")}, ${token}, ${type}, ${entityId}) on conflict (owner_token, entity_type, entity_id) do update set entity_id = excluded.entity_id returning *`; | |
| 67 | + return { item: (await toItems(rows))[0]!, created: true }; | |
| 68 | +} | |
| 69 | + | |
| 70 | +export async function removeWatch(token: string, id: string): Promise<boolean> { | |
| 71 | + const sql = pg(); | |
| 72 | + const rows = await sql`delete from watchlists where owner_token = ${token} and id = ${id} returning id`; | |
| 73 | + return rows.length > 0; | |
| 74 | +} | |
| 75 | + | |
| 76 | +/** Events for the watched entities, newest first, one row per cluster. */ | |
| 77 | +export async function watchFeed(token: string, pageNo?: number, perPage?: number): Promise<{ items: EventDTO[]; total: number; page: number; perPage: number }> { | |
| 78 | + const sql = pg(); | |
| 79 | + const pg_ = page(pageNo, perPage, 100, 50); | |
| 80 | + const rows = await sql<Row[]>` | |
| 81 | + with w as (select entity_type, entity_id from watchlists where owner_token = ${token}), | |
| 82 | + matched as ( | |
| 83 | + select distinct on (coalesce(e.cluster_id, e.id)) e.id | |
| 84 | + from events e | |
| 85 | + where e.review_status <> 'rejected' and ( | |
| 86 | + exists (select 1 from w where w.entity_type = 'operator' and (e.operator_id = w.entity_id or (e.entity_type = 'operator' and e.entity_id = w.entity_id))) | |
| 87 | + or exists (select 1 from w where w.entity_type = 'country' and e.country_iso2 = w.entity_id) | |
| 88 | + or exists (select 1 from w where w.entity_type = 'project' and (e.project_id = w.entity_id or (e.entity_type = 'project' and e.entity_id = w.entity_id))) | |
| 89 | + or exists (select 1 from w where w.entity_type = 'facility' and e.entity_type = 'facility' and e.entity_id = w.entity_id) | |
| 90 | + or exists (select 1 from w where w.entity_type = 'metro' and (e.metro_id = w.entity_id or (e.entity_type = 'facility' and e.entity_id in (select id from facilities where metro_id = w.entity_id)))) | |
| 91 | + ) | |
| 92 | + order by coalesce(e.cluster_id, e.id), e.significance desc, e.detected_at asc | |
| 93 | + ) | |
| 94 | + select ${eventCols(sql)}, count(*) over() as total from events e ${eventJoins(sql)} | |
| 95 | + where e.id in (select id from matched) | |
| 96 | + order by e.detected_at desc, e.id desc limit ${pg_.perPage} offset ${pg_.offset}`; | |
| 97 | + const total = rows.length ? Number(rows[0]!.total) : 0; | |
| 98 | + return { items: await toEventDtos(rows), total, page: pg_.page, perPage: pg_.perPage }; | |
| 99 | +} | |
added
apps/api/src/routes/admin/claims.ts
+24 −0
@@ -0,0 +1,24 @@ | ||
| 1 | +/** Admin claim workbench: list + status changes. */ | |
| 2 | +import type { FastifyInstance } from "fastify"; | |
| 3 | +import { z } from "zod"; | |
| 4 | +import { envelope, notFound, parseBody, parseQuery } from "../../lib/http.js"; | |
| 5 | +import { intParam, pageParam, strParam } from "../../lib/params.js"; | |
| 6 | +import { listClaims, setClaimStatus } from "../../repositories/admin/claims.js"; | |
| 7 | +import { invalidate } from "../../cache.js"; | |
| 8 | + | |
| 9 | +export async function claimAdminRoutes(app: FastifyInstance): Promise<void> { | |
| 10 | + app.get("/claims", { schema: { summary: "Claims (ClaimDTO[]) ?subject_type=&subject_id=&status=current|superseded|rejected|review|unscoped|all&predicate=&page=&per_page=" } }, async (req) => { | |
| 11 | + const q = parseQuery(z.object({ subject_type: strParam, subject_id: strParam, status: strParam, predicate: strParam, page: pageParam, per_page: intParam }), req.query); | |
| 12 | + const res = await listClaims({ subjectType: q.subject_type, subjectId: q.subject_id, status: q.status, predicate: q.predicate, page: q.page, perPage: q.per_page }); | |
| 13 | + return envelope(res.items, { total: res.total, page: res.page, perPage: res.perPage }); | |
| 14 | + }); | |
| 15 | + | |
| 16 | + app.post("/claims/:id/status", { schema: { summary: "Set a claim status {status: current|rejected|review, reason?}. Never rewrites the entity column: when a rejected claim was backing a displayed facility MW value a quality flag claim_rejected_backing_value is raised for the next reconciliation" } }, async (req) => { | |
| 17 | + const { id } = req.params as { id: string }; | |
| 18 | + const body = parseBody(z.object({ status: z.enum(["current", "rejected", "review"]), reason: z.string().max(500).optional() }), req.body); | |
| 19 | + const res = await setClaimStatus(id, body.status, body.reason ?? null); | |
| 20 | + if (!res) throw notFound("claim"); | |
| 21 | + await Promise.all([invalidate("/datacenters"), invalidate("/projects")]); | |
| 22 | + return envelope(res); | |
| 23 | + }); | |
| 24 | +} | |
modified
apps/api/src/routes/admin/connectors.ts
+25 −0
@@ -10,6 +10,7 @@ import { strParam, intParam } from "../../lib/params.js"; | ||
| 10 | 10 | import { pg } from "../../lib/sql.js"; |
| 11 | 11 | import { enqueueCrawl } from "../../queues.js"; |
| 12 | 12 | import { getConnectorAdmin, getRun, listConnectorHealth, listRuns } from "../../repositories/admin/connectors.js"; |
| 13 | +import { rollbackRun, runChanges } from "../../repositories/admin/runs.js"; | |
| 13 | 14 | import { invalidate } from "../../cache.js"; |
| 14 | 15 | |
| 15 | 16 | const runBody = z.object({ task: z.enum(["crawl", "discover", "full", "reprocess"]).default("crawl"), group: z.string().min(1).max(64).optional(), limit: z.number().int().min(1).max(100000).optional(), force: z.boolean().optional() }); |
@@ -89,6 +90,15 @@ export async function connectorAdminRoutes(app: FastifyInstance): Promise<void> | ||
| 89 | 90 | return envelope(items, { total: items.length }); |
| 90 | 91 | }); |
| 91 | 92 | |
| 93 | + app.post("/connectors/:id/quarantine", { schema: { summary: "Quarantine {on: boolean}: the connector extracts and previews but publishes nothing; health = quarantine while on" } }, async (req) => { | |
| 94 | + const { id } = req.params as { id: string }; | |
| 95 | + const body = parseBody(z.object({ on: z.boolean() }), req.body); | |
| 96 | + const sql = pg(); | |
| 97 | + const rows = await sql`update connectors set quarantine = ${body.on}, health = case when ${body.on} then 'quarantine' when paused then 'paused' when last_run_at is null then 'never_run' else 'ok' end, updated_at = now() where id = ${id} returning id, quarantine, health`; | |
| 98 | + if (!rows.length) throw notFound("connector"); | |
| 99 | + return envelope(rows[0]); | |
| 100 | + }); | |
| 101 | + | |
| 92 | 102 | app.get("/runs/:id", { schema: { summary: "Run detail with log" } }, async (req) => { |
| 93 | 103 | const { id } = req.params as { id: string }; |
| 94 | 104 | const r = await getRun(id); |
@@ -96,6 +106,21 @@ export async function connectorAdminRoutes(app: FastifyInstance): Promise<void> | ||
| 96 | 106 | return envelope(r); |
| 97 | 107 | }); |
| 98 | 108 | |
| 109 | + app.get("/runs/:id/changes", { schema: { summary: "Everything a run wrote: provenance rows, claims, events, document versions (run_id = id; 2 000 rows per list)" } }, async (req) => { | |
| 110 | + const { id } = req.params as { id: string }; | |
| 111 | + const r = await runChanges(id); | |
| 112 | + if (!r) throw notFound("run"); | |
| 113 | + return envelope(r, { ...r.counts }); | |
| 114 | + }); | |
| 115 | + | |
| 116 | + app.post("/runs/:id/rollback", { schema: { summary: "Roll a run back: its claims → rejected (rollback), its provenance rows → not current (winner restored to the latest remaining row per field), its events → rejected. Entity columns are re-derived by the worker's next reconciliation" } }, async (req) => { | |
| 117 | + const { id } = req.params as { id: string }; | |
| 118 | + const r = await rollbackRun(id); | |
| 119 | + if (!r) throw notFound("run"); | |
| 120 | + await Promise.all([invalidate("/datacenters"), invalidate("/projects"), invalidate("/events"), invalidate("/dashboard")]); | |
| 121 | + return envelope(r); | |
| 122 | + }); | |
| 123 | + | |
| 99 | 124 | app.post("/cache/invalidate", { schema: { summary: "Invalidate API cache entries by route prefix (body {prefix}) — default all" } }, async (req) => { |
| 100 | 125 | const body = parseBody(z.object({ prefix: z.string().max(200).default("") }), req.body); |
| 101 | 126 | const n = await invalidate(body.prefix); |
modified
apps/api/src/routes/admin/curation.ts
+9 −1
@@ -5,7 +5,7 @@ import { FACILITY_STATUSES, FACILITY_TYPES, GEO_PRECISIONS, newId, normalizeName | ||
| 5 | 5 | import { envelope, notFound, parseBody, badRequest } from "../../lib/http.js"; |
| 6 | 6 | import { pg } from "../../lib/sql.js"; |
| 7 | 7 | import { str, type Row } from "../../lib/rows.js"; |
| 8 | −import { ensureManualSource, mergeFacility } from "../../repositories/admin/merge.js"; | |
| 8 | +import { ensureManualSource, mergeFacility, setFacilityParent } from "../../repositories/admin/merge.js"; | |
| 9 | 9 | import { getFacilityDetail } from "../../repositories/facilities.js"; |
| 10 | 10 | import { invalidate } from "../../cache.js"; |
| 11 | 11 | |
@@ -100,6 +100,14 @@ export async function curationAdminRoutes(app: FastifyInstance): Promise<void> { | ||
| 100 | 100 | return envelope(res); |
| 101 | 101 | }); |
| 102 | 102 | |
| 103 | + app.post("/facilities/:id/parent", { schema: { summary: "Containment link {parentId | null}: the facility becomes a building of the parent campus (record_scope building / campus); aggregates never count both" } }, async (req) => { | |
| 104 | + const { id } = req.params as { id: string }; | |
| 105 | + const body = parseBody(z.object({ parentId: z.string().min(1).nullable() }), req.body); | |
| 106 | + const res = await setFacilityParent(id, body.parentId); | |
| 107 | + await Promise.all([invalidate("/datacenters"), invalidate("/dashboard"), invalidate("/map")]); | |
| 108 | + return envelope(res); | |
| 109 | + }); | |
| 110 | + | |
| 103 | 111 | app.patch("/events/:id", { schema: { summary: "Review an event {reviewStatus: auto|pending|approved|rejected}" } }, async (req) => { |
| 104 | 112 | const { id } = req.params as { id: string }; |
| 105 | 113 | const body = parseBody(z.object({ reviewStatus: z.enum(["auto", "pending", "approved", "rejected"]) }), req.body); |
modified
apps/api/src/routes/admin/documents.ts
+2 −2
@@ -2,7 +2,7 @@ import type { FastifyInstance } from "fastify"; | ||
| 2 | 2 | import { z } from "zod"; |
| 3 | 3 | import { envelope, notFound, parseBody, parseQuery } from "../../lib/http.js"; |
| 4 | 4 | import { intParam, pageParam, strParam } from "../../lib/params.js"; |
| 5 | −import { pg, page, andAll, type Fragment } from "../../lib/sql.js"; | |
| 5 | +import { pg, page, andAll, likePattern, type Fragment } from "../../lib/sql.js"; | |
| 6 | 6 | import { int, iso, json, num, reqStr, str, type Row } from "../../lib/rows.js"; |
| 7 | 7 | import { resolveEntityRefs } from "../../lib/resolve.js"; |
| 8 | 8 | import { enqueueCrawl, manualJobId } from "../../queues.js"; |
@@ -34,7 +34,7 @@ export async function documentAdminRoutes(app: FastifyInstance): Promise<void> { | ||
| 34 | 34 | if (q.status === "error") c.push(sql`d.error is not null`); |
| 35 | 35 | if (q.status === "changed") c.push(sql`d.change_count > 0`); |
| 36 | 36 | if (q.status === "quarantined") c.push(sql`d.quarantined`); |
| 37 | − if (q.q) c.push(sql`(d.url ilike ${"%" + q.q + "%"} or d.title ilike ${"%" + q.q + "%"})`); | |
| 37 | + if (q.q) c.push(sql`(d.url ilike ${likePattern(q.q)} or d.title ilike ${likePattern(q.q)})`); | |
| 38 | 38 | const rows = await sql<Row[]>`select d.*, count(*) over() as total from documents d where ${andAll(sql, c)} order by coalesce(d.last_changed, d.last_fetched, d.first_seen) desc limit ${pg_.perPage} offset ${pg_.offset}`; |
| 39 | 39 | return envelope(rows.map(docDto), { total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }); |
| 40 | 40 | }); |
modified
apps/api/src/routes/admin/index.ts
+9 −1
@@ -12,6 +12,10 @@ import { matchAdminRoutes } from "./matches.js"; | ||
| 12 | 12 | import { curationAdminRoutes } from "./curation.js"; |
| 13 | 13 | import { devtoolAdminRoutes } from "./devtool.js"; |
| 14 | 14 | import { opsAdminRoutes } from "./ops.js"; |
| 15 | +import { qualityAdminRoutes } from "./quality.js"; | |
| 16 | +import { proxyAdminRoutes } from "./proxy.js"; | |
| 17 | +import { claimAdminRoutes } from "./claims.js"; | |
| 18 | +import { projectAdminRoutes } from "./projects.js"; | |
| 15 | 19 | |
| 16 | 20 | export function tokenMatches(provided: string | undefined, expected: string | null): boolean { |
| 17 | 21 | if (!expected || !provided) return false; |
@@ -22,7 +26,7 @@ export function tokenMatches(provided: string | undefined, expected: string | nu | ||
| 22 | 26 | |
| 23 | 27 | export function adminAuth(req: FastifyRequest, _reply: FastifyReply, done: (err?: Error) => void): void { |
| 24 | 28 | const expected = getEnv().adminToken; |
| 25 | − if (!expected) return done(new HttpError(503, "admin API disabled: DCI_ADMIN_TOKEN is not configured")); | |
| 29 | + if (!expected) return done(new HttpError(503, "admin API disabled: DCI_ADMIN_TOKEN is not configured (missing, empty or the \"change-me\" placeholder)")); | |
| 26 | 30 | const header = req.headers["x-dci-admin-token"]; |
| 27 | 31 | const provided = Array.isArray(header) ? header[0] : header; |
| 28 | 32 | if (!tokenMatches(provided, expected)) return done(new HttpError(401, "unauthorized")); |
@@ -43,4 +47,8 @@ export async function adminRoutes(app: FastifyInstance): Promise<void> { | ||
| 43 | 47 | await app.register(curationAdminRoutes); |
| 44 | 48 | await app.register(devtoolAdminRoutes); |
| 45 | 49 | await app.register(opsAdminRoutes); |
| 50 | + await app.register(qualityAdminRoutes); | |
| 51 | + await app.register(proxyAdminRoutes); | |
| 52 | + await app.register(claimAdminRoutes); | |
| 53 | + await app.register(projectAdminRoutes); | |
| 46 | 54 | } |
modified
apps/api/src/routes/admin/matches.ts
+101 −17
@@ -1,42 +1,97 @@ | ||
| 1 | +/** | |
| 2 | + * Entity match workbench: pending candidates with BOTH facility rows side by side (name, operator, city, address, | |
| 3 | + * coordinates + distance, facility codes, external ids, source counts) and the matcher's score / reasons. | |
| 4 | + * Decisions: approve (merge), reject (keep separate), related-campus (candidate is a building of the matched campus), | |
| 5 | + * defer. | |
| 6 | + */ | |
| 1 | 7 | import type { FastifyInstance } from "fastify"; |
| 2 | 8 | import { z } from "zod"; |
| 3 | −import { envelope, notFound, parseQuery, badRequest } from "../../lib/http.js"; | |
| 9 | +import { extractFacilityCodes } from "@dci/core"; | |
| 10 | +import { envelope, notFound, parseQuery, parseBody, badRequest } from "../../lib/http.js"; | |
| 4 | 11 | import { intParam, pageParam, strParam } from "../../lib/params.js"; |
| 5 | 12 | import { pg, page } from "../../lib/sql.js"; |
| 6 | −import { int, iso, json, num, reqStr, str, strArray, type Row } from "../../lib/rows.js"; | |
| 13 | +import { int, iso, json, num, record, reqStr, str, strArray, type Row } from "../../lib/rows.js"; | |
| 7 | 14 | import { facilitiesByIds } from "../../repositories/facilities.js"; |
| 8 | −import { ensureManualSource, mergeFacility } from "../../repositories/admin/merge.js"; | |
| 15 | +import { ensureManualSource, mergeFacility, setFacilityParent } from "../../repositories/admin/merge.js"; | |
| 9 | 16 | import { invalidate } from "../../cache.js"; |
| 10 | 17 | |
| 18 | +const STATUSES = ["pending", "approved", "rejected", "related_campus", "deferred", "auto_merged", "auto_created", "all"] as const; | |
| 19 | + | |
| 11 | 20 | function matchDto(r: Row): Record<string, unknown> { |
| 12 | 21 | return { id: reqStr(r.id), connectorId: reqStr(r.connector_id), candidateKey: reqStr(r.candidate_key), candidate: json(r.candidate, {}), matchedFacilityId: str(r.matched_facility_id), score: num(r.score), reasons: strArray(r.reasons), status: reqStr(r.status), decidedBy: str(r.decided_by), decidedAt: iso(r.decided_at), createdAt: iso(r.created_at), candidateFacilityId: str(r.candidate_facility_id) }; |
| 13 | 22 | } |
| 14 | 23 | |
| 24 | +interface Side { id: string; slug: string; name: string; operator: string | null; city: string | null; address: string | null; lat: number | null; lng: number | null; codes: string[]; externalIds: Record<string, string | number>; sourceCount: number; status: string; recordScope: string; parentFacilityId: string | null } | |
| 25 | + | |
| 26 | +function side(r: Row): Side { | |
| 27 | + return { id: reqStr(r.id), slug: reqStr(r.slug), name: reqStr(r.name), operator: str(r.op_name), city: str(r.city), address: str(r.address), lat: num(r.lat), lng: num(r.lng), codes: extractFacilityCodes(reqStr(r.name)), externalIds: record(r.external_ids), sourceCount: int(r.source_count), status: reqStr(r.status), recordScope: reqStr(r.record_scope, "facility"), parentFacilityId: str(r.parent_facility_id) }; | |
| 28 | +} | |
| 29 | + | |
| 30 | +function haversineKm(a: Side, b: Side): number | null { | |
| 31 | + if (a.lat == null || a.lng == null || b.lat == null || b.lng == null) return null; | |
| 32 | + const R = 6371.0088, toRad = (d: number) => (d * Math.PI) / 180; | |
| 33 | + const dLat = toRad(b.lat - a.lat), dLng = toRad(b.lng - a.lng); | |
| 34 | + const h = Math.sin(dLat / 2) ** 2 + Math.cos(toRad(a.lat)) * Math.cos(toRad(b.lat)) * Math.sin(dLng / 2) ** 2; | |
| 35 | + return Math.round(2 * R * Math.asin(Math.sqrt(Math.min(1, h))) * 1000) / 1000; | |
| 36 | +} | |
| 37 | + | |
| 38 | +/** The facility created from the candidate (entity key or candidate.createdFacilityId) — for a pending row. */ | |
| 39 | +function candidateFacilityId(r: Row): string | null { | |
| 40 | + return str(r.candidate_facility_id) ?? str((json<Record<string, unknown>>(r.candidate, {}) as Record<string, unknown>).createdFacilityId); | |
| 41 | +} | |
| 42 | + | |
| 15 | 43 | export async function matchAdminRoutes(app: FastifyInstance): Promise<void> { |
| 16 | − app.get("/matches", { schema: { summary: "Entity match queue (?status=pending) with candidate + matched facility summaries" } }, async (req) => { | |
| 17 | − const q = parseQuery(z.object({ status: strParam, connector: strParam, page: pageParam, per_page: intParam }), req.query); | |
| 44 | + app.get("/matches", { schema: { summary: "Entity match queue (?status=pending|approved|rejected|related_campus|deferred|auto_merged|auto_created|all&connector=) with both facility rows (pair: name, operator, city, address, coords + distance, codes, external ids, source counts), score and reasons" } }, async (req) => { | |
| 45 | + const q = parseQuery(z.object({ status: z.preprocess((v) => (v === "" || v == null ? undefined : v), z.enum(STATUSES).optional()), connector: strParam, page: pageParam, per_page: intParam }), req.query); | |
| 18 | 46 | const sql = pg(); |
| 19 | 47 | const pg_ = page(q.page, q.per_page, 200, 50); |
| 20 | 48 | const status = q.status ?? "pending"; |
| 21 | 49 | const rows = await sql<Row[]>` |
| 22 | − select m.*, k.entity_id as candidate_facility_id, count(*) over() as total | |
| 50 | + select m.*, coalesce(k.entity_id, m.candidate->>'createdFacilityId') as candidate_facility_id, count(*) over() as total | |
| 23 | 51 | from entity_matches m |
| 24 | 52 | left join entity_keys k on k.key = m.candidate_key and k.entity_type = 'facility' |
| 25 | 53 | where ${status === "all" ? sql`true` : sql`m.status = ${status}`} and ${q.connector ? sql`m.connector_id = ${q.connector}` : sql`true`} |
| 26 | 54 | order by m.score desc, m.created_at desc limit ${pg_.perPage} offset ${pg_.offset}`; |
| 27 | − const ids = [...new Set(rows.flatMap((r) => [str(r.matched_facility_id), str(r.candidate_facility_id)]).filter((x): x is string => Boolean(x)))]; | |
| 28 | − const facs = await facilitiesByIds(ids); | |
| 55 | + const ids = [...new Set(rows.flatMap((r) => [str(r.matched_facility_id), candidateFacilityId(r)]).filter((x): x is string => Boolean(x)))]; | |
| 56 | + const [facs, sides] = await Promise.all([ | |
| 57 | + facilitiesByIds(ids), | |
| 58 | + ids.length ? sql<Row[]>`select f.id, f.slug, f.name, f.city, f.address, f.lat, f.lng, f.external_ids, f.source_count, f.status, f.record_scope, f.parent_facility_id, o.name as op_name from facilities f left join operators o on o.id = f.operator_id where f.id = any(${ids})` : Promise.resolve([] as Row[]), | |
| 59 | + ]); | |
| 29 | 60 | const by = new Map(facs.map((f) => [f.id, f])); |
| 30 | − return envelope(rows.map((r) => ({ ...matchDto(r), matched: r.matched_facility_id ? by.get(String(r.matched_facility_id)) ?? null : null, candidateFacility: r.candidate_facility_id ? by.get(String(r.candidate_facility_id)) ?? null : null })), { total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }); | |
| 61 | + const sideBy = new Map(sides.map((r) => [reqStr(r.id), side(r)])); | |
| 62 | + const items = rows.map((r) => { | |
| 63 | + const matchedId = str(r.matched_facility_id); | |
| 64 | + const candId = candidateFacilityId(r); | |
| 65 | + const cand = json<Record<string, unknown>>(r.candidate, {}); | |
| 66 | + const a = candId ? sideBy.get(candId) ?? null : null; | |
| 67 | + const b = matchedId ? sideBy.get(matchedId) ?? null : null; | |
| 68 | + // when the candidate was never persisted as a facility, describe it from the candidate payload itself | |
| 69 | + const candidateSide: Partial<Side> | null = a ?? (Object.keys(cand).length ? { id: "", slug: "", name: str(cand.name) ?? "", operator: str(cand.operatorName ?? cand.operator), city: str(cand.city), address: str(cand.address), lat: num((cand.geo as Record<string, unknown> | undefined)?.lat ?? cand.lat), lng: num((cand.geo as Record<string, unknown> | undefined)?.lng ?? cand.lng), codes: extractFacilityCodes(str(cand.name) ?? ""), externalIds: record(cand.externalIds), sourceCount: 0, status: str(cand.status) ?? "unknown", recordScope: "facility", parentFacilityId: null } : null); | |
| 70 | + const distanceKm = candidateSide && b && candidateSide.lat != null && candidateSide.lng != null ? haversineKm(candidateSide as Side, b) : null; | |
| 71 | + return { | |
| 72 | + ...matchDto(r), | |
| 73 | + candidateFacilityId: candId, | |
| 74 | + matched: matchedId ? by.get(matchedId) ?? null : null, | |
| 75 | + candidateFacility: candId ? by.get(candId) ?? null : null, | |
| 76 | + pair: { candidate: candidateSide, matched: b, distanceKm, sameCodes: candidateSide && b ? candidateSide.codes!.filter((c) => b.codes.includes(c)) : [] }, | |
| 77 | + }; | |
| 78 | + }); | |
| 79 | + return envelope(items, { total: rows.length ? int(rows[0]!.total) : 0, page: pg_.page, perPage: pg_.perPage }); | |
| 31 | 80 | }); |
| 32 | 81 | |
| 33 | − app.post("/matches/:id/approve", { schema: { summary: "Approve: point the candidate key at the matched facility and merge any separately-created facility into it" } }, async (req) => { | |
| 34 | − const { id } = req.params as { id: string }; | |
| 82 | + async function pendingMatch(id: string): Promise<Row> { | |
| 35 | 83 | const sql = pg(); |
| 36 | − const rows = await sql<Row[]>`select * from entity_matches where id = ${id}`; | |
| 84 | + const rows = await sql<Row[]>`select m.*, coalesce(k.entity_id, m.candidate->>'createdFacilityId') as candidate_facility_id from entity_matches m left join entity_keys k on k.key = m.candidate_key and k.entity_type = 'facility' where m.id = ${id}`; | |
| 37 | 85 | const m = rows[0]; |
| 38 | 86 | if (!m) throw notFound("match"); |
| 39 | − if (m.status !== "pending") throw badRequest(`match is already ${String(m.status)}`); | |
| 87 | + if (m.status !== "pending" && m.status !== "deferred") throw badRequest(`match is already ${String(m.status)}`); | |
| 88 | + return m; | |
| 89 | + } | |
| 90 | + | |
| 91 | + app.post("/matches/:id/approve", { schema: { summary: "Approve: point the candidate key at the matched facility and merge any separately-created facility into it" } }, async (req) => { | |
| 92 | + const { id } = req.params as { id: string }; | |
| 93 | + const sql = pg(); | |
| 94 | + const m = await pendingMatch(id); | |
| 40 | 95 | const matched = str(m.matched_facility_id); |
| 41 | 96 | if (!matched) throw badRequest("match has no matched_facility_id"); |
| 42 | 97 | const target = await sql<Row[]>`select id, merged_into from facilities where id = ${matched}`; |
@@ -44,11 +99,11 @@ export async function matchAdminRoutes(app: FastifyInstance): Promise<void> { | ||
| 44 | 99 | const survivor = str(target[0].merged_into) ?? matched; |
| 45 | 100 | await ensureManualSource(); |
| 46 | 101 | const key = reqStr(m.candidate_key); |
| 47 | − const existing = await sql<Row[]>`select entity_id from entity_keys where key = ${key}`; | |
| 48 | − const separate = str(existing[0]?.entity_id); | |
| 102 | + const existing = await sql<Row[]>`select entity_id from entity_keys where key = ${key} and entity_type = 'facility'`; | |
| 103 | + const separate = str(existing[0]?.entity_id) ?? candidateFacilityId(m); | |
| 49 | 104 | let merge: Awaited<ReturnType<typeof mergeFacility>> | null = null; |
| 50 | 105 | if (separate && separate !== survivor) merge = await mergeFacility(separate, survivor); |
| 51 | − await sql`insert into entity_keys (key, entity_type, entity_id, connector_id) values (${key}, 'facility', ${survivor}, ${reqStr(m.connector_id)}) on conflict (key) do update set entity_id = ${survivor}`; | |
| 106 | + await sql`insert into entity_keys (key, entity_type, entity_id, connector_id) values (${key}, 'facility', ${survivor}, ${reqStr(m.connector_id)}) on conflict (key, entity_type) do update set entity_id = ${survivor}`; | |
| 52 | 107 | await sql`update entity_matches set status = 'approved', decided_by = 'admin', decided_at = now() where id = ${id}`; |
| 53 | 108 | await invalidate("/datacenters"); |
| 54 | 109 | return envelope({ id, status: "approved", facilityId: survivor, candidateKey: key, merged: merge }); |
@@ -57,7 +112,7 @@ export async function matchAdminRoutes(app: FastifyInstance): Promise<void> { | ||
| 57 | 112 | app.post("/matches/:id/reject", { schema: { summary: "Reject: keep the candidate as a separate facility" } }, async (req) => { |
| 58 | 113 | const { id } = req.params as { id: string }; |
| 59 | 114 | const sql = pg(); |
| 60 | − const rows = await sql<Row[]>`update entity_matches set status = 'rejected', decided_by = 'admin', decided_at = now() where id = ${id} and status = 'pending' returning id`; | |
| 115 | + const rows = await sql<Row[]>`update entity_matches set status = 'rejected', decided_by = 'admin', decided_at = now() where id = ${id} and status in ('pending', 'deferred') returning id`; | |
| 61 | 116 | if (!rows[0]) { |
| 62 | 117 | const exists = await sql`select status from entity_matches where id = ${id}`; |
| 63 | 118 | if (!exists.length) throw notFound("match"); |
@@ -65,4 +120,33 @@ export async function matchAdminRoutes(app: FastifyInstance): Promise<void> { | ||
| 65 | 120 | } |
| 66 | 121 | return envelope({ id, status: "rejected" }); |
| 67 | 122 | }); |
| 123 | + | |
| 124 | + app.post("/matches/:id/related-campus", { schema: { summary: "Related campus: the candidate facility is a BUILDING of the matched campus (parent_facility_id set, record_scope building / campus); both rows stay, aggregates never count both" } }, async (req) => { | |
| 125 | + const { id } = req.params as { id: string }; | |
| 126 | + const sql = pg(); | |
| 127 | + const m = await pendingMatch(id); | |
| 128 | + const matched = str(m.matched_facility_id); | |
| 129 | + const building = candidateFacilityId(m); | |
| 130 | + if (!matched) throw badRequest("match has no matched_facility_id"); | |
| 131 | + if (!building) throw badRequest("the candidate was never persisted as a facility — approve or reject instead"); | |
| 132 | + const parentRow = (await sql<Row[]>`select id, merged_into from facilities where id = ${matched}`)[0]; | |
| 133 | + if (!parentRow) throw badRequest("matched facility no longer exists"); | |
| 134 | + const campus = str(parentRow.merged_into) ?? matched; | |
| 135 | + const res = await setFacilityParent(building, campus); | |
| 136 | + await sql`update entity_matches set status = 'related_campus', decided_by = 'admin', decided_at = now() where id = ${id}`; | |
| 137 | + await Promise.all([invalidate("/datacenters"), invalidate("/dashboard"), invalidate("/map")]); | |
| 138 | + return envelope({ id, status: "related_campus", building, campus, containment: res }); | |
| 139 | + }); | |
| 140 | + | |
| 141 | + app.post("/matches/:id/defer", { schema: { summary: "Defer the decision (status deferred, keeps the row in the workbench under ?status=deferred)" } }, async (req) => { | |
| 142 | + const { id } = req.params as { id: string }; | |
| 143 | + const sql = pg(); | |
| 144 | + const rows = await sql<Row[]>`update entity_matches set status = 'deferred', decided_by = 'admin', decided_at = now() where id = ${id} and status = 'pending' returning id`; | |
| 145 | + if (!rows[0]) { | |
| 146 | + const exists = await sql`select status from entity_matches where id = ${id}`; | |
| 147 | + if (!exists.length) throw notFound("match"); | |
| 148 | + throw badRequest(`match is already ${String(exists[0]!.status)}`); | |
| 149 | + } | |
| 150 | + return envelope({ id, status: "deferred" }); | |
| 151 | + }); | |
| 68 | 152 | } |
modified
apps/api/src/routes/admin/ops.ts
+7 −5
@@ -89,12 +89,14 @@ export async function opsAdminRoutes(app: FastifyInstance): Promise<void> { | ||
| 89 | 89 | }); |
| 90 | 90 | }); |
| 91 | 91 | |
| 92 | − app.post("/maintenance/:task", { schema: { summary: "Enqueue a maintenance job (rankings | metrics | refresh-stats) on dci:maintenance" } }, async (req) => { | |
| 92 | + // Note for deploy: the compose healthcheck of the api service should probe GET /api/ready (Postgres ping), not /api/health. | |
| 93 | + const maintenanceBody = z.object({ limit: z.number().int().min(1).max(100000).optional(), dryRun: z.boolean().optional(), connector: z.string().max(64).optional(), reason: z.string().max(500).optional() }).strict(); | |
| 94 | + app.post("/maintenance/:task", { schema: { summary: "Enqueue a maintenance job (rankings | metrics | refresh-stats | quality | snapshot | cleanup) on dci:maintenance; body {limit?, dryRun?, connector?, reason?} (strict)" } }, async (req) => { | |
| 93 | 95 | const { task } = req.params as { task: string }; |
| 94 | − const t = z.enum(["rankings", "metrics", "refresh-stats"]).safeParse(task); | |
| 95 | − if (!t.success) return envelope({ enqueued: false, error: "unknown task (rankings | metrics | refresh-stats)" }); | |
| 96 | − const body = parseBody(z.record(z.string(), z.unknown()).optional(), req.body ?? {}); | |
| 97 | − const job = await enqueueMaintenance(t.data, body ?? {}); | |
| 96 | + const t = z.enum(["rankings", "metrics", "refresh-stats", "quality", "snapshot", "cleanup"]).safeParse(task); | |
| 97 | + if (!t.success) return envelope({ enqueued: false, error: "unknown task (rankings | metrics | refresh-stats | quality | snapshot | cleanup)" }); | |
| 98 | + const body = parseBody(maintenanceBody, req.body ?? {}); | |
| 99 | + const job = await enqueueMaintenance(t.data, body); | |
| 98 | 100 | return envelope({ enqueued: true, job }); |
| 99 | 101 | }); |
| 100 | 102 | |
added
apps/api/src/routes/admin/projects.ts
+88 −0
@@ -0,0 +1,88 @@ | ||
| 1 | +/** Project curation routes: hide / unhide / merge / PATCH. */ | |
| 2 | +import type { FastifyInstance } from "fastify"; | |
| 3 | +import { z } from "zod"; | |
| 4 | +import { AI_EVIDENCE_LEVELS, FACILITY_STATUSES, GEO_PRECISIONS, PROJECT_CLASSES, parsePartialDate } from "@dci/core"; | |
| 5 | +import { envelope, notFound, parseBody, badRequest } from "../../lib/http.js"; | |
| 6 | +import { pg } from "../../lib/sql.js"; | |
| 7 | +import { invalidate } from "../../cache.js"; | |
| 8 | +import { mergeProject, patchProject, setProjectHidden } from "../../repositories/admin/projects.js"; | |
| 9 | + | |
| 10 | +const partialDate = z.string().max(20).nullable().refine((v) => v == null || parsePartialDate(v) != null, "not a (partial) date"); | |
| 11 | + | |
| 12 | +const patchSchema = z.object({ | |
| 13 | + name: z.string().min(2).max(200).optional(), | |
| 14 | + status: z.enum(FACILITY_STATUSES).optional(), | |
| 15 | + planned_mw: z.number().positive().max(20000).nullable().optional(), | |
| 16 | + investment_usd: z.number().positive().nullable().optional(), | |
| 17 | + expected_opening: partialDate.optional(), | |
| 18 | + announced_on: partialDate.optional(), | |
| 19 | + construction_started_on: partialDate.optional(), | |
| 20 | + approved_on: partialDate.optional(), | |
| 21 | + permit_filed_on: partialDate.optional(), | |
| 22 | + opened_on: partialDate.optional(), | |
| 23 | + operator_id: z.string().nullable().optional(), | |
| 24 | + country_iso2: z.string().length(2).nullable().optional(), | |
| 25 | + metro_id: z.string().nullable().optional(), | |
| 26 | + city: z.string().max(200).nullable().optional(), | |
| 27 | + lat: z.number().min(-90).max(90).nullable().optional(), | |
| 28 | + lng: z.number().min(-180).max(180).nullable().optional(), | |
| 29 | + geo_precision: z.enum(GEO_PRECISIONS).optional(), | |
| 30 | + project_class: z.enum(PROJECT_CLASSES).nullable().optional(), | |
| 31 | + ai_evidence: z.enum(AI_EVIDENCE_LEVELS).optional(), | |
| 32 | + is_ai: z.boolean().optional(), | |
| 33 | + description: z.string().max(5000).nullable().optional(), | |
| 34 | + note: z.string().max(500).optional(), | |
| 35 | +}).strict(); | |
| 36 | + | |
| 37 | +const CACHE_PREFIXES = ["/projects", "/dashboard", "/map", "/explore", "/pulse"]; | |
| 38 | +async function bust(): Promise<void> { await Promise.all(CACHE_PREFIXES.map((p) => invalidate(p))); } | |
| 39 | + | |
| 40 | +export async function projectAdminRoutes(app: FastifyInstance): Promise<void> { | |
| 41 | + app.post("/projects/:id/hide", { schema: { summary: "Hide a project (false positive kept for audit, never listed) {reason}; writes src_manual provenance + a project_status_changed event {hidden}" } }, async (req) => { | |
| 42 | + const { id } = req.params as { id: string }; | |
| 43 | + const body = parseBody(z.object({ reason: z.string().min(1).max(500) }), req.body); | |
| 44 | + const res = await setProjectHidden(id, true, body.reason); | |
| 45 | + if (!res) throw notFound("project"); | |
| 46 | + await bust(); | |
| 47 | + return envelope(res); | |
| 48 | + }); | |
| 49 | + | |
| 50 | + app.post("/projects/:id/unhide", { schema: { summary: "Restore a hidden project {reason?}" } }, async (req) => { | |
| 51 | + const { id } = req.params as { id: string }; | |
| 52 | + const body = parseBody(z.object({ reason: z.string().max(500).optional() }), req.body ?? {}); | |
| 53 | + const res = await setProjectHidden(id, false, body.reason ?? null); | |
| 54 | + if (!res) throw notFound("project"); | |
| 55 | + await bust(); | |
| 56 | + return envelope(res); | |
| 57 | + }); | |
| 58 | + | |
| 59 | + app.post("/projects/:id/merge", { schema: { summary: "Merge this project into another {into}: events, timeline, claims, provenance, entity keys move to the survivor; the duplicate keeps merged_into" } }, async (req) => { | |
| 60 | + const { id } = req.params as { id: string }; | |
| 61 | + const body = parseBody(z.object({ into: z.string().min(1) }), req.body); | |
| 62 | + let res: Awaited<ReturnType<typeof mergeProject>>; | |
| 63 | + try { res = await mergeProject(id, body.into); } catch (e) { | |
| 64 | + const msg = (e as Error).message; | |
| 65 | + if (msg === "project not found") throw notFound("project"); | |
| 66 | + throw badRequest(msg); | |
| 67 | + } | |
| 68 | + await bust(); | |
| 69 | + return envelope(res); | |
| 70 | + }); | |
| 71 | + | |
| 72 | + app.patch("/projects/:id", { schema: { summary: "Curate project fields (strict body); src_manual provenance on every field; events only for status (project_status_changed) and planned_mw (planned_capacity_changed)" } }, async (req) => { | |
| 73 | + const { id } = req.params as { id: string }; | |
| 74 | + const body = parseBody(patchSchema, req.body); | |
| 75 | + const { note, ...fields } = body; | |
| 76 | + const entries = Object.entries(fields).filter(([, v]) => v !== undefined); | |
| 77 | + if (!entries.length) throw badRequest("no fields to update"); | |
| 78 | + if ((fields.lat === undefined) !== (fields.lng === undefined) || (fields.lat == null) !== (fields.lng == null)) throw badRequest("lat and lng must be set together"); | |
| 79 | + const sql = pg(); | |
| 80 | + if (fields.operator_id) { const r = await sql`select id from operators where id = ${fields.operator_id}`; if (!r.length) throw badRequest("operator_id does not exist"); } | |
| 81 | + if (fields.country_iso2) { fields.country_iso2 = fields.country_iso2.toUpperCase(); const r = await sql`select iso2 from countries where iso2 = ${fields.country_iso2}`; if (!r.length) throw badRequest("country_iso2 does not exist"); } | |
| 82 | + if (fields.metro_id) { const r = await sql`select id from metros where id = ${fields.metro_id}`; if (!r.length) throw badRequest("metro_id does not exist"); } | |
| 83 | + const res = await patchProject(id, fields, note ?? null); | |
| 84 | + if (!res) throw notFound("project"); | |
| 85 | + if (res.changed.length) await bust(); | |
| 86 | + return envelope({ ...res, message: res.changed.length ? undefined : "no changes" }); | |
| 87 | + }); | |
| 88 | +} | |
added
apps/api/src/routes/admin/proxy.ts
+39 −0
@@ -0,0 +1,39 @@ | ||
| 1 | +/** | |
| 2 | + * Proxies to the worker's internal HTTP server (DCI_WORKER_URL, default http://127.0.0.1:8320; compose: | |
| 3 | + * http://worker-maint:8320): GET /data-gaps and GET /trace/<documentId> (extraction debugger). 10 s timeout, 502 on failure. | |
| 4 | + */ | |
| 5 | +import type { FastifyInstance } from "fastify"; | |
| 6 | +import { z } from "zod"; | |
| 7 | +import { getEnv } from "../../env.js"; | |
| 8 | +import { envelope, HttpError, parseQuery } from "../../lib/http.js"; | |
| 9 | +import { strParam } from "../../lib/params.js"; | |
| 10 | + | |
| 11 | +async function workerGet(path: string): Promise<unknown> { | |
| 12 | + const base = getEnv().workerUrl.replace(/\/+$/, ""); | |
| 13 | + const url = `${base}${path}`; | |
| 14 | + let host = base; | |
| 15 | + try { host = new URL(base).host; } catch { /* keep raw */ } | |
| 16 | + try { | |
| 17 | + const res = await fetch(url, { headers: { accept: "application/json" }, signal: AbortSignal.timeout(10_000) }); | |
| 18 | + const text = await res.text(); | |
| 19 | + let body: unknown = text; | |
| 20 | + try { body = JSON.parse(text); } catch { /* keep text */ } | |
| 21 | + if (!res.ok) throw new HttpError(res.status === 404 ? 404 : 502, `worker answered HTTP ${res.status}`, { worker: host, path, body }); | |
| 22 | + return body; | |
| 23 | + } catch (e) { | |
| 24 | + if (e instanceof HttpError) throw e; | |
| 25 | + const msg = (e as Error).name === "TimeoutError" ? "timeout after 10 s" : (e as Error).message; | |
| 26 | + throw new HttpError(502, `worker unreachable: ${msg}`, { worker: host, path }); | |
| 27 | + } | |
| 28 | +} | |
| 29 | + | |
| 30 | +export async function proxyAdminRoutes(app: FastifyInstance): Promise<void> { | |
| 31 | + app.get("/data-gaps", { schema: { summary: "Data gaps counters (proxied from the worker GET /data-gaps; 502 when the worker is unreachable)" } }, async () => envelope(await workerGet("/data-gaps"), { source: getEnv().workerUrl })); | |
| 32 | + | |
| 33 | + app.get("/documents/:id/trace", { schema: { summary: "Extraction debugger trace for a document (proxied from the worker GET /trace/<id>?live=1 re-fetches the page)" } }, async (req) => { | |
| 34 | + const { id } = req.params as { id: string }; | |
| 35 | + const q = parseQuery(z.object({ live: strParam }), req.query); | |
| 36 | + const live = q.live === "1" || q.live === "true" ? "1" : ""; | |
| 37 | + return envelope(await workerGet(`/trace/${encodeURIComponent(id)}${live ? "?live=1" : ""}`), { source: getEnv().workerUrl, live: live === "1" }); | |
| 38 | + }); | |
| 39 | +} | |
added
apps/api/src/routes/admin/quality.ts
+32 −0
@@ -0,0 +1,32 @@ | ||
| 1 | +/** Admin data-quality dashboard: overview, flag list, resolve / dismiss. */ | |
| 2 | +import type { FastifyInstance } from "fastify"; | |
| 3 | +import { z } from "zod"; | |
| 4 | +import { envelope, notFound, parseBody, parseQuery } from "../../lib/http.js"; | |
| 5 | +import { intParam, pageParam, strParam } from "../../lib/params.js"; | |
| 6 | +import { listFlags, qualityOverview, setFlagStatus } from "../../repositories/admin/quality.js"; | |
| 7 | + | |
| 8 | +export async function qualityAdminRoutes(app: FastifyInstance): Promise<void> { | |
| 9 | + app.get("/quality", { schema: { summary: "Data-quality overview (QualityOverview): open flags by severity / code, largest published values, review queue, alerts" } }, async () => envelope(await qualityOverview(), { methodology: "Counts are over open quality_flags, claims by status and live (not hidden / merged) projects. 'Largest' lists are plain ORDER BY on published figures; pipeline MW per operator / metro is containment-aware (campus vs buildings never both) plus site-scoped project MW." })); | |
| 10 | + | |
| 11 | + app.get("/quality/flags", { schema: { summary: "Quality flags (QualityFlagDTO[]) ?status=open|resolved|dismissed|all&code=&entity_type=&min_priority=&page=&per_page=" } }, async (req) => { | |
| 12 | + const q = parseQuery(z.object({ status: strParam, code: strParam, entity_type: strParam, min_priority: intParam, page: pageParam, per_page: intParam }), req.query); | |
| 13 | + const res = await listFlags({ status: q.status ?? "open", code: q.code, entityType: q.entity_type, minPriority: q.min_priority, page: q.page, perPage: q.per_page }); | |
| 14 | + return envelope(res.items, { total: res.total, page: res.page, perPage: res.perPage }); | |
| 15 | + }); | |
| 16 | + | |
| 17 | + app.post("/quality/flags/:id/resolve", { schema: { summary: "Resolve a quality flag {resolution}" } }, async (req) => { | |
| 18 | + const { id } = req.params as { id: string }; | |
| 19 | + const body = parseBody(z.object({ resolution: z.string().min(1).max(500) }), req.body); | |
| 20 | + const flag = await setFlagStatus(id, "resolved", body.resolution); | |
| 21 | + if (!flag) throw notFound("quality flag"); | |
| 22 | + return envelope(flag); | |
| 23 | + }); | |
| 24 | + | |
| 25 | + app.post("/quality/flags/:id/dismiss", { schema: { summary: "Dismiss a quality flag {reason?}" } }, async (req) => { | |
| 26 | + const { id } = req.params as { id: string }; | |
| 27 | + const body = parseBody(z.object({ reason: z.string().max(500).optional() }), req.body ?? {}); | |
| 28 | + const flag = await setFlagStatus(id, "dismissed", body.reason ?? null); | |
| 29 | + if (!flag) throw notFound("quality flag"); | |
| 30 | + return envelope(flag); | |
| 31 | + }); | |
| 32 | +} | |
modified
apps/api/src/routes/public/activity.ts
+34 −18
@@ -4,44 +4,60 @@ import { z } from "zod"; | ||
| 4 | 4 | import { publicGet } from "../../lib/route.js"; |
| 5 | 5 | import { TTL, csv, notFound } from "../../lib/http.js"; |
| 6 | 6 | import { boolParam, csvParam, intParam, numParam, orderParam, pageParam, strParam } from "../../lib/params.js"; |
| 7 | −import { getProjectDetail, listProjects, projectPipeline } from "../../repositories/projects.js"; | |
| 8 | −import { getEvent, listEvents } from "../../repositories/events.js"; | |
| 7 | +import { sourcesForEntities, sourcesForIds } from "../../lib/source-history.js"; | |
| 8 | +import { getProjectClaims, getProjectDetail, getProjectHistory, listProjects, projectPipeline } from "../../repositories/projects.js"; | |
| 9 | +import { EVENTS_DEDUPE_METHODOLOGY, getEvent, listEvents } from "../../repositories/events.js"; | |
| 9 | 10 | import { listNews } from "../../repositories/misc.js"; |
| 10 | 11 | |
| 11 | 12 | const empty = (v: unknown) => (v === "" || v === null ? undefined : v); |
| 13 | +const PROJECTS_NOTE = "Project records only (never facilities). False positives hidden by review and merged duplicates are excluded from every list, count and sum. plannedMw / investmentUsd are site-scoped published figures; company-wide or portfolio totals stay as claims (see /projects/:slug/claims)."; | |
| 12 | 14 | |
| 13 | 15 | export async function activityRoutes(app: FastifyInstance): Promise<void> { |
| 14 | 16 | // ---- projects (static /pipeline before /:slug — find-my-way prefers static anyway) |
| 15 | − publicGet(app, { url: "/projects/pipeline", ttl: TTL.list, summary: "Project pipeline aggregates: by status, by expected year, top 15 countries", tags: ["projects"] }, async () => { | |
| 17 | + publicGet(app, { url: "/projects/pipeline", ttl: TTL.list, summary: "Project pipeline aggregates: by status, by expected year, top 15 countries", tags: ["projects"], response: { type: "PipelineAggregates" } }, async () => { | |
| 16 | 18 | const data = await projectPipeline(); |
| 17 | − return { data, meta: { methodology: "count and SUM(planned_mw) over projects; byYear uses the year prefix of expected_opening and excludes cancelled/closed" } }; | |
| 19 | + return { data, meta: { methodology: `count and SUM(planned_mw) over live projects (hidden / merged excluded); byYear uses the year prefix of expected_opening and excludes cancelled/closed. ${PROJECTS_NOTE}` } }; | |
| 18 | 20 | }); |
| 19 | 21 | |
| 20 | − const projectsQuery = z.object({ status: csvParam, country: strParam, operator: strParam, metro: strParam, min_mw: numParam, ai: boolParam, q: strParam, sort: z.preprocess(empty, z.enum(["updated", "mw", "announced", "opening"]).optional()), order: orderParam, page: pageParam, per_page: intParam }); | |
| 21 | − publicGet(app, { url: "/projects", ttl: TTL.list, query: projectsQuery, summary: "List projects (ProjectSummary[])", tags: ["projects"] }, async (q) => { | |
| 22 | − const res = await listProjects({ ...q, status: csv(q.status) }); | |
| 23 | − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage } }; | |
| 22 | + const projectsQuery = z.object({ status: csvParam, project_status: csvParam, country: strParam, operator: strParam, metro: strParam, min_mw: numParam, max_mw: numParam, ai: boolParam, project_class: csvParam, evidence_level: csvParam, expected_from: intParam, expected_to: intParam, announced_since: strParam, q: strParam, sort: z.preprocess(empty, z.enum(["updated", "mw", "announced", "opening"]).optional()), order: orderParam, page: pageParam, per_page: intParam }); | |
| 23 | + publicGet(app, { url: "/projects", ttl: TTL.list, query: projectsQuery, summary: "List projects (ProjectSummary[]) — status/project_status, country, operator, metro, min/max_mw, ai, project_class, evidence_level, expected_from/to, announced_since, q", tags: ["projects"], response: { type: "ProjectSummary[]" } }, async (q) => { | |
| 24 | + const res = await listProjects({ ...q, status: [...csv(q.status), ...csv(q.project_status)], project_class: csv(q.project_class), evidence_level: csv(q.evidence_level) }); | |
| 25 | + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: PROJECTS_NOTE }, sources: await sourcesForEntities("project", res.items.map((p) => p.id)) }; | |
| 26 | + }); | |
| 27 | + publicGet(app, { url: "/projects/:slug", ttl: TTL.detail, query: z.object({ radius_km: numParam }), summary: "Project detail (ProjectDetail): lifecycle stages + velocity, claims, history, nearby infrastructure (?radius_km), related events, data quality, provenance, sourceHistory", tags: ["projects"], params: { slug: "project slug or prj_… id" }, response: { type: "ProjectDetail" } }, async (q, params) => { | |
| 28 | + const res = await getProjectDetail(params.slug!, { radiusKm: q.radius_km != null ? Math.min(200, Math.max(0.1, q.radius_km)) : undefined }); | |
| 29 | + if (!res) throw notFound("project"); | |
| 30 | + return { data: res.detail, sources: res.sources, meta: { methodology: `${PROJECTS_NOTE} Stages are dated from the project's dated columns, its timeline and status-change events (earliest evidence wins); velocityDays are day counts between dated stages, null when a date is unknown.` } }; | |
| 24 | 31 | }); |
| 25 | − publicGet(app, { url: "/projects/:slug", ttl: TTL.detail, summary: "Project detail (ProjectDetail) with timeline, status history, provenance, events, sourceHistory", tags: ["projects"], params: { slug: "project slug or prj_… id" } }, async (_q, params) => { | |
| 26 | − const res = await getProjectDetail(params.slug!); | |
| 32 | + publicGet(app, { url: "/projects/:slug/history", ttl: TTL.detail, summary: "Project field history (EntityHistory)", tags: ["projects"], params: { slug: "project slug or prj_… id" }, response: { type: "EntityHistory" } }, async (_q, params) => { | |
| 33 | + const res = await getProjectHistory(params.slug!); | |
| 27 | 34 | if (!res) throw notFound("project"); |
| 28 | − return { data: res.detail, sources: res.sources }; | |
| 35 | + return { data: res.history, sources: res.sources }; | |
| 36 | + }); | |
| 37 | + publicGet(app, { url: "/projects/:slug/claims", ttl: TTL.detail, query: z.object({ status: csvParam, predicate: strParam }), summary: "Claims about a project (ClaimDTO[]) — capacity and investment figures with scope and evidence", tags: ["projects"], params: { slug: "project slug or prj_… id" }, response: { type: "ClaimDTO[]" } }, async (q, params) => { | |
| 38 | + const res = await getProjectClaims(params.slug!, { status: csv(q.status), predicate: q.predicate }); | |
| 39 | + if (!res) throw notFound("project"); | |
| 40 | + return { data: res.claims, sources: res.sources, meta: { total: res.claims.length, projectId: res.id } }; | |
| 29 | 41 | }); |
| 30 | 42 | |
| 31 | 43 | // ---- events |
| 32 | − const eventsQuery = z.object({ type: csvParam, country: strParam, operator: strParam, metro: strParam, project: strParam, entity_type: strParam, entity_id: strParam, min_significance: intParam, since: strParam, page: pageParam, per_page: intParam }); | |
| 33 | − publicGet(app, { url: "/events", ttl: TTL.events, query: eventsQuery, summary: "Change feed (EventDTO[]) — newest first", tags: ["events"] }, async (q) => { | |
| 34 | − const res = await listEvents({ type: csv(q.type), country: q.country, operator: q.operator, metro: q.metro, project: q.project, entityType: q.entity_type, entityId: q.entity_id, minSignificance: q.min_significance, since: q.since, page: q.page, perPage: q.per_page }); | |
| 35 | − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage } }; | |
| 44 | + const eventsQuery = z.object({ | |
| 45 | + type: csvParam, country: strParam, metro: strParam, operator: strParam, project: strParam, entity_type: strParam, entity_id: strParam, | |
| 46 | + min_significance: intParam, significance: z.preprocess(empty, z.enum(["major", "medium", "minor"]).optional()), confidence: csvParam, source_kind: csvParam, ai: boolParam, | |
| 47 | + since: strParam, until: strParam, q: strParam, dedupe: boolParam, page: pageParam, per_page: intParam, | |
| 48 | + }); | |
| 49 | + publicGet(app, { url: "/events", ttl: TTL.events, query: eventsQuery, summary: "Change feed (EventDTO[]) — newest first; filters: type, country, metro, operator, project, entity, significance band, confidence, source_kind, ai, since/until, q; dedupe=true collapses clustered coverage", tags: ["events"], response: { type: "EventDTO[]" } }, async (q) => { | |
| 50 | + const res = await listEvents({ type: csv(q.type), country: q.country, operator: q.operator, metro: q.metro, project: q.project, entityType: q.entity_type, entityId: q.entity_id, minSignificance: q.min_significance, significance: q.significance, confidence: csv(q.confidence), sourceKind: csv(q.source_kind), ai: q.ai, since: q.since, until: q.until, q: q.q, dedupe: q.dedupe !== false, page: q.page, perPage: q.per_page }); | |
| 51 | + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, dedupe: q.dedupe !== false, methodology: EVENTS_DEDUPE_METHODOLOGY }, sources: await sourcesForIds(res.items.map((e) => e.sourceId)) }; | |
| 36 | 52 | }); |
| 37 | − publicGet(app, { url: "/events/:id", ttl: TTL.detail, summary: "Single event (EventDTO)", tags: ["events"], params: { id: "evt_… id" } }, async (_q, params) => { | |
| 53 | + publicGet(app, { url: "/events/:id", ttl: TTL.detail, summary: "Single event (EventDTO) with otherSources", tags: ["events"], params: { id: "evt_… id" }, response: { type: "EventDTO" } }, async (_q, params) => { | |
| 38 | 54 | const e = await getEvent(params.id!); |
| 39 | 55 | if (!e) throw notFound("event"); |
| 40 | − return { data: e }; | |
| 56 | + return { data: e, sources: await sourcesForIds([e.sourceId]) }; | |
| 41 | 57 | }); |
| 42 | 58 | |
| 43 | 59 | // ---- news |
| 44 | − publicGet(app, { url: "/news", ttl: TTL.events, query: z.object({ country: strParam, operator: strParam, since: strParam, q: strParam, page: pageParam, per_page: intParam }), summary: "News items (crawled announcements) — newest first", tags: ["events"] }, async (q) => { | |
| 60 | + publicGet(app, { url: "/news", ttl: TTL.events, query: z.object({ country: strParam, operator: strParam, since: strParam, q: strParam, page: pageParam, per_page: intParam }), summary: "News items (crawled announcements) — newest first", tags: ["events"], response: { type: "NewsItemDTO[]" } }, async (q) => { | |
| 45 | 61 | const res = await listNews(q); |
| 46 | 62 | return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage } }; |
| 47 | 63 | }); |
modified
apps/api/src/routes/public/discovery.ts
+21 −8
@@ -2,11 +2,15 @@ | ||
| 2 | 2 | import type { FastifyInstance } from "fastify"; |
| 3 | 3 | import { z } from "zod"; |
| 4 | 4 | import { publicGet } from "../../lib/route.js"; |
| 5 | +import { DENSITY_VIEWS, MAP_LAYERS } from "@dci/core"; | |
| 5 | 6 | import { TTL, badRequest, csv, notFound } from "../../lib/http.js"; |
| 6 | 7 | import { boolParam, csvParam, intParam, numParam, pageParam, strParam } from "../../lib/params.js"; |
| 7 | 8 | import { METHODOLOGY_MW } from "../../lib/sql.js"; |
| 8 | −import { mapData, MAX_POINTS, type Bbox } from "../../repositories/map.js"; | |
| 9 | −import { search } from "../../repositories/search.js"; | |
| 9 | +import { sourcesForEntities } from "../../lib/source-history.js"; | |
| 10 | +import { mapData, MAP_METHODOLOGY, MAX_POINTS, type Bbox } from "../../repositories/map.js"; | |
| 11 | +import { facilityIdsOf, search } from "../../repositories/search.js"; | |
| 12 | + | |
| 13 | +const empty = (v: unknown) => (v === "" || v === null ? undefined : v); | |
| 10 | 14 | import { getRanking, listRankings } from "../../repositories/rankings.js"; |
| 11 | 15 | import { dashboard } from "../../repositories/dashboard.js"; |
| 12 | 16 | import { availableMetrics, getSource, listSources, sitemap, SITEMAP_KINDS, timeseries, type SitemapKind } from "../../repositories/misc.js"; |
@@ -21,15 +25,24 @@ function parseBbox(s: string | undefined): Bbox | undefined { | ||
| 21 | 25 | } |
| 22 | 26 | |
| 23 | 27 | export async function discoveryRoutes(app: FastifyInstance): Promise<void> { |
| 24 | − const mapQuery = z.object({ zoom: numParam, bbox: strParam, status: csvParam, type: csvParam, operator: strParam, country: csvParam, min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, cloud_regions: boolParam }); | |
| 25 | − publicGet(app, { url: "/map", ttl: TTL.map, query: mapQuery, summary: "Map data: country clusters (zoom < 5), grid clusters (5–8), points (≥ 9, capped 5 000 → clusters)", tags: ["map"] }, async (q) => { | |
| 26 | − const data = await mapData({ zoom: q.zoom ?? 2, bbox: parseBbox(q.bbox), status: csv(q.status), type: csv(q.type), operator: q.operator, country: csv(q.country).map((c) => c.toUpperCase()), min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, cloud_regions: q.cloud_regions }); | |
| 27 | − return { data, meta: { maxPoints: MAX_POINTS, methodology: "Only facilities with coordinates; `p` is the geo precision (city-level points are approximate). mw = best known figure." } }; | |
| 28 | + const mapQuery = z.object({ | |
| 29 | + zoom: numParam, bbox: strParam.describe("w,s,e,n"), status: csvParam, type: csvParam, operator: strParam, metro: strParam, country: csvParam, q: strParam, min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, has_mw: boolParam, confidence: csvParam, opened_from: intParam, opened_to: intParam, cloud_regions: boolParam.describe("legacy: add the cloud-region overlay"), | |
| 30 | + layer: z.preprocess(empty, z.enum(MAP_LAYERS).optional()).describe(`thematic layer: ${MAP_LAYERS.join(" | ")} (default facilities)`), | |
| 31 | + density: z.preprocess(empty, z.enum(DENSITY_VIEWS).optional()).describe(`density grid weighted by ${DENSITY_VIEWS.join(" | ")}`), | |
| 32 | + year: intParam.describe("time machine: facilities with an opening date ≤ year"), | |
| 33 | + project_status: csvParam, expected_from: intParam, expected_to: intParam, location_precision: csvParam, ai_evidence: csvParam.describe("confirmed | likely | associated | unknown"), | |
| 34 | + }); | |
| 35 | + publicGet(app, { url: "/map", ttl: TTL.map, query: mapQuery, summary: "Map data (MapResponse): country clusters (zoom < 5), grid clusters (5–8), points (≥ 9, capped 5 000 → clusters + degraded); layers (capacity, projects, ai, cloud, ixps, connectivity, power, pipeline), density grids and a year filter", tags: ["map"], response: { type: "MapResponse", example: { zoom: 4, mode: "clusters", layer: "facilities", clusters: [{ key: "US", lat: 39.8, lng: -98.6, count: 2600, mw: 12000, statuses: { operational: 2400 }, label: "United States" }], total: 2600, year: null, yearCoverage: null } } }, async (q) => { | |
| 36 | + const data = await mapData({ zoom: q.zoom ?? 2, bbox: parseBbox(q.bbox), status: csv(q.status), type: csv(q.type), operator: q.operator, metro: q.metro, country: csv(q.country).map((c) => c.toUpperCase()), q: q.q, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, confidence: csv(q.confidence), opened_from: q.opened_from, opened_to: q.opened_to, cloud_regions: q.cloud_regions, layer: q.layer, density: q.density, year: q.year, project_status: csv(q.project_status), expected_from: q.expected_from, expected_to: q.expected_to, location_precision: csv(q.location_precision), ai_evidence: csv(q.ai_evidence) }); | |
| 37 | + const facilityIds = data.mode === "points" ? (data.points ?? []).filter((p) => !p.k).map((p) => p.id) : []; | |
| 38 | + const sources = facilityIds.length ? await sourcesForEntities("facility", facilityIds) : []; | |
| 39 | + return { data, meta: { maxPoints: MAX_POINTS, layer: data.layer, methodology: MAP_METHODOLOGY }, sources }; | |
| 28 | 40 | }); |
| 29 | 41 | |
| 30 | − publicGet(app, { url: "/search", ttl: TTL.search, query: z.object({ q: strParam, type: strParam }), summary: "Universal search: interprets the query (country, operator, status, MW, type) and returns ranked hits + matching facilities", tags: ["search"] }, async (q) => { | |
| 42 | + publicGet(app, { url: "/search", ttl: TTL.search, query: z.object({ q: strParam, type: strParam }), summary: "Universal search (SearchResponse): interprets the query (country, metro, operator, status, min / max MW, type, AI, projects, opening year) and returns ranked hits + matching facilities or projects", tags: ["search"], response: { type: "SearchResponse", example: { query: "projects over 500 MW in texas", interpreted: { entity: "project", minMw: 500, text: "texas" }, hits: [] } } }, async (q) => { | |
| 31 | 43 | const data = await search(q.q ?? ""); |
| 32 | − return { data, meta: { total: data.hits.length } }; | |
| 44 | + const sources = await sourcesForEntities("facility", facilityIdsOf(data)); | |
| 45 | + return { data, meta: { total: data.hits.length, interpreted: data.interpreted }, sources }; | |
| 33 | 46 | }); |
| 34 | 47 | |
| 35 | 48 | publicGet(app, { url: "/rankings", ttl: TTL.rankings, summary: "Current rankings (key, label, scope, unit, methodology, computedAt, total)", tags: ["rankings"] }, async () => { |
modified
apps/api/src/routes/public/docs-meta.ts
+39 −3
@@ -1,6 +1,42 @@ | ||
| 1 | −/** /docs-meta — endpoint catalogue generated from the route registry for the web API docs page. */ | |
| 1 | +/** /docs-meta — endpoint catalogue (ApiEndpointDoc[]) generated from the live route registry for the web API docs page. */ | |
| 2 | 2 | import type { FastifyInstance } from "fastify"; |
| 3 | +import type { ApiEndpointDoc } from "@dci/core"; | |
| 4 | +import { getEnv } from "../../env.js"; | |
| 5 | +import { publicGet, routeRegistry, type RegisteredRoute } from "../../lib/route.js"; | |
| 6 | +import { TTL } from "../../lib/http.js"; | |
| 3 | 7 | |
| 4 | −export async function docsMetaRoutes(_app: FastifyInstance): Promise<void> { | |
| 5 | − // filled in by the intelligence implementation | |
| 8 | +const SAMPLE: Record<string, string> = { idOrSlug: "equinix-dc2", slug: "equinix", slugOrIso2: "us", key: "facilities", id: "evt_example", kind: "facilities", lat: "39.04", lng: "-77.49", radius_km: "25", country: "US", q: "hyperscale texas", zoom: "4", window: "7d", slugs: "equinix,digital-realty", format: "csv", entity: "facilities", status: "operational", per_page: "10", page: "1", layer: "facilities", year: "2020", types: "facilities,ixps", metric: "facilities_total" }; | |
| 9 | + | |
| 10 | +function sampleFor(p: RegisteredRoute["params"][number]): string { | |
| 11 | + if (p.example) return p.example; | |
| 12 | + if (SAMPLE[p.name]) return SAMPLE[p.name]!; | |
| 13 | + if (p.type === "integer" || p.type === "number") return "1"; | |
| 14 | + if (p.type === "boolean") return "true"; | |
| 15 | + return "value"; | |
| 16 | +} | |
| 17 | + | |
| 18 | +export function endpointDoc(r: RegisteredRoute, siteUrl: string): ApiEndpointDoc { | |
| 19 | + let path = r.path; | |
| 20 | + for (const p of r.params.filter((x) => x.in === "path")) path = path.replace(`:${p.name}`, encodeURIComponent(sampleFor(p))); | |
| 21 | + const qs = r.params.filter((x) => x.in === "query").slice(0, 2).map((p) => `${p.name}=${encodeURIComponent(sampleFor(p))}`).join("&"); | |
| 22 | + const url = `${siteUrl}${path}${qs && r.method === "GET" ? `?${qs}` : ""}`; | |
| 23 | + const method: ApiEndpointDoc["method"] = r.method === "PATCH" ? "POST" : r.method; | |
| 24 | + const bodyObj = r.method === "POST" ? Object.fromEntries(r.params.filter((x) => x.in === "query").slice(0, 2).map((p) => [p.name, sampleFor(p)])) : null; | |
| 25 | + const body = bodyObj ? JSON.stringify(bodyObj) : null; | |
| 26 | + const curl = method === "GET" ? `curl -s "${url}"` : method === "DELETE" ? `curl -s -X DELETE "${url}"` : `curl -s -X POST "${url}" -H 'content-type: application/json' -d '${body}'`; | |
| 27 | + const js = method === "GET" | |
| 28 | + ? `const res = await fetch("${url}");\nconst { data, meta, sources } = await res.json();` | |
| 29 | + : `const res = await fetch("${url}", { method: "${method}"${body ? `, headers: { "content-type": "application/json" }, body: JSON.stringify(${body})` : ""}, credentials: "include" });\nconst { data } = await res.json();`; | |
| 30 | + const python = method === "GET" | |
| 31 | + ? `import requests\nr = requests.get("${url}")\npayload = r.json()\ndata, meta, sources = payload["data"], payload.get("meta"), payload.get("sources")` | |
| 32 | + : `import requests\nr = requests.${method.toLowerCase()}("${url}"${body ? `, json=${body}` : ""})\ndata = r.json()["data"]`; | |
| 33 | + return { method, path: r.path, summary: r.summary, group: r.group, params: r.params.map((p) => ({ name: p.name, in: p.in, type: p.type, description: p.description, ...(p.example ? { example: p.example } : {}) })), example: { curl, js, python }, responseSchema: r.responseType + (r.responseDescription ? ` — ${r.responseDescription}` : "") }; | |
| 34 | +} | |
| 35 | + | |
| 36 | +export async function docsMetaRoutes(app: FastifyInstance): Promise<void> { | |
| 37 | + publicGet(app, { url: "/docs-meta", ttl: TTL.sitemap, tags: ["docs"], summary: "Endpoint catalogue (ApiEndpointDoc[]) generated from the live route table: method, path, params, curl / JS / Python examples, response type", response: { type: "ApiEndpointDoc[]", example: [{ method: "GET", path: "/api/v1/datacenters", summary: "List facilities", group: "facilities", params: [{ name: "country", in: "query", type: "string", description: "ISO-3166 alpha-2" }], example: { curl: "curl -s https://www.datacenterindex.io/api/v1/datacenters?country=US", js: "await fetch(…)", python: "requests.get(…)" }, responseSchema: "FacilitySummary[]" }] } }, async () => { | |
| 38 | + const siteUrl = getEnv().siteUrl.replace(/\/$/, ""); | |
| 39 | + const data = routeRegistry.map((r) => endpointDoc(r, siteUrl)).sort((a, b) => a.group.localeCompare(b.group) || a.path.localeCompare(b.path) || a.method.localeCompare(b.method)); | |
| 40 | + return { data, meta: { total: data.length, methodology: "Generated from the Fastify route table at boot; every 200 response is the envelope { data, meta, sources } unless the summary says otherwise (downloads stream raw rows)." } }; | |
| 41 | + }); | |
| 6 | 42 | } |
modified
apps/api/src/routes/public/download.ts
+41 −3
@@ -1,6 +1,44 @@ | ||
| 1 | −/** Dataset downloads (CSV / JSON / GeoJSON) with license gating. */ | |
| 1 | +/** Dataset downloads (CSV / JSON / GeoJSON) with license gating — streamed, no envelope. */ | |
| 2 | +import { Readable } from "node:stream"; | |
| 2 | 3 | import type { FastifyInstance } from "fastify"; |
| 4 | +import { z } from "zod"; | |
| 5 | +import { publicGet, registerRoute } from "../../lib/route.js"; | |
| 6 | +import { TTL, HttpError, parseQuery } from "../../lib/http.js"; | |
| 7 | +import { boolParam, csvParam, intParam, numParam, strParam } from "../../lib/params.js"; | |
| 8 | +import { contentType, DOWNLOAD_KEYS, DOWNLOAD_LICENSE, listDatasets, MAX_ROWS, restrictedSources, specFor, streamDataset, type DownloadFormat, type DownloadKey } from "../../repositories/download.js"; | |
| 3 | 9 | |
| 4 | −export async function downloadRoutes(_app: FastifyInstance): Promise<void> { | |
| 5 | − // filled in by the intelligence implementation | |
| 10 | +const empty = (v: unknown) => (v === "" || v === null ? undefined : v); | |
| 11 | +const downloadQuery = z.object({ | |
| 12 | + format: z.preprocess(empty, z.enum(["csv", "json", "geojson"]).optional()), | |
| 13 | + country: csvParam, status: csvParam, project_status: csvParam, type: csvParam, operator: strParam, metro: strParam, | |
| 14 | + min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, has_mw: boolParam, | |
| 15 | + expected_from: intParam, expected_to: intParam, event_type: csvParam, type_: strParam.optional(), | |
| 16 | + since: strParam, until: strParam, min_significance: intParam, | |
| 17 | +}); | |
| 18 | + | |
| 19 | +export async function downloadRoutes(app: FastifyInstance): Promise<void> { | |
| 20 | + publicGet(app, { url: "/download/datasets", ttl: TTL.list, tags: ["download"], summary: "Downloadable datasets (DownloadDataset[]): formats, accepted filters, live row counts, licence and excluded (non-redistributable) sources", response: { type: "DownloadDataset[]", example: [{ key: "facilities", label: "Facilities", description: "…", formats: ["csv", "json", "geojson"], filters: ["country", "status"], rows: 8469, license: "…", attribution: [], excludedSources: [] }] } }, async () => { | |
| 21 | + const data = await listDatasets(); | |
| 22 | + return { data, meta: { total: data.length, maxRows: MAX_ROWS, methodology: DOWNLOAD_LICENSE } }; | |
| 23 | + }); | |
| 24 | + | |
| 25 | + registerRoute({ method: "GET", path: "/api/v1/download/:key", summary: `Stream a dataset (${DOWNLOAD_KEYS.join(" | ")}) as CSV, JSON or GeoJSON (≤ ${MAX_ROWS} rows, license-gated, Content-Disposition attachment; no envelope)`, group: "download", params: [{ name: "key", in: "path", type: "string", description: DOWNLOAD_KEYS.join(" | "), example: "facilities" }, { name: "format", in: "query", type: "csv|json|geojson", description: "output format (default csv)", example: "csv" }, { name: "country", in: "query", type: "string", description: "ISO-3166 alpha-2 (comma list)", example: "US" }], responseType: "text/csv | { data: rows[], meta } | GeoJSON FeatureCollection" }); | |
| 26 | + app.get("/download/:key", { schema: { summary: `Stream a dataset (${DOWNLOAD_KEYS.join(" | ")}) as csv | json | geojson — ≤ ${MAX_ROWS} rows, license-gated (rows whose only sources forbid redistribution are excluded and listed in x-dci-excluded-sources / meta.excludedSources), no envelope`, tags: ["download"], params: { type: "object", properties: { key: { type: "string", enum: [...DOWNLOAD_KEYS] } } }, querystring: { type: "object", properties: { format: { type: "string", enum: ["csv", "json", "geojson"] }, country: { type: "string" }, status: { type: "string" }, operator: { type: "string" }, since: { type: "string" } } }, response: { 200: { description: "Dataset rows (CSV text, JSON { data, meta } or GeoJSON FeatureCollection)", type: "string" } } } }, async (req, reply) => { | |
| 27 | + const { key } = req.params as { key: string }; | |
| 28 | + const spec = specFor(key); | |
| 29 | + if (!spec) throw new HttpError(404, "dataset not found"); | |
| 30 | + const q = parseQuery(downloadQuery, req.query); | |
| 31 | + const format: DownloadFormat = q.format ?? "csv"; | |
| 32 | + if (!spec.formats.includes(format)) throw new HttpError(400, `format ${format} is not available for ${key} (${spec.formats.join(", ")})`); | |
| 33 | + const excluded = spec.gated ? await restrictedSources() : []; | |
| 34 | + const filters = { country: q.country, status: q.status, project_status: q.project_status, type: q.type, operator: q.operator, metro: q.metro, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, expected_from: q.expected_from, expected_to: q.expected_to, event_type: q.event_type, since: q.since, until: q.until, min_significance: q.min_significance }; | |
| 35 | + const day = new Date().toISOString().slice(0, 10); | |
| 36 | + reply.header("content-type", contentType(format)); | |
| 37 | + reply.header("content-disposition", `attachment; filename="dci-${key}-${day}.${format}"`); | |
| 38 | + reply.header("cache-control", "public, max-age=300"); | |
| 39 | + reply.header("x-dci-license", "attribution required; see /api/v1/download/datasets"); | |
| 40 | + reply.header("x-dci-excluded-sources", excluded.map((s) => s.id).join(",") || "none"); | |
| 41 | + reply.header("x-dci-max-rows", String(MAX_ROWS)); | |
| 42 | + return reply.send(Readable.from(streamDataset(spec.key as DownloadKey, format, filters, excluded))); | |
| 43 | + }); | |
| 6 | 44 | } |
modified
apps/api/src/routes/public/facilities.ts
+33 −9
@@ -1,19 +1,43 @@ | ||
| 1 | 1 | import type { FastifyInstance } from "fastify"; |
| 2 | +import { z } from "zod"; | |
| 2 | 3 | import { publicGet } from "../../lib/route.js"; |
| 3 | −import { TTL, notFound } from "../../lib/http.js"; | |
| 4 | −import { facilityFiltersSchema } from "../../lib/params.js"; | |
| 5 | −import { METHODOLOGY_MW } from "../../lib/sql.js"; | |
| 6 | −import { getFacilityDetail, listFacilities, toFacilityQuery } from "../../repositories/facilities.js"; | |
| 4 | +import { TTL, csv, notFound } from "../../lib/http.js"; | |
| 5 | +import { facilityFiltersSchema, csvParam, numParam, strParam } from "../../lib/params.js"; | |
| 6 | +import { METHODOLOGY_MW, METHODOLOGY_CONTAINMENT } from "../../lib/sql.js"; | |
| 7 | +import { sourcesForEntities } from "../../lib/source-history.js"; | |
| 8 | +import { getFacilityClaims, getFacilityDetail, getFacilityHistory, getFacilityProvenance, listFacilities, toFacilityQuery } from "../../repositories/facilities.js"; | |
| 9 | + | |
| 10 | +const LIST_NOTE = `${METHODOLOGY_MW} ${METHODOLOGY_CONTAINMENT} List rows may show both a campus (recordScope=campus) and its buildings (parentFacility set) — aggregates never count both.`; | |
| 11 | +const radiusQuery = z.object({ radius_km: numParam }); | |
| 12 | +const claimsQuery = z.object({ status: csvParam, predicate: strParam }); | |
| 7 | 13 | |
| 8 | 14 | export async function facilityRoutes(app: FastifyInstance): Promise<void> { |
| 9 | − publicGet(app, { url: "/datacenters", ttl: TTL.list, query: facilityFiltersSchema, summary: "List facilities (FacilitySummary[]) with filters, sorting and pagination", tags: ["facilities"] }, async (q) => { | |
| 15 | + publicGet(app, { url: "/datacenters", ttl: TTL.list, query: facilityFiltersSchema, summary: "List facilities (FacilitySummary[]) with filters, sorting and pagination", tags: ["facilities"], response: { type: "FacilitySummary[]", description: "Facilities matching the filters; `sources` = distinct sources behind the returned rows (≤ 30).", example: [{ id: "fac_…", slug: "example-dc1", name: "Example DC1", status: "operational", itCapacityMw: 12.5, recordScope: "facility", aiEvidence: "unknown" }] } }, async (q) => { | |
| 10 | 16 | const res = await listFacilities(toFacilityQuery(q)); |
| 11 | − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: METHODOLOGY_MW } }; | |
| 17 | + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: LIST_NOTE }, sources: await sourcesForEntities("facility", res.items.map((f) => f.id)) }; | |
| 18 | + }); | |
| 19 | + | |
| 20 | + publicGet(app, { url: "/datacenters/:idOrSlug", ttl: TTL.detail, query: radiusQuery, summary: "Facility detail by slug or id (FacilityDetail): buildings, tenants, claims, capacity history, nearby infrastructure (?radius_km ≤ 200, default 25), power context, data quality", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "FacilityDetail" } }, async (q, params) => { | |
| 21 | + const res = await getFacilityDetail(params.idOrSlug!, { radiusKm: q.radius_km != null ? Math.min(200, Math.max(0.1, q.radius_km)) : undefined }); | |
| 22 | + if (!res) throw notFound("facility"); | |
| 23 | + return { data: res.detail, sources: res.sources, meta: { methodology: `${METHODOLOGY_MW} Claims list every figure published about this site with its scope; only site-scoped, non-rejected claims may back the displayed values (isWinner).` } }; | |
| 24 | + }); | |
| 25 | + | |
| 26 | + publicGet(app, { url: "/datacenters/:idOrSlug/history", ttl: TTL.detail, summary: "Field history (EntityHistory): dated observations, claims and value changes per field", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "EntityHistory", example: { entityType: "facility", entityId: "fac_…", fields: { itCapacityMw: [{ date: "2025-03-01T00:00:00.000Z", field: "itCapacityMw", value: 12.5, kind: "observed", sourceId: "src_…", sourceName: "Operator site", sourceKind: "operator", url: "https://…" }] }, changes: [] } } }, async (_q, params) => { | |
| 27 | + const res = await getFacilityHistory(params.idOrSlug!); | |
| 28 | + if (!res) throw notFound("facility"); | |
| 29 | + return { data: res.history, sources: res.sources, meta: { methodology: "observed = a source page stated the value (first observation date); claim = a figure asserted by a document (published date when known); changed = a detected value change with old → new. Dates are ISO or partial (YYYY, YYYY-MM)." } }; | |
| 30 | + }); | |
| 31 | + | |
| 32 | + publicGet(app, { url: "/datacenters/:idOrSlug/claims", ttl: TTL.detail, query: claimsQuery, summary: "Claims about a facility (ClaimDTO[]) — every published figure with scope, evidence sentence, authority tier and status (?status=current,unscoped&predicate=)", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "ClaimDTO[]" } }, async (q, params) => { | |
| 33 | + const res = await getFacilityClaims(params.idOrSlug!, { status: csv(q.status), predicate: q.predicate }); | |
| 34 | + if (!res) throw notFound("facility"); | |
| 35 | + return { data: res.claims, sources: res.sources, meta: { total: res.claims.length, facilityId: res.id, methodology: "A claim is one figure asserted by one document about this site. Scope building/facility/campus may back the displayed value (isWinner); portfolio/company/country/metro/unknown scopes are kept as claims only (status unscoped)." } }; | |
| 12 | 36 | }); |
| 13 | 37 | |
| 14 | − publicGet(app, { url: "/datacenters/:idOrSlug", ttl: TTL.detail, summary: "Facility detail by slug or id (FacilityDetail)", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" } }, async (_q, params) => { | |
| 15 | − const res = await getFacilityDetail(params.idOrSlug!); | |
| 38 | + publicGet(app, { url: "/datacenters/:idOrSlug/provenance", ttl: TTL.detail, summary: "All provenance observations for a facility (ProvenanceDTO[]), current and superseded, with winner flag", tags: ["facilities"], params: { idOrSlug: "facility slug or fac_… id" }, response: { type: "ProvenanceDTO[]" } }, async (_q, params) => { | |
| 39 | + const res = await getFacilityProvenance(params.idOrSlug!); | |
| 16 | 40 | if (!res) throw notFound("facility"); |
| 17 | − return { data: res.detail, sources: res.sources, meta: { methodology: METHODOLOGY_MW } }; | |
| 41 | + return { data: res.provenance, sources: res.sources, meta: { total: res.provenance.length, current: res.current, superseded: res.provenance.length - res.current, facilityId: res.id } }; | |
| 18 | 42 | }); |
| 19 | 43 | } |
modified
apps/api/src/routes/public/graph.ts
+21 −19
@@ -4,7 +4,7 @@ import { z } from "zod"; | ||
| 4 | 4 | import { publicGet } from "../../lib/route.js"; |
| 5 | 5 | import { TTL, notFound } from "../../lib/http.js"; |
| 6 | 6 | import { boolParam, intParam, orderParam, pageParam, strParam } from "../../lib/params.js"; |
| 7 | −import { METHODOLOGY_MW } from "../../lib/sql.js"; | |
| 7 | +import { METHODOLOGY_MW, METHODOLOGY_CONTAINMENT } from "../../lib/sql.js"; | |
| 8 | 8 | import { getOperatorDetail, listOperators } from "../../repositories/operators.js"; |
| 9 | 9 | import { getCountryDetail, listCountries } from "../../repositories/countries.js"; |
| 10 | 10 | import { getMetroDetail, listMetros } from "../../repositories/metros.js"; |
@@ -13,61 +13,63 @@ import { getIxp, listIxps } from "../../repositories/ixps.js"; | ||
| 13 | 13 | |
| 14 | 14 | const empty = (v: unknown) => (v === "" || v === null ? undefined : v); |
| 15 | 15 | const fPage = z.object({ fPage: pageParam }); |
| 16 | +const METHODOLOGY = `${METHODOLOGY_MW} ${METHODOLOGY_CONTAINMENT} mwCoverage (0..1) is the share of counted facilities with a published MW figure — compare MW totals only when coverage is comparable.`; | |
| 17 | +const HHI_NOTE = "concentration.hhi = Σ (share × 100)² over operators (0–10 000), computed on facility counts and separately on known MW with its coverage; momentum components are shown separately and never collapsed into a score."; | |
| 16 | 18 | |
| 17 | 19 | export async function graphRoutes(app: FastifyInstance): Promise<void> { |
| 18 | 20 | // ---- operators |
| 19 | 21 | const operatorsQuery = z.object({ q: strParam, kind: strParam, country: strParam, sort: z.preprocess(empty, z.enum(["facilities", "name", "mw"]).optional()), order: orderParam, page: pageParam, per_page: intParam }); |
| 20 | − publicGet(app, { url: "/operators", ttl: TTL.list, query: operatorsQuery, summary: "List operators (OperatorSummary[]) with live facility aggregates", tags: ["operators"] }, async (q) => { | |
| 22 | + publicGet(app, { url: "/operators", ttl: TTL.list, query: operatorsQuery, summary: "List operators (OperatorSummary[]) with containment-aware facility aggregates", tags: ["operators"], response: { type: "OperatorSummary[]", example: [{ id: "op_x", slug: "equinix", name: "Equinix", kind: "colocation", facilityCount: 262, countryCount: 33, metroCount: 71, knownMw: 1180.5, plannedMw: null, constructionMw: 60, projectCount: 4, aiCount: 0, cloudRegionCount: 0, mwCoverage: 0.61, isCloudProvider: false, isCarrier: false }] } }, async (q) => { | |
| 21 | 23 | const res = await listOperators(q); |
| 22 | − return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: METHODOLOGY_MW } }; | |
| 24 | + return { data: res.items, meta: { total: res.total, page: res.page, perPage: res.perPage, methodology: METHODOLOGY } }; | |
| 23 | 25 | }); |
| 24 | − publicGet(app, { url: "/operators/:slug", ttl: TTL.detail, query: fPage, summary: "Operator detail (OperatorDetail); facilities paginated with ?fPage (50 per page)", tags: ["operators"], params: { slug: "operator slug or op_… id" } }, async (q, params) => { | |
| 26 | + publicGet(app, { url: "/operators/:slug", ttl: TTL.detail, query: fPage, summary: "Operator detail (OperatorDetail): pipeline, expansion velocity, top countries / metros, AI facilities, cloud regions, corporate events, claims, data quality; facilities paginated with ?fPage (50 per page)", tags: ["operators"], params: { slug: "operator slug or op_… id" }, response: { type: "OperatorDetail", description: "OperatorSummary + pipeline (PipelineBreakdown), velocity (12m / 3y / 5y windows, countriesOverTime with basis), topCountries / topMetros with share, aiFacilities, cloudRegions, corporateEvents, claims, dataQuality and the legacy lists." } }, async (q, params) => { | |
| 25 | 27 | const res = await getOperatorDetail(params.slug!, q.fPage); |
| 26 | 28 | if (!res) throw notFound("operator"); |
| 27 | − return { data: res.detail, sources: res.sources, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: res.detail.facilityCount } }; | |
| 29 | + return { data: res.detail, sources: res.sources, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: res.detail.facilityCount, methodology: `${METHODOLOGY} Velocity windows use opened_on (else first indexed) for facilities and announced_on (else first indexed) for projects — the basis is labelled per point. Corporate events (acquisitions, financing, partnerships, executive changes) are never mixed with physical projects.` } }; | |
| 28 | 30 | }); |
| 29 | 31 | |
| 30 | 32 | // ---- countries |
| 31 | − publicGet(app, { url: "/countries", ttl: TTL.list, query: z.object({ all: boolParam, region: strParam }), summary: "Countries with ≥1 facility or cloud region (CountrySummary[]); ?all=1 for every country", tags: ["countries"] }, async (q) => { | |
| 33 | + publicGet(app, { url: "/countries", ttl: TTL.list, query: z.object({ all: boolParam, region: strParam }), summary: "Countries with ≥1 facility or cloud region (CountrySummary[]); ?all=1 for every country", tags: ["countries"], response: { type: "CountrySummary[]", example: [{ iso2: "US", iso3: "USA", slug: "united-states", name: "United States", facilityCount: 3200, operationalCount: 2900, constructionCount: 60, plannedCount: 110, knownMw: 9800, projectCount: 120, projectPlannedMw: 14000, ixpCount: 140, mwCoverage: 0.22 }] } }, async (q) => { | |
| 32 | 34 | const items = await listCountries({ all: q.all === true, region: q.region }); |
| 33 | − return { data: items, meta: { total: items.length, methodology: METHODOLOGY_MW } }; | |
| 35 | + return { data: items, meta: { total: items.length, methodology: METHODOLOGY } }; | |
| 34 | 36 | }); |
| 35 | − publicGet(app, { url: "/countries/:slugOrIso2", ttl: TTL.detail, query: fPage, summary: "Country detail (CountryDetail); facilities paginated with ?fPage", tags: ["countries"], params: { slugOrIso2: "country slug, ISO-3166 alpha-2 or alpha-3" } }, async (q, params) => { | |
| 37 | + publicGet(app, { url: "/countries/:slugOrIso2", ttl: TTL.detail, query: fPage, summary: "Country detail (CountryDetail): IXPs, grid constraints, energy context, AI facilities / projects, pipeline, coverage, claims; facilities paginated with ?fPage", tags: ["countries"], params: { slugOrIso2: "country slug, ISO-3166 alpha-2 or alpha-3" }, response: { type: "CountryDetail", description: "CountrySummary + ixps, gridConstraints, energy (national grid averages with a note — they do not describe a facility's contracted electricity), aiFacilities, aiProjects, pipeline, coverage (CoverageRow), claims and the legacy lists." } }, async (q, params) => { | |
| 36 | 38 | const d = await getCountryDetail(params.slugOrIso2!, q.fPage); |
| 37 | 39 | if (!d) throw notFound("country"); |
| 38 | − return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: METHODOLOGY_MW } }; | |
| 40 | + return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: `${METHODOLOGY} announcedInvestmentUsd sums site-scoped project investments only (company capex, deal values and national programmes are kept as claims).` } }; | |
| 39 | 41 | }); |
| 40 | 42 | |
| 41 | 43 | // ---- metros |
| 42 | − publicGet(app, { url: "/metros", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Metros / markets with counts (MetroSummary[])", tags: ["metros"] }, async (q) => { | |
| 44 | + publicGet(app, { url: "/metros", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Metros / markets with containment-aware counts (MetroSummary[])", tags: ["metros"], response: { type: "MetroSummary[]", example: [{ id: "met_x", slug: "northern-virginia", name: "Northern Virginia", countryIso2: "US", lat: 39.04, lng: -77.49, facilityCount: 310, operationalCount: 280, constructionCount: 20, plannedCount: 10, knownMw: 4200, operatorCount: 40, cloudRegionCount: 6, ixpCount: 4, projectCount: 25, projectPlannedMw: 5600, aiCount: 12, mwCoverage: 0.48 }] } }, async (q) => { | |
| 43 | 45 | const items = await listMetros({ country: q.country, q: q.q }); |
| 44 | − return { data: items, meta: { total: items.length, methodology: METHODOLOGY_MW } }; | |
| 46 | + return { data: items, meta: { total: items.length, methodology: METHODOLOGY } }; | |
| 45 | 47 | }); |
| 46 | − publicGet(app, { url: "/metros/:slug", ttl: TTL.detail, query: fPage, summary: "Metro detail (MetroDetail); facilities paginated with ?fPage", tags: ["metros"], params: { slug: "metro slug or met_… id" } }, async (q, params) => { | |
| 48 | + publicGet(app, { url: "/metros/:slug", ttl: TTL.detail, query: fPage, summary: "Metro detail (MetroDetail): concentration (HHI), 12-month momentum components, pipeline, grid constraints, AI facilities, opening timeline, coverage, claims; facilities paginated with ?fPage", tags: ["metros"], params: { slug: "metro slug or met_… id" }, response: { type: "MetroDetail", description: "MetroSummary + concentration (MarketConcentration), momentum (MarketMomentum, window 12m), pipeline, gridConstraints (grid_constraints rows + grid / utility / power events), aiFacilities, openingTimeline, coverage, claims and the legacy lists." } }, async (q, params) => { | |
| 47 | 49 | const d = await getMetroDetail(params.slug!, q.fPage); |
| 48 | 50 | if (!d) throw notFound("metro"); |
| 49 | − return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: METHODOLOGY_MW } }; | |
| 51 | + return { data: d, meta: { facilitiesPage: q.fPage, facilitiesPerPage: 50, facilitiesTotal: d.facilityCount, methodology: `${METHODOLOGY} ${HHI_NOTE}` } }; | |
| 50 | 52 | }); |
| 51 | 53 | |
| 52 | 54 | // ---- cloud regions |
| 53 | − publicGet(app, { url: "/cloud-regions", ttl: TTL.list, query: z.object({ provider: strParam, country: strParam, status: strParam }), summary: "Cloud regions (CloudRegionSummary[])", tags: ["cloud-regions"] }, async (q) => { | |
| 55 | + publicGet(app, { url: "/cloud-regions", ttl: TTL.list, query: z.object({ provider: strParam, country: strParam, status: strParam }), summary: "Cloud regions (CloudRegionSummary[])", tags: ["cloud-regions"], response: { type: "CloudRegionSummary[]" } }, async (q) => { | |
| 54 | 56 | const items = await listCloudRegions({ provider: q.provider, country: q.country, status: q.status }); |
| 55 | 57 | return { data: items, meta: { total: items.length } }; |
| 56 | 58 | }); |
| 57 | − publicGet(app, { url: "/cloud-regions/:slug", ttl: TTL.detail, summary: "Cloud region detail (CloudRegionSummary + metro, siblings, facilities in metro)", tags: ["cloud-regions"], params: { slug: "cloud region slug or cr_… id" } }, async (_q, params) => { | |
| 59 | + publicGet(app, { url: "/cloud-regions/:slug", ttl: TTL.detail, summary: "Cloud region detail (CloudRegionDetail): host facilities (publicly verified only) or market facilities, siblings, events, provenance", tags: ["cloud-regions"], params: { slug: "cloud region slug or cr_… id" }, response: { type: "CloudRegionDetail" } }, async (_q, params) => { | |
| 58 | 60 | const d = await getCloudRegion(params.slug!); |
| 59 | 61 | if (!d) throw notFound("cloud region"); |
| 60 | − return { data: d }; | |
| 62 | + return { data: d, meta: { methodology: "hostFacilities lists only facilities with an explicit public tenancy link to the provider; otherwise the region is associated with its market (marketFacilities) and never pinned to a building." } }; | |
| 61 | 63 | }); |
| 62 | 64 | |
| 63 | 65 | // ---- ixps |
| 64 | − publicGet(app, { url: "/ixps", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Internet exchange points (IxpSummary[])", tags: ["ixps"] }, async (q) => { | |
| 66 | + publicGet(app, { url: "/ixps", ttl: TTL.list, query: z.object({ country: strParam, q: strParam }), summary: "Internet exchange points (IxpSummary[])", tags: ["ixps"], response: { type: "IxpSummary[]" } }, async (q) => { | |
| 65 | 67 | const items = await listIxps({ country: q.country, q: q.q }); |
| 66 | 68 | return { data: items, meta: { total: items.length } }; |
| 67 | 69 | }); |
| 68 | − publicGet(app, { url: "/ixps/:slug", ttl: TTL.detail, summary: "IXP detail (IxpSummary + facilities)", tags: ["ixps"], params: { slug: "ixp slug or ix_… id" } }, async (_q, params) => { | |
| 70 | + publicGet(app, { url: "/ixps/:slug", ttl: TTL.detail, summary: "IXP detail (IxpDetail): facilities, operators, nearby facilities (metro coordinates), source history", tags: ["ixps"], params: { slug: "ixp slug or ix_… id" }, response: { type: "IxpDetail" } }, async (_q, params) => { | |
| 69 | 71 | const d = await getIxp(params.slug!); |
| 70 | 72 | if (!d) throw notFound("ixp"); |
| 71 | − return { data: d }; | |
| 73 | + return { data: d, meta: { methodology: "IXPs carry no published coordinates; nearby facilities are computed from the metro reference point when the IXP is assigned to a metro." } }; | |
| 72 | 74 | }); |
| 73 | 75 | } |
modified
apps/api/src/routes/public/intelligence.ts
+76 −2
@@ -1,6 +1,80 @@ | ||
| 1 | 1 | /** Nearby, explore, AI index, power, connectivity, time machine. */ |
| 2 | 2 | import type { FastifyInstance } from "fastify"; |
| 3 | +import { z } from "zod"; | |
| 4 | +import { publicGet } from "../../lib/route.js"; | |
| 5 | +import { TTL, badRequest } from "../../lib/http.js"; | |
| 6 | +import { boolParam, csvParam, intParam, numParam, orderParam, pageParam, strParam } from "../../lib/params.js"; | |
| 7 | +import { METHODOLOGY_CONTAINMENT } from "../../lib/sql.js"; | |
| 8 | +import { nearby, parseNearbyTypes, NEARBY_MAX_RADIUS_KM, NEARBY_TYPES } from "../../lib/nearby.js"; | |
| 9 | +import { explore, EXPLORE_METHODOLOGY } from "../../repositories/explore.js"; | |
| 10 | +import { aiIndex, AI_EVIDENCE_NOTE } from "../../repositories/ai.js"; | |
| 11 | +import { powerOverview, POWER_NOTE } from "../../repositories/power.js"; | |
| 12 | +import { connectivityOverview, CONNECTIVITY_NOTE } from "../../repositories/connectivity.js"; | |
| 13 | +import { timeMachine, TIME_MACHINE_NOTE } from "../../repositories/time-machine.js"; | |
| 3 | 14 | |
| 4 | −export async function intelligenceRoutes(_app: FastifyInstance): Promise<void> { | |
| 5 | − // filled in by the intelligence implementation | |
| 15 | +const empty = (v: unknown) => (v === "" || v === null ? undefined : v); | |
| 16 | + | |
| 17 | +const nearbyQuery = z.object({ lat: numParam.describe("latitude (-90..90)"), lng: numParam.describe("longitude (-180..180)"), radius_km: numParam.describe(`radius in km (default 25, max ${NEARBY_MAX_RADIUS_KM})`), types: csvParam.describe(`comma list of ${NEARBY_TYPES.join("|")} (default all)`) }); | |
| 18 | + | |
| 19 | +export const exploreQuery = z.object({ | |
| 20 | + entity: z.preprocess(empty, z.enum(["facilities", "projects"]).optional()), | |
| 21 | + status: csvParam, | |
| 22 | + project_status: csvParam, | |
| 23 | + country: csvParam, | |
| 24 | + metro: strParam, | |
| 25 | + operator: strParam, | |
| 26 | + type: csvParam, | |
| 27 | + min_mw: numParam, | |
| 28 | + max_mw: numParam, | |
| 29 | + ai: z.preprocess(empty, z.enum(["confirmed", "likely", "associated", "any"]).optional()), | |
| 30 | + hyperscale: boolParam, | |
| 31 | + expected_before: intParam, | |
| 32 | + expected_after: intParam, | |
| 33 | + opened_from: intParam, | |
| 34 | + opened_to: intParam, | |
| 35 | + announced_since: strParam, | |
| 36 | + confidence: csvParam, | |
| 37 | + location_precision: csvParam, | |
| 38 | + has_mw: boolParam, | |
| 39 | + project_class: csvParam, | |
| 40 | + q: strParam, | |
| 41 | + sort: z.preprocess(empty, z.enum(["name", "mw", "updated", "opened", "completeness", "announced", "opening"]).optional()), | |
| 42 | + order: orderParam, | |
| 43 | + page: pageParam, | |
| 44 | + per_page: intParam, | |
| 45 | + view: z.preprocess(empty, z.enum(["table", "map", "charts"]).optional()), | |
| 46 | +}).strict(); | |
| 47 | + | |
| 48 | +export async function intelligenceRoutes(app: FastifyInstance): Promise<void> { | |
| 49 | + publicGet(app, { url: "/nearby", ttl: TTL.detail, query: nearbyQuery, tags: ["nearby"], summary: "Infrastructure around a point (NearbyInfrastructure): facilities, live projects, IXPs, cloud regions, metros within radius_km", description: "Great-circle distances on a bbox pre-filter. Layers without a connector (landing stations, substations, power plants) are empty with a note.", response: { type: "NearbyInfrastructure", example: { center: { lat: 39.04, lng: -77.49 }, radiusKm: 25, facilities: [], projects: [], ixps: [], cloudRegions: [], metros: [], landingStations: [], substations: [], powerPlants: [], note: "…" } } }, async (q) => { | |
| 50 | + if (q.lat == null || q.lng == null || Math.abs(q.lat) > 90 || Math.abs(q.lng) > 180) throw badRequest("lat and lng are required (lat -90..90, lng -180..180)"); | |
| 51 | + if (q.radius_km != null && (q.radius_km <= 0 || q.radius_km > NEARBY_MAX_RADIUS_KM)) throw badRequest(`radius_km must be between 0 and ${NEARBY_MAX_RADIUS_KM}`); | |
| 52 | + const data = await nearby({ lat: q.lat, lng: q.lng, radiusKm: q.radius_km, types: parseNearbyTypes(q.types) }); | |
| 53 | + return { data, meta: { total: data.facilities.length + data.projects.length + data.ixps.length + data.cloudRegions.length + data.metros.length, methodology: data.note } }; | |
| 54 | + }); | |
| 55 | + | |
| 56 | + publicGet(app, { url: "/explore", ttl: TTL.list, query: exploreQuery, tags: ["explore"], summary: "Structured explorer (ExploreResponse): facilities or projects with facets, charts and map points; unknown query keys are rejected (400)", response: { type: "ExploreResponse", example: { query: { entity: "facilities", country: "US" }, total: 0, items: [], facets: { status: [], country: [], operator: [], type: [], ai: [] }, charts: { byStatus: [], byCountry: [], byYear: [] }, map: { points: [], total: 0, degraded: false }, mwCoverage: 0 } } }, async (q) => { | |
| 57 | + const data = await explore(q); | |
| 58 | + return { data, meta: { total: data.total, page: q.page, perPage: q.per_page ?? 24, mwCoverage: data.mwCoverage, methodology: `${EXPLORE_METHODOLOGY} ${METHODOLOGY_CONTAINMENT}` } }; | |
| 59 | + }); | |
| 60 | + | |
| 61 | + publicGet(app, { url: "/ai-infrastructure", ttl: TTL.detail, tags: ["ai"], summary: "AI / HPC infrastructure index (AiIndex): confirmed + likely facilities and live projects, top operators / metros / countries, pipeline by stage", response: { type: "AiIndex", example: { stats: { facilities: 0, confirmed: 0, likely: 0, associated: 0, projects: 0, plannedMw: null, constructionMw: null, countries: 0, operators: 0, mwCoverage: 0 }, topOperators: [], topMetros: [], topCountries: [], pipelineByStage: [], recentAnnouncements: [], recentProjects: [], facilities: [], evidenceNote: "…" } } }, async () => { | |
| 62 | + const data = await aiIndex(); | |
| 63 | + return { data, meta: { total: data.stats.facilities, mwCoverage: data.stats.mwCoverage, methodology: AI_EVIDENCE_NOTE } }; | |
| 64 | + }); | |
| 65 | + | |
| 66 | + publicGet(app, { url: "/power", ttl: TTL.detail, tags: ["power"], summary: "Power overview (PowerOverview): grid constraints, power / grid events, large loads (utility, grid, planned ≥ 200 MW), national energy context, utilities", response: { type: "PowerOverview", example: { gridConstraints: [], powerEvents: [], largeLoads: [], countryEnergy: [], utilities: [], note: "…" } } }, async () => { | |
| 67 | + const data = await powerOverview(); | |
| 68 | + return { data, meta: { total: data.largeLoads.length, methodology: POWER_NOTE } }; | |
| 69 | + }); | |
| 70 | + | |
| 71 | + publicGet(app, { url: "/connectivity", ttl: TTL.detail, tags: ["connectivity"], summary: "Connectivity overview (ConnectivityOverview): IXPs, cloud regions, carrier hotels, per-metro interconnection density", response: { type: "ConnectivityOverview", example: { ixps: [], cloudRegions: [], carrierHotels: [], landingStations: [], byMetro: [], note: "…" } } }, async () => { | |
| 72 | + const data = await connectivityOverview(); | |
| 73 | + return { data, meta: { total: data.ixps.length, methodology: CONNECTIVITY_NOTE } }; | |
| 74 | + }); | |
| 75 | + | |
| 76 | + publicGet(app, { url: "/time-machine", ttl: TTL.detail, tags: ["time-machine"], summary: "Yearly frames of the index from published opening dates (TimeMachine) with earliestReliableYear and openingDateCoverage", response: { type: "TimeMachine", example: { frames: [{ year: 2020, facilities: 120, knownMw: 950.5, announced: 12, construction: 4, opened: 9 }], earliestReliableYear: 2005, openingDateCoverage: 0.12, note: "…" } } }, async () => { | |
| 77 | + const data = await timeMachine(); | |
| 78 | + return { data, meta: { total: data.frames.length, openingDateCoverage: data.openingDateCoverage, methodology: TIME_MACHINE_NOTE } }; | |
| 79 | + }); | |
| 6 | 80 | } |
modified
apps/api/src/routes/public/markets.ts
+51 −3
@@ -1,6 +1,54 @@ | ||
| 1 | −/** Pulse, operator comparison, coverage report (owned by the aggregates group). */ | |
| 1 | +/** Pulse, operator comparison, coverage report. */ | |
| 2 | 2 | import type { FastifyInstance } from "fastify"; |
| 3 | +import { z } from "zod"; | |
| 4 | +import { publicGet } from "../../lib/route.js"; | |
| 5 | +import { TTL, badRequest, csv } from "../../lib/http.js"; | |
| 6 | +import { csvParam } from "../../lib/params.js"; | |
| 7 | +import { pulse, PULSE_METHODOLOGY } from "../../repositories/pulse.js"; | |
| 8 | +import { compareOperators, COMPARE_METHODOLOGY } from "../../repositories/compare.js"; | |
| 9 | +import { coverageReport, sourceCoverage, COVERAGE_METHODOLOGY } from "../../repositories/coverage.js"; | |
| 3 | 10 | |
| 4 | −export async function marketRoutes(_app: FastifyInstance): Promise<void> { | |
| 5 | − // filled in by the aggregates implementation | |
| 11 | +const empty = (v: unknown) => (v === "" || v === null ? undefined : v); | |
| 12 | + | |
| 13 | +export async function marketRoutes(app: FastifyInstance): Promise<void> { | |
| 14 | + publicGet(app, { | |
| 15 | + url: "/pulse", ttl: TTL.dashboard, tags: ["pulse"], | |
| 16 | + query: z.object({ window: z.preprocess(empty, z.enum(["24h", "7d", "30d"]).default("24h")) }), | |
| 17 | + summary: "Infrastructure pulse: what measurably changed in the last 24h / 7d / 30d (Pulse)", | |
| 18 | + description: "Deterministic counts (events, new projects, construction starts, openings, new markets per operator, power / grid events) — no scoring.", | |
| 19 | + response: { type: "Pulse", description: "Counts over the window plus the major events and per-country / per-operator breakdowns.", example: { window: "7d", since: "2026-09-05T00:00:00.000Z", newProjects: 12, projectsEnteredConstruction: 3, facilitiesOpened: 2, newlyAnnouncedMw: 850, eventsTotal: 310, operatorsNewMarkets: [{ operator: { id: "op_x", slug: "example", name: "Example" }, market: { id: "met_x", slug: "ashburn", name: "Northern Virginia" }, countryIso2: "US" }] } }, | |
| 20 | + }, async (q) => { | |
| 21 | + const data = await pulse(q.window); | |
| 22 | + return { data, meta: { window: q.window, methodology: PULSE_METHODOLOGY } }; | |
| 23 | + }); | |
| 24 | + | |
| 25 | + publicGet(app, { | |
| 26 | + url: "/compare/operators", ttl: TTL.detail, tags: ["compare"], | |
| 27 | + query: z.object({ slugs: csvParam }), | |
| 28 | + summary: "Compare 2–5 operators side by side (OperatorComparison)", | |
| 29 | + response: { type: "OperatorComparison", description: "One column per operator: summary, pipeline, velocity windows, new markets (12 m), recent projects, MW coverage.", example: { operators: [{ id: "op_x", slug: "equinix", name: "Equinix", facilityCount: 260, countryCount: 33, knownMw: 1200, mwCoverage: 0.62, pipeline: { operational: { count: 240, mw: 1200 } } }], generatedAt: "2026-09-12T00:00:00.000Z" } }, | |
| 30 | + }, async (q) => { | |
| 31 | + const slugs = [...new Set(csv(q.slugs))]; | |
| 32 | + if (slugs.length < 2 || slugs.length > 5) throw badRequest("slugs must list 2 to 5 operator slugs (comma-separated)"); | |
| 33 | + const data = await compareOperators(slugs); | |
| 34 | + return { data, meta: { methodology: COMPARE_METHODOLOGY } }; | |
| 35 | + }); | |
| 36 | + | |
| 37 | + publicGet(app, { | |
| 38 | + url: "/coverage", ttl: TTL.detail, tags: ["coverage"], | |
| 39 | + summary: "Coverage report: share of known fields per country / metro / operator (≥ 5 facilities) / field / source kind (CoverageReport)", | |
| 40 | + response: { type: "CoverageReport", description: "Containment-aware shares 0..1; a low share means unknown, not zero.", example: { global: { key: "global", name: "Global", slug: "global", facilities: 8400, capacityCoverage: 0.18, operatorCoverage: 0.9, preciseLocationCoverage: 0.55 }, fields: [{ field: "it_capacity_mw", label: "IT capacity (MW)", coverage: 0.12, count: 1000 }] } }, | |
| 41 | + }, async () => { | |
| 42 | + const data = await coverageReport(); | |
| 43 | + return { data, meta: { methodology: COVERAGE_METHODOLOGY } }; | |
| 44 | + }); | |
| 45 | + | |
| 46 | + publicGet(app, { | |
| 47 | + url: "/coverage/sources", ttl: TTL.detail, tags: ["coverage"], | |
| 48 | + summary: "Per-source contribution, freshness, authority tier, failure rate and licence (SourceCoverage[])", | |
| 49 | + response: { type: "SourceCoverage[]", description: "One row per source; uniqueRecords = entities only this source documents.", example: [{ id: "src_x", name: "Equinix", kind: "operator", recordsContributed: 262, fieldsContributed: 3100, uniqueRecords: 40, lastSuccessfulCrawl: "2026-09-11T02:00:00.000Z", freshnessDays: 1, authority: "A", failureRate: 0, license: "Terms of use", redistribution: "attribution" }] }, | |
| 50 | + }, async () => { | |
| 51 | + const items = await sourceCoverage(); | |
| 52 | + return { data: items, meta: { total: items.length, methodology: "recordsContributed / fieldsContributed count current provenance rows; authority is the identity-field tier of the source kind (packages/core claims.ts FIELD_AUTHORITY); failureRate = failed or aborted runs ÷ finished runs over 30 days." } }; | |
| 53 | + }); | |
| 6 | 54 | } |
modified
apps/api/src/routes/public/watchlist.ts
+63 −4
@@ -1,6 +1,65 @@ | ||
| 1 | −/** Private cookie-scoped watchlist (no accounts). */ | |
| 2 | −import type { FastifyInstance } from "fastify"; | |
| 1 | +/** Private cookie-scoped watchlist (no accounts): GET / POST / DELETE /watchlist, GET /watchlist/feed. Never cached. */ | |
| 2 | +import type { FastifyInstance, FastifyReply, FastifyRequest } from "fastify"; | |
| 3 | +import { z } from "zod"; | |
| 4 | +import { getEnv } from "../../env.js"; | |
| 5 | +import { envelope, HttpError, parseBody, parseQuery } from "../../lib/http.js"; | |
| 6 | +import { intParam, pageParam } from "../../lib/params.js"; | |
| 7 | +import { registerRoute } from "../../lib/route.js"; | |
| 8 | +import { addWatch, listWatchlist, newWatchToken, parseCookies, removeWatch, resolveWatchEntity, watchFeed, WATCH_COOKIE, WATCH_ENTITY_TYPES, WATCH_MAX_ITEMS } from "../../repositories/watchlist.js"; | |
| 3 | 9 | |
| 4 | −export async function watchlistRoutes(_app: FastifyInstance): Promise<void> { | |
| 5 | − // filled in by the intelligence implementation | |
| 10 | +const addBody = z.object({ entityType: z.enum(WATCH_ENTITY_TYPES), entityId: z.string().min(1).max(120).optional(), slug: z.string().min(1).max(200).optional() }).strict().refine((b) => b.entityId || b.slug, "entityId or slug is required"); | |
| 11 | + | |
| 12 | +/** Token from the cookie, minting (and setting) a new one when absent. */ | |
| 13 | +function tokenFor(req: FastifyRequest, reply: FastifyReply, create: boolean): string | null { | |
| 14 | + const existing = parseCookies(req.headers.cookie)[WATCH_COOKIE]; | |
| 15 | + if (existing) return existing; | |
| 16 | + if (!create) return null; | |
| 17 | + const token = newWatchToken(); | |
| 18 | + const secure = getEnv().siteUrl.startsWith("https://") ? "; Secure" : ""; | |
| 19 | + reply.header("set-cookie", `${WATCH_COOKIE}=${token}; Path=/; HttpOnly; SameSite=Lax; Max-Age=31536000${secure}`); | |
| 20 | + return token; | |
| 21 | +} | |
| 22 | + | |
| 23 | +const NO_STORE = { "cache-control": "no-store", vary: "cookie" }; | |
| 24 | + | |
| 25 | +export async function watchlistRoutes(app: FastifyInstance): Promise<void> { | |
| 26 | + registerRoute({ method: "GET", path: "/api/v1/watchlist", summary: "Your private watchlist (WatchlistItem[]) — keyed by the httpOnly dci_watch cookie, created on first use; no accounts", group: "watchlist", params: [], responseType: "WatchlistItem[]" }); | |
| 27 | + app.get("/watchlist", { schema: { summary: "Private watchlist (WatchlistItem[]) keyed by the dci_watch cookie", tags: ["watchlist"] } }, async (req, reply) => { | |
| 28 | + reply.headers(NO_STORE); | |
| 29 | + const token = tokenFor(req, reply, true)!; | |
| 30 | + const items = await listWatchlist(token); | |
| 31 | + return envelope(items, { total: items.length, max: WATCH_MAX_ITEMS }); | |
| 32 | + }); | |
| 33 | + | |
| 34 | + registerRoute({ method: "POST", path: "/api/v1/watchlist", summary: "Watch an entity: body { entityType: operator|metro|country|project|facility, entityId | slug } → WatchlistItem (201 when created)", group: "watchlist", params: [{ name: "entityType", in: "query", type: "enum", description: WATCH_ENTITY_TYPES.join(" | "), example: "operator" }, { name: "slug", in: "query", type: "string", description: "entity slug (or entityId)", example: "equinix" }], responseType: "WatchlistItem" }); | |
| 35 | + app.post("/watchlist", { schema: { summary: "Add an entity to the private watchlist { entityType, entityId | slug }", tags: ["watchlist"], body: { type: "object", properties: { entityType: { type: "string", enum: [...WATCH_ENTITY_TYPES] }, entityId: { type: "string" }, slug: { type: "string" } }, required: ["entityType"] } } }, async (req, reply) => { | |
| 36 | + reply.headers(NO_STORE); | |
| 37 | + const body = parseBody(addBody, req.body); | |
| 38 | + const id = await resolveWatchEntity(body.entityType, body.entityId ?? body.slug!); | |
| 39 | + if (!id) throw new HttpError(404, `${body.entityType} not found`); | |
| 40 | + const token = tokenFor(req, reply, true)!; | |
| 41 | + const res = await addWatch(token, body.entityType, id); | |
| 42 | + if ("error" in res) throw new HttpError(409, `watchlist is full (${WATCH_MAX_ITEMS} items)`); | |
| 43 | + reply.code(res.created ? 201 : 200); | |
| 44 | + return envelope(res.item, { created: res.created }); | |
| 45 | + }); | |
| 46 | + | |
| 47 | + registerRoute({ method: "DELETE", path: "/api/v1/watchlist/:id", summary: "Stop watching (only your own rows) → { deleted }", group: "watchlist", params: [{ name: "id", in: "path", type: "string", description: "watchlist item id (wtc_…)" }], responseType: "{ deleted: boolean }" }); | |
| 48 | + app.delete("/watchlist/:id", { schema: { summary: "Remove a watchlist item (owner only)", tags: ["watchlist"], params: { type: "object", properties: { id: { type: "string" } } } } }, async (req, reply) => { | |
| 49 | + reply.headers(NO_STORE); | |
| 50 | + const token = tokenFor(req, reply, false); | |
| 51 | + const { id } = req.params as { id: string }; | |
| 52 | + const deleted = token ? await removeWatch(token, id) : false; | |
| 53 | + return envelope({ id, deleted }); | |
| 54 | + }); | |
| 55 | + | |
| 56 | + registerRoute({ method: "GET", path: "/api/v1/watchlist/feed", summary: "Events for your watched entities (EventDTO[]), newest first, one row per announcement cluster", group: "watchlist", params: [{ name: "page", in: "query", type: "integer", description: "page (default 1)" }, { name: "per_page", in: "query", type: "integer", description: "rows per page (default 50, max 100)" }], responseType: "EventDTO[]" }); | |
| 57 | + app.get("/watchlist/feed", { schema: { summary: "Change feed for the watched entities (EventDTO[])", tags: ["watchlist"], querystring: { type: "object", properties: { page: { type: "integer" }, per_page: { type: "integer" } } } } }, async (req, reply) => { | |
| 58 | + reply.headers(NO_STORE); | |
| 59 | + const q = parseQuery(z.object({ page: pageParam, per_page: intParam }), req.query); | |
| 60 | + const token = tokenFor(req, reply, false); | |
| 61 | + if (!token) return envelope([], { total: 0, page: q.page, perPage: q.per_page ?? 50 }); | |
| 62 | + const res = await watchFeed(token, q.page, q.per_page); | |
| 63 | + return envelope(res.items, { total: res.total, page: res.page, perPage: res.perPage }); | |
| 64 | + }); | |
| 6 | 65 | } |
modified
apps/api/src/search/interpret.test.ts
+43 −0
@@ -24,12 +24,53 @@ describe("interpretQuery", () => { | ||
| 24 | 24 | expect(interpretQuery("at least 50mw germany")).toMatchObject({ minMw: 50, countryIso2: "DE" }); |
| 25 | 25 | }); |
| 26 | 26 | |
| 27 | + it("extracts maximum MW thresholds", () => { | |
| 28 | + expect(interpretQuery("<500 MW")).toMatchObject({ maxMw: 500 }); | |
| 29 | + expect(interpretQuery("under 500 MW announced")).toMatchObject({ maxMw: 500, status: "announced" }); | |
| 30 | + expect(interpretQuery("below 1 GW").maxMw).toBe(1000); | |
| 31 | + expect(interpretQuery("less than 50mw in france")).toMatchObject({ maxMw: 50, countryIso2: "FR" }); | |
| 32 | + expect(interpretQuery("less than 50mw").minMw).toBeUndefined(); | |
| 33 | + expect(interpretQuery(">500 MW announced")).toMatchObject({ minMw: 500, status: "announced" }); | |
| 34 | + expect(interpretQuery("500+ MW").minMw).toBe(500); | |
| 35 | + }); | |
| 36 | + | |
| 27 | 37 | it("extracts status and type words", () => { |
| 28 | 38 | expect(interpretQuery("under construction data centers")).toMatchObject({ status: "under_construction" }); |
| 29 | 39 | expect(interpretQuery("planned AI data center")).toMatchObject({ status: "announced", facilityType: "ai" }); |
| 30 | 40 | expect(interpretQuery("operational colocation in Canada")).toMatchObject({ status: "operational", facilityType: "colocation", countryIso2: "CA" }); |
| 31 | 41 | }); |
| 32 | 42 | |
| 43 | + it("flags AI queries", () => { | |
| 44 | + expect(interpretQuery("ai data centers in texas")).toMatchObject({ ai: true, facilityType: "ai" }); | |
| 45 | + expect(interpretQuery("gpu campus")).toMatchObject({ ai: true }); | |
| 46 | + expect(interpretQuery("colocation in Paris").ai).toBeUndefined(); | |
| 47 | + }); | |
| 48 | + | |
| 49 | + it("switches to projects when the query says projects", () => { | |
| 50 | + const i = interpretQuery("projects over 500 MW in texas"); | |
| 51 | + expect(i).toMatchObject({ entity: "project", minMw: 500 }); | |
| 52 | + expect(i.text).toBe("texas"); | |
| 53 | + expect(interpretQuery("announced data centers").entity).toBeUndefined(); | |
| 54 | + expect(interpretQuery("planned campus in Ohio").entity).toBeUndefined(); | |
| 55 | + }); | |
| 56 | + | |
| 57 | + it("extracts opening year phrases", () => { | |
| 58 | + expect(interpretQuery("projects opening 2028")).toMatchObject({ entity: "project", year: 2028, yearOp: "in" }); | |
| 59 | + expect(interpretQuery("before 2028")).toMatchObject({ year: 2028, yearOp: "before" }); | |
| 60 | + expect(interpretQuery("after 2027 hyperscale")).toMatchObject({ year: 2027, yearOp: "after", facilityType: "hyperscale" }); | |
| 61 | + expect(interpretQuery("opens in 2026").year).toBe(2026); | |
| 62 | + expect(interpretQuery("by 2030")).toMatchObject({ year: 2030, yearOp: "before" }); | |
| 63 | + expect(interpretQuery("500 MW").year).toBeUndefined(); | |
| 64 | + }); | |
| 65 | + | |
| 66 | + it("matches metro names and aliases supplied by the caller", () => { | |
| 67 | + const metros = [{ slug: "northern-virginia", name: "Northern Virginia", aliases: ["NoVA", "Ashburn"] }, { slug: "dallas-fort-worth", name: "Dallas-Fort Worth", aliases: ["DFW"] }]; | |
| 68 | + expect(interpretQuery("hyperscale in Northern Virginia", { metros })).toMatchObject({ metro: "northern-virginia", facilityType: "hyperscale" }); | |
| 69 | + expect(interpretQuery("ashburn colocation", { metros })).toMatchObject({ metro: "northern-virginia" }); | |
| 70 | + expect(interpretQuery("DFW over 50 MW", { metros })).toMatchObject({ metro: "dallas-fort-worth", minMw: 50 }); | |
| 71 | + expect(interpretQuery("virginia", { metros }).metro).toBeUndefined(); | |
| 72 | + }); | |
| 73 | + | |
| 33 | 74 | it("recognises an operator supplied by the caller and keeps the rest as text", () => { |
| 34 | 75 | const i = interpretQuery("Equinix Paris operational", { operators: [{ slug: "equinix", name: "Equinix", similarity: 0.5 }] }); |
| 35 | 76 | expect(i.operator).toBe("equinix"); |
@@ -53,5 +94,7 @@ describe("interpretQuery", () => { | ||
| 53 | 94 | const i = interpretQuery("equinix operational 100 MW in france", { operators: [{ slug: "equinix", name: "Equinix", similarity: 0.4 }] }); |
| 54 | 95 | expect(i).toEqual({ minMw: 100, operator: "equinix", countryIso2: "FR", status: "operational" }); |
| 55 | 96 | expect(hasFilters(i)).toBe(true); |
| 97 | + expect(hasFilters({ entity: "project" })).toBe(true); | |
| 98 | + expect(hasFilters({ year: 2028 })).toBe(true); | |
| 56 | 99 | }); |
| 57 | 100 | }); |
modified
apps/api/src/search/interpret.ts
+54 −14
@@ -1,6 +1,7 @@ | ||
| 1 | 1 | /** |
| 2 | − * Query interpretation for /search: pulls structured filters (country, operator, status, min MW, facility | |
| 3 | − * type) out of free text and returns the remaining text. Pure function — unit-tested. | |
| 2 | + * Query interpretation for /search: pulls structured filters (country, operator, metro, status, min / max MW, | |
| 3 | + * facility type, AI, entity = project, opening year) out of free text and returns the remaining text. | |
| 4 | + * Pure function — unit-tested. | |
| 4 | 5 | */ |
| 5 | 6 | import type { FacilityStatus, FacilityType, SearchResponse } from "@dci/core"; |
| 6 | 7 | import { COUNTRY_ALIASES, normalizeStatus, parseMw } from "@dci/core"; |
@@ -12,6 +13,8 @@ export interface InterpretOptions { | ||
| 12 | 13 | operators?: Array<{ slug: string; name: string; similarity?: number }>; |
| 13 | 14 | /** extra country names from the countries table: [{ iso2, name }] */ |
| 14 | 15 | countries?: Array<{ iso2: string; name: string }>; |
| 16 | + /** metro / market names (and aliases) from the metros table */ | |
| 17 | + metros?: Array<{ slug: string; name: string; aliases?: string[] }>; | |
| 15 | 18 | } |
| 16 | 19 | |
| 17 | 20 | const STATUS_PHRASES: Array<[RegExp, FacilityStatus]> = [ |
@@ -44,7 +47,14 @@ const TYPE_PHRASES: Array<[RegExp, FacilityType]> = [ | ||
| 44 | 47 | [/\b(government)\b/i, "government"], |
| 45 | 48 | ]; |
| 46 | 49 | |
| 47 | −const STOP = new Set(["data", "center", "centers", "centre", "centres", "datacenter", "datacenters", "datacentre", "datacentres", "dc", "facility", "facilities", "in", "at", "near", "the", "of", "and", "with", "over", "above", "more", "than", "least", "min", "minimum", "campus", "campuses", "site", "sites"]); | |
| 50 | +const STOP = new Set(["data", "center", "centers", "centre", "centres", "datacenter", "datacenters", "datacentre", "datacentres", "dc", "facility", "facilities", "in", "at", "near", "the", "of", "and", "with", "over", "above", "more", "than", "least", "min", "minimum", "campus", "campuses", "site", "sites", "under", "below", "less", "max", "maximum", "up", "to", "by", "opening", "opens", "open", "project", "projects"]); | |
| 51 | + | |
| 52 | +const MW_UNIT = String.raw`(\d+(?:[.,]\d+)?\s*\+?\s*(?:gw|gigawatts?|mw|megawatts?))\b`; | |
| 53 | +const MAX_RE = new RegExp(String.raw`(?:<=?|under|below|less than|at most|max(?:imum)?|up to|smaller than|no more than)\s*${MW_UNIT}`, "i"); | |
| 54 | +const MIN_RE = new RegExp(String.raw`(?:>=?|over|above|more than|at least|min(?:imum)?|\+|larger than|bigger than)?\s*${MW_UNIT}`, "i"); | |
| 55 | +const AI_RE = /\b(ai|artificial intelligence|gpu|gpus|ai[- ]ready|ai[- ]factory|ai[- ]factories)\b/i; | |
| 56 | +const PROJECT_RE = /\b(projects?|pipeline projects?)\b/i; | |
| 57 | +const YEAR_RE = /\b(before|by|until|prior to|pre|after|from|since|post|opening|opens|open(?:ed|ing)? in|in|for|due)?\s*((?:19|20)\d{2})\b/i; | |
| 48 | 58 | |
| 49 | 59 | function esc(s: string): string { |
| 50 | 60 | return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); |
@@ -59,15 +69,33 @@ export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interp | ||
| 59 | 69 | let text = raw.replace(/\s+/g, " ").trim(); |
| 60 | 70 | if (!text) return out; |
| 61 | 71 | |
| 62 | − // 1. Power figure: "100 MW", "over 50mw", "> 1 GW", "100+ MW" | |
| 63 | − const mwMatch = text.match(/(?:>=?|over|above|more than|at least|min(?:imum)?|\+)?\s*(\d+(?:[.,]\d+)?\s*\+?\s*(?:gw|gigawatts?|mw|megawatts?))\b/i); | |
| 72 | + // 1. Power figures: "<500 MW" / "under 500 MW" → maxMw; "100 MW", "over 50mw", "> 1 GW", "100+ MW" → minMw | |
| 73 | + const maxMatch = text.match(MAX_RE); | |
| 74 | + if (maxMatch) { | |
| 75 | + const mw = parseMw(maxMatch[1]!.replace("+", "")); | |
| 76 | + if (mw != null && mw > 0) out.maxMw = mw; | |
| 77 | + text = strip(text, new RegExp(esc(maxMatch[0]), "i")); | |
| 78 | + } | |
| 79 | + const mwMatch = text.match(MIN_RE); | |
| 64 | 80 | if (mwMatch) { |
| 65 | 81 | const mw = parseMw(mwMatch[1]!.replace("+", "")); |
| 66 | 82 | if (mw != null && mw > 0) out.minMw = mw; |
| 67 | 83 | text = strip(text, new RegExp(esc(mwMatch[0]), "i")); |
| 68 | 84 | } |
| 69 | 85 | |
| 70 | − // 2. Operator (candidates supplied by the caller from a trigram lookup on the full query) | |
| 86 | + // 2. Opening year phrases: "opening 2028", "before 2028", "after 2027" (a year right after a MW figure was consumed above) | |
| 87 | + const yearMatch = text.match(YEAR_RE); | |
| 88 | + if (yearMatch) { | |
| 89 | + const y = Number(yearMatch[2]); | |
| 90 | + const cue = (yearMatch[1] ?? "").toLowerCase(); | |
| 91 | + if (y >= 1990 && y <= 2060) { | |
| 92 | + out.year = y; | |
| 93 | + out.yearOp = /^(before|by|until|prior to|pre)$/.test(cue) ? "before" : /^(after|from|since|post)$/.test(cue) ? "after" : "in"; | |
| 94 | + text = strip(text, new RegExp(esc(yearMatch[0]), "i")); | |
| 95 | + } | |
| 96 | + } | |
| 97 | + | |
| 98 | + // 3. Operator (candidates supplied by the caller from a trigram lookup on the full query) | |
| 71 | 99 | for (const op of opts.operators ?? []) { |
| 72 | 100 | const re = new RegExp(`(^|[^a-z0-9])${esc(op.name)}([^a-z0-9]|$)`, "i"); |
| 73 | 101 | if (re.test(text) || (op.similarity ?? 0) >= 0.8) { |
@@ -78,11 +106,17 @@ export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interp | ||
| 78 | 106 | } |
| 79 | 107 | } |
| 80 | 108 | |
| 81 | − // 3. Country: bare ISO2 token in caps, known aliases, or the countries table | |
| 82 | − const iso2Token = raw.match(/(^|\s)([A-Z]{2})(\s|$)/); | |
| 83 | − if (iso2Token && !/^(AI|DC|MW|GW|IX|HQ|IT|AZ|US)$/.test(iso2Token[2]!)) { | |
| 84 | − // AZ/IT/US ambiguous in lower-case; treat "US"/"IT" explicitly below via aliases | |
| 109 | + // 4. Metro / market names (longest first, word boundaries, aliases included) | |
| 110 | + if (opts.metros?.length) { | |
| 111 | + const cands = opts.metros.flatMap((m) => [m.name, ...(m.aliases ?? [])].filter((n) => n && n.length >= 3).map((n) => ({ slug: m.slug, n }))).sort((a, b) => b.n.length - a.n.length); | |
| 112 | + for (const c of cands) { | |
| 113 | + const re = new RegExp(`(^|[^\\p{L}\\p{N}])${esc(c.n)}([^\\p{L}\\p{N}]|$)`, "iu"); | |
| 114 | + if (re.test(text)) { out.metro = c.slug; text = strip(text, re); break; } | |
| 115 | + } | |
| 85 | 116 | } |
| 117 | + | |
| 118 | + // 5. Country: bare ISO2 token in caps, known aliases, or the countries table | |
| 119 | + const iso2Token = raw.match(/(^|\s)([A-Z]{2})(\s|$)/); | |
| 86 | 120 | if (iso2Token && /^(US|UK|CA|DE|FR|NL|GB|IE|SG|JP|AU|IN|BR|ES|IT|SE|NO|FI|DK|PL|CH|AT|BE|PT|MX|ZA|AE|SA|KR|CN|HK|TW|MY|ID|TH|VN|PH|NZ|CL|AR|CO|IL|TR|QA|KE|NG|EG)$/.test(iso2Token[2]!)) { |
| 87 | 121 | const code = iso2Token[2]! === "UK" ? "GB" : iso2Token[2]!; |
| 88 | 122 | out.countryIso2 = code; |
@@ -103,22 +137,28 @@ export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interp | ||
| 103 | 137 | } |
| 104 | 138 | } |
| 105 | 139 | |
| 106 | − // 4. Status | |
| 140 | + // 6. Entity: "projects" (plural is the strongest signal) — "announced" / "planned" alone only set the status | |
| 141 | + if (PROJECT_RE.test(text)) { out.entity = "project"; text = strip(text, PROJECT_RE); } | |
| 142 | + | |
| 143 | + // 7. Status | |
| 107 | 144 | for (const [re, st] of STATUS_PHRASES) { |
| 108 | 145 | if (re.test(text)) { out.status = normalizeStatus(st) ?? st; text = strip(text, re); break; } |
| 109 | 146 | } |
| 110 | 147 | |
| 111 | − // 5. Facility type | |
| 148 | + // 8. AI flag (kept alongside the facility type so callers can match is_ai / ai_evidence rather than the type column) | |
| 149 | + if (AI_RE.test(text)) out.ai = true; | |
| 150 | + | |
| 151 | + // 9. Facility type | |
| 112 | 152 | for (const [re, ty] of TYPE_PHRASES) { |
| 113 | 153 | if (re.test(text)) { out.facilityType = ty; text = strip(text, re); break; } |
| 114 | 154 | } |
| 115 | 155 | |
| 116 | − // 6. Remaining text (drop generic words) | |
| 156 | + // 10. Remaining text (drop generic words) | |
| 117 | 157 | const rest = text.split(" ").filter((t) => t && !STOP.has(t.toLowerCase())).join(" ").trim(); |
| 118 | 158 | if (rest) out.text = rest; |
| 119 | 159 | return out; |
| 120 | 160 | } |
| 121 | 161 | |
| 122 | 162 | export function hasFilters(i: Interpreted): boolean { |
| 123 | − return Boolean(i.countryIso2 || i.operator || i.status || i.minMw != null || i.facilityType); | |
| 163 | + return Boolean(i.countryIso2 || i.operator || i.metro || i.status || i.minMw != null || i.maxMw != null || i.facilityType || i.ai || i.entity || i.year != null); | |
| 124 | 164 | } |
modified
apps/web/src/lib/labels.ts
+10 −0
@@ -176,6 +176,16 @@ export const EVENT_LABEL: Record<EventType, string> = { | ||
| 176 | 176 | incident: "Incident", |
| 177 | 177 | page_changed: "Page changed", |
| 178 | 178 | news: "News", |
| 179 | + land_acquired: "Land acquired", | |
| 180 | + grid_connection: "Grid connection", | |
| 181 | + grid_constraint: "Grid constraint", | |
| 182 | + utility_event: "Utility event", | |
| 183 | + operator_expansion: "Operator expansion", | |
| 184 | + customer_agreement: "Customer agreement", | |
| 185 | + partnership: "Partnership", | |
| 186 | + executive_change: "Executive change", | |
| 187 | + project_delayed: "Project delayed", | |
| 188 | + project_cancelled: "Project cancelled", | |
| 179 | 189 | }; |
| 180 | 190 | |
| 181 | 191 | /** Event families → tone, so the feed reads by colour without a rainbow. */ |
| 182 | 192 | |