/** Dataset downloads (CSV / JSON / GeoJSON) with license gating — streamed, no envelope. */ import { Readable } from "node:stream"; import type { FastifyInstance } from "fastify"; import { z } from "zod"; import { publicGet, registerRoute } from "../../lib/route.js"; import { TTL, HttpError, parseQuery } from "../../lib/http.js"; import { boolParam, csvParam, intParam, numParam, strParam } from "../../lib/params.js"; import { contentType, DOWNLOAD_KEYS, DOWNLOAD_LICENSE, listDatasets, MAX_ROWS, restrictedSources, specFor, streamDataset, type DownloadFormat, type DownloadKey } from "../../repositories/download.js"; const empty = (v: unknown) => (v === "" || v === null ? undefined : v); const downloadQuery = z.object({ format: z.preprocess(empty, z.enum(["csv", "json", "geojson"]).optional()), country: csvParam, status: csvParam, project_status: csvParam, type: csvParam, operator: strParam, metro: strParam, min_mw: numParam, max_mw: numParam, ai: boolParam, hyperscale: boolParam, has_mw: boolParam, expected_from: intParam, expected_to: intParam, event_type: csvParam, type_: strParam.optional(), since: strParam, until: strParam, min_significance: intParam, }); export async function downloadRoutes(app: FastifyInstance): Promise { publicGet(app, { url: "/download/datasets", ttl: TTL.list, tags: ["download"], summary: "Downloadable datasets (DownloadDataset[]): formats, accepted filters, live row counts, licence and excluded (non-redistributable) sources", response: { type: "DownloadDataset[]", example: [{ key: "facilities", label: "Facilities", description: "…", formats: ["csv", "json", "geojson"], filters: ["country", "status"], rows: 8469, license: "…", attribution: [], excludedSources: [] }] } }, async () => { const data = await listDatasets(); return { data, meta: { total: data.length, maxRows: MAX_ROWS, methodology: DOWNLOAD_LICENSE } }; }); registerRoute({ method: "GET", path: "/api/v1/download/:key", summary: `Stream a dataset (${DOWNLOAD_KEYS.join(" | ")}) as CSV, JSON or GeoJSON (≤ ${MAX_ROWS} rows, license-gated, Content-Disposition attachment; no envelope)`, group: "download", params: [{ name: "key", in: "path", type: "string", description: DOWNLOAD_KEYS.join(" | "), example: "facilities" }, { name: "format", in: "query", type: "csv|json|geojson", description: "output format (default csv)", example: "csv" }, { name: "country", in: "query", type: "string", description: "ISO-3166 alpha-2 (comma list)", example: "US" }], responseType: "text/csv | { data: rows[], meta } | GeoJSON FeatureCollection" }); app.get("/download/:key", { schema: { summary: `Stream a dataset (${DOWNLOAD_KEYS.join(" | ")}) as csv | json | geojson — ≤ ${MAX_ROWS} rows, license-gated (rows whose only sources forbid redistribution are excluded and listed in x-dci-excluded-sources / meta.excludedSources), no envelope`, tags: ["download"], params: { type: "object", properties: { key: { type: "string", enum: [...DOWNLOAD_KEYS] } } }, querystring: { type: "object", properties: { format: { type: "string", enum: ["csv", "json", "geojson"] }, country: { type: "string" }, status: { type: "string" }, operator: { type: "string" }, since: { type: "string" } } }, response: { 200: { description: "Dataset rows (CSV text, JSON { data, meta } or GeoJSON FeatureCollection)", type: "string" } } } }, async (req, reply) => { const { key } = req.params as { key: string }; const spec = specFor(key); if (!spec) throw new HttpError(404, "dataset not found"); const q = parseQuery(downloadQuery, req.query); const format: DownloadFormat = q.format ?? "csv"; if (!spec.formats.includes(format)) throw new HttpError(400, `format ${format} is not available for ${key} (${spec.formats.join(", ")})`); const excluded = spec.gated ? await restrictedSources() : []; const filters = { country: q.country, status: q.status, project_status: q.project_status, type: q.type, operator: q.operator, metro: q.metro, min_mw: q.min_mw, max_mw: q.max_mw, ai: q.ai, hyperscale: q.hyperscale, has_mw: q.has_mw, expected_from: q.expected_from, expected_to: q.expected_to, event_type: q.event_type, since: q.since, until: q.until, min_significance: q.min_significance }; const day = new Date().toISOString().slice(0, 10); reply.header("content-type", contentType(format)); reply.header("content-disposition", `attachment; filename="dci-${key}-${day}.${format}"`); reply.header("cache-control", "public, max-age=300"); reply.header("x-dci-license", "attribution required; see /api/v1/download/datasets"); reply.header("x-dci-excluded-sources", excluded.map((s) => s.id).join(",") || "none"); reply.header("x-dci-max-rows", String(MAX_ROWS)); return reply.send(Readable.from(streamDataset(spec.key as DownloadKey, format, filters, excluded))); }); }