adopt: état déployé sur M4M64a adopté comme source de vérité (remote-first, 2026-10-02)
16 changed files +2,339 −0
added
.gitignore
+6 −0
@@ -0,0 +1,6 @@ | ||
| 1 | +.venv/ | |
| 2 | +data/ | |
| 3 | +__pycache__/ | |
| 4 | +*.pyc | |
| 5 | +.DS_Store | |
| 6 | +.env | |
added
requirements-semantic.txt
+3 −0
@@ -0,0 +1,3 @@ | ||
| 1 | +# Optionnel : recherche sémantique en texte libre (encodage des requêtes). | |
| 2 | +# La similarité entre projets (vecteurs stockés) fonctionne sans ce paquet. | |
| 3 | +sentence-transformers>=2.5 | |
added
requirements.txt
+4 −0
@@ -0,0 +1,4 @@ | ||
| 1 | +fastapi>=0.115 | |
| 2 | +uvicorn[standard]>=0.30 | |
| 3 | +duckdb>=1.1 | |
| 4 | +numpy>=1.26 | |
added
server/__init__.py
+0 −0
added
server/db.py
+223 −0
@@ -0,0 +1,223 @@ | ||
| 1 | +"""Accès DuckDB en lecture seule + garde-fous pour le bac à sable SQL.""" | |
| 2 | +from __future__ import annotations | |
| 3 | + | |
| 4 | +import os | |
| 5 | +import re | |
| 6 | +import threading | |
| 7 | +import time | |
| 8 | +from pathlib import Path | |
| 9 | + | |
| 10 | +import duckdb | |
| 11 | +import numpy as np | |
| 12 | + | |
| 13 | +ROOT = Path(__file__).resolve().parent.parent | |
| 14 | +DB_PATH = os.environ.get("SPID_DB", str(ROOT / "data" / "spid.duckdb")) | |
| 15 | +QUERY_TIMEOUT_S = float(os.environ.get("SQL_TIMEOUT_S", "20")) | |
| 16 | +MAX_ROWS = int(os.environ.get("SQL_MAX_ROWS", "5000")) | |
| 17 | + | |
| 18 | +TABLES = ["projects", "project_mentions", "project_timeline", "kg_nodes", "kg_edges", "project_embeddings", "spid_sections"] | |
| 19 | + | |
| 20 | +_ALIAS = {"yr": "year", "loc": "location", "lbl": "label"} | |
| 21 | +_lock = threading.Lock() | |
| 22 | +_conn: duckdb.DuckDBPyConnection | None = None | |
| 23 | +_schema_cache: list[dict] | None = None | |
| 24 | +_emb_cache: dict | None = None | |
| 25 | + | |
| 26 | + | |
| 27 | +def connect() -> duckdb.DuckDBPyConnection: | |
| 28 | + global _conn | |
| 29 | + with _lock: | |
| 30 | + if _conn is None: | |
| 31 | + if not os.path.exists(DB_PATH): | |
| 32 | + raise RuntimeError(f"Base introuvable : {DB_PATH}") | |
| 33 | + _conn = duckdb.connect( | |
| 34 | + DB_PATH, | |
| 35 | + read_only=True, | |
| 36 | + config={ | |
| 37 | + "enable_external_access": "false", # pas de read_parquet/read_csv sur le disque | |
| 38 | + "threads": os.environ.get("DUCKDB_THREADS", "4"), | |
| 39 | + "memory_limit": os.environ.get("DUCKDB_MEM", "4GB"), | |
| 40 | + }, | |
| 41 | + ) | |
| 42 | + try: | |
| 43 | + _conn.execute("SET lock_configuration = true") | |
| 44 | + except Exception: | |
| 45 | + pass | |
| 46 | + return _conn | |
| 47 | + | |
| 48 | + | |
| 49 | +def cursor() -> duckdb.DuckDBPyConnection: | |
| 50 | + """Connexion dupliquée (thread-safe) sur la connexion principale.""" | |
| 51 | + return connect().cursor() | |
| 52 | + | |
| 53 | + | |
| 54 | +def rows(sql: str, params: list | tuple = ()) -> list[dict]: | |
| 55 | + cur = cursor() | |
| 56 | + try: | |
| 57 | + res = cur.execute(sql, list(params)) | |
| 58 | + cols = [_ALIAS.get(d[0], d[0]) for d in res.description] # alias courts : year/location/label sont réservés en DuckDB | |
| 59 | + return [dict(zip(cols, _jsonable(r))) for r in res.fetchall()] | |
| 60 | + finally: | |
| 61 | + cur.close() | |
| 62 | + | |
| 63 | + | |
| 64 | +def one(sql: str, params: list | tuple = ()) -> dict | None: | |
| 65 | + r = rows(sql, params) | |
| 66 | + return r[0] if r else None | |
| 67 | + | |
| 68 | + | |
| 69 | +def scalar(sql: str, params: list | tuple = ()): | |
| 70 | + cur = cursor() | |
| 71 | + try: | |
| 72 | + return cur.execute(sql, list(params)).fetchone()[0] | |
| 73 | + finally: | |
| 74 | + cur.close() | |
| 75 | + | |
| 76 | + | |
| 77 | +def _jsonable(row): | |
| 78 | + out = [] | |
| 79 | + for v in row: | |
| 80 | + if hasattr(v, "isoformat"): | |
| 81 | + v = v.isoformat() | |
| 82 | + elif isinstance(v, (np.floating,)): | |
| 83 | + v = float(v) | |
| 84 | + elif isinstance(v, (np.integer,)): | |
| 85 | + v = int(v) | |
| 86 | + elif isinstance(v, float) and (v != v): # NaN | |
| 87 | + v = None | |
| 88 | + out.append(v) | |
| 89 | + return out | |
| 90 | + | |
| 91 | + | |
| 92 | +# ---------------------------------------------------------------- schéma | |
| 93 | +def schema() -> list[dict]: | |
| 94 | + global _schema_cache | |
| 95 | + if _schema_cache is None: | |
| 96 | + cur = cursor() | |
| 97 | + try: | |
| 98 | + out = [] | |
| 99 | + for t in TABLES: | |
| 100 | + cols = cur.execute( | |
| 101 | + "select column_name, data_type from information_schema.columns where table_name=? order by ordinal_position", [t] | |
| 102 | + ).fetchall() | |
| 103 | + n = cur.execute(f'select count(*) from "{t}"').fetchone()[0] | |
| 104 | + out.append({"table": t, "rows": n, "columns": [{"name": c, "type": d} for c, d in cols]}) | |
| 105 | + _schema_cache = out | |
| 106 | + finally: | |
| 107 | + cur.close() | |
| 108 | + return _schema_cache | |
| 109 | + | |
| 110 | + | |
| 111 | +# ---------------------------------------------------------------- bac à sable SQL | |
| 112 | +_FORBIDDEN = re.compile( | |
| 113 | + r"\b(attach|detach|copy|export|import|install|load|pragma|create|insert|update|delete|drop|alter|call|set|reset|" | |
| 114 | + r"checkpoint|vacuum|force|begin|commit|rollback|grant|revoke|use|read_\w+|glob|getenv|current_setting|" | |
| 115 | + r"sqlite_\w+|postgres_\w+|http\w*|write_\w+|list_files|duckdb_secrets|secrets?|parquet_\w+|iceberg_\w+|delta_\w+|" | |
| 116 | + r"json_extract_path_text|read_json\w*|read_text|read_blob)\b", | |
| 117 | + re.IGNORECASE, | |
| 118 | +) | |
| 119 | + | |
| 120 | + | |
| 121 | +def _strip_comments(sql: str) -> str: | |
| 122 | + sql = re.sub(r"/\*.*?\*/", " ", sql, flags=re.S) | |
| 123 | + sql = re.sub(r"--[^\n]*", " ", sql) | |
| 124 | + return sql.strip() | |
| 125 | + | |
| 126 | + | |
| 127 | +def guard_sql(sql: str) -> str: | |
| 128 | + s = _strip_comments(sql).rstrip(";").strip() | |
| 129 | + if not s: | |
| 130 | + raise ValueError("Requête vide.") | |
| 131 | + if ";" in s: | |
| 132 | + raise ValueError("Une seule instruction à la fois.") | |
| 133 | + if not re.match(r"^(select|with|from|describe|show|summarize|explain)\b", s, re.IGNORECASE): | |
| 134 | + raise ValueError("Seules les requêtes de lecture (SELECT / WITH / DESCRIBE / SUMMARIZE / EXPLAIN) sont autorisées.") | |
| 135 | + m = _FORBIDDEN.search(s) | |
| 136 | + if m: | |
| 137 | + raise ValueError(f"Mot-clé interdit dans le bac à sable : {m.group(0)}") | |
| 138 | + return s | |
| 139 | + | |
| 140 | + | |
| 141 | +def run_sql(sql: str, limit: int = 500) -> dict: | |
| 142 | + """Exécute une requête de lecture avec délai maximal et plafond de lignes.""" | |
| 143 | + s = guard_sql(sql) | |
| 144 | + limit = max(1, min(int(limit), MAX_ROWS)) | |
| 145 | + wrapped = s if re.match(r"^(describe|show|summarize|explain)\b", s, re.IGNORECASE) else f"select * from ({s}) as _q limit {limit + 1}" | |
| 146 | + cur = cursor() | |
| 147 | + result: dict = {} | |
| 148 | + err: list[Exception] = [] | |
| 149 | + | |
| 150 | + def work(): | |
| 151 | + try: | |
| 152 | + t0 = time.perf_counter() | |
| 153 | + res = cur.execute(wrapped) | |
| 154 | + cols = [d[0] for d in res.description] | |
| 155 | + data = res.fetchall() | |
| 156 | + result["columns"] = cols | |
| 157 | + result["truncated"] = len(data) > limit | |
| 158 | + result["rows"] = [_jsonable(r) for r in data[:limit]] | |
| 159 | + result["elapsed_ms"] = round((time.perf_counter() - t0) * 1000, 1) | |
| 160 | + except Exception as e: # noqa: BLE001 | |
| 161 | + err.append(e) | |
| 162 | + | |
| 163 | + th = threading.Thread(target=work, daemon=True) | |
| 164 | + th.start() | |
| 165 | + th.join(QUERY_TIMEOUT_S) | |
| 166 | + if th.is_alive(): | |
| 167 | + try: | |
| 168 | + cur.interrupt() | |
| 169 | + except Exception: | |
| 170 | + pass | |
| 171 | + th.join(2) | |
| 172 | + cur.close() | |
| 173 | + raise TimeoutError(f"Requête interrompue après {QUERY_TIMEOUT_S:.0f} s.") | |
| 174 | + cur.close() | |
| 175 | + if err: | |
| 176 | + raise ValueError(str(err[0]).split("\n")[0][:400]) | |
| 177 | + result["sql"] = s | |
| 178 | + result["limit"] = limit | |
| 179 | + return result | |
| 180 | + | |
| 181 | + | |
| 182 | +# ---------------------------------------------------------------- embeddings (similarité entre projets) | |
| 183 | +def embeddings() -> dict: | |
| 184 | + global _emb_cache | |
| 185 | + if _emb_cache is None: | |
| 186 | + cur = cursor() | |
| 187 | + try: | |
| 188 | + data = cur.execute("select project_id, vector from project_embeddings").fetchall() | |
| 189 | + finally: | |
| 190 | + cur.close() | |
| 191 | + ids = [d[0] for d in data] | |
| 192 | + mat = np.asarray([d[1] for d in data], dtype=np.float32) | |
| 193 | + norms = np.linalg.norm(mat, axis=1, keepdims=True) | |
| 194 | + norms[norms == 0] = 1 | |
| 195 | + mat = mat / norms | |
| 196 | + _emb_cache = {"ids": ids, "index": {pid: i for i, pid in enumerate(ids)}, "mat": mat} | |
| 197 | + return _emb_cache | |
| 198 | + | |
| 199 | + | |
| 200 | +def similar_ids(project_id: str, k: int = 10) -> list[tuple[str, float]]: | |
| 201 | + e = embeddings() | |
| 202 | + i = e["index"].get(project_id) | |
| 203 | + if i is None: | |
| 204 | + return [] | |
| 205 | + sims = e["mat"] @ e["mat"][i] | |
| 206 | + order = np.argsort(-sims) | |
| 207 | + out = [] | |
| 208 | + for j in order: | |
| 209 | + if j == i: | |
| 210 | + continue | |
| 211 | + out.append((e["ids"][j], float(sims[j]))) | |
| 212 | + if len(out) >= k: | |
| 213 | + break | |
| 214 | + return out | |
| 215 | + | |
| 216 | + | |
| 217 | +def nearest_to_vector(vec: np.ndarray, k: int = 20) -> list[tuple[str, float]]: | |
| 218 | + e = embeddings() | |
| 219 | + v = np.asarray(vec, dtype=np.float32) | |
| 220 | + v = v / (np.linalg.norm(v) or 1) | |
| 221 | + sims = e["mat"] @ v | |
| 222 | + order = np.argsort(-sims)[:k] | |
| 223 | + return [(e["ids"][j], float(sims[j])) for j in order] | |
added
server/main.py
+521 −0
@@ -0,0 +1,521 @@ | ||
| 1 | +"""PDB API — explorateur web + API REST de la base SPID (SEC Project Intelligence Database). | |
| 2 | + | |
| 3 | +- /v1/* : API publique, clé requise (en-tête `X-API-Key` ou `Authorization: Bearer …`). | |
| 4 | +- /app/* : mêmes ressources pour l'interface web (même origine, sans clé — la clé n'est jamais envoyée au navigateur). | |
| 5 | +- / : interface web (SPA statique dans web/). | |
| 6 | +- /docs : documentation OpenAPI interactive. | |
| 7 | +""" | |
| 8 | +from __future__ import annotations | |
| 9 | + | |
| 10 | +import csv | |
| 11 | +import io | |
| 12 | +import os | |
| 13 | +import threading | |
| 14 | +import time | |
| 15 | +from collections import defaultdict, deque | |
| 16 | +from pathlib import Path | |
| 17 | +from typing import Optional | |
| 18 | + | |
| 19 | +from fastapi import APIRouter, Depends, FastAPI, HTTPException, Query, Request | |
| 20 | +from fastapi.responses import FileResponse, JSONResponse, StreamingResponse | |
| 21 | +from fastapi.staticfiles import StaticFiles | |
| 22 | +from pydantic import BaseModel, Field | |
| 23 | + | |
| 24 | +from . import db | |
| 25 | + | |
| 26 | +ROOT = Path(__file__).resolve().parent.parent | |
| 27 | +WEB = ROOT / "web" | |
| 28 | +API_KEY = os.environ.get("API_KEY", "").strip() | |
| 29 | +PUBLIC_URL = os.environ.get("PUBLIC_URL", "https://www.pdb-api.co").rstrip("/") | |
| 30 | +RATE_PER_MIN = int(os.environ.get("RATE_PER_MIN", "240")) | |
| 31 | + | |
| 32 | +TYPE_LABELS = { | |
| 33 | + "ma_integration": "Intégration M&A", "plant_construction": "Construction d'usine", "rd_program": "Programme de R&D", | |
| 34 | + "product_launch": "Lancement de produit", "manufacturing_expansion": "Expansion manufacturière", "partnership": "Partenariat", | |
| 35 | + "technology_deployment": "Déploiement technologique", "digital_transformation": "Transformation numérique", | |
| 36 | + "infrastructure": "Infrastructure", "capex_program": "Programme de capex", "ai_initiative": "Initiative IA", | |
| 37 | + "geographic_expansion": "Expansion géographique", "cost_reduction": "Réduction des coûts", "sustainability": "Durabilité", | |
| 38 | + "energy_transition": "Transition énergétique", "supply_chain": "Chaîne d'approvisionnement", "drug_pipeline": "Pipeline de médicaments", | |
| 39 | + "data_center": "Centre de données", "cloud_migration": "Migration infonuagique", "automation": "Automatisation", | |
| 40 | + "erp_implementation": "Implantation ERP", | |
| 41 | +} | |
| 42 | +STATUSES = ["planned", "in_progress", "completed", "mentioned"] | |
| 43 | +PROJECT_SORTS = { | |
| 44 | + "amount": "coalesce(total_amount_usd,0)", "mentions": "n_mentions", "filings": "n_filings", "first_seen": "first_seen", | |
| 45 | + "last_seen": "last_seen", "confidence": "avg_confidence", "name": "project_name", "ticker": "ticker", "type": "project_type", | |
| 46 | +} | |
| 47 | +MENTION_SORTS = {"date": "filing_date", "amount": "coalesce(amount_usd,0)", "confidence": "confidence", "ticker": "ticker"} | |
| 48 | + | |
| 49 | +app = FastAPI( | |
| 50 | + title="PDB API — SEC Project Intelligence Database", | |
| 51 | + version="1.0.0", | |
| 52 | + description=( | |
| 53 | + "API REST de la base SPID : 19 227 projets stratégiques et 39 930 mentions extraits des filings SEC " | |
| 54 | + "(10-K, 10-Q, 8-K) de 500 sociétés du S&P 500, 2010–2026. Toutes les routes `/v1/*` exigent une clé d'API " | |
| 55 | + "dans l'en-tête `X-API-Key` (ou `Authorization: Bearer <clé>`). Les réponses paginées renvoient " | |
| 56 | + "`{total, limit, offset, items}`." | |
| 57 | + ), | |
| 58 | + docs_url="/docs", redoc_url="/redoc", openapi_url="/openapi.json", | |
| 59 | +) | |
| 60 | + | |
| 61 | +# ---------------------------------------------------------------- sécurité | |
| 62 | +_buckets: dict[str, deque] = defaultdict(deque) | |
| 63 | +_bl = threading.Lock() | |
| 64 | + | |
| 65 | + | |
| 66 | +def _rate_limit(request: Request, scope: str): | |
| 67 | + ip = request.headers.get("x-forwarded-for", request.client.host if request.client else "?").split(",")[0].strip() | |
| 68 | + key = f"{scope}:{ip}" | |
| 69 | + now = time.time() | |
| 70 | + with _bl: | |
| 71 | + q = _buckets[key] | |
| 72 | + while q and q[0] < now - 60: | |
| 73 | + q.popleft() | |
| 74 | + if len(q) >= RATE_PER_MIN: | |
| 75 | + raise HTTPException(429, "Trop de requêtes : limite de %d par minute." % RATE_PER_MIN) | |
| 76 | + q.append(now) | |
| 77 | + | |
| 78 | + | |
| 79 | +def require_key(request: Request): | |
| 80 | + _rate_limit(request, "v1") | |
| 81 | + if not API_KEY: | |
| 82 | + raise HTTPException(503, "API non configurée (clé absente côté serveur).") | |
| 83 | + key = request.headers.get("x-api-key") or "" | |
| 84 | + auth = request.headers.get("authorization") or "" | |
| 85 | + if not key and auth.lower().startswith("bearer "): | |
| 86 | + key = auth[7:].strip() | |
| 87 | + if not key: | |
| 88 | + key = request.query_params.get("api_key", "") | |
| 89 | + if key != API_KEY: | |
| 90 | + raise HTTPException(401, "Clé d'API manquante ou invalide (en-tête X-API-Key).") | |
| 91 | + | |
| 92 | + | |
| 93 | +def require_same_origin(request: Request): | |
| 94 | + _rate_limit(request, "app") | |
| 95 | + sfs = request.headers.get("sec-fetch-site") | |
| 96 | + if sfs and sfs not in ("same-origin", "none"): | |
| 97 | + raise HTTPException(403, "Accès réservé à l'interface web.") | |
| 98 | + origin = request.headers.get("origin") or request.headers.get("referer") | |
| 99 | + host = request.headers.get("x-forwarded-host") or request.headers.get("host") or "" | |
| 100 | + if origin: | |
| 101 | + from urllib.parse import urlparse | |
| 102 | + if urlparse(origin).netloc.split(":")[0] != host.split(":")[0]: | |
| 103 | + raise HTTPException(403, "Accès réservé à l'interface web.") | |
| 104 | + | |
| 105 | + | |
| 106 | +# ---------------------------------------------------------------- helpers | |
| 107 | +def _like(v: str) -> str: | |
| 108 | + return f"%{v.lower().strip()}%" | |
| 109 | + | |
| 110 | + | |
| 111 | +def _page(limit: int, offset: int) -> tuple[int, int]: | |
| 112 | + return max(1, min(limit, 500)), max(0, offset) | |
| 113 | + | |
| 114 | + | |
| 115 | +def _project_filters(q, type_, sector, status, ticker, cik, location, tech, partner, min_amount, max_amount, | |
| 116 | + year_from, year_to, min_conf, has_amount): | |
| 117 | + where, params = [], [] | |
| 118 | + if q: | |
| 119 | + where.append("(lower(project_name) like ? or lower(description) like ? or lower(company_name) like ? or lower(ticker) = ?)") | |
| 120 | + params += [_like(q), _like(q), _like(q), q.lower().strip()] | |
| 121 | + if type_: | |
| 122 | + ts = [t.strip() for t in type_.split(",") if t.strip()] | |
| 123 | + where.append("project_type in (" + ",".join("?" * len(ts)) + ")"); params += ts | |
| 124 | + if sector: | |
| 125 | + where.append("lower(sector) like ?"); params.append(_like(sector)) | |
| 126 | + if status: | |
| 127 | + ss = [s.strip() for s in status.split(",") if s.strip()] | |
| 128 | + where.append("status in (" + ",".join("?" * len(ss)) + ")"); params += ss | |
| 129 | + if ticker: | |
| 130 | + where.append("upper(ticker) = ?"); params.append(ticker.upper().strip()) | |
| 131 | + if cik: | |
| 132 | + where.append("cik = ?"); params.append(cik.zfill(10)) | |
| 133 | + if location: | |
| 134 | + where.append("lower(canonical_location) like ?"); params.append(_like(location)) | |
| 135 | + if tech: | |
| 136 | + where.append("exists (select 1 from unnest(technologies) as u(t) where lower(t) like ?)"); params.append(_like(tech)) | |
| 137 | + if partner: | |
| 138 | + where.append("project_id in (select md5(cik||'|'||project_type||'|'||coalesce(nullif(lower(coalesce(try(locations[1]),'')),''), regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{4,})',1), 'general')) from project_mentions, unnest(partners) as u(p) where lower(p) like ?)") | |
| 139 | + params.append(_like(partner)) | |
| 140 | + if min_amount is not None: | |
| 141 | + where.append("total_amount_usd >= ?"); params.append(min_amount) | |
| 142 | + if max_amount is not None: | |
| 143 | + where.append("total_amount_usd <= ?"); params.append(max_amount) | |
| 144 | + if year_from is not None: | |
| 145 | + where.append("year(last_seen) >= ?"); params.append(year_from) | |
| 146 | + if year_to is not None: | |
| 147 | + where.append("year(first_seen) <= ?"); params.append(year_to) | |
| 148 | + if min_conf is not None: | |
| 149 | + where.append("avg_confidence >= ?"); params.append(min_conf) | |
| 150 | + if has_amount: | |
| 151 | + where.append("total_amount_usd > 0") | |
| 152 | + return (" where " + " and ".join(where)) if where else "", params | |
| 153 | + | |
| 154 | + | |
| 155 | +def _mention_filters(q, type_, sector, status, ticker, cik, form, date_from, date_to, min_amount, min_conf, accession, section): | |
| 156 | + where, params = [], [] | |
| 157 | + if q: | |
| 158 | + where.append("(lower(project_name) like ? or lower(description) like ? or lower(company_name) like ?)") | |
| 159 | + params += [_like(q)] * 3 | |
| 160 | + if type_: | |
| 161 | + ts = [t.strip() for t in type_.split(",") if t.strip()] | |
| 162 | + where.append("project_type in (" + ",".join("?" * len(ts)) + ")"); params += ts | |
| 163 | + if sector: | |
| 164 | + where.append("lower(sector) like ?"); params.append(_like(sector)) | |
| 165 | + if status: | |
| 166 | + where.append("status = ?"); params.append(status) | |
| 167 | + if ticker: | |
| 168 | + where.append("upper(ticker) = ?"); params.append(ticker.upper().strip()) | |
| 169 | + if cik: | |
| 170 | + where.append("cik = ?"); params.append(cik.zfill(10)) | |
| 171 | + if form: | |
| 172 | + fs = [f.strip() for f in form.split(",") if f.strip()] | |
| 173 | + where.append("form_type in (" + ",".join("?" * len(fs)) + ")"); params += fs | |
| 174 | + if date_from: | |
| 175 | + where.append("filing_date >= ?"); params.append(date_from) | |
| 176 | + if date_to: | |
| 177 | + where.append("filing_date <= ?"); params.append(date_to) | |
| 178 | + if min_amount is not None: | |
| 179 | + where.append("amount_usd >= ?"); params.append(min_amount) | |
| 180 | + if min_conf is not None: | |
| 181 | + where.append("confidence >= ?"); params.append(min_conf) | |
| 182 | + if accession: | |
| 183 | + where.append("accession_number = ?"); params.append(accession) | |
| 184 | + if section: | |
| 185 | + where.append("section_id = ?"); params.append(section) | |
| 186 | + return (" where " + " and ".join(where)) if where else "", params | |
| 187 | + | |
| 188 | + | |
| 189 | +class SqlBody(BaseModel): | |
| 190 | + sql: str = Field(..., description="Requête SQL DuckDB de lecture (SELECT / WITH / DESCRIBE / SUMMARIZE).") | |
| 191 | + limit: int = Field(500, ge=1, le=5000, description="Nombre maximal de lignes renvoyées.") | |
| 192 | + | |
| 193 | + | |
| 194 | +# ---------------------------------------------------------------- routes (fabrique, montée deux fois) | |
| 195 | +def make_router(tag: str) -> APIRouter: | |
| 196 | + r = APIRouter(tags=[tag]) | |
| 197 | + | |
| 198 | + @r.get("/stats", summary="Vue d'ensemble : compteurs et répartitions") | |
| 199 | + def stats(): | |
| 200 | + ov = db.one("""select count(*) projects, sum(n_mentions) mentions, count(distinct cik) companies, | |
| 201 | + count(distinct sector) sectors, count(*) filter (where total_amount_usd>0) with_amount, | |
| 202 | + median(total_amount_usd) median_amount_usd, sum(total_amount_usd) total_amount_usd, | |
| 203 | + min(first_seen) first_seen, max(last_seen) last_seen from projects""") | |
| 204 | + ov["mentions"] = db.scalar("select count(*) from project_mentions") | |
| 205 | + ov["filings"] = db.scalar("select count(distinct accession_number) from project_mentions") | |
| 206 | + ov["kg_nodes"] = db.scalar("select count(*) from kg_nodes") | |
| 207 | + ov["kg_edges"] = db.scalar("select count(*) from kg_edges") | |
| 208 | + ov["sections_10k"] = db.scalar("select count(*) from spid_sections") | |
| 209 | + by_type = db.rows("""select project_type, count(*) n, sum(total_amount_usd) amount_usd, median(total_amount_usd) median_usd, | |
| 210 | + round(avg(avg_confidence),3) confidence from projects group by 1 order by n desc""") | |
| 211 | + for t in by_type: | |
| 212 | + t["label"] = TYPE_LABELS.get(t["project_type"], t["project_type"]) | |
| 213 | + return { | |
| 214 | + "overview": ov, | |
| 215 | + "by_type": by_type, | |
| 216 | + "by_sector": db.rows("select sector, count(*) n, count(distinct cik) companies, sum(total_amount_usd) amount_usd from projects where sector is not null group by 1 order by n desc"), | |
| 217 | + "by_status": db.rows("select status, count(*) n from projects group by 1 order by n desc"), | |
| 218 | + "by_form": db.rows("select form_type, count(*) n from project_mentions group by 1 order by n desc"), | |
| 219 | + "mentions_by_year": db.rows("""select year(filing_date) as yr, form_type, count(*) n from project_mentions | |
| 220 | + where filing_date >= '2010-01-01' group by 1,2 order by 1,2"""), | |
| 221 | + "projects_by_year": db.rows("select year(first_seen) as yr, count(*) n, sum(total_amount_usd) amount_usd from projects where first_seen >= '2010-01-01' group by 1 order by 1"), | |
| 222 | + "themes_by_year": db.rows("""select year(filing_date) as yr, project_type, count(*) n from project_mentions | |
| 223 | + where project_type in ('ai_initiative','data_center','cloud_migration','digital_transformation','sustainability','energy_transition') | |
| 224 | + and filing_date >= '2010-01-01' group by 1,2 order by 1,2"""), | |
| 225 | + "sector_type": db.rows("select sector, project_type, count(*) n from projects where sector is not null group by 1,2"), | |
| 226 | + "top_companies": db.rows("select ticker, any_value(company_name) company_name, any_value(sector) sector, count(*) n, sum(total_amount_usd) amount_usd from projects group by 1 order by n desc limit 20"), | |
| 227 | + "top_technologies": db.rows("select lower(trim(t)) technology, count(*) n from projects, unnest(technologies) as u(t) where t<>'' group by 1 order by n desc limit 25"), | |
| 228 | + "top_locations": db.rows("select canonical_location as loc, count(*) n from projects where canonical_location is not null and canonical_location<>'' group by 1 order by n desc limit 25"), | |
| 229 | + } | |
| 230 | + | |
| 231 | + @r.get("/taxonomy", summary="Les 21 types de projets (libellés et effectifs)") | |
| 232 | + def taxonomy(): | |
| 233 | + counts = {x["project_type"]: x for x in db.rows("select project_type, count(*) n, sum(total_amount_usd) amount_usd from projects group by 1")} | |
| 234 | + return [{"project_type": k, "label": v, "n": counts.get(k, {}).get("n", 0), "amount_usd": counts.get(k, {}).get("amount_usd")} for k, v in TYPE_LABELS.items()] | |
| 235 | + | |
| 236 | + @r.get("/sectors", summary="Secteurs GICS") | |
| 237 | + def sectors(): | |
| 238 | + return db.rows("select sector, count(*) n, count(distinct cik) companies, sum(total_amount_usd) amount_usd from projects where sector is not null group by 1 order by n desc") | |
| 239 | + | |
| 240 | + @r.get("/technologies", summary="Technologies citées") | |
| 241 | + def technologies(q: Optional[str] = None, limit: int = Query(50, le=500)): | |
| 242 | + w, p = ("where lower(t) like ?", [_like(q)]) if q else ("", []) | |
| 243 | + return db.rows(f"select lower(trim(t)) technology, count(*) n from projects, unnest(technologies) as u(t) {w} {'and' if w else 'where'} t<>'' group by 1 order by n desc limit {int(limit)}", p) | |
| 244 | + | |
| 245 | + @r.get("/locations", summary="Localisations canoniques") | |
| 246 | + def locations(q: Optional[str] = None, limit: int = Query(50, le=500)): | |
| 247 | + w, p = ("and lower(canonical_location) like ?", [_like(q)]) if q else ("", []) | |
| 248 | + return db.rows(f"select canonical_location as loc, count(*) n, sum(total_amount_usd) amount_usd from projects where canonical_location is not null and canonical_location<>'' {w} group by 1 order by n desc limit {int(limit)}", p) | |
| 249 | + | |
| 250 | + @r.get("/partners", summary="Partenaires cités dans les mentions") | |
| 251 | + def partners(q: Optional[str] = None, limit: int = Query(50, le=500)): | |
| 252 | + w, p = ("and lower(p) like ?", [_like(q)]) if q else ("", []) | |
| 253 | + return db.rows(f"select lower(trim(p)) partner, any_value(p) as lbl, count(*) n, count(distinct cik) companies from project_mentions, unnest(partners) as u(p) where p<>'' {w} group by 1 order by n desc limit {int(limit)}", p) | |
| 254 | + | |
| 255 | + @r.get("/projects", summary="Rechercher des projets (filtres + pagination)") | |
| 256 | + def projects( | |
| 257 | + q: Optional[str] = Query(None, description="Texte libre : nom, description, entreprise, ticker"), | |
| 258 | + type: Optional[str] = Query(None, description="project_type (liste séparée par des virgules)"), | |
| 259 | + sector: Optional[str] = None, status: Optional[str] = Query(None, description="planned,in_progress,completed,mentioned"), | |
| 260 | + ticker: Optional[str] = None, cik: Optional[str] = None, location: Optional[str] = None, tech: Optional[str] = None, | |
| 261 | + partner: Optional[str] = None, | |
| 262 | + min_amount: Optional[float] = None, max_amount: Optional[float] = None, | |
| 263 | + year_from: Optional[int] = None, year_to: Optional[int] = None, min_confidence: Optional[float] = None, | |
| 264 | + has_amount: bool = False, | |
| 265 | + sort: str = Query("mentions", description="amount | mentions | filings | first_seen | last_seen | confidence | name | ticker | type"), | |
| 266 | + order: str = Query("desc", pattern="^(asc|desc)$"), | |
| 267 | + limit: int = Query(50, ge=1, le=500), offset: int = Query(0, ge=0), | |
| 268 | + ): | |
| 269 | + w, p = _project_filters(q, type, sector, status, ticker, cik, location, tech, partner, min_amount, max_amount, year_from, year_to, min_confidence, has_amount) | |
| 270 | + limit, offset = _page(limit, offset) | |
| 271 | + col = PROJECT_SORTS.get(sort, "n_mentions") | |
| 272 | + total = db.scalar(f"select count(*) from projects{w}", p) | |
| 273 | + items = db.rows(f"select * from projects{w} order by {col} {order} nulls last, project_id limit {limit} offset {offset}", p) | |
| 274 | + for it in items: | |
| 275 | + it["type_label"] = TYPE_LABELS.get(it["project_type"], it["project_type"]) | |
| 276 | + return {"total": total, "limit": limit, "offset": offset, "items": items} | |
| 277 | + | |
| 278 | + @r.get("/projects/export.csv", summary="Exporter les projets filtrés en CSV (max 50 000 lignes)") | |
| 279 | + def projects_csv( | |
| 280 | + q: Optional[str] = None, type: Optional[str] = None, sector: Optional[str] = None, status: Optional[str] = None, | |
| 281 | + ticker: Optional[str] = None, cik: Optional[str] = None, location: Optional[str] = None, tech: Optional[str] = None, | |
| 282 | + partner: Optional[str] = None, min_amount: Optional[float] = None, max_amount: Optional[float] = None, | |
| 283 | + year_from: Optional[int] = None, year_to: Optional[int] = None, min_confidence: Optional[float] = None, has_amount: bool = False, | |
| 284 | + ): | |
| 285 | + w, p = _project_filters(q, type, sector, status, ticker, cik, location, tech, partner, min_amount, max_amount, year_from, year_to, min_confidence, has_amount) | |
| 286 | + items = db.rows(f"select * from projects{w} order by n_mentions desc limit 50000", p) | |
| 287 | + | |
| 288 | + def gen(): | |
| 289 | + buf = io.StringIO(); wr = csv.writer(buf) | |
| 290 | + cols = list(items[0].keys()) if items else ["project_id"] | |
| 291 | + wr.writerow(cols); yield buf.getvalue(); buf.seek(0); buf.truncate() | |
| 292 | + for it in items: | |
| 293 | + wr.writerow(["|".join(map(str, v)) if isinstance(v, list) else v for v in it.values()]) | |
| 294 | + yield buf.getvalue(); buf.seek(0); buf.truncate() | |
| 295 | + return StreamingResponse(gen(), media_type="text/csv", headers={"Content-Disposition": "attachment; filename=spid_projects.csv"}) | |
| 296 | + | |
| 297 | + @r.get("/projects/{project_id}", summary="Fiche complète d'un projet (chronologie, mentions, graphe, similaires)") | |
| 298 | + def project(project_id: str): | |
| 299 | + pr = db.one("select * from projects where project_id = ?", [project_id]) | |
| 300 | + if not pr: | |
| 301 | + raise HTTPException(404, "Projet introuvable.") | |
| 302 | + pr["type_label"] = TYPE_LABELS.get(pr["project_type"], pr["project_type"]) | |
| 303 | + timeline = db.rows("select * from project_timeline where project_id = ? order by filing_date", [project_id]) | |
| 304 | + # mentions rattachées : même clé de résolution que spid/resolve.py | |
| 305 | + mentions = db.rows("""with m as (select *, lower(coalesce(try(locations[1]),'')) loc1, | |
| 306 | + regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{4,})',1) name_tok from project_mentions) | |
| 307 | + select * exclude (loc1, name_tok) from m | |
| 308 | + where md5(cik||'|'||project_type||'|'||coalesce(nullif(loc1,''), nullif(name_tok,''), 'general')) = ? | |
| 309 | + order by filing_date""", [project_id]) | |
| 310 | + edges = db.rows("""select e.rel, e.src, e.dst, n.node_type, n.label from kg_edges e join kg_nodes n on n.node_id = case when e.src = ? then e.dst else e.src end | |
| 311 | + where e.src = ? or e.dst = ?""", ["P:" + project_id] * 3) | |
| 312 | + sims = db.similar_ids(project_id, 8) | |
| 313 | + similar = [] | |
| 314 | + if sims: | |
| 315 | + ids = [s[0] for s in sims] | |
| 316 | + found = {x["project_id"]: x for x in db.rows("select project_id, ticker, company_name, project_type, project_name, status, total_amount_usd, first_seen from projects where project_id in (" + ",".join("?" * len(ids)) + ")", ids)} | |
| 317 | + for pid, sc in sims: | |
| 318 | + if pid in found: | |
| 319 | + found[pid]["score"] = round(sc, 4); found[pid]["type_label"] = TYPE_LABELS.get(found[pid]["project_type"]); similar.append(found[pid]) | |
| 320 | + return {"project": pr, "timeline": timeline, "mentions": mentions, "graph": edges, "similar": similar} | |
| 321 | + | |
| 322 | + @r.get("/projects/{project_id}/similar", summary="Projets sémantiquement proches (vecteurs MiniLM stockés)") | |
| 323 | + def project_similar(project_id: str, k: int = Query(10, le=50)): | |
| 324 | + sims = db.similar_ids(project_id, k) | |
| 325 | + if not sims and not db.one("select 1 from projects where project_id=?", [project_id]): | |
| 326 | + raise HTTPException(404, "Projet introuvable.") | |
| 327 | + ids = [s[0] for s in sims] | |
| 328 | + found = {x["project_id"]: x for x in db.rows("select * from projects where project_id in (" + ",".join("?" * len(ids)) + ")", ids)} if ids else {} | |
| 329 | + return [dict(found[pid], score=round(sc, 4)) for pid, sc in sims if pid in found] | |
| 330 | + | |
| 331 | + @r.get("/mentions", summary="Rechercher des mentions (unité d'extraction, une par section et par projet)") | |
| 332 | + def mentions( | |
| 333 | + q: Optional[str] = None, type: Optional[str] = None, sector: Optional[str] = None, status: Optional[str] = None, | |
| 334 | + ticker: Optional[str] = None, cik: Optional[str] = None, form: Optional[str] = Query(None, description="10-K,10-Q,8-K"), | |
| 335 | + date_from: Optional[str] = None, date_to: Optional[str] = None, min_amount: Optional[float] = None, | |
| 336 | + min_confidence: Optional[float] = None, accession: Optional[str] = None, section: Optional[str] = None, | |
| 337 | + sort: str = Query("date", description="date | amount | confidence | ticker"), order: str = Query("desc", pattern="^(asc|desc)$"), | |
| 338 | + limit: int = Query(50, ge=1, le=500), offset: int = Query(0, ge=0), | |
| 339 | + ): | |
| 340 | + w, p = _mention_filters(q, type, sector, status, ticker, cik, form, date_from, date_to, min_amount, min_confidence, accession, section) | |
| 341 | + limit, offset = _page(limit, offset) | |
| 342 | + col = MENTION_SORTS.get(sort, "filing_date") | |
| 343 | + total = db.scalar(f"select count(*) from project_mentions{w}", p) | |
| 344 | + items = db.rows(f"""select *, md5(cik||'|'||project_type||'|'||coalesce(nullif(lower(coalesce(try(locations[1]),'')),''), | |
| 345 | + nullif(regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{{4,}})',1),''), 'general')) project_id | |
| 346 | + from project_mentions{w} order by {col} {order} nulls last, mention_id limit {limit} offset {offset}""", p) | |
| 347 | + for it in items: | |
| 348 | + it["type_label"] = TYPE_LABELS.get(it["project_type"], it["project_type"]) | |
| 349 | + return {"total": total, "limit": limit, "offset": offset, "items": items} | |
| 350 | + | |
| 351 | + @r.get("/mentions/{mention_id}", summary="Une mention et sa section source (10-K)") | |
| 352 | + def mention(mention_id: str): | |
| 353 | + m = db.one("select * from project_mentions where mention_id = ?", [mention_id]) | |
| 354 | + if not m: | |
| 355 | + raise HTTPException(404, "Mention introuvable.") | |
| 356 | + m["type_label"] = TYPE_LABELS.get(m["project_type"], m["project_type"]) | |
| 357 | + m["project_id"] = db.scalar("""select md5(cik||'|'||project_type||'|'||coalesce(nullif(lower(coalesce(try(locations[1]),'')),''), | |
| 358 | + nullif(regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{4,})',1),''), 'general')) | |
| 359 | + from project_mentions where mention_id = ?""", [mention_id]) | |
| 360 | + sec = db.one("select section_id, accession_number, section_name, word_count from spid_sections where section_id = ?", [m["section_id"]]) | |
| 361 | + return {"mention": m, "section": sec, "section_text_url": f"/sections/{m['section_id']}" if sec else None} | |
| 362 | + | |
| 363 | + @r.get("/sections/{section_id}", summary="Texte d'une section 10-K ingérée par SPID") | |
| 364 | + def section(section_id: str, highlight: Optional[str] = Query(None, description="Mot à repérer (renvoie les positions)")): | |
| 365 | + s = db.one("select * from spid_sections where section_id = ?", [section_id]) | |
| 366 | + if not s: | |
| 367 | + raise HTTPException(404, "Section introuvable (seules les sections 10-K sont stockées dans la base ; les sections 10-Q/8-K restent dans le corpus parquet amont).") | |
| 368 | + if highlight: | |
| 369 | + import re | |
| 370 | + s["highlights"] = [m.start() for m in re.finditer(re.escape(highlight), s["text"], re.IGNORECASE)][:200] | |
| 371 | + return s | |
| 372 | + | |
| 373 | + @r.get("/companies", summary="Entreprises et taille de leur portefeuille de projets") | |
| 374 | + def companies(q: Optional[str] = None, sector: Optional[str] = None, sort: str = Query("projects", description="projects | amount | mentions | ticker"), | |
| 375 | + order: str = Query("desc", pattern="^(asc|desc)$"), limit: int = Query(100, ge=1, le=500), offset: int = Query(0, ge=0)): | |
| 376 | + where, p = [], [] | |
| 377 | + if q: | |
| 378 | + where.append("(lower(company_name) like ? or lower(ticker) like ?)"); p += [_like(q), _like(q)] | |
| 379 | + if sector: | |
| 380 | + where.append("lower(sector) like ?"); p.append(_like(sector)) | |
| 381 | + w = (" where " + " and ".join(where)) if where else "" | |
| 382 | + col = {"projects": "n_projects", "amount": "coalesce(amount_usd,0)", "mentions": "n_mentions", "ticker": "ticker"}.get(sort, "n_projects") | |
| 383 | + limit, offset = _page(limit, offset) | |
| 384 | + base = f"""with c as (select cik, any_value(ticker) ticker, any_value(company_name) company_name, any_value(sector) sector, | |
| 385 | + count(*) n_projects, sum(n_mentions) n_mentions, sum(total_amount_usd) amount_usd, min(first_seen) first_seen, max(last_seen) last_seen, | |
| 386 | + count(distinct project_type) n_types from projects group by cik) select * from c{w}""" | |
| 387 | + total = db.scalar(f"select count(*) from ({base})", p) | |
| 388 | + items = db.rows(f"{base} order by {col} {order} nulls last limit {limit} offset {offset}", p) | |
| 389 | + return {"total": total, "limit": limit, "offset": offset, "items": items} | |
| 390 | + | |
| 391 | + @r.get("/companies/{ident}", summary="Profil d'une entreprise (ticker ou CIK)") | |
| 392 | + def company(ident: str): | |
| 393 | + ident = ident.strip() | |
| 394 | + cond, val = ("cik = ?", ident.zfill(10)) if ident.isdigit() else ("upper(ticker) = ?", ident.upper()) | |
| 395 | + prof = db.one(f"""select cik, any_value(ticker) ticker, any_value(company_name) company_name, any_value(sector) sector, | |
| 396 | + count(*) n_projects, sum(n_mentions) n_mentions, sum(total_amount_usd) amount_usd, min(first_seen) first_seen, max(last_seen) last_seen | |
| 397 | + from projects where {cond} group by cik""", [val]) | |
| 398 | + if not prof: | |
| 399 | + raise HTTPException(404, "Entreprise introuvable.") | |
| 400 | + cik = prof["cik"] | |
| 401 | + return { | |
| 402 | + "company": prof, | |
| 403 | + "by_type": [dict(x, label=TYPE_LABELS.get(x["project_type"])) for x in db.rows("select project_type, count(*) n, sum(total_amount_usd) amount_usd from projects where cik=? group by 1 order by n desc", [cik])], | |
| 404 | + "by_status": db.rows("select status, count(*) n from projects where cik=? group by 1", [cik]), | |
| 405 | + "by_year": db.rows("select year(filing_date) as yr, count(*) n from project_mentions where cik=? and filing_date>='2010-01-01' group by 1 order by 1", [cik]), | |
| 406 | + "by_form": db.rows("select form_type, count(*) n from project_mentions where cik=? group by 1", [cik]), | |
| 407 | + "locations": db.rows("select canonical_location as loc, count(*) n from projects where cik=? and canonical_location<>'' group by 1 order by n desc limit 15", [cik]), | |
| 408 | + "technologies": db.rows("select lower(t) technology, count(*) n from projects, unnest(technologies) as u(t) where cik=? group by 1 order by n desc limit 15", [cik]), | |
| 409 | + "partners": db.rows("select any_value(p) partner, count(*) n from project_mentions, unnest(partners) as u(p) where cik=? and p<>'' group by lower(p) order by n desc limit 15", [cik]), | |
| 410 | + "projects": [dict(x, type_label=TYPE_LABELS.get(x["project_type"])) for x in db.rows("select * from projects where cik=? order by n_mentions desc, total_amount_usd desc nulls last limit 500", [cik])], | |
| 411 | + } | |
| 412 | + | |
| 413 | + @r.get("/graph/search", summary="Chercher un nœud du graphe (entreprise, projet, lieu, technologie, partenaire)") | |
| 414 | + def graph_search(q: str, type: Optional[str] = Query(None, description="Company | Project | Location | Technology | Partner"), limit: int = Query(30, le=200)): | |
| 415 | + w, p = "where lower(label) like ?", [_like(q)] | |
| 416 | + if type: | |
| 417 | + w += " and node_type = ?"; p.append(type) | |
| 418 | + return db.rows(f"""select n.node_id, n.node_type, n.label, n.props, (select count(*) from kg_edges e where e.src=n.node_id or e.dst=n.node_id) degree | |
| 419 | + from kg_nodes n {w} order by degree desc limit {int(limit)}""", p) | |
| 420 | + | |
| 421 | + @r.get("/graph/node/{node_id:path}", summary="Un nœud et son voisinage") | |
| 422 | + def graph_node(node_id: str, limit: int = Query(200, le=2000)): | |
| 423 | + n = db.one("select * from kg_nodes where node_id = ?", [node_id]) | |
| 424 | + if not n: | |
| 425 | + raise HTTPException(404, "Nœud introuvable.") | |
| 426 | + edges = db.rows(f"""select e.rel, e.src, e.dst, e.props, case when e.src = ? then 'out' else 'in' end direction, | |
| 427 | + m.node_type neighbor_type, m.label neighbor_label, m.node_id neighbor_id, m.props neighbor_props | |
| 428 | + from kg_edges e join kg_nodes m on m.node_id = case when e.src = ? then e.dst else e.src end | |
| 429 | + where e.src = ? or e.dst = ? limit {int(limit)}""", [node_id] * 4) | |
| 430 | + degree = db.scalar("select count(*) from kg_edges where src = ? or dst = ?", [node_id, node_id]) | |
| 431 | + return {"node": n, "degree": degree, "edges": edges} | |
| 432 | + | |
| 433 | + @r.get("/search/semantic", summary="Recherche sémantique en texte libre (modèle all-MiniLM-L6-v2 côté serveur)") | |
| 434 | + def semantic(q: str, k: int = Query(20, le=100), type: Optional[str] = None, sector: Optional[str] = None): | |
| 435 | + from . import semantic as sem | |
| 436 | + try: | |
| 437 | + vec = sem.encode(q) | |
| 438 | + except sem.Unavailable as e: | |
| 439 | + raise HTTPException(501, str(e)) | |
| 440 | + hits = db.nearest_to_vector(vec, k * 5) | |
| 441 | + ids = [h[0] for h in hits] | |
| 442 | + where, p = ["project_id in (" + ",".join("?" * len(ids)) + ")"], list(ids) | |
| 443 | + if type: | |
| 444 | + where.append("project_type = ?"); p.append(type) | |
| 445 | + if sector: | |
| 446 | + where.append("lower(sector) like ?"); p.append(_like(sector)) | |
| 447 | + found = {x["project_id"]: x for x in db.rows("select * from projects where " + " and ".join(where), p)} | |
| 448 | + out = [] | |
| 449 | + for pid, sc in hits: | |
| 450 | + if pid in found: | |
| 451 | + out.append(dict(found[pid], score=round(sc, 4), type_label=TYPE_LABELS.get(found[pid]["project_type"]))) | |
| 452 | + if len(out) >= k: | |
| 453 | + break | |
| 454 | + return {"query": q, "items": out} | |
| 455 | + | |
| 456 | + @r.get("/schema", summary="Tables et colonnes de la base") | |
| 457 | + def schema(): | |
| 458 | + return db.schema() | |
| 459 | + | |
| 460 | + @r.post("/sql", summary="Bac à sable SQL en lecture seule (DuckDB)") | |
| 461 | + def sql(body: SqlBody): | |
| 462 | + try: | |
| 463 | + return db.run_sql(body.sql, body.limit) | |
| 464 | + except TimeoutError as e: | |
| 465 | + raise HTTPException(408, str(e)) | |
| 466 | + except ValueError as e: | |
| 467 | + raise HTTPException(400, str(e)) | |
| 468 | + | |
| 469 | + return r | |
| 470 | + | |
| 471 | + | |
| 472 | +app.include_router(make_router("API v1 (clé requise)"), prefix="/v1", dependencies=[Depends(require_key)]) | |
| 473 | +app.include_router(make_router("Interface web (même origine)"), prefix="/app", dependencies=[Depends(require_same_origin)], include_in_schema=False) | |
| 474 | + | |
| 475 | + | |
| 476 | +@app.get("/health", include_in_schema=False) | |
| 477 | +@app.get("/v1/health", summary="État du service (sans clé)", tags=["Service"]) | |
| 478 | +def health(): | |
| 479 | + try: | |
| 480 | + n = db.scalar("select count(*) from projects") | |
| 481 | + return {"status": "ok", "projects": n, "db": os.path.basename(db.DB_PATH), "semantic": _semantic_state()} | |
| 482 | + except Exception as e: # noqa: BLE001 | |
| 483 | + return JSONResponse({"status": "error", "detail": str(e)[:200]}, status_code=503) | |
| 484 | + | |
| 485 | + | |
| 486 | +def _semantic_state(): | |
| 487 | + try: | |
| 488 | + from . import semantic as sem | |
| 489 | + return sem.state() | |
| 490 | + except Exception: | |
| 491 | + return "unavailable" | |
| 492 | + | |
| 493 | + | |
| 494 | +@app.get("/v1/meta", summary="Métadonnées de l'API (routes, limites, exemples)", tags=["Service"]) | |
| 495 | +def meta(): | |
| 496 | + return { | |
| 497 | + "name": "PDB API", "version": app.version, "base_url": PUBLIC_URL + "/v1", | |
| 498 | + "auth": "En-tête X-API-Key: <clé> (ou Authorization: Bearer <clé>). La clé est fournie par l'équipe UQO ; elle n'est publiée nulle part sur le site.", | |
| 499 | + "rate_limit_per_minute": RATE_PER_MIN, "pagination": "limit (≤500) / offset ; réponses {total, limit, offset, items}", | |
| 500 | + "routes": [f"{r.methods and list(r.methods)[0]} {r.path}" for r in app.routes if getattr(r, "path", "").startswith("/v1")], | |
| 501 | + "docs": PUBLIC_URL + "/docs", | |
| 502 | + } | |
| 503 | + | |
| 504 | + | |
| 505 | +@app.on_event("startup") | |
| 506 | +def _warm(): | |
| 507 | + def w(): | |
| 508 | + try: | |
| 509 | + db.connect(); db.schema(); db.embeddings() | |
| 510 | + except Exception as e: # noqa: BLE001 | |
| 511 | + print("warmup:", e) | |
| 512 | + threading.Thread(target=w, daemon=True).start() | |
| 513 | + | |
| 514 | + | |
| 515 | +# ---------------------------------------------------------------- statique | |
| 516 | +if (WEB / "report.pdf").exists(): | |
| 517 | + @app.get("/report.pdf", include_in_schema=False) | |
| 518 | + def report(): | |
| 519 | + return FileResponse(WEB / "report.pdf", media_type="application/pdf") | |
| 520 | + | |
| 521 | +app.mount("/", StaticFiles(directory=str(WEB), html=True), name="web") | |
added
server/semantic.py
+55 −0
@@ -0,0 +1,55 @@ | ||
| 1 | +"""Encodage des requêtes en texte libre avec all-MiniLM-L6-v2 (optionnel, chargé paresseusement).""" | |
| 2 | +from __future__ import annotations | |
| 3 | + | |
| 4 | +import os | |
| 5 | +import threading | |
| 6 | + | |
| 7 | +MODEL_NAME = os.environ.get("EMBED_MODEL", "all-MiniLM-L6-v2") | |
| 8 | +_model = None | |
| 9 | +_state = "idle" | |
| 10 | +_lock = threading.Lock() | |
| 11 | + | |
| 12 | + | |
| 13 | +class Unavailable(Exception): | |
| 14 | + pass | |
| 15 | + | |
| 16 | + | |
| 17 | +def state() -> str: | |
| 18 | + return _state | |
| 19 | + | |
| 20 | + | |
| 21 | +def _load(): | |
| 22 | + global _model, _state | |
| 23 | + with _lock: | |
| 24 | + if _model is not None: | |
| 25 | + return _model | |
| 26 | + try: | |
| 27 | + _state = "loading" | |
| 28 | + from sentence_transformers import SentenceTransformer # type: ignore | |
| 29 | + _model = SentenceTransformer(MODEL_NAME) | |
| 30 | + _state = "ready" | |
| 31 | + except Exception as e: # noqa: BLE001 | |
| 32 | + _state = f"unavailable: {type(e).__name__}" | |
| 33 | + raise Unavailable("Recherche sémantique indisponible sur ce serveur (paquet sentence-transformers absent ou modèle non téléchargé). " | |
| 34 | + "Utilisez /projects?q= ou /projects/{id}/similar.") from e | |
| 35 | + return _model | |
| 36 | + | |
| 37 | + | |
| 38 | +def encode(text: str): | |
| 39 | + m = _load() | |
| 40 | + return m.encode([text], normalize_embeddings=True)[0] | |
| 41 | + | |
| 42 | + | |
| 43 | +def preload(): | |
| 44 | + threading.Thread(target=lambda: _safe(_load), daemon=True).start() | |
| 45 | + | |
| 46 | + | |
| 47 | +def _safe(f): | |
| 48 | + try: | |
| 49 | + f() | |
| 50 | + except Exception: | |
| 51 | + pass | |
| 52 | + | |
| 53 | + | |
| 54 | +if os.environ.get("SEMANTIC_PRELOAD", "1") == "1": | |
| 55 | + preload() | |
added
web/app.js
+414 −0
@@ -0,0 +1,414 @@ | ||
| 1 | +/* PDB — explorateur SPID. SPA sans dépendance (Chart.js chargé par CDN). */ | |
| 2 | +(() => { | |
| 3 | + const $ = (s, r = document) => r.querySelector(s); | |
| 4 | + const view = $("#view"); | |
| 5 | + const API = "/app"; | |
| 6 | + const TYPE = {}; | |
| 7 | + const STATUT = { planned: "Planifié", in_progress: "En cours", completed: "Terminé", mentioned: "Mentionné" }; | |
| 8 | + const COL = { bleu: "#003E7E", bleu2: "#0066B3", or: "#C6A300", teal: "#007979", orange: "#D67A00", vert: "#008046", violet: "#5E35B1", rouge: "#B42318", gris: "#8a8c90" }; | |
| 9 | + const PAL = [COL.bleu, COL.or, COL.teal, COL.orange, COL.vert, COL.violet, COL.rouge, COL.bleu2, COL.gris, "#8DA9C4", "#E0C36B"]; | |
| 10 | + let charts = []; | |
| 11 | + const fmtN = (n) => n == null ? "—" : Number(n).toLocaleString("fr-CA"); | |
| 12 | + const fmtUSD = (v) => v == null ? "—" : v >= 1e9 ? (v / 1e9).toLocaleString("fr-CA", { maximumFractionDigits: 1 }) + " G$" : v >= 1e6 ? (v / 1e6).toLocaleString("fr-CA", { maximumFractionDigits: 0 }) + " M$" : Math.round(v).toLocaleString("fr-CA") + " $"; | |
| 13 | + const esc = (s) => String(s ?? "").replace(/[&<>"']/g, (c) => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" }[c])); | |
| 14 | + const tl = (t) => TYPE[t] || t; | |
| 15 | + const stPill = (s) => `<span class="st st-${esc(s)}">${esc(STATUT[s] || s)}</span>`; | |
| 16 | + const qs = (o) => Object.entries(o).filter(([, v]) => v !== "" && v != null && v !== false).map(([k, v]) => `${encodeURIComponent(k)}=${encodeURIComponent(v)}`).join("&"); | |
| 17 | + const toast = (m) => { const t = $("#toast"); t.textContent = m; t.hidden = false; clearTimeout(t._h); t._h = setTimeout(() => (t.hidden = true), 2600); }; | |
| 18 | + | |
| 19 | + async function api(path, params) { | |
| 20 | + const url = API + path + (params ? "?" + qs(params) : ""); | |
| 21 | + const r = await fetch(url, { headers: { Accept: "application/json" } }); | |
| 22 | + if (!r.ok) { let d = ""; try { d = (await r.json()).detail; } catch { } throw new Error(d || `${r.status} ${r.statusText}`); } | |
| 23 | + return r.json(); | |
| 24 | + } | |
| 25 | + function destroyCharts() { charts.forEach((c) => c.destroy()); charts = []; } | |
| 26 | + function chart(el, cfg) { if (!window.Chart) return; Chart.defaults.font.family = getComputedStyle(document.body).fontFamily; Chart.defaults.color = "#58595B"; const c = new Chart(el, cfg); charts.push(c); return c; } | |
| 27 | + const hashParams = () => { const h = location.hash.slice(2); const [p, q] = h.split("?"); return { path: p || "", parts: (p || "").split("/"), q: Object.fromEntries(new URLSearchParams(q || "")) }; }; | |
| 28 | + const setQuery = (obj) => { const { parts } = hashParams(); location.hash = "#/" + parts[0] + (Object.keys(obj).length ? "?" + qs(obj) : ""); }; | |
| 29 | + | |
| 30 | + /* ------------------------------------------------------------ tableau générique */ | |
| 31 | + function table(cols, rows, opts = {}) { | |
| 32 | + const th = cols.map((c) => `<th class="${c.num ? "num" : ""} ${opts.sort === c.key ? "sorted " + (opts.order || "") : ""}" data-sort="${c.sortKey || ""}">${esc(c.label)}</th>`).join(""); | |
| 33 | + const tr = rows.map((r) => `<tr class="${opts.href ? "click" : ""}" data-href="${opts.href ? esc(opts.href(r)) : ""}">${cols.map((c) => `<td class="${c.num ? "num" : ""} ${c.trunc ? "trunc" : ""}" title="${c.trunc ? esc(c.title ? c.title(r) : r[c.key]) : ""}">${c.render ? c.render(r) : esc(r[c.key])}</td>`).join("")}</tr>`).join(""); | |
| 34 | + return `<div class="table-wrap"><table><thead><tr>${th}</tr></thead><tbody>${tr || `<tr><td colspan="${cols.length}" class="muted">Aucun résultat.</td></tr>`}</tbody></table></div>`; | |
| 35 | + } | |
| 36 | + function bindTable(root, onSort) { | |
| 37 | + root.querySelectorAll("tr[data-href]").forEach((tr) => tr.addEventListener("click", (e) => { if (e.target.closest("a")) return; if (tr.dataset.href) location.hash = tr.dataset.href; })); | |
| 38 | + if (onSort) root.querySelectorAll("th[data-sort]").forEach((th) => th.addEventListener("click", () => th.dataset.sort && onSort(th.dataset.sort))); | |
| 39 | + } | |
| 40 | + function pager(total, limit, offset) { | |
| 41 | + const page = Math.floor(offset / limit) + 1, pages = Math.max(1, Math.ceil(total / limit)); | |
| 42 | + return `<div class="pager"><span>${fmtN(total)} résultats · page ${page} / ${fmtN(pages)}</span><span class="sp"></span> | |
| 43 | + <button class="btn btn-ghost btn-sm" data-off="${Math.max(0, offset - limit)}" ${offset === 0 ? "disabled" : ""}>← Précédent</button> | |
| 44 | + <button class="btn btn-ghost btn-sm" data-off="${offset + limit}" ${offset + limit >= total ? "disabled" : ""}>Suivant →</button></div>`; | |
| 45 | + } | |
| 46 | + | |
| 47 | + /* ------------------------------------------------------------ tableau de bord */ | |
| 48 | + async function dashboard() { | |
| 49 | + const s = await api("/stats"); | |
| 50 | + const ov = s.overview; | |
| 51 | + view.innerHTML = ` | |
| 52 | + <div class="hero"><div><h1>SEC Project Intelligence Database</h1> | |
| 53 | + <p>${fmtN(ov.projects)} projets stratégiques et ${fmtN(ov.mentions)} mentions extraits par LLM des filings SEC (10-K, 10-Q, 8-K) de ${fmtN(ov.companies)} sociétés du S&P 500, ${ov.first_seen?.slice(0, 4)}–${ov.last_seen?.slice(0, 4)}. Explorez, filtrez, interrogez en SQL, ou branchez-vous sur l'API.</p></div> | |
| 54 | + <div><a class="btn" href="#/projects">Explorer les projets</a> <a class="btn btn-ghost" href="#/api" style="color:#fff;border-color:rgba(255,255,255,.5);background:transparent">API</a></div></div> | |
| 55 | + <div class="kpis"> | |
| 56 | + ${kpi(ov.projects, "projets résolus")}${kpi(ov.mentions, "mentions extraites")}${kpi(ov.filings, "filings avec projets")}${kpi(ov.companies, "entreprises")} | |
| 57 | + ${kpi(ov.with_amount, "projets avec montant")}${kpi(fmtUSD(ov.median_amount_usd), "montant médian")}${kpi(ov.kg_nodes, "nœuds du graphe")}${kpi(ov.sections_10k, "sections 10-K")} | |
| 58 | + </div> | |
| 59 | + <div class="grid g2"> | |
| 60 | + <div class="card"><h3>Mentions par année et formulaire</h3><div class="chart"><canvas id="c-year"></canvas></div></div> | |
| 61 | + <div class="card"><h3>Statut des projets</h3><div class="chart"><canvas id="c-status"></canvas></div></div> | |
| 62 | + <div class="card" style="grid-column:1/-1"><h3>Les 21 types de projets (nombre et capital divulgué)</h3><div class="chart tall"><canvas id="c-type"></canvas></div></div> | |
| 63 | + <div class="card"><h3>Projets par secteur GICS</h3><div class="chart tall"><canvas id="c-sector"></canvas></div></div> | |
| 64 | + <div class="card"><h3>Thèmes émergents — part des mentions annuelles</h3><div class="chart tall"><canvas id="c-themes"></canvas></div></div> | |
| 65 | + <div class="card"><h3>Entreprises les plus prolifiques</h3>${table([ | |
| 66 | + { key: "ticker", label: "Ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` }, | |
| 67 | + { key: "company_name", label: "Entreprise" }, { key: "sector", label: "Secteur" }, | |
| 68 | + { key: "n", label: "Projets", num: true, render: (r) => fmtN(r.n) }, { key: "amount_usd", label: "Capital", num: true, render: (r) => fmtUSD(r.amount_usd) }, | |
| 69 | + ], s.top_companies.slice(0, 12))}</div> | |
| 70 | + <div class="card"><h3>Technologies et localisations les plus citées</h3><div class="grid g2"> | |
| 71 | + <div class="tags">${s.top_technologies.slice(0, 20).map((t) => `<a class="tag t-tech" href="#/projects?tech=${encodeURIComponent(t.technology)}">${esc(t.technology)} <b>${fmtN(t.n)}</b></a>`).join("")}</div> | |
| 72 | + <div class="tags">${s.top_locations.slice(0, 20).map((t) => `<a class="tag t-loc" href="#/projects?location=${encodeURIComponent(t.location)}">${esc(t.location)} <b>${fmtN(t.n)}</b></a>`).join("")}</div></div></div> | |
| 73 | + </div>`; | |
| 74 | + // graphiques | |
| 75 | + const years = [...new Set(s.mentions_by_year.map((r) => r.year))].sort(); | |
| 76 | + const forms = ["10-K", "10-Q", "8-K"]; | |
| 77 | + chart($("#c-year"), { type: "bar", data: { labels: years, datasets: forms.map((f, i) => ({ label: f, data: years.map((y) => s.mentions_by_year.find((r) => r.year === y && r.form_type === f)?.n || 0), backgroundColor: [COL.bleu, COL.or, COL.teal][i] })) }, options: { responsive: true, maintainAspectRatio: false, scales: { x: { stacked: true }, y: { stacked: true } }, plugins: { legend: { position: "bottom" } } } }); | |
| 78 | + chart($("#c-status"), { type: "doughnut", data: { labels: s.by_status.map((r) => STATUT[r.status] || r.status), datasets: [{ data: s.by_status.map((r) => r.n), backgroundColor: [COL.bleu2, COL.vert, COL.or, COL.gris] }] }, options: { responsive: true, maintainAspectRatio: false, cutout: "58%", plugins: { legend: { position: "right" } } } }); | |
| 79 | + chart($("#c-type"), { type: "bar", data: { labels: s.by_type.map((r) => r.label), datasets: [{ label: "Projets", data: s.by_type.map((r) => r.n), backgroundColor: COL.bleu, yAxisID: "y" }, { label: "Capital (G$)", data: s.by_type.map((r) => (r.amount_usd || 0) / 1e9), backgroundColor: COL.or, yAxisID: "y2" }] }, options: { responsive: true, maintainAspectRatio: false, scales: { x: { ticks: { autoSkip: false, maxRotation: 60, minRotation: 45 } }, y: { position: "left", title: { display: true, text: "projets" } }, y2: { position: "right", grid: { drawOnChartArea: false }, title: { display: true, text: "G$" } } }, plugins: { legend: { position: "bottom" } }, onClick: (e, els) => { if (els[0]) location.hash = "#/projects?type=" + s.by_type[els[0].index].project_type; } } }); | |
| 80 | + chart($("#c-sector"), { type: "bar", data: { labels: s.by_sector.map((r) => r.sector), datasets: [{ label: "Projets", data: s.by_sector.map((r) => r.n), backgroundColor: COL.bleu2 }] }, options: { indexAxis: "y", responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } }, onClick: (e, els) => { if (els[0]) location.hash = "#/projects?sector=" + encodeURIComponent(s.by_sector[els[0].index].sector); } } }); | |
| 81 | + const tot = {}; s.mentions_by_year.forEach((r) => (tot[r.year] = (tot[r.year] || 0) + r.n)); | |
| 82 | + const themes = [...new Set(s.themes_by_year.map((r) => r.project_type))]; | |
| 83 | + chart($("#c-themes"), { type: "line", data: { labels: years, datasets: themes.map((t, i) => ({ label: tl(t), data: years.map((y) => { const n = s.themes_by_year.find((r) => r.year === y && r.project_type === t)?.n || 0; return tot[y] ? +(100 * n / tot[y]).toFixed(2) : 0; }), borderColor: [COL.rouge, COL.violet, COL.bleu2, COL.teal, COL.vert, COL.or][i], backgroundColor: "transparent", tension: .25, pointRadius: 2 })) }, options: { responsive: true, maintainAspectRatio: false, scales: { y: { title: { display: true, text: "% des mentions" } } }, plugins: { legend: { position: "bottom" } } } }); | |
| 84 | + } | |
| 85 | + const kpi = (v, l) => `<div class="card kpi"><b>${typeof v === "number" ? fmtN(v) : v}</b><span>${l}</span></div>`; | |
| 86 | + | |
| 87 | + /* ------------------------------------------------------------ projets */ | |
| 88 | + const PCOLS = [ | |
| 89 | + { key: "ticker", label: "Ticker", sortKey: "ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` }, | |
| 90 | + { key: "project_name", label: "Projet", sortKey: "name", trunc: true, render: (r) => `<a href="#/project/${esc(r.project_id)}">${esc(r.project_name)}</a><div class="muted small">${esc(r.company_name)}</div>` }, | |
| 91 | + { key: "project_type", label: "Type", sortKey: "type", render: (r) => `<span class="pill pill-blue">${esc(tl(r.project_type))}</span>` }, | |
| 92 | + { key: "status", label: "Statut", render: (r) => stPill(r.status) }, | |
| 93 | + { key: "canonical_location", label: "Lieu" }, | |
| 94 | + { key: "total_amount_usd", label: "Montant", num: true, sortKey: "amount", render: (r) => fmtUSD(r.total_amount_usd) }, | |
| 95 | + { key: "n_mentions", label: "Mentions", num: true, sortKey: "mentions", render: (r) => fmtN(r.n_mentions) }, | |
| 96 | + { key: "first_seen", label: "Première / dernière", sortKey: "first_seen", render: (r) => `${esc(r.first_seen?.slice(0, 7))} → ${esc(r.last_seen?.slice(0, 7))}` }, | |
| 97 | + { key: "avg_confidence", label: "Conf.", num: true, sortKey: "confidence", render: (r) => (r.avg_confidence ?? 0).toFixed(2) }, | |
| 98 | + ]; | |
| 99 | + async function projects() { | |
| 100 | + const { q } = hashParams(); | |
| 101 | + const f = { q: q.q || "", type: q.type || "", sector: q.sector || "", status: q.status || "", ticker: q.ticker || "", location: q.location || "", tech: q.tech || "", partner: q.partner || "", min_amount: q.min_amount || "", year_from: q.year_from || "", year_to: q.year_to || "", min_confidence: q.min_confidence || "", has_amount: q.has_amount || "", sort: q.sort || "mentions", order: q.order || "desc", limit: 50, offset: +q.offset || 0 }; | |
| 102 | + const [tax, secs] = await Promise.all([api("/taxonomy"), api("/sectors")]); | |
| 103 | + view.innerHTML = `<div class="page-head"><div><h1>Projets</h1><p>${fmtN(0)} — chargement…</p></div> | |
| 104 | + <div><a class="btn btn-ghost btn-sm" id="csv" href="${API}/projects/export.csv?${qs(f)}" download>Exporter CSV</a></div></div> | |
| 105 | + <form class="filters" id="pf"> | |
| 106 | + <label class="span2">Recherche<input name="q" value="${esc(f.q)}" placeholder="nom, description, entreprise, ticker"></label> | |
| 107 | + <label>Type<select name="type"><option value="">Tous</option>${tax.map((t) => `<option value="${t.project_type}" ${f.type === t.project_type ? "selected" : ""}>${esc(t.label)} (${fmtN(t.n)})</option>`).join("")}</select></label> | |
| 108 | + <label>Secteur<select name="sector"><option value="">Tous</option>${secs.map((s) => `<option ${f.sector === s.sector ? "selected" : ""}>${esc(s.sector)}</option>`).join("")}</select></label> | |
| 109 | + <label>Statut<select name="status"><option value="">Tous</option>${Object.entries(STATUT).map(([k, v]) => `<option value="${k}" ${f.status === k ? "selected" : ""}>${v}</option>`).join("")}</select></label> | |
| 110 | + <label>Ticker<input name="ticker" value="${esc(f.ticker)}"></label> | |
| 111 | + <label>Lieu<input name="location" value="${esc(f.location)}"></label> | |
| 112 | + <label>Technologie<input name="tech" value="${esc(f.tech)}"></label> | |
| 113 | + <label>Partenaire<input name="partner" value="${esc(f.partner)}"></label> | |
| 114 | + <label>Montant min (US$)<input name="min_amount" type="number" value="${esc(f.min_amount)}" placeholder="ex. 1e9"></label> | |
| 115 | + <label>Année de<input name="year_from" type="number" value="${esc(f.year_from)}" min="2010" max="2026"></label> | |
| 116 | + <label>Année à<input name="year_to" type="number" value="${esc(f.year_to)}" min="2010" max="2026"></label> | |
| 117 | + <label>Confiance min<input name="min_confidence" type="number" step="0.05" min="0" max="1" value="${esc(f.min_confidence)}"></label> | |
| 118 | + <label>Avec montant<select name="has_amount"><option value="">Indifférent</option><option value="true" ${f.has_amount ? "selected" : ""}>Oui</option></select></label> | |
| 119 | + <label> <span><button class="btn btn-sm">Filtrer</button> <button type="button" class="btn btn-ghost btn-sm" id="reset">Effacer</button></span></label> | |
| 120 | + </form><div id="res"><div class="loading">Chargement…</div></div>`; | |
| 121 | + $("#pf").addEventListener("submit", (e) => { e.preventDefault(); const d = Object.fromEntries(new FormData(e.target)); setQuery({ ...d, sort: f.sort, order: f.order }); }); | |
| 122 | + $("#reset").addEventListener("click", () => (location.hash = "#/projects")); | |
| 123 | + const data = await api("/projects", f); | |
| 124 | + $(".page-head p").textContent = `${fmtN(data.total)} projets correspondent aux filtres.`; | |
| 125 | + const res = $("#res"); | |
| 126 | + res.innerHTML = table(PCOLS, data.items, { href: (r) => `#/project/${r.project_id}`, sort: f.sort, order: f.order }) + pager(data.total, f.limit, f.offset); | |
| 127 | + bindTable(res, (k) => setQuery({ ...f, offset: 0, sort: k, order: f.sort === k && f.order === "desc" ? "asc" : "desc" })); | |
| 128 | + res.querySelectorAll("[data-off]").forEach((b) => b.addEventListener("click", () => setQuery({ ...f, offset: b.dataset.off }))); | |
| 129 | + } | |
| 130 | + | |
| 131 | + /* ------------------------------------------------------------ fiche projet */ | |
| 132 | + async function project(id) { | |
| 133 | + const d = await api("/projects/" + id); | |
| 134 | + const p = d.project; | |
| 135 | + const list = (arr, cls) => (arr || []).length ? `<div class="tags">${arr.map((x) => `<span class="tag ${cls}">${esc(x)}</span>`).join("")}</div>` : `<span class="muted">—</span>`; | |
| 136 | + view.innerHTML = ` | |
| 137 | + <div class="page-head"><div><div class="muted small"><a href="#/projects">Projets</a> › <a href="#/company/${esc(p.ticker)}">${esc(p.company_name)}</a></div> | |
| 138 | + <h1>${esc(p.project_name)}</h1><p><span class="pill pill-blue">${esc(tl(p.project_type))}</span> ${stPill(p.status)} <span class="muted">${esc(p.sector || "")}</span></p></div> | |
| 139 | + <div><a class="btn btn-ghost btn-sm" href="#/graph?node=${encodeURIComponent("P:" + p.project_id)}">Voir dans le graphe</a> <a class="btn btn-ghost btn-sm" href="/docs#/API%20v1%20(cl%C3%A9%20requise)/project_v1_projects__project_id__get" target="_blank">API ↗</a></div></div> | |
| 140 | + <div class="grid g3"> | |
| 141 | + <div class="card" style="grid-column:span 2"><h3>Description</h3><p>${esc(p.description)}</p> | |
| 142 | + <dl class="dl"><dt>Entreprise</dt><dd><a href="#/company/${esc(p.ticker)}">${esc(p.company_name)} (${esc(p.ticker)})</a> · CIK ${esc(p.cik)}</dd> | |
| 143 | + <dt>Localisation</dt><dd>${p.canonical_location ? `<a href="#/projects?location=${encodeURIComponent(p.canonical_location)}">${esc(p.canonical_location)}</a>` : "—"}</dd> | |
| 144 | + <dt>Montant (max divulgué)</dt><dd>${fmtUSD(p.total_amount_usd)}</dd> | |
| 145 | + <dt>Observé</dt><dd>${esc(p.first_seen)} → ${esc(p.last_seen)} · ${fmtN(p.n_mentions)} mentions dans ${fmtN(p.n_filings)} filings</dd> | |
| 146 | + <dt>Années citées</dt><dd>${p.first_year ?? "—"} → ${p.last_year ?? "—"}</dd> | |
| 147 | + <dt>Confiance moyenne</dt><dd>${(p.avg_confidence ?? 0).toFixed(3)}</dd> | |
| 148 | + <dt>Technologies</dt><dd>${list(p.technologies, "t-tech")}</dd> | |
| 149 | + <dt>Identifiant</dt><dd class="mono">${esc(p.project_id)}</dd></dl></div> | |
| 150 | + <div class="card"><h3>Chronologie (${d.timeline.length})</h3><ul class="timeline">${d.timeline.map((t) => `<li><div><b>${esc(t.form_type)}</b> · ${stPill(t.status)} ${t.amount_usd ? `· ${fmtUSD(t.amount_usd)}` : ""}</div><div class="when">${esc(t.filing_date)} · conf. ${(t.confidence ?? 0).toFixed(2)} · <span class="mono">${esc(t.accession_number)}</span></div><div class="snip">${esc(t.snippet)}</div></li>`).join("")}</ul></div> | |
| 151 | + <div class="card" style="grid-column:span 2"><h3>Mentions extraites (${d.mentions.length})</h3> | |
| 152 | + ${d.mentions.map((m) => `<div class="mention"><div class="mh"><b>${esc(m.form_type)}</b><span class="muted">${esc(m.filing_date)}</span><span class="pill pill-gris">${esc(m.section_name)}</span>${stPill(m.status)}${m.amount_usd ? `<span class="pill pill-or">${fmtUSD(m.amount_usd)}</span>` : ""}<span class="muted small">conf. ${(m.confidence ?? 0).toFixed(2)}</span><span class="sp" style="flex:1"></span><a class="small" href="#/mention/${esc(m.mention_id)}">détail</a></div> | |
| 153 | + <p><b>${esc(m.project_name)}</b> — ${esc(m.description)}</p>${m.objective ? `<p class="muted"><i>Objectif :</i> ${esc(m.objective)}</p>` : ""} | |
| 154 | + <div class="grid g2" style="gap:6px">${m.locations?.length ? `<div>${list(m.locations, "t-loc")}</div>` : ""}${m.technologies?.length ? `<div>${list(m.technologies, "t-tech")}</div>` : ""}${m.partners?.length ? `<div><span class="small muted">Partenaires</span> ${list(m.partners, "t-part")}</div>` : ""}${m.benefits?.length ? `<div><span class="small muted">Bénéfices</span> ${list(m.benefits, "t-ben")}</div>` : ""}${m.risks?.length ? `<div><span class="small muted">Risques</span> ${list(m.risks, "t-risk")}</div>` : ""}${m.years?.length ? `<div><span class="small muted">Années</span> ${list(m.years, "")}</div>` : ""}</div></div>`).join("")}</div> | |
| 155 | + <div class="card"><h3>Projets similaires (embeddings)</h3>${d.similar.length ? d.similar.map((s) => `<div style="padding:6px 0;border-top:1px solid var(--ligne)"><a href="#/project/${esc(s.project_id)}">${esc(s.project_name)}</a><div class="small muted">${esc(s.ticker)} · ${esc(tl(s.project_type))} · <span class="score">${(s.score * 100).toFixed(0)} %</span></div></div>`).join("") : `<span class="muted">—</span>`} | |
| 156 | + <h3 style="margin-top:14px">Voisins dans le graphe</h3><div>${d.graph.map((g) => `<a class="node-chip nt-${esc(g.node_type)}" href="#/graph?node=${encodeURIComponent(g.src === "P:" + p.project_id ? g.dst : g.src)}"><span class="nt">${esc(g.rel)}</span>${esc(g.label)}</a>`).join("") || `<span class="muted">—</span>`}</div></div> | |
| 157 | + </div>`; | |
| 158 | + } | |
| 159 | + | |
| 160 | + /* ------------------------------------------------------------ mention */ | |
| 161 | + async function mention(id) { | |
| 162 | + const d = await api("/mentions/" + id); | |
| 163 | + const m = d.mention; | |
| 164 | + view.innerHTML = `<div class="page-head"><div><div class="muted small"><a href="#/mentions">Mentions</a> › <a href="#/project/${esc(m.project_id)}">projet résolu</a></div><h1>${esc(m.project_name)}</h1> | |
| 165 | + <p><span class="pill pill-blue">${esc(tl(m.project_type))}</span> ${stPill(m.status)} · <a href="#/company/${esc(m.ticker)}">${esc(m.company_name)}</a> · ${esc(m.form_type)} du ${esc(m.filing_date)} · section <b>${esc(m.section_name)}</b></p></div></div> | |
| 166 | + <div class="grid g2"><div class="card"><h3>Extraction</h3><dl class="dl"> | |
| 167 | + <dt>Description</dt><dd>${esc(m.description)}</dd><dt>Objectif</dt><dd>${esc(m.objective) || "—"}</dd> | |
| 168 | + <dt>Montant</dt><dd>${fmtUSD(m.amount_usd)} ${m.amount_raw ? `<span class="muted">(${esc(m.amount_raw)})</span>` : ""}</dd> | |
| 169 | + <dt>Localisations</dt><dd>${(m.locations || []).join(", ") || "—"}</dd><dt>Technologies</dt><dd>${(m.technologies || []).join(", ") || "—"}</dd> | |
| 170 | + <dt>Partenaires</dt><dd>${(m.partners || []).join(", ") || "—"}</dd><dt>Fournisseurs</dt><dd>${(m.suppliers || []).join(", ") || "—"}</dd> | |
| 171 | + <dt>Bénéfices</dt><dd>${(m.benefits || []).join(" · ") || "—"}</dd><dt>Risques</dt><dd>${(m.risks || []).join(" · ") || "—"}</dd> | |
| 172 | + <dt>Années</dt><dd>${(m.years || []).join(", ") || "—"}</dd><dt>Confiance</dt><dd>${(m.confidence ?? 0).toFixed(3)} (backend ${esc(m.backend)})</dd> | |
| 173 | + <dt>Accession</dt><dd class="mono">${esc(m.accession_number)}</dd><dt>Section</dt><dd class="mono">${esc(m.section_id)}</dd><dt>Mention</dt><dd class="mono">${esc(m.mention_id)}</dd></dl></div> | |
| 174 | + <div class="card"><h3>Section source</h3>${d.section ? `<p class="small muted">${esc(d.section.section_name)} · ${fmtN(d.section.word_count)} mots · 10-K</p><button class="btn btn-sm" id="load-sec">Afficher le texte de la section</button><div id="sec"></div>` : `<div class="warn">Le texte source n'est disponible dans la base que pour les sections 10-K ingérées par SPID. Cette mention provient d'un ${esc(m.form_type)} dont la section reste dans le corpus parquet amont.</div>`} | |
| 175 | + <p class="small muted" style="margin-top:12px">Filing sur EDGAR : <a target="_blank" rel="noopener" href="https://www.sec.gov/cgi-bin/browse-edgar?action=getcompany&CIK=${esc(m.cik)}&type=${esc(m.form_type)}&dateb=&owner=include&count=40">liste des ${esc(m.form_type)} de ${esc(m.ticker)} ↗</a></p></div></div>`; | |
| 176 | + $("#load-sec")?.addEventListener("click", async (e) => { | |
| 177 | + e.target.disabled = true; | |
| 178 | + const s = await api("/sections/" + m.section_id); | |
| 179 | + const key = (m.project_name || "").split(/\s+/).filter((w) => w.length > 4)[0]; | |
| 180 | + let txt = esc(s.text); | |
| 181 | + if (key) txt = txt.replace(new RegExp(esc(key).replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "gi"), (x) => `<mark>${x}</mark>`); | |
| 182 | + $("#sec").innerHTML = `<div class="sec-text">${txt}</div>`; | |
| 183 | + }); | |
| 184 | + } | |
| 185 | + | |
| 186 | + /* ------------------------------------------------------------ mentions */ | |
| 187 | + async function mentions() { | |
| 188 | + const { q } = hashParams(); | |
| 189 | + const f = { q: q.q || "", type: q.type || "", form: q.form || "", ticker: q.ticker || "", status: q.status || "", date_from: q.date_from || "", date_to: q.date_to || "", min_amount: q.min_amount || "", min_confidence: q.min_confidence || "", sort: q.sort || "date", order: q.order || "desc", limit: 50, offset: +q.offset || 0 }; | |
| 190 | + const tax = await api("/taxonomy"); | |
| 191 | + view.innerHTML = `<div class="page-head"><div><h1>Mentions</h1><p>Unité d'extraction : un projet cité dans une section d'un filing.</p></div></div> | |
| 192 | + <form class="filters" id="mf"> | |
| 193 | + <label class="span2">Recherche<input name="q" value="${esc(f.q)}"></label> | |
| 194 | + <label>Type<select name="type"><option value="">Tous</option>${tax.map((t) => `<option value="${t.project_type}" ${f.type === t.project_type ? "selected" : ""}>${esc(t.label)}</option>`).join("")}</select></label> | |
| 195 | + <label>Formulaire<select name="form"><option value="">Tous</option>${["10-K", "10-Q", "8-K"].map((x) => `<option ${f.form === x ? "selected" : ""}>${x}</option>`).join("")}</select></label> | |
| 196 | + <label>Statut<select name="status"><option value="">Tous</option>${Object.entries(STATUT).map(([k, v]) => `<option value="${k}" ${f.status === k ? "selected" : ""}>${v}</option>`).join("")}</select></label> | |
| 197 | + <label>Ticker<input name="ticker" value="${esc(f.ticker)}"></label> | |
| 198 | + <label>Du<input type="date" name="date_from" value="${esc(f.date_from)}"></label><label>Au<input type="date" name="date_to" value="${esc(f.date_to)}"></label> | |
| 199 | + <label>Montant min<input type="number" name="min_amount" value="${esc(f.min_amount)}"></label> | |
| 200 | + <label>Confiance min<input type="number" step="0.05" name="min_confidence" value="${esc(f.min_confidence)}"></label> | |
| 201 | + <label> <span><button class="btn btn-sm">Filtrer</button> <button type="button" class="btn btn-ghost btn-sm" id="reset">Effacer</button></span></label></form> | |
| 202 | + <div id="res"><div class="loading">Chargement…</div></div>`; | |
| 203 | + $("#mf").addEventListener("submit", (e) => { e.preventDefault(); setQuery({ ...Object.fromEntries(new FormData(e.target)), sort: f.sort, order: f.order }); }); | |
| 204 | + $("#reset").addEventListener("click", () => (location.hash = "#/mentions")); | |
| 205 | + const data = await api("/mentions", f); | |
| 206 | + const cols = [ | |
| 207 | + { key: "filing_date", label: "Date", sortKey: "date" }, { key: "form_type", label: "Form." }, | |
| 208 | + { key: "ticker", label: "Ticker", sortKey: "ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` }, | |
| 209 | + { key: "project_name", label: "Projet / description", trunc: true, title: (r) => r.description, render: (r) => `<a href="#/mention/${esc(r.mention_id)}">${esc(r.project_name)}</a><div class="muted small">${esc(r.description)}</div>` }, | |
| 210 | + { key: "project_type", label: "Type", render: (r) => `<span class="pill pill-blue">${esc(tl(r.project_type))}</span>` }, | |
| 211 | + { key: "status", label: "Statut", render: (r) => stPill(r.status) }, { key: "section_name", label: "Section" }, | |
| 212 | + { key: "amount_usd", label: "Montant", num: true, sortKey: "amount", render: (r) => fmtUSD(r.amount_usd) }, | |
| 213 | + { key: "confidence", label: "Conf.", num: true, sortKey: "confidence", render: (r) => (r.confidence ?? 0).toFixed(2) }, | |
| 214 | + ]; | |
| 215 | + const res = $("#res"); | |
| 216 | + res.innerHTML = `<p class="muted">${fmtN(data.total)} mentions.</p>` + table(cols, data.items, { href: (r) => `#/mention/${r.mention_id}`, sort: f.sort, order: f.order }) + pager(data.total, f.limit, f.offset); | |
| 217 | + bindTable(res, (k) => setQuery({ ...f, offset: 0, sort: k, order: f.sort === k && f.order === "desc" ? "asc" : "desc" })); | |
| 218 | + res.querySelectorAll("[data-off]").forEach((b) => b.addEventListener("click", () => setQuery({ ...f, offset: b.dataset.off }))); | |
| 219 | + } | |
| 220 | + | |
| 221 | + /* ------------------------------------------------------------ entreprises */ | |
| 222 | + async function companies() { | |
| 223 | + const { q } = hashParams(); | |
| 224 | + const f = { q: q.q || "", sector: q.sector || "", sort: q.sort || "projects", order: q.order || "desc", limit: 100, offset: +q.offset || 0 }; | |
| 225 | + const secs = await api("/sectors"); | |
| 226 | + view.innerHTML = `<div class="page-head"><div><h1>Entreprises</h1><p>500 sociétés du S&P 500 et la taille de leur portefeuille de projets divulgués.</p></div></div> | |
| 227 | + <form class="filters" id="cf"><label class="span2">Recherche<input name="q" value="${esc(f.q)}" placeholder="nom ou ticker"></label> | |
| 228 | + <label>Secteur<select name="sector"><option value="">Tous</option>${secs.map((s) => `<option ${f.sector === s.sector ? "selected" : ""}>${esc(s.sector)}</option>`).join("")}</select></label> | |
| 229 | + <label> <span><button class="btn btn-sm">Filtrer</button></span></label></form><div id="res"></div>`; | |
| 230 | + $("#cf").addEventListener("submit", (e) => { e.preventDefault(); setQuery({ ...Object.fromEntries(new FormData(e.target)), sort: f.sort, order: f.order }); }); | |
| 231 | + const data = await api("/companies", f); | |
| 232 | + const cols = [ | |
| 233 | + { key: "ticker", label: "Ticker", sortKey: "ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` }, | |
| 234 | + { key: "company_name", label: "Entreprise" }, { key: "sector", label: "Secteur" }, | |
| 235 | + { key: "n_projects", label: "Projets", num: true, sortKey: "projects", render: (r) => fmtN(r.n_projects) }, | |
| 236 | + { key: "n_mentions", label: "Mentions", num: true, sortKey: "mentions", render: (r) => fmtN(r.n_mentions) }, | |
| 237 | + { key: "n_types", label: "Types", num: true }, { key: "amount_usd", label: "Capital divulgué", num: true, sortKey: "amount", render: (r) => fmtUSD(r.amount_usd) }, | |
| 238 | + { key: "first_seen", label: "Période", render: (r) => `${esc(r.first_seen?.slice(0, 4))}–${esc(r.last_seen?.slice(0, 4))}` }, | |
| 239 | + ]; | |
| 240 | + const res = $("#res"); | |
| 241 | + res.innerHTML = table(cols, data.items, { href: (r) => `#/company/${r.ticker}`, sort: f.sort, order: f.order }) + pager(data.total, f.limit, f.offset); | |
| 242 | + bindTable(res, (k) => setQuery({ ...f, offset: 0, sort: k, order: f.sort === k && f.order === "desc" ? "asc" : "desc" })); | |
| 243 | + res.querySelectorAll("[data-off]").forEach((b) => b.addEventListener("click", () => setQuery({ ...f, offset: b.dataset.off }))); | |
| 244 | + } | |
| 245 | + async function company(ident) { | |
| 246 | + const d = await api("/companies/" + encodeURIComponent(ident)); | |
| 247 | + const c = d.company; | |
| 248 | + view.innerHTML = `<div class="page-head"><div><div class="muted small"><a href="#/companies">Entreprises</a></div><h1>${esc(c.company_name)} <span class="muted">(${esc(c.ticker)})</span></h1> | |
| 249 | + <p>${esc(c.sector || "")} · CIK ${esc(c.cik)} · ${fmtN(c.n_projects)} projets · ${fmtN(c.n_mentions)} mentions · ${esc(c.first_seen?.slice(0, 4))}–${esc(c.last_seen?.slice(0, 4))}</p></div> | |
| 250 | + <div><a class="btn btn-ghost btn-sm" href="#/projects?ticker=${esc(c.ticker)}">Filtrer les projets</a> <a class="btn btn-ghost btn-sm" href="#/graph?node=${encodeURIComponent("C:" + c.cik)}">Graphe</a></div></div> | |
| 251 | + <div class="kpis">${kpi(c.n_projects, "projets")}${kpi(c.n_mentions, "mentions")}${kpi(fmtUSD(c.amount_usd), "capital divulgué (Σ max)")}${kpi(d.by_type.length, "types de projets")}</div> | |
| 252 | + <div class="grid g3"> | |
| 253 | + <div class="card"><h3>Par type</h3><div class="chart"><canvas id="c1"></canvas></div></div> | |
| 254 | + <div class="card"><h3>Mentions par année</h3><div class="chart"><canvas id="c2"></canvas></div></div> | |
| 255 | + <div class="card"><h3>Lieux, technologies, partenaires</h3> | |
| 256 | + <div class="tags">${d.locations.map((x) => `<a class="tag t-loc" href="#/projects?ticker=${esc(c.ticker)}&location=${encodeURIComponent(x.location)}">${esc(x.location)} ${x.n}</a>`).join("")}</div><br> | |
| 257 | + <div class="tags">${d.technologies.map((x) => `<a class="tag t-tech" href="#/projects?ticker=${esc(c.ticker)}&tech=${encodeURIComponent(x.technology)}">${esc(x.technology)} ${x.n}</a>`).join("")}</div><br> | |
| 258 | + <div class="tags">${d.partners.map((x) => `<span class="tag t-part">${esc(x.partner)} ${x.n}</span>`).join("")}</div></div> | |
| 259 | + </div> | |
| 260 | + <h2 style="margin-top:18px">Portefeuille de projets (${fmtN(d.projects.length)})</h2><div id="res"></div>`; | |
| 261 | + chart($("#c1"), { type: "bar", data: { labels: d.by_type.map((r) => r.label), datasets: [{ data: d.by_type.map((r) => r.n), backgroundColor: COL.bleu }] }, options: { indexAxis: "y", responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } } } }); | |
| 262 | + chart($("#c2"), { type: "bar", data: { labels: d.by_year.map((r) => r.year), datasets: [{ data: d.by_year.map((r) => r.n), backgroundColor: COL.or }] }, options: { responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } } } }); | |
| 263 | + const res = $("#res"); res.innerHTML = table(PCOLS.filter((c) => c.key !== "ticker"), d.projects, { href: (r) => `#/project/${r.project_id}` }); bindTable(res); | |
| 264 | + } | |
| 265 | + | |
| 266 | + /* ------------------------------------------------------------ graphe */ | |
| 267 | + async function graph() { | |
| 268 | + const { q } = hashParams(); | |
| 269 | + view.innerHTML = `<div class="page-head"><div><h1>Graphe de connaissances</h1><p>32 226 nœuds (entreprises, projets, lieux, technologies, partenaires) et 33 176 arêtes <span class="mono">owns · located_in · uses</span>. Cherchez un nœud puis naviguez de voisin en voisin.</p></div></div> | |
| 270 | + <form class="filters" id="gf"><label class="span2">Nœud<input name="q" value="${esc(q.q || "")}" placeholder="Texas, Tesla, batterie, AI…"></label> | |
| 271 | + <label>Type<select name="type"><option value="">Tous</option>${["Company", "Project", "Location", "Technology", "Partner"].map((t) => `<option ${q.type === t ? "selected" : ""}>${t}</option>`).join("")}</select></label> | |
| 272 | + <label> <span><button class="btn btn-sm">Chercher</button></span></label></form><div id="hits"></div><div id="node"></div>`; | |
| 273 | + $("#gf").addEventListener("submit", (e) => { e.preventDefault(); const d = Object.fromEntries(new FormData(e.target)); location.hash = "#/graph?" + qs(d); }); | |
| 274 | + if (q.node) await showNode(q.node); | |
| 275 | + else if (q.q) { | |
| 276 | + const hits = await api("/graph/search", { q: q.q, type: q.type || "", limit: 60 }); | |
| 277 | + $("#hits").innerHTML = `<div class="card"><h3>${hits.length} nœuds</h3>${hits.map((h) => `<a class="node-chip nt-${esc(h.node_type)}" href="#/graph?node=${encodeURIComponent(h.node_id)}"><span class="nt">${esc(h.node_type)}</span>${esc(h.label)} <span class="muted small">· ${fmtN(h.degree)}</span></a>`).join("") || "<span class='muted'>Aucun résultat.</span>"}</div>`; | |
| 278 | + } else { | |
| 279 | + $("#hits").innerHTML = `<div class="card"><h3>Points d'entrée</h3><div>${[["L:texas", "Location"], ["L:china", "Location"], ["L:united states", "Location"], ["T:AI", "Technology"], ["T:Renewable Energy", "Technology"], ["T:Battery/Storage", "Technology"], ["T:Data Center", "Technology"], ["C:0001318605", "Company"], ["C:0000045012", "Company"]].map(([n, t]) => `<a class="node-chip nt-${t}" href="#/graph?node=${encodeURIComponent(n)}"><span class="nt">${t}</span>${esc(n.slice(2))}</a>`).join("")}</div></div>`; | |
| 280 | + } | |
| 281 | + } | |
| 282 | + async function showNode(nodeId) { | |
| 283 | + const d = await api("/graph/node/" + nodeId, { limit: 400 }); | |
| 284 | + const n = d.node; const props = safeJson(n.props); | |
| 285 | + const groups = {}; d.edges.forEach((e) => ((groups[e.neighbor_type] ||= []).push(e))); | |
| 286 | + const target = n.node_type === "Project" ? `<a class="btn btn-sm" href="#/project/${esc(n.node_id.slice(2))}">Fiche du projet</a>` : n.node_type === "Company" ? `<a class="btn btn-sm" href="#/company/${esc(n.node_id.slice(2))}">Fiche de l'entreprise</a>` : n.node_type === "Location" ? `<a class="btn btn-sm" href="#/projects?location=${encodeURIComponent(n.label)}">Projets à cet endroit</a>` : n.node_type === "Technology" ? `<a class="btn btn-sm" href="#/projects?tech=${encodeURIComponent(n.label)}">Projets avec cette technologie</a>` : `<a class="btn btn-sm" href="#/projects?partner=${encodeURIComponent(n.label)}">Projets avec ce partenaire</a>`; | |
| 287 | + $("#node").innerHTML = `<div class="card"><div class="page-head"><div><span class="pill pill-blue">${esc(n.node_type)}</span> <h2 style="display:inline">${esc(n.label)}</h2><div class="muted small mono">${esc(n.node_id)} · degré ${fmtN(d.degree)}</div> | |
| 288 | + ${props ? `<div class="small muted">${Object.entries(props).filter(([, v]) => v != null).map(([k, v]) => `${esc(k)} = <b>${esc(k.includes("amount") ? fmtUSD(v) : v)}</b>`).join(" · ")}</div>` : ""}</div><div>${target}</div></div> | |
| 289 | + <svg class="svg-graph" id="svg"></svg> | |
| 290 | + ${Object.entries(groups).map(([t, es]) => `<h3 style="margin-top:12px">${esc(t)} (${es.length})</h3><div>${es.slice(0, 200).map((e) => `<a class="node-chip nt-${esc(t)}" href="#/graph?node=${encodeURIComponent(e.neighbor_id)}"><span class="nt">${esc(e.rel)} ${e.direction === "out" ? "→" : "←"}</span>${esc(e.neighbor_label)}</a>`).join("")}${es.length > 200 ? `<span class="muted small"> … ${es.length - 200} de plus</span>` : ""}</div>`).join("") || "<p class='muted'>Ce nœud n'a pas d'arête (les partenaires ne sont pas encore reliés dans cette version du graphe).</p>"}</div>`; | |
| 291 | + drawRadial($("#svg"), n, d.edges.slice(0, 48)); | |
| 292 | + } | |
| 293 | + function safeJson(s) { if (!s) return null; try { return typeof s === "string" ? JSON.parse(s) : s; } catch { return null; } } | |
| 294 | + function drawRadial(svg, center, edges) { | |
| 295 | + const W = svg.clientWidth || 900, H = 520, cx = W / 2, cy = H / 2; | |
| 296 | + const colors = { Company: COL.bleu, Project: COL.violet, Location: COL.orange, Technology: COL.teal, Partner: COL.or }; | |
| 297 | + const n = edges.length; const R = Math.min(W, H) / 2 - 70; | |
| 298 | + let s = `<g>`; | |
| 299 | + edges.forEach((e, i) => { const a = (2 * Math.PI * i) / n - Math.PI / 2; const x = cx + R * Math.cos(a), y = cy + R * Math.sin(a); s += `<line x1="${cx}" y1="${cy}" x2="${x}" y2="${y}"></line>`; }); | |
| 300 | + edges.forEach((e, i) => { const a = (2 * Math.PI * i) / n - Math.PI / 2; const x = cx + R * Math.cos(a), y = cy + R * Math.sin(a); const lab = e.neighbor_label.length > 28 ? e.neighbor_label.slice(0, 26) + "…" : e.neighbor_label; const anchor = Math.cos(a) > 0.2 ? "start" : Math.cos(a) < -0.2 ? "end" : "middle"; s += `<a href="#/graph?node=${encodeURIComponent(e.neighbor_id)}"><circle cx="${x}" cy="${y}" r="6" fill="${colors[e.neighbor_type] || COL.gris}"></circle><text x="${x + 10 * Math.cos(a)}" y="${y + 10 * Math.sin(a) + 4}" text-anchor="${anchor}">${esc(lab)}</text></a>`; }); | |
| 301 | + s += `<circle cx="${cx}" cy="${cy}" r="14" fill="${colors[center.node_type] || COL.gris}" stroke="#fff" stroke-width="3"></circle><text x="${cx}" y="${cy + 30}" text-anchor="middle" font-weight="700">${esc(center.label.slice(0, 40))}</text></g>`; | |
| 302 | + svg.setAttribute("viewBox", `0 0 ${W} ${H}`); svg.innerHTML = s; | |
| 303 | + } | |
| 304 | + | |
| 305 | + /* ------------------------------------------------------------ recherche sémantique */ | |
| 306 | + async function semantic() { | |
| 307 | + const { q } = hashParams(); | |
| 308 | + view.innerHTML = `<div class="page-head"><div><h1>Recherche sémantique</h1><p>Décrivez un projet en langage naturel (anglais recommandé, langue des filings) : la requête est encodée avec all-MiniLM-L6-v2 et comparée aux 19 227 vecteurs de projets.</p></div></div> | |
| 309 | + <form class="filters" id="sf"><label class="span2" style="grid-column:span 3">Requête<input name="q" value="${esc(q.q || "")}" placeholder="ex. battery cell factory in the United States, AI platform for banks, LNG export terminal…"></label> | |
| 310 | + <label> <span><button class="btn btn-sm">Chercher</button></span></label></form><div id="res"></div>`; | |
| 311 | + $("#sf").addEventListener("submit", (e) => { e.preventDefault(); location.hash = "#/semantic?" + qs(Object.fromEntries(new FormData(e.target))); }); | |
| 312 | + if (!q.q) { $("#res").innerHTML = `<div class="examples">${["gigafactory for battery cells", "generative AI assistant for customers", "offshore wind farm", "ERP migration to SAP S/4HANA", "new hospital or clinical trial for oncology drug", "data center campus in Virginia", "restructuring plan with plant closures"].map((x) => `<button data-q="${esc(x)}">${esc(x)}</button>`).join("")}</div>`; $("#res").querySelectorAll("button").forEach((b) => b.addEventListener("click", () => (location.hash = "#/semantic?q=" + encodeURIComponent(b.dataset.q)))); return; } | |
| 313 | + $("#res").innerHTML = `<div class="loading">Encodage et recherche…</div>`; | |
| 314 | + try { | |
| 315 | + const d = await api("/search/semantic", { q: q.q, k: 40 }); | |
| 316 | + const cols = [{ key: "score", label: "Score", num: true, render: (r) => `<span class="score">${(r.score * 100).toFixed(1)} %</span>` }, ...PCOLS.filter((c) => !["avg_confidence", "first_seen"].includes(c.key))]; | |
| 317 | + const res = $("#res"); res.innerHTML = table(cols, d.items, { href: (r) => `#/project/${r.project_id}` }); bindTable(res); | |
| 318 | + } catch (e) { $("#res").innerHTML = `<div class="warn">${esc(e.message)}</div><p class="muted">La similarité entre projets reste disponible sur chaque fiche (« Projets similaires »), car elle n'utilise que les vecteurs déjà stockés.</p>`; } | |
| 319 | + } | |
| 320 | + | |
| 321 | + /* ------------------------------------------------------------ SQL */ | |
| 322 | + const SQL_EXAMPLES = [ | |
| 323 | + ["Projets par type", "select project_type, count(*) n, round(sum(total_amount_usd)/1e9,1) capital_gusd\nfrom projects group by 1 order by n desc"], | |
| 324 | + ["Centres de données > 500 M$", "select ticker, project_name, canonical_location, total_amount_usd, first_seen\nfrom projects where project_type='data_center' and total_amount_usd > 5e8\norder by total_amount_usd desc"], | |
| 325 | + ["Mentions IA par année", "select year(filing_date) y, count(*) n\nfrom project_mentions where project_type='ai_initiative' group by 1 order by 1"], | |
| 326 | + ["Technologies les plus citées", "select lower(t) technology, count(*) n\nfrom projects, unnest(technologies) as u(t) group by 1 order by n desc limit 25"], | |
| 327 | + ["Partenaires de Tesla", "select p partner, count(*) n from project_mentions, unnest(partners) as u(p)\nwhere ticker='TSLA' group by 1 order by n desc"], | |
| 328 | + ["Durée de suivi médiane par type", "select project_type, median(last_year-first_year) mediane_annees, count(*) n\nfrom projects where last_year>=first_year group by 1 order by 2 desc"], | |
| 329 | + ["Graphe : lieux les plus connectés", "select n.label, count(*) degre from kg_edges e join kg_nodes n on n.node_id=e.dst\nwhere e.rel='located_in' group by 1 order by 2 desc limit 20"], | |
| 330 | + ["Texte 10-K : sections les plus longues", "select s.accession_number, s.section_name, s.word_count\nfrom spid_sections s order by word_count desc limit 10"], | |
| 331 | + ]; | |
| 332 | + async function sqlPage() { | |
| 333 | + const schema = await api("/schema"); | |
| 334 | + const saved = localStorage.getItem("pdb_sql") || SQL_EXAMPLES[0][1]; | |
| 335 | + view.innerHTML = `<div class="page-head"><div><h1>Bac à sable SQL</h1><p>DuckDB en lecture seule sur les 7 tables de la base. SELECT / WITH / DESCRIBE / SUMMARIZE ; 20 s et 5 000 lignes au maximum. ⌘⏎ pour exécuter.</p></div></div> | |
| 336 | + <div class="sql-layout"><div class="card schema"><h3>Schéma</h3>${schema.map((t) => `<details ${t.table === "projects" ? "open" : ""}><summary>${esc(t.table)} <span class="muted">(${fmtN(t.rows)})</span></summary><ul>${t.columns.map((c) => `<li>${esc(c.name)} <span>${esc(c.type)}</span></li>`).join("")}</ul></details>`).join("")}</div> | |
| 337 | + <div><textarea class="sql" id="sql">${esc(saved)}</textarea> | |
| 338 | + <div class="examples">${SQL_EXAMPLES.map((e, i) => `<button data-i="${i}">${esc(e[0])}</button>`).join("")}</div> | |
| 339 | + <div style="display:flex;gap:10px;align-items:center;margin:6px 0 12px"><button class="btn" id="run">Exécuter</button><label class="small muted">Limite <input class="inp" id="lim" type="number" value="500" min="1" max="5000" style="width:90px"></label><button class="btn btn-ghost btn-sm" id="csv" disabled>Télécharger CSV</button><span class="muted small" id="info"></span></div> | |
| 340 | + <div id="out"></div></div></div>`; | |
| 341 | + const ta = $("#sql"); | |
| 342 | + view.querySelectorAll(".examples button").forEach((b) => b.addEventListener("click", () => { ta.value = SQL_EXAMPLES[b.dataset.i][1]; run(); })); | |
| 343 | + let last = null; | |
| 344 | + async function run() { | |
| 345 | + const sql = ta.value; localStorage.setItem("pdb_sql", sql); | |
| 346 | + $("#out").innerHTML = `<div class="loading">Exécution…</div>`; $("#info").textContent = ""; | |
| 347 | + try { | |
| 348 | + const r = await fetch(API + "/sql", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ sql, limit: +$("#lim").value || 500 }) }); | |
| 349 | + const d = await r.json(); | |
| 350 | + if (!r.ok) throw new Error(d.detail || r.statusText); | |
| 351 | + last = d; | |
| 352 | + $("#info").textContent = `${fmtN(d.rows.length)} lignes${d.truncated ? " (tronqué)" : ""} · ${d.elapsed_ms} ms`; $("#csv").disabled = false; | |
| 353 | + $("#out").innerHTML = `<div class="table-wrap"><table><thead><tr>${d.columns.map((c) => `<th>${esc(c)}</th>`).join("")}</tr></thead><tbody>${d.rows.map((row) => `<tr>${row.map((v) => `<td class="${typeof v === "number" ? "num" : ""}">${esc(Array.isArray(v) ? v.join(" | ") : typeof v === "object" && v ? JSON.stringify(v) : v)}</td>`).join("")}</tr>`).join("")}</tbody></table></div>`; | |
| 354 | + } catch (e) { $("#out").innerHTML = `<div class="err">${esc(e.message)}</div>`; } | |
| 355 | + } | |
| 356 | + $("#run").addEventListener("click", run); | |
| 357 | + ta.addEventListener("keydown", (e) => { if ((e.metaKey || e.ctrlKey) && e.key === "Enter") run(); }); | |
| 358 | + $("#csv").addEventListener("click", () => { if (!last) return; const csv = [last.columns.join(",")].concat(last.rows.map((r) => r.map((v) => `"${String(Array.isArray(v) ? v.join("|") : v ?? "").replace(/"/g, '""')}"`).join(","))).join("\n"); const a = document.createElement("a"); a.href = URL.createObjectURL(new Blob([csv], { type: "text/csv" })); a.download = "pdb_query.csv"; a.click(); }); | |
| 359 | + run(); | |
| 360 | + } | |
| 361 | + | |
| 362 | + /* ------------------------------------------------------------ à propos */ | |
| 363 | + async function about() { | |
| 364 | + view.innerHTML = `<div class="prose"><h1>La base SPID et son pipeline</h1> | |
| 365 | + <p><b>SPID</b> (<i>SEC Project Intelligence Database</i>) reconstruit le portefeuille de projets stratégiques des grandes sociétés cotées américaines à partir du texte brut de leurs documents réglementaires (10-K annuels, 10-Q trimestriels, 8-K d'événements) déposés auprès de la SEC sur EDGAR. Plutôt que de lire ces filings comme des documents financiers, SPID les traite comme une source d'intelligence sur les <i>projets</i> : usines, centres de données, acquisitions, programmes de R&D, initiatives d'IA, transition énergétique…</p> | |
| 366 | + <h2>D'où viennent les données</h2> | |
| 367 | + <ul><li><b>Univers</b> : les 500 sociétés du S&P 500 (instantané 2026), chacune reliée à son identifiant SEC (CIK).</li> | |
| 368 | + <li><b>Corpus</b> : 134 124 dépôts (103 927 8-K, 22 542 10-Q, 7 655 10-K) téléchargés depuis l'API publique de la SEC, janvier 2010 → mai 2026, soit 88 Go de HTML brut, nettoyés et découpés en sections.</li> | |
| 369 | + <li><b>Sections analysées</b> : 153 894 sections d'au moins 30 mots (les 10-K sont découpés par SPID en Items 1, 1A, 2, 7 et 7A ; les 8-K sont traités en une seule section).</li></ul> | |
| 370 | + <h2>Comment fonctionne le pipeline</h2> | |
| 371 | + <ol><li><b>Ingestion</b> des 10-K bruts (nettoyage HTML, découpage en Items).</li> | |
| 372 | + <li><b>Extraction</b> par grand modèle de langage (gpt-4o-mini, température 0, JSON strict) : un pré-filtre lexical écarte les sections sans signal de projet, une fenêtre de focalisation limite le texte envoyé ; le modèle renvoie nom, type (taxonomie de 21 catégories), description, objectif, statut, montant, lieux, technologies, partenaires, bénéfices, risques, années et confiance. Les mentions sous 0,45 de confiance sont écartées.</li> | |
| 373 | + <li><b>Résolution</b> : les mentions sont fusionnées en projets par entreprise × type × ancre (première localisation ou premier mot significatif du nom) ; chaque projet reçoit une chronologie, un statut courant, le montant maximal divulgué et des dates de première et dernière observation.</li> | |
| 374 | + <li><b>Graphe de connaissances</b> : entreprises, projets, lieux, technologies et partenaires reliés par <span class="mono">owns</span>, <span class="mono">located_in</span>, <span class="mono">uses</span>.</li> | |
| 375 | + <li><b>Embeddings</b> : chaque projet est encodé en 384 dimensions (all-MiniLM-L6-v2) pour la recherche sémantique et les projets similaires.</li></ol> | |
| 376 | + <h2>Ce que contient la base</h2> | |
| 377 | + <p>39 930 mentions, 19 227 projets, 39 930 points de chronologie, 32 226 nœuds et 33 176 arêtes, 19 227 vecteurs, 27 164 sections 10-K en texte intégral. Un fichier DuckDB de 2,9 Go, interrogeable ici en SQL.</p> | |
| 378 | + <h2>Limites à garder en tête</h2> | |
| 379 | + <ul><li>SPID mesure la <i>divulgation</i> de projets, non les projets eux-mêmes ; les incitations à divulguer varient selon les secteurs.</li> | |
| 380 | + <li>Les montants sont auto-déclarés, hétérogènes en portée et absents pour la moitié des projets ; quelques valeurs aberrantes subsistent.</li> | |
| 381 | + <li>L'extraction par LLM comporte des faux positifs et négatifs ; la résolution est heuristique (elle peut scinder ou fusionner à tort).</li> | |
| 382 | + <li>Les années citées (first_year / last_year) peuvent inclure des horizons projetés ; préférez first_seen / last_seen pour dater l'observation.</li></ul> | |
| 383 | + <p>Le <a href="/report.pdf" target="_blank">rapport technique complet (PDF)</a> détaille la provenance, chaque étape du pipeline, le schéma et le portrait statistique. Équipe : Manel Kammoun, Charli Tandja Mbianda, Simon-Pierre Boucher — Département des sciences administratives, Université du Québec en Outaouais.</p></div>`; | |
| 384 | + } | |
| 385 | + | |
| 386 | + /* ------------------------------------------------------------ routeur */ | |
| 387 | + const routes = { "": dashboard, projects, companies, mentions, graph, semantic, sql: sqlPage, api: () => window.PDBPlayground.render({ view, api, esc, fmtN, fmtUSD, qs, toast, tl }), about }; | |
| 388 | + async function render() { | |
| 389 | + destroyCharts(); | |
| 390 | + const { parts } = hashParams(); | |
| 391 | + const r = parts[0]; | |
| 392 | + document.querySelectorAll("#nav a[data-route]").forEach((a) => a.classList.toggle("active", a.dataset.route === r)); | |
| 393 | + $("#burger") && $(".side").classList.remove("open"); | |
| 394 | + window.scrollTo(0, 0); | |
| 395 | + try { | |
| 396 | + if (r === "project" && parts[1]) await project(parts[1]); | |
| 397 | + else if (r === "company" && parts[1]) await company(parts[1]); | |
| 398 | + else if (r === "mention" && parts[1]) await mention(parts[1]); | |
| 399 | + else if (routes[r]) await routes[r](); | |
| 400 | + else view.innerHTML = `<div class="card"><h2>Page introuvable</h2><a href="#/">Retour au tableau de bord</a></div>`; | |
| 401 | + } catch (e) { view.innerHTML = `<div class="err">Erreur : ${esc(e.message)}</div>`; } | |
| 402 | + } | |
| 403 | + async function init() { | |
| 404 | + try { const tax = await api("/taxonomy"); tax.forEach((t) => (TYPE[t.project_type] = t.label)); } catch { } | |
| 405 | + try { const h = await fetch("/health").then((r) => r.json()); $("#kpi-projects").textContent = fmtN(h.projects) + " projets"; $("#side-status").textContent = `service ${h.status} · sémantique : ${h.semantic}`; } catch { $("#side-status").textContent = "service indisponible"; } | |
| 406 | + fetch("/report.pdf", { method: "HEAD" }).then((r) => { if (!r.ok) $("#report-link").remove(); }).catch(() => { }); | |
| 407 | + $("#global-search").addEventListener("submit", (e) => { e.preventDefault(); const v = $("#global-q").value.trim(); if (v) location.hash = "#/projects?q=" + encodeURIComponent(v); }); | |
| 408 | + document.addEventListener("keydown", (e) => { if ((e.metaKey || e.ctrlKey) && e.key.toLowerCase() === "k") { e.preventDefault(); $("#global-q").focus(); } }); | |
| 409 | + $("#burger").addEventListener("click", () => $(".side").classList.toggle("open")); | |
| 410 | + window.addEventListener("hashchange", render); | |
| 411 | + render(); | |
| 412 | + } | |
| 413 | + init(); | |
| 414 | +})(); | |
added
web/index.html
+62 −0
@@ -0,0 +1,62 @@ | ||
| 1 | +<!doctype html> | |
| 2 | +<html lang="fr"> | |
| 3 | +<head> | |
| 4 | +<meta charset="utf-8"> | |
| 5 | +<meta name="viewport" content="width=device-width, initial-scale=1"> | |
| 6 | +<title>PDB — SEC Project Intelligence Database · Explorateur</title> | |
| 7 | +<meta name="description" content="Explorateur et API de la base SPID : 19 227 projets stratégiques extraits des filings SEC de 500 sociétés du S&P 500 (2010–2026). UQO — Département des sciences administratives."> | |
| 8 | +<link rel="icon" href="data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 32 32'%3E%3Crect width='32' height='32' rx='7' fill='%23003E7E'/%3E%3Ctext x='16' y='22' font-family='Helvetica,Arial' font-size='15' font-weight='700' fill='%23C6A300' text-anchor='middle'%3EPDB%3C/text%3E%3C/svg%3E"> | |
| 9 | +<link rel="preconnect" href="https://cdn.jsdelivr.net"> | |
| 10 | +<link rel="stylesheet" href="/style.css"> | |
| 11 | +<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.4/dist/chart.umd.min.js" defer></script> | |
| 12 | +<script src="/playground.js" defer></script> | |
| 13 | +<script src="/app.js" defer></script> | |
| 14 | +</head> | |
| 15 | +<body> | |
| 16 | +<div class="shell"> | |
| 17 | + <aside class="side"> | |
| 18 | + <a class="brand" href="#/"> | |
| 19 | + <span class="brand-mark">PDB</span> | |
| 20 | + <span class="brand-text"><b>SEC Project Intelligence</b><small>Database · UQO</small></span> | |
| 21 | + </a> | |
| 22 | + <nav class="nav" id="nav"> | |
| 23 | + <a href="#/" data-route="">Tableau de bord</a> | |
| 24 | + <a href="#/projects" data-route="projects">Projets</a> | |
| 25 | + <a href="#/companies" data-route="companies">Entreprises</a> | |
| 26 | + <a href="#/mentions" data-route="mentions">Mentions</a> | |
| 27 | + <a href="#/graph" data-route="graph">Graphe</a> | |
| 28 | + <a href="#/semantic" data-route="semantic">Recherche sémantique</a> | |
| 29 | + <div class="nav-sep">Développeurs</div> | |
| 30 | + <a href="#/sql" data-route="sql">Bac à sable SQL</a> | |
| 31 | + <a href="#/api" data-route="api">API & playground</a> | |
| 32 | + <a href="/docs" target="_blank" rel="noopener">Documentation OpenAPI ↗</a> | |
| 33 | + <div class="nav-sep">À propos</div> | |
| 34 | + <a href="#/about" data-route="about">La base et le pipeline</a> | |
| 35 | + <a href="/report.pdf" target="_blank" rel="noopener" id="report-link">Rapport technique (PDF) ↗</a> | |
| 36 | + </nav> | |
| 37 | + <div class="side-foot"> | |
| 38 | + <div>Université du Québec en Outaouais<br>Département des sciences administratives</div> | |
| 39 | + <div class="muted" id="side-status">connexion…</div> | |
| 40 | + </div> | |
| 41 | + </aside> | |
| 42 | + <main class="main"> | |
| 43 | + <header class="topbar"> | |
| 44 | + <button class="burger" id="burger" aria-label="Menu">☰</button> | |
| 45 | + <form class="search" id="global-search" autocomplete="off"> | |
| 46 | + <input id="global-q" type="search" placeholder="Rechercher un projet, une entreprise, un ticker… (⌘K)"> | |
| 47 | + </form> | |
| 48 | + <div class="topbar-right"> | |
| 49 | + <span class="pill pill-blue" id="kpi-projects">…</span> | |
| 50 | + <a class="btn btn-ghost" href="#/api">Obtenir l'API</a> | |
| 51 | + </div> | |
| 52 | + </header> | |
| 53 | + <section id="view" class="view"><div class="loading">Chargement…</div></section> | |
| 54 | + <footer class="foot"> | |
| 55 | + <span>PDB API · données SPID (filings SEC 10-K · 10-Q · 8-K, 500 sociétés, 2010–2026) · extraction par LLM, résolution, graphe, embeddings.</span> | |
| 56 | + <span>Kammoun · Tandja Mbianda · Boucher — UQO</span> | |
| 57 | + </footer> | |
| 58 | + </main> | |
| 59 | +</div> | |
| 60 | +<div id="toast" class="toast" hidden></div> | |
| 61 | +</body> | |
| 62 | +</html> | |
added
web/playground.js
+337 −0
@@ -0,0 +1,337 @@ | ||
| 1 | +/* PDB — page API & playground (démarrage rapide, playground, recettes, référence, SDK). */ | |
| 2 | +window.PDBPlayground = (() => { | |
| 3 | + const LANGS = [["curl", "curl"], ["py", "Python"], ["js", "JavaScript"], ["r", "R"]]; | |
| 4 | + const KEY_PH = "VOTRE_CLE"; | |
| 5 | + const key = () => localStorage.getItem("pdb_key") || ""; | |
| 6 | + const lang = () => localStorage.getItem("pdb_lang") || "py"; | |
| 7 | + const esc = (s) => String(s ?? "").replace(/[&<>"']/g, (c) => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" }[c])); | |
| 8 | + const qs = (o) => Object.entries(o).filter(([, v]) => v !== "" && v != null).map(([k, v]) => `${encodeURIComponent(k)}=${encodeURIComponent(v)}`).join("&"); | |
| 9 | + const fmtN = (n) => n == null ? "—" : Number(n).toLocaleString("fr-CA"); | |
| 10 | + | |
| 11 | + /* ------------------------------------------------------------------ catalogue des routes */ | |
| 12 | + const P = (name, desc, opts = {}) => ({ name, desc, ...opts }); | |
| 13 | + const PROJECT_FILTERS = [ | |
| 14 | + P("q", "texte libre : nom, description, entreprise, ticker"), P("type", "project_type, liste séparée par des virgules", { enum: "types" }), | |
| 15 | + P("sector", "secteur GICS (contient)", { enum: "sectors" }), P("status", "planned | in_progress | completed | mentioned (liste possible)"), | |
| 16 | + P("ticker", "ticker exact"), P("cik", "CIK SEC (10 chiffres ou moins)"), P("location", "localisation canonique (contient)"), | |
| 17 | + P("tech", "technologie (contient)"), P("partner", "partenaire cité dans une mention (contient)"), | |
| 18 | + P("min_amount", "montant divulgué minimal, US$", { type: "number" }), P("max_amount", "montant maximal, US$", { type: "number" }), | |
| 19 | + P("year_from", "dernière observation ≥ année", { type: "number" }), P("year_to", "première observation ≤ année", { type: "number" }), | |
| 20 | + P("min_confidence", "confiance moyenne minimale (0–1)", { type: "number" }), P("has_amount", "true : seulement les projets avec montant"), | |
| 21 | + ]; | |
| 22 | + const PAGE = [P("sort", "clé de tri"), P("order", "asc | desc"), P("limit", "≤ 500 (défaut 50)", { type: "number" }), P("offset", "décalage", { type: "number" })]; | |
| 23 | + const ENDPOINTS = [ | |
| 24 | + { g: "Découverte", m: "GET", p: "/v1/health", key: false, d: "État du service : compteur de projets, état du modèle sémantique. Sans clé.", params: [], ex: [{ l: "État", v: {} }] }, | |
| 25 | + { g: "Découverte", m: "GET", p: "/v1/stats", d: "Vue d'ensemble : compteurs, répartitions par type, secteur, statut, formulaire, année, thèmes émergents, top entreprises/technologies/lieux. C'est la source du tableau de bord.", params: [], ex: [{ l: "Tout", v: {} }] }, | |
| 26 | + { g: "Découverte", m: "GET", p: "/v1/taxonomy", d: "Les 21 types de projets (clé, libellé, nombre de projets, capital divulgué).", params: [], ex: [{ l: "Liste", v: {} }] }, | |
| 27 | + { g: "Découverte", m: "GET", p: "/v1/sectors", d: "Les 11 secteurs GICS avec nombre de projets, d'entreprises et capital.", params: [], ex: [{ l: "Liste", v: {} }] }, | |
| 28 | + { g: "Découverte", m: "GET", p: "/v1/technologies", d: "Technologies citées dans les projets, par fréquence.", params: [P("q", "filtre texte"), P("limit", "≤ 500", { type: "number" })], ex: [{ l: "Top 25", v: { limit: 25 } }, { l: "Batteries", v: { q: "batter" } }] }, | |
| 29 | + { g: "Découverte", m: "GET", p: "/v1/locations", d: "Localisations canoniques, par fréquence, avec capital.", params: [P("q", "filtre texte"), P("limit", "≤ 500", { type: "number" })], ex: [{ l: "Top 25", v: { limit: 25 } }, { l: "Texas", v: { q: "texas" } }] }, | |
| 30 | + { g: "Découverte", m: "GET", p: "/v1/partners", d: "Partenaires cités dans les mentions (coentreprises, clients, fournisseurs nommés).", params: [P("q", "filtre texte"), P("limit", "≤ 500", { type: "number" })], ex: [{ l: "Top 25", v: { limit: 25 } }, { l: "Microsoft", v: { q: "microsoft" } }] }, | |
| 31 | + { g: "Projets", m: "GET", p: "/v1/projects", d: "Recherche de projets avec filtres combinables. Réponse paginée {total, limit, offset, items}. Tri : amount, mentions, filings, first_seen, last_seen, confidence, name, ticker, type.", params: [...PROJECT_FILTERS, ...PAGE], ex: [ | |
| 32 | + { l: "Centres de données > 500 M$", v: { type: "data_center", min_amount: "5e8", sort: "amount", limit: 10 } }, | |
| 33 | + { l: "IA depuis 2024", v: { type: "ai_initiative", year_from: 2024, sort: "last_seen", limit: 20 } }, | |
| 34 | + { l: "Usines au Texas", v: { type: "plant_construction,manufacturing_expansion", location: "texas", limit: 20 } }, | |
| 35 | + { l: "Batteries, terminées", v: { tech: "battery", status: "completed", limit: 20 } }, | |
| 36 | + { l: "Tesla", v: { ticker: "TSLA", sort: "mentions", limit: 20 } }, | |
| 37 | + { l: "Partenaire NVIDIA", v: { partner: "nvidia", limit: 20 } }, | |
| 38 | + ] }, | |
| 39 | + { g: "Projets", m: "GET", p: "/v1/projects/export.csv", d: "Export CSV des projets filtrés (mêmes filtres que /v1/projects, ≤ 50 000 lignes, listes jointes par |).", params: PROJECT_FILTERS, raw: true, ex: [{ l: "Tous les centres de données", v: { type: "data_center" } }, { l: "Services publics avec montant", v: { sector: "Utilities", has_amount: "true" } }] }, | |
| 40 | + { g: "Projets", m: "GET", p: "/v1/projects/{project_id}", d: "Fiche complète d'un projet : attributs, chronologie (une ligne par mention datée), mentions détaillées (lieux, technologies, partenaires, bénéfices, risques), voisins du graphe et projets similaires par embeddings.", params: [P("project_id", "identifiant MD5 (voir /v1/projects)", { path: true })], ex: [{ l: "Tesla — Energy Storage Products", v: { project_id: "TSLA_ENERGY" } }, { l: "WEC — Data Center Investments", v: { project_id: "46fe2966ba0cbf6228807faca0645b54" } }] }, | |
| 41 | + { g: "Projets", m: "GET", p: "/v1/projects/{project_id}/similar", d: "Projets sémantiquement proches (similarité cosinus des vecteurs MiniLM 384-d stockés).", params: [P("project_id", "identifiant", { path: true }), P("k", "≤ 50", { type: "number" })], ex: [{ l: "10 voisins", v: { project_id: "46fe2966ba0cbf6228807faca0645b54", k: 10 } }] }, | |
| 42 | + { g: "Mentions & sections", m: "GET", p: "/v1/mentions", d: "Mentions individuelles (une par projet et par section de filing) : l'unité d'extraction brute, avec formulaire, date, section, montant, confiance. Chaque mention porte le project_id du projet résolu.", params: [P("q", "texte libre"), P("type", "project_type", { enum: "types" }), P("sector", "", { enum: "sectors" }), P("status", ""), P("ticker", ""), P("cik", ""), P("form", "10-K, 10-Q, 8-K (liste possible)"), P("date_from", "AAAA-MM-JJ"), P("date_to", "AAAA-MM-JJ"), P("min_amount", "US$", { type: "number" }), P("min_confidence", "0–1", { type: "number" }), P("accession", "numéro d'accession SEC"), P("sort", "date | amount | confidence | ticker"), P("order", "asc | desc"), P("limit", "≤ 500", { type: "number" }), P("offset", "", { type: "number" })], ex: [ | |
| 43 | + { l: "8-K de NVIDIA", v: { ticker: "NVDA", form: "8-K", limit: 20 } }, | |
| 44 | + { l: "IA en 2025, confiance ≥ 0,9", v: { type: "ai_initiative", date_from: "2025-01-01", min_confidence: 0.9, limit: 20 } }, | |
| 45 | + { l: "Gros montants 10-K", v: { form: "10-K", min_amount: "1e10", sort: "amount", limit: 20 } }, | |
| 46 | + ] }, | |
| 47 | + { g: "Mentions & sections", m: "GET", p: "/v1/mentions/{mention_id}", d: "Une mention complète et, si elle vient d'un 10-K, la section source (section_text_url).", params: [P("mention_id", "identifiant SHA-1 (20 car.)", { path: true })], ex: [{ l: "Exemple", v: { mention_id: "MENTION_10K" } }] }, | |
| 48 | + { g: "Mentions & sections", m: "GET", p: "/v1/sections/{section_id}", d: "Texte intégral d'une section 10-K ingérée par SPID (Items 1, 1A, 2, 7, 7A). Les sections 10-Q/8-K ne sont pas stockées dans la base.", params: [P("section_id", "UUID de section", { path: true }), P("highlight", "mot à repérer : renvoie les positions")], ex: [{ l: "Exemple", v: { section_id: "SECTION_10K" } }] }, | |
| 49 | + { g: "Entreprises", m: "GET", p: "/v1/companies", d: "Les 500 entreprises avec taille de portefeuille (projets, mentions, types, capital, période). Tri : projects, amount, mentions, ticker.", params: [P("q", "nom ou ticker"), P("sector", "", { enum: "sectors" }), P("sort", ""), P("order", ""), P("limit", "≤ 500", { type: "number" }), P("offset", "", { type: "number" })], ex: [{ l: "Top 20", v: { sort: "projects", limit: 20 } }, { l: "Santé par capital", v: { sector: "Health Care", sort: "amount", limit: 20 } }] }, | |
| 50 | + { g: "Entreprises", m: "GET", p: "/v1/companies/{ident}", d: "Profil d'une entreprise (ticker ou CIK) : répartitions par type, statut, année, formulaire ; lieux, technologies, partenaires ; portefeuille complet (≤ 500 projets).", params: [P("ident", "ticker ou CIK", { path: true })], ex: [{ l: "Tesla", v: { ident: "TSLA" } }, { l: "Duke Energy", v: { ident: "DUK" } }, { l: "Microsoft (CIK)", v: { ident: "789019" } }] }, | |
| 51 | + { g: "Graphe", m: "GET", p: "/v1/graph/search", d: "Chercher un nœud du graphe de connaissances par libellé, avec son degré.", params: [P("q", "texte"), P("type", "Company | Project | Location | Technology | Partner"), P("limit", "≤ 200", { type: "number" })], ex: [{ l: "Texas", v: { q: "texas" } }, { l: "Technologies IA", v: { q: "ai", type: "Technology" } }] }, | |
| 52 | + { g: "Graphe", m: "GET", p: "/v1/graph/node/{node_id}", d: "Un nœud et ses arêtes (owns, located_in, uses) avec les nœuds voisins. Identifiants : C:<cik>, P:<project_id>, L:<lieu en minuscules>, T:<technologie>, PR:<partenaire>.", params: [P("node_id", "ex. T:AI, L:texas, C:0001318605", { path: true }), P("limit", "≤ 2000", { type: "number" })], ex: [{ l: "T:AI", v: { node_id: "T:AI", limit: 50 } }, { l: "L:texas", v: { node_id: "L:texas", limit: 50 } }, { l: "C:Tesla", v: { node_id: "C:0001318605", limit: 50 } }] }, | |
| 53 | + { g: "Recherche", m: "GET", p: "/v1/search/semantic", d: "Recherche en langage naturel : la requête est encodée avec all-MiniLM-L6-v2 côté serveur et comparée aux 19 227 vecteurs de projets. Anglais recommandé (langue des filings).", params: [P("q", "description en langage naturel"), P("k", "≤ 100", { type: "number" }), P("type", "filtre project_type", { enum: "types" }), P("sector", "filtre secteur", { enum: "sectors" })], ex: [{ l: "Usine de cellules de batteries", v: { q: "battery cell factory", k: 10 } }, { l: "IA générative pour clients", v: { q: "generative AI assistant for customers", k: 10 } }, { l: "Terminal GNL", v: { q: "LNG export terminal", k: 10 } }] }, | |
| 54 | + { g: "SQL", m: "GET", p: "/v1/schema", d: "Tables, colonnes et types de la base DuckDB.", params: [], ex: [{ l: "Schéma", v: {} }] }, | |
| 55 | + { g: "SQL", m: "POST", p: "/v1/sql", d: "Bac à sable SQL DuckDB en lecture seule (SELECT, WITH, DESCRIBE, SUMMARIZE). Corps JSON {sql, limit}. 20 s et 5 000 lignes maximum. Réponse {columns, rows, truncated, elapsed_ms}.", params: [P("sql", "requête SQL", { body: true, textarea: true }), P("limit", "≤ 5000", { body: true, type: "number" })], ex: [ | |
| 56 | + { l: "Projets par type", v: { sql: "select project_type, count(*) n, round(sum(total_amount_usd)/1e9,1) capital_gusd\nfrom projects group by 1 order by n desc", limit: 25 } }, | |
| 57 | + { l: "Mentions IA par année", v: { sql: "select year(filing_date) as yr, count(*) n\nfrom project_mentions where project_type='ai_initiative' group by 1 order by 1", limit: 30 } }, | |
| 58 | + { l: "Panel entreprise × année", v: { sql: "select cik, any_value(ticker) ticker, year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital\nfrom projects group by 1,3 order by 1,3", limit: 5000 } }, | |
| 59 | + { l: "Technologies co-citées", v: { sql: "select a.t t1, b.t t2, count(*) n\nfrom (select project_id, unnest(technologies) t from projects) a\njoin (select project_id, unnest(technologies) t from projects) b on a.project_id=b.project_id and a.t<b.t\ngroup by 1,2 order by n desc", limit: 50 } }, | |
| 60 | + ] }, | |
| 61 | + ]; | |
| 62 | + const GROUPS = [...new Set(ENDPOINTS.map((e) => e.g))]; | |
| 63 | + | |
| 64 | + /* ------------------------------------------------------------------ générateurs de code */ | |
| 65 | + function buildRequest(e, vals) { | |
| 66 | + let path = e.p; const query = {}; let body = null; | |
| 67 | + for (const prm of e.params) { | |
| 68 | + const v = vals[prm.name]; if (v === undefined || v === "") continue; | |
| 69 | + if (prm.path) path = path.replace(`{${prm.name}}`, encodeURIComponent(v)); | |
| 70 | + else if (prm.body) (body ||= {})[prm.name] = prm.type === "number" ? Number(v) : v; | |
| 71 | + else query[prm.name] = v; | |
| 72 | + } | |
| 73 | + path = path.replace(/\{[^}]+\}/g, ""); | |
| 74 | + return { method: e.m, path, query, body, url: location.origin + path + (Object.keys(query).length ? "?" + qs(query) : "") }; | |
| 75 | + } | |
| 76 | + const pyDict = (o) => "{" + Object.entries(o).map(([k, v]) => `${JSON.stringify(k)}: ${typeof v === "number" ? v : JSON.stringify(v)}`).join(", ") + "}"; | |
| 77 | + const rList = (o) => "list(" + Object.entries(o).map(([k, v]) => `${/^[a-z_]+$/i.test(k) ? k : "`" + k + "`"} = ${typeof v === "number" ? v : JSON.stringify(v)}`).join(", ") + ")"; | |
| 78 | + function snippet(l, r, k = KEY_PH, nokey = false) { | |
| 79 | + const base = location.origin + r.path; | |
| 80 | + if (nokey) return { curl: `curl '${r.url}'`, py: `import requests\n\nr = requests.get("${r.url}")\nr.raise_for_status()\nprint(r.json())`, js: `const r = await fetch("${r.url}");\nconsole.log(await r.json());`, r: `library(httr2)\nrequest("${r.url}") |> req_perform() |> resp_body_json()` }[l]; | |
| 81 | + if (l === "curl") return r.method === "POST" | |
| 82 | + ? `curl -X POST '${base}' \\\n -H 'X-API-Key: ${k}' -H 'Content-Type: application/json' \\\n -d '${JSON.stringify(r.body || {})}'` | |
| 83 | + : `curl '${r.url}' -H 'X-API-Key: ${k}'`; | |
| 84 | + if (l === "py") return r.method === "POST" | |
| 85 | + ? `import requests\n\nr = requests.post("${base}",\n headers={"X-API-Key": "${k}"},\n json=${pyDict(r.body || {})})\nr.raise_for_status()\ndata = r.json()\nprint(data)` | |
| 86 | + : `import requests\n\nr = requests.get("${base}",\n headers={"X-API-Key": "${k}"},${Object.keys(r.query).length ? `\n params=${pyDict(r.query)},` : ""}\n)\nr.raise_for_status()\ndata = r.json()\nprint(data)`; | |
| 87 | + if (l === "js") return `const r = await fetch("${r.url}", {\n method: "${r.method}",\n headers: { "X-API-Key": "${k}"${r.body ? ', "Content-Type": "application/json"' : ""} },${r.body ? `\n body: JSON.stringify(${JSON.stringify(r.body)}),` : ""}\n});\nif (!r.ok) throw new Error(\`HTTP \${r.status}\`);\nconst data = await r.json();\nconsole.log(data);`; | |
| 88 | + if (l === "r") return `library(httr2)\n\nresp <- request("${base}") |>\n req_headers(\`X-API-Key\` = "${k}") |>${Object.keys(r.query).length ? `\n req_url_query(!!!${rList(r.query)}) |>` : ""}${r.body ? `\n req_body_json(${rList(r.body)}) |>` : ""}\n req_perform()\ndata <- resp_body_json(resp)\nstr(data, max.level = 1)`; | |
| 89 | + } | |
| 90 | + | |
| 91 | + /* ------------------------------------------------------------------ recettes */ | |
| 92 | + const RECIPES = [ | |
| 93 | + { t: "Premier appel : compter les projets", d: "Vérifier la clé et lire les compteurs globaux.", ep: "/v1/stats", v: {}, code: { | |
| 94 | + curl: `curl https://www.pdb-api.co/v1/stats -H 'X-API-Key: ${KEY_PH}' | python3 -m json.tool | head -30`, | |
| 95 | + py: `import requests\nBASE, H = "https://www.pdb-api.co/v1", {"X-API-Key": "${KEY_PH}"}\n\nov = requests.get(f"{BASE}/stats", headers=H).json()["overview"]\nprint(f"{ov['projects']:,} projets · {ov['mentions']:,} mentions · {ov['companies']} entreprises")`, | |
| 96 | + js: `const BASE = "https://www.pdb-api.co/v1", H = { "X-API-Key": "${KEY_PH}" };\nconst { overview } = await (await fetch(\`\${BASE}/stats\`, { headers: H })).json();\nconsole.log(overview.projects, "projets", overview.mentions, "mentions");`, | |
| 97 | + r: `library(httr2)\nBASE <- "https://www.pdb-api.co/v1"; KEY <- "${KEY_PH}"\nov <- request(paste0(BASE, "/stats")) |> req_headers(\`X-API-Key\` = KEY) |> req_perform() |> resp_body_json()\nov$overview$projects` } }, | |
| 98 | + { t: "Filtrer et trier des projets", d: "Centres de données de plus de 500 M$, du plus gros au plus petit.", ep: "/v1/projects", v: { type: "data_center", min_amount: "5e8", sort: "amount", limit: 10 }, code: { | |
| 99 | + curl: `curl 'https://www.pdb-api.co/v1/projects?type=data_center&min_amount=5e8&sort=amount&limit=10' -H 'X-API-Key: ${KEY_PH}'`, | |
| 100 | + py: `r = requests.get(f"{BASE}/projects", headers=H, params={\n "type": "data_center", "min_amount": 5e8, "sort": "amount", "limit": 10})\nfor p in r.json()["items"]:\n print(f"{p['ticker']:6s} {p['project_name'][:45]:45s} {p['total_amount_usd']/1e9:6.1f} G$ {p['canonical_location']}")`, | |
| 101 | + js: `const q = new URLSearchParams({ type: "data_center", min_amount: 5e8, sort: "amount", limit: 10 });\nconst { total, items } = await (await fetch(\`\${BASE}/projects?\${q}\`, { headers: H })).json();\nitems.forEach(p => console.log(p.ticker, p.project_name, p.total_amount_usd / 1e9, "G$"));`, | |
| 102 | + r: `res <- request(paste0(BASE, "/projects")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(type = "data_center", min_amount = 5e8, sort = "amount", limit = 10) |>\n req_perform() |> resp_body_json()\ndo.call(rbind, lapply(res$items, \\(p) data.frame(ticker = p$ticker, projet = p$project_name, gusd = p$total_amount_usd / 1e9)))` } }, | |
| 103 | + { t: "Tout récupérer avec la pagination → DataFrame", d: "Boucler sur limit/offset (≤ 500 par page) pour construire un jeu de données complet.", ep: "/v1/projects", v: { sector: "Utilities", limit: 500 }, code: { | |
| 104 | + curl: `# page 1, 2, 3… : incrémenter offset de 500 jusqu'à atteindre total\nfor off in 0 500 1000 1500; do\n curl -s "https://www.pdb-api.co/v1/projects?sector=Utilities&limit=500&offset=$off" -H 'X-API-Key: ${KEY_PH}' > utilities_$off.json\ndone`, | |
| 105 | + py: `import pandas as pd\n\ndef fetch_all(path, **filters):\n offset, rows = 0, []\n while True:\n page = requests.get(f"{BASE}/{path}", headers=H, params={**filters, "limit": 500, "offset": offset}).json()\n rows += page["items"]\n offset += len(page["items"])\n if not page["items"] or offset >= page["total"]:\n return rows\n\ndf = pd.DataFrame(fetch_all("projects", sector="Utilities"))\ndf["technologies"] = df["technologies"].str.join(" | ")\nprint(df.shape)\ndf.groupby("project_type")["total_amount_usd"].agg(["count", "median"]).sort_values("count", ascending=False)`, | |
| 106 | + js: `async function fetchAll(path, filters) {\n const rows = []; let offset = 0;\n for (;;) {\n const q = new URLSearchParams({ ...filters, limit: 500, offset });\n const page = await (await fetch(\`\${BASE}/\${path}?\${q}\`, { headers: H })).json();\n rows.push(...page.items); offset += page.items.length;\n if (!page.items.length || offset >= page.total) return rows;\n }\n}\nconst utilities = await fetchAll("projects", { sector: "Utilities" });\nconsole.log(utilities.length, "projets");`, | |
| 107 | + r: `fetch_all <- function(path, ...) {\n out <- list(); offset <- 0\n repeat {\n page <- request(paste0(BASE, "/", path)) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(..., limit = 500, offset = offset) |> req_perform() |> resp_body_json()\n out <- c(out, page$items); offset <- offset + length(page$items)\n if (length(page$items) == 0 || offset >= page$total) break\n }\n out\n}\nprojets <- fetch_all("projects", sector = "Utilities")\ndf <- dplyr::bind_rows(lapply(projets, \\(p) as.data.frame(p[c("ticker","project_type","project_name","status","total_amount_usd","first_seen")])))` } }, | |
| 108 | + { t: "Export CSV direct", d: "Le plus simple pour Excel, R ou pandas : un seul appel, jusqu'à 50 000 lignes.", ep: "/v1/projects/export.csv", v: { type: "data_center" }, code: { | |
| 109 | + curl: `curl 'https://www.pdb-api.co/v1/projects/export.csv?type=data_center' -H 'X-API-Key: ${KEY_PH}' -o data_centers.csv`, | |
| 110 | + py: `import io, pandas as pd\ncsv_text = requests.get(f"{BASE}/projects/export.csv", headers=H, params={"type": "data_center"}).text\ndf = pd.read_csv(io.StringIO(csv_text))\ndf.head()`, | |
| 111 | + js: `const csv = await (await fetch(\`\${BASE}/projects/export.csv?type=data_center\`, { headers: H })).text();\nrequire("fs").writeFileSync("data_centers.csv", csv);`, | |
| 112 | + r: `csv <- request(paste0(BASE, "/projects/export.csv")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(type = "data_center") |> req_perform() |> resp_body_string()\ndf <- read.csv(text = csv)` } }, | |
| 113 | + { t: "Chronologie d'un projet", d: "Suivre un projet de son annonce (8-K) à ses mentions ultérieures (10-K, 10-Q).", ep: "/v1/projects/{project_id}", v: { project_id: "46fe2966ba0cbf6228807faca0645b54" }, code: { | |
| 114 | + curl: `PID=$(curl -s 'https://www.pdb-api.co/v1/projects?ticker=TSLA&q=energy%20storage&limit=1' -H 'X-API-Key: ${KEY_PH}' | python3 -c 'import json,sys; print(json.load(sys.stdin)["items"][0]["project_id"])')\ncurl "https://www.pdb-api.co/v1/projects/$PID" -H 'X-API-Key: ${KEY_PH}'`, | |
| 115 | + py: `pid = requests.get(f"{BASE}/projects", headers=H, params={"ticker": "TSLA", "q": "energy storage", "limit": 1}).json()["items"][0]["project_id"]\nfiche = requests.get(f"{BASE}/projects/{pid}", headers=H).json()\nprint(fiche["project"]["project_name"], fiche["project"]["status"])\nfor t in fiche["timeline"]:\n print(t["filing_date"], t["form_type"], t["status"], t["amount_usd"], "—", t["snippet"][:80])\nprint("similaires :", [s["project_name"] for s in fiche["similar"][:5]])`, | |
| 116 | + js: `const first = (await (await fetch(\`\${BASE}/projects?ticker=TSLA&q=energy%20storage&limit=1\`, { headers: H })).json()).items[0];\nconst fiche = await (await fetch(\`\${BASE}/projects/\${first.project_id}\`, { headers: H })).json();\nfiche.timeline.forEach(t => console.log(t.filing_date, t.form_type, t.status, t.amount_usd));`, | |
| 117 | + r: `pid <- (request(paste0(BASE, "/projects")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(ticker = "TSLA", q = "energy storage", limit = 1) |> req_perform() |> resp_body_json())$items[[1]]$project_id\nfiche <- request(paste0(BASE, "/projects/", pid)) |> req_headers(\`X-API-Key\` = KEY) |> req_perform() |> resp_body_json()\nsapply(fiche$timeline, \\(t) paste(t$filing_date, t$form_type, t$status))` } }, | |
| 118 | + { t: "Profil d'une entreprise", d: "Répartitions par type, année et formulaire, plus le portefeuille complet.", ep: "/v1/companies/{ident}", v: { ident: "DUK" }, code: { | |
| 119 | + curl: `curl https://www.pdb-api.co/v1/companies/DUK -H 'X-API-Key: ${KEY_PH}'`, | |
| 120 | + py: `duk = requests.get(f"{BASE}/companies/DUK", headers=H).json()\nprint(duk["company"])\npd.DataFrame(duk["by_type"]).set_index("label")["n"].plot.barh(title="Duke Energy — projets par type")`, | |
| 121 | + js: `const duk = await (await fetch(\`\${BASE}/companies/DUK\`, { headers: H })).json();\nconsole.table(duk.by_type.map(x => ({ type: x.label, n: x.n })));`, | |
| 122 | + r: `duk <- request(paste0(BASE, "/companies/DUK")) |> req_headers(\`X-API-Key\` = KEY) |> req_perform() |> resp_body_json()\nbarplot(sapply(duk$by_type, \\(x) x$n), names.arg = sapply(duk$by_type, \\(x) x$project_type), las = 2)` } }, | |
| 123 | + { t: "Recherche sémantique", d: "Décrire un projet en langage naturel et obtenir les projets les plus proches.", ep: "/v1/search/semantic", v: { q: "battery cell factory", k: 10 }, code: { | |
| 124 | + curl: `curl 'https://www.pdb-api.co/v1/search/semantic?q=battery+cell+factory&k=10' -H 'X-API-Key: ${KEY_PH}'`, | |
| 125 | + py: `hits = requests.get(f"{BASE}/search/semantic", headers=H, params={"q": "battery cell factory", "k": 10}).json()["items"]\nfor h in hits:\n print(f"{h['score']:.3f} {h['ticker']:6s} {h['project_name']}")`, | |
| 126 | + js: `const { items } = await (await fetch(\`\${BASE}/search/semantic?\${new URLSearchParams({ q: "battery cell factory", k: 10 })}\`, { headers: H })).json();\nitems.forEach(h => console.log(h.score.toFixed(3), h.ticker, h.project_name));`, | |
| 127 | + r: `hits <- request(paste0(BASE, "/search/semantic")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(q = "battery cell factory", k = 10) |> req_perform() |> resp_body_json()\nsapply(hits$items, \\(h) sprintf("%.3f %s %s", h$score, h$ticker, h$project_name))` } }, | |
| 128 | + { t: "SQL : agrégations libres", d: "Quand les filtres ne suffisent pas : une requête DuckDB de lecture, résultat en colonnes/lignes.", ep: "/v1/sql", v: { sql: "select year(filing_date) as yr, project_type, count(*) n\nfrom project_mentions\nwhere project_type in ('ai_initiative','data_center','cloud_migration')\ngroup by 1,2 order by 1,2", limit: 200 }, code: { | |
| 129 | + curl: `curl -X POST https://www.pdb-api.co/v1/sql -H 'X-API-Key: ${KEY_PH}' -H 'Content-Type: application/json' \\\n -d '{"sql":"select year(filing_date) as yr, project_type, count(*) n from project_mentions where project_type in (\\'ai_initiative\\',\\'data_center\\') group by 1,2 order by 1,2","limit":200}'`, | |
| 130 | + py: `sql = """\nselect year(filing_date) as yr, project_type, count(*) n\nfrom project_mentions\nwhere project_type in ('ai_initiative','data_center','cloud_migration')\ngroup by 1,2 order by 1,2\n"""\nres = requests.post(f"{BASE}/sql", headers=H, json={"sql": sql, "limit": 500}).json()\ndf = pd.DataFrame(res["rows"], columns=res["columns"])\ndf.pivot(index="yr", columns="project_type", values="n").plot(title="Mentions par année")`, | |
| 131 | + js: `const res = await (await fetch(\`\${BASE}/sql\`, { method: "POST", headers: { ...H, "Content-Type": "application/json" },\n body: JSON.stringify({ sql: "select project_type, count(*) n from projects group by 1 order by n desc", limit: 25 }) })).json();\nconsole.table(res.rows.map(r => Object.fromEntries(res.columns.map((c, i) => [c, r[i]]))));`, | |
| 132 | + r: `res <- request(paste0(BASE, "/sql")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_body_json(list(sql = "select project_type, count(*) n from projects group by 1 order by n desc", limit = 25)) |>\n req_perform() |> resp_body_json()\ndf <- as.data.frame(do.call(rbind, lapply(res$rows, unlist))); names(df) <- unlist(res$columns); df` } }, | |
| 133 | + { t: "Graphe : qui construit au Texas avec des batteries ?", d: "Partir d'un nœud Lieu, remonter aux projets, croiser avec une technologie.", ep: "/v1/graph/node/{node_id}", v: { node_id: "L:texas", limit: 100 }, code: { | |
| 134 | + curl: `curl 'https://www.pdb-api.co/v1/graph/node/L:texas?limit=100' -H 'X-API-Key: ${KEY_PH}'`, | |
| 135 | + py: `node = requests.get(f"{BASE}/graph/node/L:texas", headers=H, params={"limit": 500}).json()\nprojets_texas = {e["neighbor_id"][2:] for e in node["edges"] if e["neighbor_type"] == "Project"}\nbatt = requests.get(f"{BASE}/projects", headers=H, params={"location": "texas", "tech": "batter", "limit": 100}).json()["items"]\nprint(len(projets_texas), "projets au Texas ;", len(batt), "avec batteries :", [(p["ticker"], p["project_name"]) for p in batt][:5])`, | |
| 136 | + js: `const node = await (await fetch(\`\${BASE}/graph/node/L:texas?limit=500\`, { headers: H })).json();\nconsole.log(node.degree, "projets rattachés à texas");`, | |
| 137 | + r: `node <- request(paste0(BASE, "/graph/node/L:texas")) |> req_headers(\`X-API-Key\` = KEY) |> req_url_query(limit = 500) |> req_perform() |> resp_body_json()\nnode$degree` } }, | |
| 138 | + { t: "Gérer les erreurs et la limite de débit", d: "401 clé invalide, 404 introuvable, 408 SQL trop long, 429 trop de requêtes (240/min), 501 sémantique indisponible.", ep: "/v1/health", v: {}, code: { | |
| 139 | + curl: `curl -s -o /dev/null -w '%{http_code}\\n' https://www.pdb-api.co/v1/stats -H 'X-API-Key: mauvaise' # 401`, | |
| 140 | + py: `import time\n\ndef get(path, **params):\n for attempt in range(4):\n r = requests.get(f"{BASE}/{path}", headers=H, params=params, timeout=60)\n if r.status_code == 429: # limite de débit : attendre puis réessayer\n time.sleep(2 * (attempt + 1)); continue\n if r.status_code >= 400:\n raise RuntimeError(f"HTTP {r.status_code}: {r.json().get('detail')}")\n return r.json()\n raise RuntimeError("trop de tentatives")`, | |
| 141 | + js: `async function get(path, params = {}) {\n for (let i = 0; i < 4; i++) {\n const r = await fetch(\`\${BASE}/\${path}?\${new URLSearchParams(params)}\`, { headers: H });\n if (r.status === 429) { await new Promise(s => setTimeout(s, 2000 * (i + 1))); continue; }\n if (!r.ok) throw new Error(\`HTTP \${r.status}: \${(await r.json()).detail}\`);\n return r.json();\n }\n}`, | |
| 142 | + r: `resp <- request(paste0(BASE, "/stats")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_retry(max_tries = 4, is_transient = \\(r) resp_status(r) == 429) |> req_perform()` } }, | |
| 143 | + ]; | |
| 144 | + | |
| 145 | + /* ------------------------------------------------------------------ référence : dictionnaire des champs */ | |
| 146 | + const FIELDS = { | |
| 147 | + projects: [["project_id", "MD5 de cik|type|ancre ; identifiant stable"], ["cik / ticker / company_name / sector", "entreprise (GICS)"], ["project_type", "un des 21 types (voir /v1/taxonomy)"], ["project_name / description", "nom et description de la mention la plus confiante"], ["canonical_location", "première localisation (minuscules) ayant servi d'ancre"], ["technologies[]", "union des technologies citées"], ["status", "statut du dépôt le plus récent : planned, in_progress, completed, mentioned"], ["total_amount_usd", "montant MAXIMAL divulgué parmi les mentions (pas une somme)"], ["n_mentions / n_filings", "nombre de mentions et de dépôts distincts"], ["first_seen / last_seen", "dates de dépôt extrêmes (fiables pour dater)"], ["first_year / last_year", "années citées dans le texte (peuvent être projetées)"], ["avg_confidence", "confiance moyenne du modèle (0,45–1)"]], | |
| 148 | + mentions: [["mention_id", "SHA-1[0:20] de accession|section|type|ancre"], ["accession_number / section_id / section_name", "dépôt et section source"], ["form_type / filing_date / fiscal_year", "10-K, 10-Q ou 8-K ; date de dépôt (fiscal_year non renseigné)"], ["project_name / project_type / description / objective", "extraction du LLM"], ["amount_usd / amount_raw", "montant en US$ et chaîne d'origine"], ["locations[] / technologies[] / partners[] / suppliers[]", "listes extraites (≤ 8 éléments)"], ["benefits[] / risks[] / years[]", "bénéfices, risques, années citées"], ["status / confidence / backend", "statut, confiance 0–1, backend (llm)"], ["project_id", "projet résolu auquel la mention est rattachée (ajouté par l'API)"]], | |
| 149 | + timeline: [["project_id / accession_number / filing_date / form_type", "un point par mention datée"], ["status / amount_usd / confidence / snippet", "état à cette date, montant, confiance, extrait de 300 caractères"]], | |
| 150 | + }; | |
| 151 | + const ERRORS = [["200", "OK"], ["400", "Requête SQL refusée (mot-clé interdit, syntaxe) ou paramètre invalide"], ["401", "Clé absente ou invalide (en-tête X-API-Key)"], ["403", "Route /app réservée à l'interface web"], ["404", "Projet, mention, section, entreprise ou nœud introuvable"], ["408", "Requête SQL interrompue après 20 s"], ["422", "Paramètre mal typé (voir detail)"], ["429", "Plus de 240 requêtes par minute pour votre adresse"], ["501", "Recherche sémantique indisponible sur le serveur"], ["503", "Service ou clé non configurés"]]; | |
| 152 | + | |
| 153 | + /* ------------------------------------------------------------------ rendu */ | |
| 154 | + let ctx, meta = { types: [], sectors: [] }, cur = 7, tab = "start", exampleIds = {}; | |
| 155 | + function render(c) { | |
| 156 | + ctx = c; const view = c.view; | |
| 157 | + const hp = new URLSearchParams((location.hash.split("?")[1] || "")); | |
| 158 | + if (hp.get("tab")) tab = hp.get("tab"); | |
| 159 | + if (hp.get("ep")) { const i = ENDPOINTS.findIndex((e) => e.p === hp.get("ep")); if (i >= 0) { cur = i; tab = "play"; } } | |
| 160 | + view.innerHTML = `<div class="page-head"><div><h1>API & playground</h1><p>API REST JSON de la base SPID. Base : <span class="mono">${esc(location.origin)}/v1</span> · authentification par en-tête <span class="mono">X-API-Key</span> · 240 requêtes/min · pagination <span class="mono">limit/offset</span> (≤ 500).</p></div> | |
| 161 | + <div class="langsel">Langage des exemples : ${LANGS.map(([k, l]) => `<button class="${lang() === k ? "active" : ""}" data-lang="${k}">${l}</button>`).join("")}</div></div> | |
| 162 | + <div class="tabs big" id="api-tabs">${[["start", "Démarrage rapide"], ["play", "Playground"], ["recipes", "Recettes"], ["ref", "Référence"], ["sdk", "SDK Python"]].map(([k, l]) => `<button class="${tab === k ? "active" : ""}" data-tab="${k}">${l}</button>`).join("")}</div> | |
| 163 | + <div id="api-body"></div>`; | |
| 164 | + view.querySelectorAll("[data-lang]").forEach((b) => b.addEventListener("click", () => { localStorage.setItem("pdb_lang", b.dataset.lang); render(ctx); })); | |
| 165 | + view.querySelector("#api-tabs").addEventListener("click", (e) => { if (e.target.dataset.tab) { tab = e.target.dataset.tab; render(ctx); } }); | |
| 166 | + Promise.all([ctx.api("/taxonomy"), ctx.api("/sectors")]).then(([t, s]) => { meta = { types: t, sectors: s }; if (tab === "play") renderPlay(); }).catch(() => { }); | |
| 167 | + ({ start: renderStart, play: renderPlay, recipes: renderRecipes, ref: renderRef, sdk: renderSdk }[tab] || renderStart)(); | |
| 168 | + resolveExampleIds(); | |
| 169 | + } | |
| 170 | + async function resolveExampleIds() { | |
| 171 | + if (exampleIds.TSLA_ENERGY) return; | |
| 172 | + try { | |
| 173 | + const p = await ctx.api("/projects", { ticker: "TSLA", q: "energy storage", limit: 1 }); exampleIds.TSLA_ENERGY = p.items[0]?.project_id; | |
| 174 | + const m = await ctx.api("/mentions", { form: "10-K", min_confidence: 0.95, limit: 1 }); exampleIds.MENTION_10K = m.items[0]?.mention_id; exampleIds.SECTION_10K = m.items[0]?.section_id; | |
| 175 | + } catch { } | |
| 176 | + } | |
| 177 | + const sub = (v) => exampleIds[v] || v; | |
| 178 | + | |
| 179 | + const keyBox = () => `<div class="keybox"><label><b>Votre clé d'API</b><input class="inp" id="key" type="password" value="${esc(key())}" placeholder="saisir la clé (reste dans ce navigateur)" autocomplete="off"></label> | |
| 180 | + <button class="btn btn-ghost btn-sm" id="key-show">Afficher</button><button class="btn btn-sm" id="key-test">Tester la clé</button><span id="key-st" class="small muted"></span> | |
| 181 | + <div class="small muted" style="flex-basis:100%">La clé n'est pas publiée sur ce site. Elle est fournie par l'équipe UQO (<a href="mailto:simon-pierre.boucher@uqo.ca">simon-pierre.boucher@uqo.ca</a>). Elle est conservée en localStorage et envoyée uniquement à cette API.</div></div>`; | |
| 182 | + function bindKey(root) { | |
| 183 | + const inp = root.querySelector("#key"); if (!inp) return; | |
| 184 | + inp.addEventListener("input", () => { localStorage.setItem("pdb_key", inp.value); root.querySelectorAll("pre.code[data-req]").forEach(() => { }); }); | |
| 185 | + root.querySelector("#key-show").addEventListener("click", () => (inp.type = inp.type === "password" ? "text" : "password")); | |
| 186 | + root.querySelector("#key-test").addEventListener("click", async () => { | |
| 187 | + const st = root.querySelector("#key-st"); st.textContent = "…"; | |
| 188 | + const r = await fetch("/v1/taxonomy", { headers: { "X-API-Key": key() } }); | |
| 189 | + st.innerHTML = r.ok ? `<span class="ok" style="padding:3px 8px">✓ clé valide (HTTP 200)</span>` : `<span class="err" style="padding:3px 8px;display:inline-block;margin:0">✗ HTTP ${r.status} — ${esc((await r.json()).detail || "")}</span>`; | |
| 190 | + }); | |
| 191 | + } | |
| 192 | + const codeBlock = (code, id = "") => `<div class="codewrap"><button class="copy" data-copy>Copier</button><pre class="code" ${id ? `id="${id}"` : ""}>${esc(code)}</pre></div>`; | |
| 193 | + function bindCopy(root) { root.querySelectorAll("[data-copy]").forEach((b) => b.addEventListener("click", () => { navigator.clipboard.writeText(b.nextElementSibling.textContent); b.textContent = "Copié ✓"; setTimeout(() => (b.textContent = "Copier"), 1500); })); } | |
| 194 | + | |
| 195 | + /* ---- Démarrage rapide */ | |
| 196 | + function renderStart() { | |
| 197 | + const l = lang(); const body = ctx.view.querySelector("#api-body"); | |
| 198 | + const r1 = buildRequest(ENDPOINTS[0], {}); const r2 = buildRequest(ENDPOINTS[1], {}); const r3 = buildRequest(ENDPOINTS.find((e) => e.p === "/v1/projects"), { type: "ai_initiative", year_from: 2024, sort: "last_seen", limit: 5 }); | |
| 199 | + const install = { curl: "# curl est déjà installé sur macOS / Linux / Windows 10+", py: "pip install requests pandas # ou : téléchargez pdb_client.py (onglet SDK Python)", js: "# Node ≥ 18 (fetch natif) ou navigateur — aucune dépendance", r: 'install.packages(c("httr2", "dplyr"))' }[l]; | |
| 200 | + body.innerHTML = `<div class="grid g2"> | |
| 201 | + <div class="card" style="grid-column:1/-1">${keyBox()}</div> | |
| 202 | + <div class="card"><h3><span class="step">1</span> Installer</h3>${codeBlock(install)}</div> | |
| 203 | + <div class="card"><h3><span class="step">2</span> Vérifier le service (sans clé)</h3>${codeBlock(snippet(l, r1, KEY_PH, true))}</div> | |
| 204 | + <div class="card"><h3><span class="step">3</span> Premier appel authentifié</h3>${codeBlock(snippet(l, r2, key() || KEY_PH))}<p class="small muted">Réponse : <span class="mono">{overview:{projects, mentions, companies…}, by_type:[…], by_sector:[…], mentions_by_year:[…]}</span></p></div> | |
| 205 | + <div class="card"><h3><span class="step">4</span> Filtrer des projets</h3>${codeBlock(snippet(l, r3, key() || KEY_PH))}<p class="small muted">Réponse paginée : <span class="mono">{total, limit, offset, items:[{project_id, ticker, project_name, project_type, status, total_amount_usd, …}]}</span>. Pour tout récupérer, incrémentez <span class="mono">offset</span> de <span class="mono">limit</span> jusqu'à <span class="mono">total</span> (recette « pagination »).</p></div> | |
| 206 | + <div class="card" style="grid-column:1/-1"><h3>Modèle de données en 30 secondes</h3><div class="grid g3"> | |
| 207 | + <div><b>Projet</b> (19 227) — unité principale. Un projet = mentions fusionnées par entreprise × type × ancre (lieu ou premier mot du nom). <span class="mono">/v1/projects</span>, <span class="mono">/v1/projects/{id}</span>.</div> | |
| 208 | + <div><b>Mention</b> (39 930) — l'extraction brute : un projet cité dans une section d'un filing, avec date, formulaire, montant, confiance. <span class="mono">/v1/mentions</span>. La chronologie d'un projet = ses mentions triées par date.</div> | |
| 209 | + <div><b>Graphe</b> (32 226 nœuds) et <b>embeddings</b> (384-d) — navigation par lieu/technologie/partenaire et recherche par le sens. <span class="mono">/v1/graph/*</span>, <span class="mono">/v1/search/semantic</span>, <span class="mono">/v1/projects/{id}/similar</span>.</div></div> | |
| 210 | + <p class="small muted" style="margin-top:8px">Point d'attention : <span class="mono">total_amount_usd</span> est le montant <i>maximal</i> divulgué pour le projet, auto-déclaré et hétérogène ; <span class="mono">first_seen/last_seen</span> datent l'observation, <span class="mono">first_year/last_year</span> sont les années citées dans le texte (parfois projetées).</p></div> | |
| 211 | + <div class="card" style="grid-column:1/-1"><h3>Codes de réponse</h3><table class="ref"><tbody>${ERRORS.map(([c, d]) => `<tr><td class="mono"><b>${c}</b></td><td>${esc(d)}</td></tr>`).join("")}</tbody></table></div> | |
| 212 | + </div>`; | |
| 213 | + bindKey(body); bindCopy(body); | |
| 214 | + } | |
| 215 | + | |
| 216 | + /* ---- Playground */ | |
| 217 | + const HIST_KEY = "pdb_hist"; | |
| 218 | + function renderPlay() { | |
| 219 | + const body = ctx.view.querySelector("#api-body"); const e = ENDPOINTS[cur]; | |
| 220 | + body.innerHTML = `<div class="api-layout"><div class="card"><h3>Points d'accès</h3><ul class="endpoints">${GROUPS.map((g) => `<li class="grp">${esc(g)}</li>` + ENDPOINTS.map((x, i) => x.g === g ? `<li data-i="${i}" class="${i === cur ? "active" : ""}"><span class="m ${x.m === "POST" ? "post" : ""}">${x.m}</span><span class="mono">${esc(x.p.replace("/v1", ""))}</span></li>` : "").join("")).join("")}</ul> | |
| 221 | + <h3 style="margin-top:14px">Historique</h3><div id="hist" class="hist"></div></div> | |
| 222 | + <div><div class="card">${keyBox()}</div><div class="card" id="pg"></div></div></div>`; | |
| 223 | + body.querySelectorAll(".endpoints li[data-i]").forEach((li) => li.addEventListener("click", () => { cur = +li.dataset.i; renderPlay(); })); | |
| 224 | + bindKey(body); renderHist(body); | |
| 225 | + const pg = body.querySelector("#pg"); | |
| 226 | + const ex0 = e.ex?.[0]?.v || {}; | |
| 227 | + pg.innerHTML = `<h3><span class="m ${e.m === "POST" ? "post" : ""}" style="font-family:var(--mono);color:${e.m === "POST" ? "var(--orange)" : "var(--vert)"}">${e.m}</span> <span class="mono">${esc(e.p)}</span>${e.key === false ? ' <span class="pill pill-vert">sans clé</span>' : ""}</h3><p class="muted">${esc(e.d)}</p> | |
| 228 | + ${e.ex?.length ? `<div class="examples"><span class="small muted" style="align-self:center">Exemples :</span>${e.ex.map((x, i) => `<button data-ex="${i}">${esc(x.l)}</button>`).join("")}</div>` : ""} | |
| 229 | + <div class="params">${e.params.map((p) => p.textarea ? `<label style="grid-column:1/-1"><b>${esc(p.name)}</b> <span>${esc(p.desc)}</span><textarea class="sql" data-p="${esc(p.name)}" rows="5">${esc(sub(ex0[p.name] ?? ""))}</textarea></label>` : | |
| 230 | + `<label><b>${esc(p.name)}${p.path ? " *" : ""}</b><span>${esc(p.desc)}</span>${p.enum ? `<input class="inp" list="dl-${p.enum}" data-p="${esc(p.name)}" value="${esc(sub(ex0[p.name] ?? ""))}">` : `<input class="inp" type="${p.type === "number" ? "text" : "text"}" data-p="${esc(p.name)}" value="${esc(sub(ex0[p.name] ?? ""))}" placeholder="${esc(p.desc.slice(0, 40))}">`}</label>`).join("") || `<span class="muted small">Aucun paramètre.</span>`}</div> | |
| 231 | + <datalist id="dl-types">${meta.types.map((t) => `<option value="${t.project_type}">${esc(t.label)}</option>`).join("")}</datalist><datalist id="dl-sectors">${meta.sectors.map((s) => `<option value="${esc(s.sector)}">`).join("")}</datalist> | |
| 232 | + <div style="display:flex;gap:10px;align-items:center;flex-wrap:wrap"><button class="btn" id="send">▶ Envoyer</button><span class="muted small" id="st"></span><span class="sp" style="flex:1"></span><a id="explore" class="btn btn-ghost btn-sm" hidden>Ouvrir dans l'explorateur</a></div> | |
| 233 | + <div class="tabs" style="margin-top:14px" id="reqtabs">${LANGS.map(([k, l]) => `<button class="${lang() === k ? "active" : ""}" data-lang="${k}">${l}</button>`).join("")}</div> | |
| 234 | + ${codeBlock("", "req")} | |
| 235 | + <div style="display:flex;gap:10px;align-items:center;margin-top:14px"><h3 style="margin:0">Réponse</h3><div class="tabs" style="margin:0;border:0" id="resptabs"><button class="active" data-v="json">JSON</button><button data-v="table">Tableau</button></div><span class="sp" style="flex:1"></span><button class="btn btn-ghost btn-sm" id="dl" hidden>Télécharger</button></div> | |
| 236 | + <div id="resp"><pre class="code">—</pre></div>`; | |
| 237 | + const vals = () => { const o = {}; pg.querySelectorAll("[data-p]").forEach((i) => { if (i.value !== "") o[i.dataset.p] = i.value; }); return o; }; | |
| 238 | + const showReq = () => { const r = buildRequest(e, vals()); pg.querySelector("#req").textContent = snippet(lang(), r, key() || KEY_PH); const ex = exploreLink(e, vals(), r); const a = pg.querySelector("#explore"); a.hidden = !ex; if (ex) a.href = ex; }; | |
| 239 | + pg.querySelectorAll("[data-p]").forEach((i) => i.addEventListener("input", showReq)); showReq(); | |
| 240 | + pg.querySelectorAll("[data-ex]").forEach((b) => b.addEventListener("click", () => { const v = e.ex[b.dataset.ex].v; pg.querySelectorAll("[data-p]").forEach((i) => (i.value = sub(v[i.dataset.p] ?? ""))); showReq(); send(); })); | |
| 241 | + pg.querySelector("#reqtabs").addEventListener("click", (ev) => { if (ev.target.dataset.lang) { localStorage.setItem("pdb_lang", ev.target.dataset.lang); pg.querySelectorAll("#reqtabs button").forEach((b) => b.classList.toggle("active", b.dataset.lang === ev.target.dataset.lang)); ctx.view.querySelectorAll(".langsel button").forEach((b) => b.classList.toggle("active", b.dataset.lang === ev.target.dataset.lang)); showReq(); } }); | |
| 242 | + bindCopy(pg); | |
| 243 | + let last = null, lastRaw = ""; | |
| 244 | + const renderResp = () => { | |
| 245 | + const v = pg.querySelector("#resptabs .active").dataset.v; const box = pg.querySelector("#resp"); | |
| 246 | + if (!last) { box.innerHTML = `<pre class="code">${esc(lastRaw.slice(0, 60000))}</pre>`; return; } | |
| 247 | + if (v === "table") { const rows = tabular(last); box.innerHTML = rows ? rows : `<div class="warn">Réponse non tabulaire — voir JSON.</div>`; } | |
| 248 | + else { const txt = JSON.stringify(last, null, 2); box.innerHTML = `<div class="codewrap"><button class="copy" data-copy>Copier</button><pre class="code json">${hl(txt.length > 80000 ? txt.slice(0, 80000) + "\n… (tronqué)" : txt)}</pre></div>`; bindCopy(box); } | |
| 249 | + }; | |
| 250 | + pg.querySelector("#resptabs").addEventListener("click", (ev) => { if (ev.target.dataset.v) { pg.querySelectorAll("#resptabs button").forEach((b) => b.classList.toggle("active", b === ev.target)); renderResp(); } }); | |
| 251 | + async function send() { | |
| 252 | + const r = buildRequest(e, vals()); const st = pg.querySelector("#st"); st.textContent = "…"; const t0 = performance.now(); | |
| 253 | + try { | |
| 254 | + const resp = await fetch(r.url, { method: r.method, headers: { ...(e.key === false ? {} : { "X-API-Key": key() }), ...(r.body ? { "Content-Type": "application/json" } : {}) }, body: r.body ? JSON.stringify(r.body) : undefined }); | |
| 255 | + lastRaw = await resp.text(); last = null; try { last = JSON.parse(lastRaw); } catch { } | |
| 256 | + const ms = Math.round(performance.now() - t0); | |
| 257 | + st.innerHTML = `<span class="${resp.ok ? "pill pill-vert" : "pill pill-rouge"}">HTTP ${resp.status}</span> ${ms} ms · ${(lastRaw.length / 1024).toFixed(1)} ko${last?.total != null ? ` · total ${fmtN(last.total)}` : ""}`; | |
| 258 | + const dl = pg.querySelector("#dl"); dl.hidden = false; dl.onclick = () => { const a = document.createElement("a"); a.href = URL.createObjectURL(new Blob([lastRaw], { type: e.raw ? "text/csv" : "application/json" })); a.download = e.raw ? "export.csv" : "response.json"; a.click(); }; | |
| 259 | + pushHist({ ep: e.p, vals: vals(), status: resp.status, ms, t: Date.now() }); renderHist(body); renderResp(); | |
| 260 | + } catch (err) { st.textContent = ""; pg.querySelector("#resp").innerHTML = `<div class="err">${esc(String(err))}</div>`; } | |
| 261 | + } | |
| 262 | + pg.querySelector("#send").addEventListener("click", send); | |
| 263 | + } | |
| 264 | + function exploreLink(e, vals, r) { | |
| 265 | + if (e.p === "/v1/projects" || e.p === "/v1/projects/export.csv") return "#/projects?" + qs(Object.fromEntries(Object.entries(r.query).filter(([k]) => !["limit", "offset"].includes(k)))); | |
| 266 | + if (e.p.startsWith("/v1/projects/{") && vals.project_id) return "#/project/" + encodeURIComponent(vals.project_id); | |
| 267 | + if (e.p === "/v1/companies/{ident}" && vals.ident) return "#/company/" + encodeURIComponent(vals.ident); | |
| 268 | + if (e.p === "/v1/mentions") return "#/mentions?" + qs(r.query); | |
| 269 | + if (e.p === "/v1/mentions/{mention_id}" && vals.mention_id) return "#/mention/" + encodeURIComponent(vals.mention_id); | |
| 270 | + if (e.p === "/v1/graph/node/{node_id}" && vals.node_id) return "#/graph?node=" + encodeURIComponent(vals.node_id); | |
| 271 | + if (e.p === "/v1/graph/search" && vals.q) return "#/graph?q=" + encodeURIComponent(vals.q); | |
| 272 | + if (e.p === "/v1/search/semantic" && vals.q) return "#/semantic?q=" + encodeURIComponent(vals.q); | |
| 273 | + if (e.p === "/v1/sql") return "#/sql"; | |
| 274 | + return null; | |
| 275 | + } | |
| 276 | + function tabular(d) { | |
| 277 | + let rows = Array.isArray(d) ? d : d?.items || d?.edges || d?.projects || d?.timeline || null; | |
| 278 | + if (d?.columns && d?.rows) rows = d.rows.map((r) => Object.fromEntries(d.columns.map((c, i) => [c, r[i]]))); | |
| 279 | + if (!rows || !rows.length || typeof rows[0] !== "object") return null; | |
| 280 | + const cols = Object.keys(rows[0]).filter((c) => !["description", "props", "neighbor_props", "vector"].includes(c)).slice(0, 14); | |
| 281 | + return `<div class="table-wrap" style="max-height:60vh"><table><thead><tr>${cols.map((c) => `<th>${esc(c)}</th>`).join("")}</tr></thead><tbody>${rows.slice(0, 500).map((r) => `<tr>${cols.map((c) => { const v = r[c]; return `<td class="${typeof v === "number" ? "num" : ""}">${esc(Array.isArray(v) ? v.join(" | ") : v && typeof v === "object" ? JSON.stringify(v) : v)}</td>`; }).join("")}</tr>`).join("")}</tbody></table></div><div class="small muted" style="margin-top:6px">${fmtN(rows.length)} lignes${rows.length > 500 ? " (500 affichées)" : ""}</div>`; | |
| 282 | + } | |
| 283 | + const hl = (txt) => esc(txt).replace(/("(?:\\.|[^"\\])*")(\s*:)?/g, (m, s, c) => c ? `<span class="k">${s}</span>${c}` : `<span class="s">${s}</span>`).replace(/\b(-?\d+(?:\.\d+)?(?:e[+-]?\d+)?)\b/g, `<span class="n">$1</span>`).replace(/\b(true|false|null)\b/g, `<span class="b">$1</span>`); | |
| 284 | + function pushHist(h) { const a = JSON.parse(localStorage.getItem(HIST_KEY) || "[]"); a.unshift(h); localStorage.setItem(HIST_KEY, JSON.stringify(a.slice(0, 12))); } | |
| 285 | + function renderHist(root) { | |
| 286 | + const a = JSON.parse(localStorage.getItem(HIST_KEY) || "[]"); const box = root.querySelector("#hist"); if (!box) return; | |
| 287 | + box.innerHTML = a.length ? a.map((h, i) => `<div class="hist-it" data-h="${i}"><span class="pill ${h.status < 400 ? "pill-vert" : "pill-rouge"}">${h.status}</span> <span class="mono small">${esc(h.ep.replace("/v1", ""))}</span><div class="small muted">${esc(Object.entries(h.vals).map(([k, v]) => `${k}=${String(v).slice(0, 18)}`).join(" ")) || "—"} · ${h.ms} ms</div></div>`).join("") + `<button class="btn btn-ghost btn-sm" id="hist-clear">Effacer</button>` : `<span class="small muted">Aucune requête envoyée.</span>`; | |
| 288 | + box.querySelectorAll(".hist-it").forEach((d) => d.addEventListener("click", () => { const h = a[d.dataset.h]; cur = ENDPOINTS.findIndex((e) => e.p === h.ep); renderPlay(); const pg = root.querySelector("#pg"); pg.querySelectorAll("[data-p]").forEach((i) => (i.value = h.vals[i.dataset.p] ?? "")); pg.querySelector("[data-p]")?.dispatchEvent(new Event("input")); })); | |
| 289 | + box.querySelector("#hist-clear")?.addEventListener("click", () => { localStorage.removeItem(HIST_KEY); renderHist(root); }); | |
| 290 | + } | |
| 291 | + | |
| 292 | + /* ---- Recettes */ | |
| 293 | + function renderRecipes() { | |
| 294 | + const l = lang(); const body = ctx.view.querySelector("#api-body"); | |
| 295 | + const pre = { py: `# Préambule commun aux recettes Python\nimport requests, pandas as pd\nBASE = "https://www.pdb-api.co/v1"\nH = {"X-API-Key": "${key() || KEY_PH}"}`, js: `// Préambule commun (Node ≥ 18 ou navigateur)\nconst BASE = "https://www.pdb-api.co/v1";\nconst H = { "X-API-Key": "${key() || KEY_PH}" };`, r: `# Préambule commun aux recettes R\nlibrary(httr2); library(dplyr)\nBASE <- "https://www.pdb-api.co/v1"; KEY <- "${key() || KEY_PH}"`, curl: `# Remplacez ${KEY_PH} par votre clé. Ajoutez « | python3 -m json.tool » pour lire le JSON.` }[l]; | |
| 296 | + body.innerHTML = `<div class="card">${codeBlock(pre)}</div><div class="grid g2">${RECIPES.map((r, i) => `<div class="card recipe"><h3>${i + 1}. ${esc(r.t)}</h3><p class="muted small">${esc(r.d)}</p>${codeBlock(r.code[l].replace(new RegExp(KEY_PH, "g"), key() || KEY_PH))}<div style="margin-top:8px"><button class="btn btn-ghost btn-sm" data-try="${i}">Essayer dans le playground</button></div></div>`).join("")}</div>`; | |
| 297 | + bindCopy(body); | |
| 298 | + body.querySelectorAll("[data-try]").forEach((b) => b.addEventListener("click", () => { const r = RECIPES[b.dataset.try]; cur = ENDPOINTS.findIndex((e) => e.p === r.ep); tab = "play"; render(ctx); const pg = ctx.view.querySelector("#pg"); pg.querySelectorAll("[data-p]").forEach((i) => (i.value = sub(r.v[i.dataset.p] ?? ""))); pg.querySelector("[data-p]")?.dispatchEvent(new Event("input")); window.scrollTo(0, 0); })); | |
| 299 | + } | |
| 300 | + | |
| 301 | + /* ---- Référence */ | |
| 302 | + function renderRef() { | |
| 303 | + const body = ctx.view.querySelector("#api-body"); | |
| 304 | + body.innerHTML = `<div class="card"><h3>Conventions</h3><ul class="small"><li><b>Authentification</b> : en-tête <span class="mono">X-API-Key: <clé></span>, ou <span class="mono">Authorization: Bearer <clé></span>, ou paramètre <span class="mono">?api_key=</span> (déconseillé). <span class="mono">/v1/health</span> est libre.</li> | |
| 305 | + <li><b>Pagination</b> : <span class="mono">limit</span> (≤ 500) et <span class="mono">offset</span> ; réponse <span class="mono">{total, limit, offset, items}</span>.</li> | |
| 306 | + <li><b>Filtres texte</b> : insensibles à la casse, correspondance « contient » (sauf <span class="mono">ticker</span> et <span class="mono">cik</span>, exacts). Les paramètres <span class="mono">type</span>, <span class="mono">status</span>, <span class="mono">form</span> acceptent des listes séparées par des virgules.</li> | |
| 307 | + <li><b>Montants</b> en US$ (<span class="mono">5e8</span> accepté). <b>Dates</b> au format ISO <span class="mono">AAAA-MM-JJ</span>. <b>Listes</b> renvoyées comme tableaux JSON ; en CSV, jointes par <span class="mono">|</span>.</li> | |
| 308 | + <li><b>Limite</b> : 240 requêtes par minute et par adresse IP (429 au-delà). Bac à sable SQL : 20 s, 5 000 lignes, lecture seule, pas d'accès aux fichiers.</li> | |
| 309 | + <li><b>OpenAPI</b> : <a href="/openapi.json" target="_blank">openapi.json</a> · <a href="/docs" target="_blank">Swagger UI</a> · <a href="/redoc" target="_blank">ReDoc</a>.</li></ul></div> | |
| 310 | + <div class="card"><h3>Routes</h3><div class="table-wrap"><table class="ref"><thead><tr><th>Méthode</th><th>Route</th><th>Description</th><th>Paramètres</th></tr></thead><tbody>${ENDPOINTS.map((e) => `<tr><td class="mono"><b>${e.m}</b></td><td class="mono"><a href="#/api?tab=play&ep=${encodeURIComponent(e.p)}">${esc(e.p)}</a></td><td>${esc(e.d)}</td><td class="small">${e.params.map((p) => `<span class="mono">${esc(p.name)}</span>`).join(", ") || "—"}</td></tr>`).join("")}</tbody></table></div></div> | |
| 311 | + ${Object.entries(FIELDS).map(([t, f]) => `<div class="card"><h3>Champs — ${t}</h3><table class="ref"><tbody>${f.map(([k, d]) => `<tr><td class="mono">${esc(k)}</td><td>${esc(d)}</td></tr>`).join("")}</tbody></table></div>`).join("")} | |
| 312 | + <div class="card"><h3>Types de projets</h3><table class="ref"><thead><tr><th>project_type</th><th>Libellé</th><th class="num">Projets</th></tr></thead><tbody>${meta.types.map((t) => `<tr><td class="mono">${t.project_type}</td><td>${esc(t.label)}</td><td class="num">${fmtN(t.n)}</td></tr>`).join("") || "<tr><td colspan=3 class='muted'>chargement…</td></tr>"}</tbody></table></div> | |
| 313 | + <div class="card"><h3>Codes de réponse</h3><table class="ref"><tbody>${ERRORS.map(([c, d]) => `<tr><td class="mono"><b>${c}</b></td><td>${esc(d)}</td></tr>`).join("")}</tbody></table></div>`; | |
| 314 | + if (!meta.types.length) ctx.api("/taxonomy").then((t) => { meta.types = t; if (tab === "ref") renderRef(); }).catch(() => { }); | |
| 315 | + } | |
| 316 | + | |
| 317 | + /* ---- SDK */ | |
| 318 | + function renderSdk() { | |
| 319 | + const body = ctx.view.querySelector("#api-body"); const k = key() || KEY_PH; | |
| 320 | + body.innerHTML = `<div class="grid g2"> | |
| 321 | + <div class="card" style="grid-column:1/-1"><h3>pdb_client.py — client Python sans dépendance</h3><p class="muted">Un fichier à déposer à côté de votre script ou notebook. Pagination automatique, réessais sur 429, conversion pandas/CSV, toutes les routes.</p> | |
| 322 | + <a class="btn" href="/sdk/pdb_client.py" download>⬇ Télécharger pdb_client.py</a> <a class="btn btn-ghost" href="/sdk/pdb_client.py" target="_blank">Voir le source</a> <a class="btn btn-or" href="/sdk/pdb_api.R" download>⬇ Script R prêt à l'emploi (pdb_api.R)</a> <a class="btn btn-ghost" href="/sdk/pdb_api.m" download>⬇ MATLAB (pdb_api.m)</a> <a class="btn btn-ghost" href="/sdk/pdb_api.do" download>⬇ Stata (pdb_api.do)</a></div> | |
| 323 | + <div class="card" style="grid-column:1/-1"><h3>R — pdb_api.R</h3><p class="muted">Fichier R complet (httr2 + dplyr) : fonctions <span class="mono">pdb_get()</span>, <span class="mono">pdb_sql()</span>, <span class="mono">pdb_all()</span> (pagination) et dix exemples exécutables (filtres, secteur entier, CSV, chronologie, entreprise, sémantique, SQL, panel entreprise × année, graphe). Testé avec R 4.6.</p>${codeBlock(`install.packages(c("httr2", "dplyr"))\nSys.setenv(PDB_API_KEY = "${k}")\nsource("pdb_api.R") # définit pdb_get / pdb_sql / pdb_all\n\ndc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items\nutil <- pdb_all("projects", sector = "Utilities") # data.frame complet\nia <- pdb_sql("select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1")`)}</div> | |
| 324 | + <div class="card"><h3>MATLAB — pdb_api.m</h3><p class="muted">webread / webwrite natifs (R2018a+), fonctions <span class="mono">pdb_get</span>, <span class="mono">pdb_sql</span>, <span class="mono">pdb_all</span> et dix sections exécutables. Non testé sur MATLAB ici : signalez toute erreur.</p>${codeBlock(`BASE = "${location.origin}/v1"; KEY = "${k}";\nopts = weboptions("HeaderFields", ["X-API-Key", KEY], "ContentType", "json");\nr = webread(BASE + "/projects", "type", "data_center", "min_amount", "5e8", "sort", "amount", "limit", "10", opts);\ndc = struct2table(r.items);\n\nres = webwrite(BASE + "/sql", struct("sql", "select project_type, count(*) n from projects group by 1 order by n desc", "limit", 25), ...\n weboptions("HeaderFields", ["X-API-Key", KEY], "MediaType", "application/json"));\nT = cell2table(res.rows, "VariableNames", res.columns);`)}</div> | |
| 325 | + <div class="card"><h3>Stata — pdb_api.do</h3><p class="muted">Stata ne peut pas envoyer d'en-tête HTTP : le .do définit <span class="mono">pdb_projects</span>, <span class="mono">pdb_mentions</span>, <span class="mono">pdb_sql</span> via Python intégré (Stata 16+) et une voie <span class="mono">shell curl</span> + <span class="mono">import delimited</span> pour toute version. Non testé sur Stata ici.</p>${codeBlock(`global PDB_KEY "${k}"\ndo pdb_api.do // définit les programmes\npdb_projects, filters(type=data_center min_amount=5e8 sort=amount)\npdb_sql "select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1"\n\n* sans Python :\nshell curl -s "${location.origin}/v1/projects/export.csv?type=plant_construction" -H "X-API-Key: $PDB_KEY" -o usines.csv\nimport delimited using "usines.csv", clear varnames(1) encoding(utf8)`)}</div> | |
| 326 | + <div class="card"><h3>Prise en main</h3>${codeBlock(`from pdb_client import PDB\n\npdb = PDB("${k}") # ou export PDB_API_KEY=...\nprint(pdb.health())\n\nov = pdb.stats()["overview"]\nprint(ov["projects"], "projets")\n\n# une page\npage = pdb.projects(type="data_center", min_amount=5e8, sort="amount", limit=10)\nfor p in page["items"]:\n print(p["ticker"], p["project_name"], p["total_amount_usd"])`)}</div> | |
| 327 | + <div class="card"><h3>Tout un jeu de données en DataFrame</h3>${codeBlock(`# pagination automatique (500 par page)\nrows = pdb.iter_projects(sector="Utilities")\ndf = PDB.to_dataframe(rows)\nprint(df.shape)\n\n# mentions 8-K de 2025 sur l'IA\nm = PDB.to_dataframe(pdb.iter_mentions(type="ai_initiative", form="8-K", date_from="2025-01-01"))\nm.groupby("ticker").size().sort_values(ascending=False).head(10)\n\n# export CSV\nPDB.to_csv(pdb.iter_projects(type="plant_construction"), "usines.csv")`)}</div> | |
| 328 | + <div class="card"><h3>Fiche, similaires, entreprise, graphe</h3>${codeBlock(`fiche = pdb.project(page["items"][0]["project_id"])\nfiche["project"], fiche["timeline"], fiche["mentions"], fiche["similar"]\n\npdb.similar(fiche["project"]["project_id"], k=5)\n\nduk = pdb.company("DUK") # ticker ou CIK\nduk["by_type"], duk["projects"][:3]\n\nnode = pdb.graph_node("L:texas", limit=500)\n[e["neighbor_label"] for e in node["edges"]][:10]`)}</div> | |
| 329 | + <div class="card"><h3>Sémantique et SQL</h3>${codeBlock(`for h in pdb.semantic("battery cell factory", k=5):\n print(round(h["score"], 3), h["ticker"], h["project_name"])\n\nres = pdb.sql("""\n select year(filing_date) as yr, count(*) n\n from project_mentions where project_type = 'data_center'\n group by 1 order by 1\n""")\nimport pandas as pd\npd.DataFrame(res["rows"], columns=res["columns"]).set_index("yr").plot()\n\n# ou directement en enregistrements\npdb.sql_records("select ticker, count(*) n from projects group by 1 order by n desc limit 5")`)}</div> | |
| 330 | + <div class="card"><h3>Dans un notebook Jupyter / Colab</h3>${codeBlock(`!curl -sO ${location.origin}/sdk/pdb_client.py\nimport os; os.environ["PDB_API_KEY"] = "${k}"\nfrom pdb_client import PDB\npdb = PDB()`)}</div> | |
| 331 | + <div class="card"><h3>JavaScript / TypeScript (sans SDK)</h3>${codeBlock(`const BASE = "${location.origin}/v1";\nconst H = { "X-API-Key": process.env.PDB_API_KEY };\n\nexport async function pdb(path, params = {}) {\n const r = await fetch(\`\${BASE}/\${path}?\${new URLSearchParams(params)}\`, { headers: H });\n if (!r.ok) throw new Error(\`HTTP \${r.status}: \${(await r.json()).detail}\`);\n return r.json();\n}\nexport async function* iterate(path, params = {}) {\n for (let offset = 0; ; ) {\n const page = await pdb(path, { ...params, limit: 500, offset });\n yield* page.items; offset += page.items.length;\n if (!page.items.length || offset >= page.total) return;\n }\n}`)}</div> | |
| 332 | + </div>`; | |
| 333 | + bindCopy(body); | |
| 334 | + } | |
| 335 | + | |
| 336 | + return { render }; | |
| 337 | +})(); | |
added
web/report.pdf
+0 −0
Binary file not shown.
added
web/sdk/pdb_api.R
+113 −0
@@ -0,0 +1,113 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# pdb_api.R — utiliser la PDB API (SEC Project Intelligence Database, UQO) en R | |
| 3 | +# install.packages(c("httr2", "dplyr", "ggplot2")) # une seule fois | |
| 4 | +# source("pdb_api.R") ou copier-coller ce fichier dans RStudio | |
| 5 | +# ============================================================================= | |
| 6 | +library(httr2) | |
| 7 | +library(dplyr) | |
| 8 | + | |
| 9 | +BASE <- "https://www.pdb-api.co/v1" | |
| 10 | +KEY <- Sys.getenv("PDB_API_KEY", unset = "VOTRE_CLE") # ou : KEY <- "VOTRE_CLE" | |
| 11 | + | |
| 12 | +# ---- 1. Deux fonctions génériques ------------------------------------------ | |
| 13 | +pdb_get <- function(path, ...) { | |
| 14 | + q <- Filter(Negate(is.null), list(...)) | |
| 15 | + req <- request(paste0(BASE, "/", path)) |> | |
| 16 | + req_headers(`X-API-Key` = KEY) |> | |
| 17 | + req_retry(max_tries = 4, is_transient = \(r) resp_status(r) == 429) |> | |
| 18 | + req_error(body = \(r) tryCatch(resp_body_json(r)$detail, error = \(e) NULL)) | |
| 19 | + if (length(q)) req <- req |> req_url_query(!!!q) | |
| 20 | + req |> req_perform() |> resp_body_json(simplifyVector = TRUE) | |
| 21 | +} | |
| 22 | + | |
| 23 | +pdb_sql <- function(sql, limit = 500) { | |
| 24 | + res <- request(paste0(BASE, "/sql")) |> | |
| 25 | + req_headers(`X-API-Key` = KEY) |> | |
| 26 | + req_body_json(list(sql = sql, limit = limit)) |> | |
| 27 | + req_perform() |> resp_body_json(simplifyVector = TRUE) | |
| 28 | + df <- as.data.frame(res$rows, stringsAsFactors = FALSE) | |
| 29 | + if (nrow(df)) names(df) <- res$columns | |
| 30 | + df | |
| 31 | +} | |
| 32 | + | |
| 33 | +# Tout récupérer (pagination automatique, 500 par page) -> data.frame | |
| 34 | +pdb_all <- function(path, ...) { | |
| 35 | + pages <- list(); offset <- 0 | |
| 36 | + repeat { | |
| 37 | + page <- pdb_get(path, ..., limit = 500, offset = offset) | |
| 38 | + items <- as.data.frame(page$items) | |
| 39 | + pages[[length(pages) + 1]] <- items | |
| 40 | + offset <- offset + nrow(items) | |
| 41 | + if (nrow(items) == 0 || offset >= page$total) break | |
| 42 | + } | |
| 43 | + bind_rows(pages) | |
| 44 | +} | |
| 45 | + | |
| 46 | +# ---- 2. Exemples ----------------------------------------------------------- | |
| 47 | +if (sys.nframe() == 0) { # exécuté seulement si on lance le fichier directement | |
| 48 | + | |
| 49 | + # a) vérifier le service et la clé | |
| 50 | + print(pdb_get("health")) | |
| 51 | + ov <- pdb_get("stats")$overview | |
| 52 | + cat(sprintf("%s projets · %s mentions · %s entreprises\n", | |
| 53 | + format(ov$projects, big.mark = " "), format(ov$mentions, big.mark = " "), ov$companies)) | |
| 54 | + | |
| 55 | + # b) une page de projets filtrés (centres de données > 500 M$) | |
| 56 | + dc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items | |
| 57 | + print(dc[, c("ticker", "project_name", "canonical_location", "total_amount_usd", "first_seen")]) | |
| 58 | + | |
| 59 | + # c) tout un secteur en data.frame, puis agrégation dplyr | |
| 60 | + util <- pdb_all("projects", sector = "Utilities") | |
| 61 | + util |> | |
| 62 | + group_by(project_type) |> | |
| 63 | + summarise(n = n(), capital_gusd = sum(total_amount_usd, na.rm = TRUE) / 1e9, .groups = "drop") |> | |
| 64 | + arrange(desc(n)) |> | |
| 65 | + print(n = 10) | |
| 66 | + | |
| 67 | + # d) export CSV direct (le plus simple pour un jeu de données complet) | |
| 68 | + csv <- request(paste0(BASE, "/projects/export.csv")) |> | |
| 69 | + req_headers(`X-API-Key` = KEY) |> | |
| 70 | + req_url_query(type = "plant_construction") |> | |
| 71 | + req_perform() |> resp_body_string() | |
| 72 | + usines <- read.csv(text = csv) | |
| 73 | + cat("usines :", nrow(usines), "lignes\n") | |
| 74 | + | |
| 75 | + # e) fiche d'un projet : chronologie et projets similaires | |
| 76 | + pid <- pdb_get("projects", ticker = "TSLA", q = "energy storage", limit = 1)$items$project_id[1] | |
| 77 | + fiche <- pdb_get(paste0("projects/", pid)) | |
| 78 | + cat(fiche$project$project_name, "-", fiche$project$status, "\n") | |
| 79 | + print(fiche$timeline[, c("filing_date", "form_type", "status", "amount_usd")]) | |
| 80 | + print(fiche$similar[, c("score", "ticker", "project_name")]) | |
| 81 | + | |
| 82 | + # f) profil d'une entreprise | |
| 83 | + duk <- pdb_get("companies/DUK") | |
| 84 | + print(duk$by_type[, c("label", "n", "amount_usd")]) | |
| 85 | + barplot(duk$by_year$n, names.arg = duk$by_year$year, las = 2, main = "Duke Energy — mentions par année") | |
| 86 | + | |
| 87 | + # g) recherche sémantique (langage naturel, anglais recommandé) | |
| 88 | + sem <- pdb_get("search/semantic", q = "battery cell factory", k = 5)$items | |
| 89 | + print(sem[, c("score", "ticker", "project_name")]) | |
| 90 | + | |
| 91 | + # h) SQL libre (lecture seule, DuckDB) | |
| 92 | + ia <- pdb_sql(" | |
| 93 | + select year(filing_date) as yr, count(*) n | |
| 94 | + from project_mentions | |
| 95 | + where project_type = 'ai_initiative' | |
| 96 | + group by 1 order by 1") | |
| 97 | + print(ia) | |
| 98 | + plot(ia$yr, ia$n, type = "b", xlab = "année", ylab = "mentions", main = "Initiatives IA dans les filings") | |
| 99 | + | |
| 100 | + # i) panel entreprise × année pour l'économétrie | |
| 101 | + panel <- pdb_sql(" | |
| 102 | + select cik, any_value(ticker) ticker, any_value(sector) sector, | |
| 103 | + year(first_seen) as yr, count(*) n_projets, | |
| 104 | + sum(total_amount_usd) capital_usd | |
| 105 | + from projects group by 1, 4 order by 1, 4", limit = 5000) | |
| 106 | + cat("panel :", nrow(panel), "lignes (entreprise × année)\n") | |
| 107 | + # summary(lm(log1p(n_projets) ~ factor(yr) + factor(sector), data = panel)) | |
| 108 | + | |
| 109 | + # j) graphe : projets rattachés au Texas | |
| 110 | + tx <- pdb_get("graph/node/L:texas", limit = 500) | |
| 111 | + cat("degré de L:texas :", tx$degree, "\n") | |
| 112 | + print(head(tx$edges[, c("rel", "neighbor_type", "neighbor_label")])) | |
| 113 | +} | |
added
web/sdk/pdb_api.do
+150 −0
@@ -0,0 +1,150 @@ | ||
| 1 | +* ============================================================================= | |
| 2 | +* pdb_api.do — utiliser la PDB API (SEC Project Intelligence Database, UQO) dans Stata | |
| 3 | +* | |
| 4 | +* Stata ne peut pas envoyer d'en-tête HTTP avec `copy` ou `import delimited`. | |
| 5 | +* Deux voies : | |
| 6 | +* A) Stata 16+ avec Python intégré (recommandé) : `python:` appelle l'API et écrit un CSV. | |
| 7 | +* B) Toute version : `shell curl` télécharge le CSV / JSON, puis `import delimited`. | |
| 8 | +* Remplacer VOTRE_CLE ci-dessous (ou définir la variable d'environnement PDB_API_KEY). | |
| 9 | +* ============================================================================= | |
| 10 | +clear all | |
| 11 | +set more off | |
| 12 | +global PDB_BASE "https://www.pdb-api.co/v1" | |
| 13 | +global PDB_KEY "VOTRE_CLE" | |
| 14 | + | |
| 15 | +* ----------------------------------------------------------------------------- | |
| 16 | +* A) Voie Python (Stata 16+) — définit trois programmes : pdb_projects, pdb_mentions, pdb_sql | |
| 17 | +* ----------------------------------------------------------------------------- | |
| 18 | +python: | |
| 19 | +import csv, json, os, urllib.parse, urllib.request | |
| 20 | +from sfi import Macro | |
| 21 | + | |
| 22 | +BASE = Macro.getGlobal("PDB_BASE"); KEY = os.environ.get("PDB_API_KEY") or Macro.getGlobal("PDB_KEY") | |
| 23 | + | |
| 24 | +def _req(path, params=None, body=None): | |
| 25 | + url = f"{BASE}/{path}" + (f"?{urllib.parse.urlencode(params)}" if params else "") | |
| 26 | + data = json.dumps(body).encode() if body is not None else None | |
| 27 | + req = urllib.request.Request(url, data=data, headers={"X-API-Key": KEY, "Content-Type": "application/json"}) | |
| 28 | + with urllib.request.urlopen(req, timeout=120) as r: | |
| 29 | + return json.loads(r.read()) | |
| 30 | + | |
| 31 | +def _flat(rec): | |
| 32 | + return {k: (" | ".join(map(str, v)) if isinstance(v, list) else v) for k, v in rec.items()} | |
| 33 | + | |
| 34 | +def _write(rows, path): | |
| 35 | + if not rows: | |
| 36 | + open(path, "w").close(); return 0 | |
| 37 | + with open(path, "w", newline="", encoding="utf-8") as f: | |
| 38 | + w = csv.DictWriter(f, fieldnames=list(rows[0].keys())); w.writeheader() | |
| 39 | + for r in rows: w.writerow(_flat(r)) | |
| 40 | + return len(rows) | |
| 41 | + | |
| 42 | +def pdb_pages(path, csvfile, **filters): | |
| 43 | + """Tous les enregistrements paginés (500 par page) -> CSV.""" | |
| 44 | + rows, offset = [], 0 | |
| 45 | + while True: | |
| 46 | + page = _req(path, {**filters, "limit": 500, "offset": offset}) | |
| 47 | + rows += page["items"]; offset += len(page["items"]) | |
| 48 | + if not page["items"] or offset >= page["total"]: break | |
| 49 | + n = _write(rows, csvfile); Macro.setLocal("pdb_n", str(n)) | |
| 50 | + | |
| 51 | +def pdb_sql(sql, csvfile, limit=5000): | |
| 52 | + res = _req("sql", body={"sql": sql, "limit": limit}) | |
| 53 | + rows = [dict(zip(res["columns"], r)) for r in res["rows"]] | |
| 54 | + n = _write(rows, csvfile); Macro.setLocal("pdb_n", str(n)) | |
| 55 | +end | |
| 56 | + | |
| 57 | +* --- Programmes Stata enveloppant les fonctions Python ------------------------ | |
| 58 | +capture program drop pdb_projects | |
| 59 | +program define pdb_projects | |
| 60 | + * usage : pdb_projects, filters(type=data_center min_amount=5e8) [file(x.csv)] | |
| 61 | + syntax , [FILters(string) FILE(string)] | |
| 62 | + if "`file'" == "" local file "pdb_projects.csv" | |
| 63 | + local kw "" | |
| 64 | + foreach f of local filters { | |
| 65 | + gettoken k v : f, parse("=") | |
| 66 | + local v = subinstr("`v'", "=", "", 1) | |
| 67 | + local kw `"`kw' `k'="`v'","' | |
| 68 | + } | |
| 69 | + python: pdb_pages("projects", "`file'" `kw') | |
| 70 | + import delimited using "`file'", clear varnames(1) encoding(utf8) stringcols(_all) | |
| 71 | + destring total_amount_usd n_mentions n_filings first_year last_year avg_confidence, replace force | |
| 72 | + gen date_first = date(first_seen, "YMD"); format date_first %td | |
| 73 | + gen date_last = date(last_seen, "YMD"); format date_last %td | |
| 74 | + di as txt "`pdb_n' projets importés" | |
| 75 | +end | |
| 76 | + | |
| 77 | +capture program drop pdb_mentions | |
| 78 | +program define pdb_mentions | |
| 79 | + syntax , [FILters(string) FILE(string)] | |
| 80 | + if "`file'" == "" local file "pdb_mentions.csv" | |
| 81 | + local kw "" | |
| 82 | + foreach f of local filters { | |
| 83 | + gettoken k v : f, parse("=") | |
| 84 | + local v = subinstr("`v'", "=", "", 1) | |
| 85 | + local kw `"`kw' `k'="`v'","' | |
| 86 | + } | |
| 87 | + python: pdb_pages("mentions", "`file'" `kw') | |
| 88 | + import delimited using "`file'", clear varnames(1) encoding(utf8) stringcols(_all) | |
| 89 | + destring amount_usd confidence, replace force | |
| 90 | + gen date_filing = date(filing_date, "YMD"); format date_filing %td | |
| 91 | + di as txt "`pdb_n' mentions importées" | |
| 92 | +end | |
| 93 | + | |
| 94 | +capture program drop pdb_sql | |
| 95 | +program define pdb_sql | |
| 96 | + * usage : pdb_sql "select ... " [, file(x.csv) limit(5000)] | |
| 97 | + syntax anything(everything name=sql) [, FILE(string) LIMit(integer 5000)] | |
| 98 | + if "`file'" == "" local file "pdb_sql.csv" | |
| 99 | + local sql = subinstr(`"`sql'"', `"""', "", .) | |
| 100 | + python: pdb_sql("""`sql'""", "`file'", `limit') | |
| 101 | + import delimited using "`file'", clear varnames(1) encoding(utf8) | |
| 102 | + di as txt "`pdb_n' lignes importées" | |
| 103 | +end | |
| 104 | + | |
| 105 | +* ----------------------------------------------------------------------------- | |
| 106 | +* Exemples (voie A) | |
| 107 | +* ----------------------------------------------------------------------------- | |
| 108 | + | |
| 109 | +* 1. Centres de données > 500 M$ | |
| 110 | +pdb_projects, filters(type=data_center min_amount=5e8 sort=amount) | |
| 111 | +list ticker project_name canonical_location total_amount_usd first_seen in 1/10, clean | |
| 112 | + | |
| 113 | +* 2. Tout un secteur, puis agrégation | |
| 114 | +pdb_projects, filters(sector=Utilities) file(utilities.csv) | |
| 115 | +tab project_type, sort | |
| 116 | +collapse (count) n=project_id (sum) capital=total_amount_usd, by(project_type) | |
| 117 | +gsort -n | |
| 118 | +list, clean | |
| 119 | + | |
| 120 | +* 3. Mentions 8-K sur l'IA depuis 2024 | |
| 121 | +pdb_mentions, filters(type=ai_initiative form=8-K date_from=2024-01-01) | |
| 122 | +gen yr = year(date_filing) | |
| 123 | +tab yr | |
| 124 | + | |
| 125 | +* 4. SQL libre : mentions IA par année | |
| 126 | +pdb_sql "select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1" | |
| 127 | +twoway connected n yr, title("Initiatives IA dans les filings") ytitle("mentions") | |
| 128 | + | |
| 129 | +* 5. Panel entreprise x année pour l'économétrie | |
| 130 | +pdb_sql "select cik, any_value(ticker) ticker, any_value(sector) sector, year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital_usd from projects group by 1, 4 order by 1, 4", file(panel.csv) | |
| 131 | +encode cik, gen(id) | |
| 132 | +xtset id yr | |
| 133 | +xtpoisson n_projets i.yr, fe // intensité d'annonce de projets, effets fixes entreprise | |
| 134 | +* xtreg ln_capital i.yr, fe après : gen ln_capital = ln(capital_usd) | |
| 135 | + | |
| 136 | +* ----------------------------------------------------------------------------- | |
| 137 | +* B) Voie curl (toute version de Stata) : CSV d'export puis import delimited | |
| 138 | +* ----------------------------------------------------------------------------- | |
| 139 | +* Sur macOS / Linux : `shell` ; sur Windows : remplacer par `winexec` ou `shell curl.exe ...` | |
| 140 | +shell curl -s "$PDB_BASE/projects/export.csv?type=plant_construction" -H "X-API-Key: $PDB_KEY" -o usines.csv | |
| 141 | +import delimited using "usines.csv", clear varnames(1) encoding(utf8) stringcols(_all) | |
| 142 | +destring total_amount_usd n_mentions n_filings avg_confidence, replace force | |
| 143 | +gen date_first = date(first_seen, "YMD"); format date_first %td | |
| 144 | +describe, short | |
| 145 | +summarize total_amount_usd, detail | |
| 146 | + | |
| 147 | +* SQL via curl : la réponse est en JSON ; convertir en CSV avec python3 du système | |
| 148 | +shell curl -s -X POST "$PDB_BASE/sql" -H "X-API-Key: $PDB_KEY" -H "Content-Type: application/json" -d "{\"sql\":\"select project_type, count(*) n, sum(total_amount_usd) capital from projects group by 1 order by n desc\",\"limit\":100}" | python3 -c "import csv,json,sys; d=json.load(sys.stdin); w=csv.writer(sys.stdout); w.writerow(d['columns']); w.writerows(d['rows'])" > types.csv | |
| 149 | +import delimited using "types.csv", clear varnames(1) | |
| 150 | +list, clean | |
added
web/sdk/pdb_api.m
+111 −0
@@ -0,0 +1,111 @@ | ||
| 1 | +%% pdb_api.m — utiliser la PDB API (SEC Project Intelligence Database, UQO) dans MATLAB | |
| 2 | +% MATLAB R2018a ou plus récent (webread / webwrite / jsondecode). Aucun toolbox requis. | |
| 3 | +% Usage : ouvrir ce fichier, remplacer KEY, exécuter section par section (Ctrl+Entrée). | |
| 4 | +% Les fonctions pdb_get / pdb_sql / pdb_all sont définies en fin de fichier (fonctions locales de script). | |
| 5 | + | |
| 6 | +BASE = "https://www.pdb-api.co/v1"; | |
| 7 | +KEY = getenv("PDB_API_KEY"); if isempty(KEY), KEY = "VOTRE_CLE"; end | |
| 8 | + | |
| 9 | +%% 1. Vérifier le service et la clé | |
| 10 | +disp(pdb_get(BASE, KEY, "health")) | |
| 11 | +ov = pdb_get(BASE, KEY, "stats").overview; | |
| 12 | +fprintf("%d projets, %d mentions, %d entreprises\n", ov.projects, ov.mentions, ov.companies); | |
| 13 | + | |
| 14 | +%% 2. Centres de données > 500 M$, du plus gros au plus petit | |
| 15 | +r = pdb_get(BASE, KEY, "projects", type="data_center", min_amount=5e8, sort="amount", limit=10); | |
| 16 | +dc = struct2table(r.items); % tableau MATLAB | |
| 17 | +disp(dc(:, {'ticker','project_name','canonical_location','total_amount_usd','first_seen'})) | |
| 18 | + | |
| 19 | +%% 3. Tout un secteur (pagination automatique) puis agrégation | |
| 20 | +util = pdb_all(BASE, KEY, "projects", sector="Utilities"); % ~1 956 lignes | |
| 21 | +G = groupsummary(util, "project_type", "sum", "total_amount_usd"); | |
| 22 | +G = sortrows(G, "GroupCount", "descend"); | |
| 23 | +disp(G(1:10, :)) | |
| 24 | + | |
| 25 | +%% 4. Export CSV direct (le plus simple pour un jeu de données complet) | |
| 26 | +opts = weboptions("HeaderFields", ["X-API-Key", KEY], "ContentType", "text", "Timeout", 120); | |
| 27 | +csvText = webread(BASE + "/projects/export.csv", "type", "plant_construction", opts); | |
| 28 | +fid = fopen("usines.csv", "w"); fwrite(fid, csvText); fclose(fid); | |
| 29 | +usines = readtable("usines.csv", "TextType", "string"); | |
| 30 | +fprintf("usines : %d lignes\n", height(usines)); | |
| 31 | + | |
| 32 | +%% 5. Fiche d'un projet : chronologie et projets similaires | |
| 33 | +p1 = pdb_get(BASE, KEY, "projects", ticker="TSLA", q="energy storage", limit=1); | |
| 34 | +fiche = pdb_get(BASE, KEY, "projects/" + string(p1.items(1).project_id)); | |
| 35 | +fprintf("%s — %s\n", fiche.project.project_name, fiche.project.status); | |
| 36 | +disp(struct2table(fiche.timeline)(:, {'filing_date','form_type','status','amount_usd'})) | |
| 37 | +disp(struct2table(fiche.similar)(:, {'score','ticker','project_name'})) | |
| 38 | + | |
| 39 | +%% 6. Profil d'une entreprise (ticker ou CIK) | |
| 40 | +duk = pdb_get(BASE, KEY, "companies/DUK"); | |
| 41 | +disp(struct2table(duk.by_type)(:, {'label','n','amount_usd'})) | |
| 42 | +figure; bar([duk.by_year.n]); xticklabels(string([duk.by_year.year])); xticks(1:numel(duk.by_year)); | |
| 43 | +title("Duke Energy — mentions par année"); ylabel("mentions"); | |
| 44 | + | |
| 45 | +%% 7. Recherche sémantique (langage naturel, anglais recommandé) | |
| 46 | +sem = pdb_get(BASE, KEY, "search/semantic", q="battery cell factory", k=5); | |
| 47 | +disp(struct2table(sem.items)(:, {'score','ticker','project_name'})) | |
| 48 | + | |
| 49 | +%% 8. SQL libre (DuckDB, lecture seule) -> table | |
| 50 | +ia = pdb_sql(BASE, KEY, join([ ... | |
| 51 | + "select year(filing_date) as yr, count(*) n", ... | |
| 52 | + "from project_mentions where project_type = 'ai_initiative'", ... | |
| 53 | + "group by 1 order by 1"], " ")); | |
| 54 | +figure; plot(ia.yr, ia.n, "-o"); xlabel("année"); ylabel("mentions"); title("Initiatives IA dans les filings"); | |
| 55 | + | |
| 56 | +%% 9. Panel entreprise × année pour l'économétrie | |
| 57 | +panel = pdb_sql(BASE, KEY, join([ ... | |
| 58 | + "select cik, any_value(ticker) ticker, any_value(sector) sector,", ... | |
| 59 | + " year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital_usd", ... | |
| 60 | + "from projects group by 1, 4 order by 1, 4"], " "), 5000); | |
| 61 | +fprintf("panel : %d lignes\n", height(panel)); | |
| 62 | +% mdl = fitlm(panel, "n_projets ~ yr + sector"); % Statistics and Machine Learning Toolbox | |
| 63 | + | |
| 64 | +%% 10. Graphe : projets rattachés au Texas | |
| 65 | +tx = pdb_get(BASE, KEY, "graph/node/L:texas", limit=500); | |
| 66 | +fprintf("degré de L:texas : %d\n", tx.degree); | |
| 67 | +disp(struct2table(tx.edges)(1:6, {'rel','neighbor_type','neighbor_label'})) | |
| 68 | + | |
| 69 | +%% ------------------------------------------------------------------ fonctions locales | |
| 70 | +function out = pdb_get(base, key, path, varargin) | |
| 71 | +% GET générique : pdb_get(BASE, KEY, "projects", type="data_center", limit=10) | |
| 72 | + opts = weboptions("HeaderFields", ["X-API-Key", key], "ContentType", "json", "Timeout", 120); | |
| 73 | + args = {}; | |
| 74 | + for i = 1:2:numel(varargin) | |
| 75 | + v = varargin{i+1}; | |
| 76 | + if isnumeric(v), v = num2str(v, "%.10g"); end | |
| 77 | + args(end+1:end+2) = {char(varargin{i}), char(string(v))}; | |
| 78 | + end | |
| 79 | + try | |
| 80 | + out = webread(base + "/" + path, args{:}, opts); | |
| 81 | + catch e | |
| 82 | + error("PDB API : %s", e.message); % 401 = clé invalide, 404 = introuvable, 429 = trop de requêtes | |
| 83 | + end | |
| 84 | +end | |
| 85 | + | |
| 86 | +function T = pdb_sql(base, key, sql, limit) | |
| 87 | +% Requête SQL de lecture -> table MATLAB | |
| 88 | + if nargin < 4, limit = 500; end | |
| 89 | + opts = weboptions("HeaderFields", ["X-API-Key", key], "MediaType", "application/json", "ContentType", "json", "Timeout", 120); | |
| 90 | + res = webwrite(base + "/sql", struct("sql", sql, "limit", limit), opts); | |
| 91 | + if isempty(res.rows), T = table(); return; end | |
| 92 | + rows = res.rows; | |
| 93 | + if iscell(rows), rows = vertcat(rows{:}); end % cellules si types mixtes | |
| 94 | + T = cell2table(num2cell(rows), "VariableNames", matlab.lang.makeValidName(string(res.columns))); | |
| 95 | + if iscell(rows) || ~isnumeric(rows) | |
| 96 | + T = cell2table(rows, "VariableNames", matlab.lang.makeValidName(string(res.columns))); | |
| 97 | + end | |
| 98 | +end | |
| 99 | + | |
| 100 | +function T = pdb_all(base, key, path, varargin) | |
| 101 | +% Pagination automatique (500 par page) -> table complète | |
| 102 | + offset = 0; parts = {}; | |
| 103 | + while true | |
| 104 | + page = pdb_get(base, key, path, varargin{:}, "limit", 500, "offset", offset); | |
| 105 | + if isempty(page.items), break; end | |
| 106 | + parts{end+1} = struct2table(page.items, "AsArray", true); %#ok<AGROW> | |
| 107 | + offset = offset + numel(page.items); | |
| 108 | + if offset >= page.total, break; end | |
| 109 | + end | |
| 110 | + T = vertcat(parts{:}); | |
| 111 | +end | |
added
web/sdk/pdb_client.py
+203 −0
@@ -0,0 +1,203 @@ | ||
| 1 | +"""pdb_client — client Python minimal pour la PDB API (SEC Project Intelligence Database, UQO). | |
| 2 | + | |
| 3 | +Aucune dépendance obligatoire (urllib). pandas est optionnel pour `to_dataframe`. | |
| 4 | + | |
| 5 | + from pdb_client import PDB | |
| 6 | + pdb = PDB("VOTRE_CLE") # ou variable d'environnement PDB_API_KEY | |
| 7 | + pdb.stats()["overview"] | |
| 8 | + for p in pdb.iter_projects(type="data_center", min_amount=5e8): | |
| 9 | + print(p["ticker"], p["project_name"], p["total_amount_usd"]) | |
| 10 | + df = pdb.to_dataframe(pdb.iter_projects(sector="Utilities")) | |
| 11 | + pdb.sql("select project_type, count(*) n from projects group by 1 order by n desc") | |
| 12 | +""" | |
| 13 | +from __future__ import annotations | |
| 14 | + | |
| 15 | +import csv | |
| 16 | +import io | |
| 17 | +import json | |
| 18 | +import os | |
| 19 | +import time | |
| 20 | +import urllib.error | |
| 21 | +import urllib.parse | |
| 22 | +import urllib.request | |
| 23 | +from typing import Any, Iterator | |
| 24 | + | |
| 25 | +__version__ = "1.0.0" | |
| 26 | +DEFAULT_BASE = "https://www.pdb-api.co/v1" | |
| 27 | + | |
| 28 | + | |
| 29 | +class PDBError(RuntimeError): | |
| 30 | + def __init__(self, status: int, detail: str, url: str): | |
| 31 | + super().__init__(f"HTTP {status} — {detail} ({url})") | |
| 32 | + self.status, self.detail, self.url = status, detail, url | |
| 33 | + | |
| 34 | + | |
| 35 | +class PDB: | |
| 36 | + def __init__(self, api_key: str | None = None, base_url: str = DEFAULT_BASE, timeout: float = 60.0, retries: int = 3): | |
| 37 | + self.api_key = api_key or os.environ.get("PDB_API_KEY", "") | |
| 38 | + if not self.api_key: | |
| 39 | + raise ValueError("Clé d'API manquante : PDB(api_key=...) ou variable PDB_API_KEY.") | |
| 40 | + self.base_url = base_url.rstrip("/") | |
| 41 | + self.timeout = timeout | |
| 42 | + self.retries = retries | |
| 43 | + | |
| 44 | + # ------------------------------------------------------------ transport | |
| 45 | + def request(self, method: str, path: str, params: dict | None = None, body: dict | None = None, raw: bool = False) -> Any: | |
| 46 | + q = {k: (str(v).lower() if isinstance(v, bool) else v) for k, v in (params or {}).items() if v is not None and v != ""} | |
| 47 | + url = f"{self.base_url}/{path.lstrip('/')}" + (f"?{urllib.parse.urlencode(q)}" if q else "") | |
| 48 | + data = json.dumps(body).encode() if body is not None else None | |
| 49 | + headers = {"X-API-Key": self.api_key, "Accept": "application/json", "User-Agent": f"pdb_client/{__version__}"} | |
| 50 | + if data is not None: | |
| 51 | + headers["Content-Type"] = "application/json" | |
| 52 | + for attempt in range(self.retries + 1): | |
| 53 | + req = urllib.request.Request(url, data=data, method=method, headers=headers) | |
| 54 | + try: | |
| 55 | + with urllib.request.urlopen(req, timeout=self.timeout) as r: | |
| 56 | + payload = r.read() | |
| 57 | + return payload.decode() if raw else json.loads(payload) | |
| 58 | + except urllib.error.HTTPError as e: | |
| 59 | + detail = e.read().decode(errors="replace") | |
| 60 | + try: | |
| 61 | + detail = json.loads(detail).get("detail", detail) | |
| 62 | + except Exception: | |
| 63 | + pass | |
| 64 | + if e.code in (429, 502, 503, 504) and attempt < self.retries: | |
| 65 | + time.sleep(1.5 * (attempt + 1)) | |
| 66 | + continue | |
| 67 | + raise PDBError(e.code, detail, url) from None | |
| 68 | + | |
| 69 | + def get(self, path: str, **params) -> Any: | |
| 70 | + return self.request("GET", path, params) | |
| 71 | + | |
| 72 | + # ------------------------------------------------------------ découverte | |
| 73 | + def health(self) -> dict: | |
| 74 | + return self.get("health") | |
| 75 | + | |
| 76 | + def stats(self) -> dict: | |
| 77 | + return self.get("stats") | |
| 78 | + | |
| 79 | + def taxonomy(self) -> list[dict]: | |
| 80 | + return self.get("taxonomy") | |
| 81 | + | |
| 82 | + def sectors(self) -> list[dict]: | |
| 83 | + return self.get("sectors") | |
| 84 | + | |
| 85 | + def technologies(self, q: str | None = None, limit: int = 50) -> list[dict]: | |
| 86 | + return self.get("technologies", q=q, limit=limit) | |
| 87 | + | |
| 88 | + def locations(self, q: str | None = None, limit: int = 50) -> list[dict]: | |
| 89 | + return self.get("locations", q=q, limit=limit) | |
| 90 | + | |
| 91 | + def partners(self, q: str | None = None, limit: int = 50) -> list[dict]: | |
| 92 | + return self.get("partners", q=q, limit=limit) | |
| 93 | + | |
| 94 | + def schema(self) -> list[dict]: | |
| 95 | + return self.get("schema") | |
| 96 | + | |
| 97 | + # ------------------------------------------------------------ projets | |
| 98 | + def projects(self, **filters) -> dict: | |
| 99 | + """Une page : {total, limit, offset, items}. Filtres : q, type, sector, status, ticker, cik, location, tech, | |
| 100 | + partner, min_amount, max_amount, year_from, year_to, min_confidence, has_amount, sort, order, limit, offset.""" | |
| 101 | + return self.get("projects", **filters) | |
| 102 | + | |
| 103 | + def iter_projects(self, page_size: int = 500, max_items: int | None = None, **filters) -> Iterator[dict]: | |
| 104 | + """Itère sur tous les projets correspondant aux filtres (pagination automatique).""" | |
| 105 | + yield from self._paginate("projects", page_size, max_items, **filters) | |
| 106 | + | |
| 107 | + def project(self, project_id: str) -> dict: | |
| 108 | + """Fiche complète : {project, timeline, mentions, graph, similar}.""" | |
| 109 | + return self.get(f"projects/{project_id}") | |
| 110 | + | |
| 111 | + def similar(self, project_id: str, k: int = 10) -> list[dict]: | |
| 112 | + return self.get(f"projects/{project_id}/similar", k=k) | |
| 113 | + | |
| 114 | + def projects_csv(self, **filters) -> str: | |
| 115 | + return self.request("GET", "projects/export.csv", filters, raw=True) | |
| 116 | + | |
| 117 | + # ------------------------------------------------------------ mentions / sections | |
| 118 | + def mentions(self, **filters) -> dict: | |
| 119 | + return self.get("mentions", **filters) | |
| 120 | + | |
| 121 | + def iter_mentions(self, page_size: int = 500, max_items: int | None = None, **filters) -> Iterator[dict]: | |
| 122 | + yield from self._paginate("mentions", page_size, max_items, **filters) | |
| 123 | + | |
| 124 | + def mention(self, mention_id: str) -> dict: | |
| 125 | + return self.get(f"mentions/{mention_id}") | |
| 126 | + | |
| 127 | + def section(self, section_id: str, highlight: str | None = None) -> dict: | |
| 128 | + return self.get(f"sections/{section_id}", highlight=highlight) | |
| 129 | + | |
| 130 | + # ------------------------------------------------------------ entreprises | |
| 131 | + def companies(self, **filters) -> dict: | |
| 132 | + return self.get("companies", **filters) | |
| 133 | + | |
| 134 | + def iter_companies(self, page_size: int = 500, **filters) -> Iterator[dict]: | |
| 135 | + yield from self._paginate("companies", page_size, None, **filters) | |
| 136 | + | |
| 137 | + def company(self, ticker_or_cik: str) -> dict: | |
| 138 | + return self.get(f"companies/{ticker_or_cik}") | |
| 139 | + | |
| 140 | + # ------------------------------------------------------------ graphe / recherche / SQL | |
| 141 | + def graph_search(self, q: str, type: str | None = None, limit: int = 30) -> list[dict]: | |
| 142 | + return self.get("graph/search", q=q, type=type, limit=limit) | |
| 143 | + | |
| 144 | + def graph_node(self, node_id: str, limit: int = 200) -> dict: | |
| 145 | + return self.get(f"graph/node/{urllib.parse.quote(node_id, safe=':')}", limit=limit) | |
| 146 | + | |
| 147 | + def semantic(self, q: str, k: int = 20, type: str | None = None, sector: str | None = None) -> list[dict]: | |
| 148 | + return self.get("search/semantic", q=q, k=k, type=type, sector=sector)["items"] | |
| 149 | + | |
| 150 | + def sql(self, query: str, limit: int = 500) -> dict: | |
| 151 | + """Requête de lecture DuckDB. Retour : {columns, rows, truncated, elapsed_ms}.""" | |
| 152 | + return self.request("POST", "sql", body={"sql": query, "limit": limit}) | |
| 153 | + | |
| 154 | + def sql_records(self, query: str, limit: int = 500) -> list[dict]: | |
| 155 | + r = self.sql(query, limit) | |
| 156 | + return [dict(zip(r["columns"], row)) for row in r["rows"]] | |
| 157 | + | |
| 158 | + # ------------------------------------------------------------ utilitaires | |
| 159 | + def _paginate(self, path: str, page_size: int, max_items: int | None, **filters) -> Iterator[dict]: | |
| 160 | + offset, n = int(filters.pop("offset", 0) or 0), 0 | |
| 161 | + page_size = max(1, min(page_size, 500)) | |
| 162 | + while True: | |
| 163 | + page = self.get(path, limit=page_size, offset=offset, **filters) | |
| 164 | + items = page.get("items", []) | |
| 165 | + for it in items: | |
| 166 | + yield it | |
| 167 | + n += 1 | |
| 168 | + if max_items and n >= max_items: | |
| 169 | + return | |
| 170 | + offset += len(items) | |
| 171 | + if not items or offset >= page.get("total", 0): | |
| 172 | + return | |
| 173 | + | |
| 174 | + @staticmethod | |
| 175 | + def to_dataframe(records): | |
| 176 | + """Convertit une liste/itérateur de dicts en DataFrame pandas (listes jointes par ' | ').""" | |
| 177 | + import pandas as pd # optionnel | |
| 178 | + rows = [] | |
| 179 | + for r in records: | |
| 180 | + rows.append({k: (" | ".join(map(str, v)) if isinstance(v, list) else v) for k, v in r.items()}) | |
| 181 | + return pd.DataFrame(rows) | |
| 182 | + | |
| 183 | + @staticmethod | |
| 184 | + def to_csv(records, path: str) -> int: | |
| 185 | + rows = list(records) | |
| 186 | + if not rows: | |
| 187 | + return 0 | |
| 188 | + with open(path, "w", newline="", encoding="utf-8") as f: | |
| 189 | + w = csv.DictWriter(f, fieldnames=list(rows[0].keys())) | |
| 190 | + w.writeheader() | |
| 191 | + for r in rows: | |
| 192 | + w.writerow({k: (" | ".join(map(str, v)) if isinstance(v, list) else v) for k, v in r.items()}) | |
| 193 | + return len(rows) | |
| 194 | + | |
| 195 | + | |
| 196 | +if __name__ == "__main__": # petit test : python pdb_client.py VOTRE_CLE | |
| 197 | + import sys | |
| 198 | + c = PDB(sys.argv[1] if len(sys.argv) > 1 else None) | |
| 199 | + print(json.dumps(c.health(), indent=2)) | |
| 200 | + ov = c.stats()["overview"] | |
| 201 | + print(f"{ov['projects']:,} projets, {ov['mentions']:,} mentions, {ov['companies']} entreprises") | |
| 202 | + for p in c.iter_projects(type="data_center", min_amount=5e8, sort="amount", max_items=5): | |
| 203 | + print(f" {p['ticker']:6s} {p['project_name'][:50]:50s} {p['total_amount_usd']/1e9:6.1f} G$") | |
added
web/style.css
+137 −0
@@ -0,0 +1,137 @@ | ||
| 1 | +:root{ | |
| 2 | + --bleu:#003E7E;--bleu2:#0066B3;--or:#C6A300;--gris:#58595B;--gris2:#8a8c90;--ligne:#e3e7ee;--bg:#f4f6fa;--card:#fff; | |
| 3 | + --vert:#008046;--rouge:#B42318;--orange:#D67A00;--teal:#007979;--violet:#5E35B1;--encre:#14213d; | |
| 4 | + --mono:ui-monospace,SFMono-Regular,Menlo,Consolas,monospace; | |
| 5 | + --sans:-apple-system,BlinkMacSystemFont,"Segoe UI",Inter,Roboto,Helvetica,Arial,sans-serif; | |
| 6 | +} | |
| 7 | +*{box-sizing:border-box} | |
| 8 | +html,body{margin:0;height:100%;font-family:var(--sans);color:var(--encre);background:var(--bg);font-size:14.5px;line-height:1.45} | |
| 9 | +a{color:var(--bleu2);text-decoration:none}a:hover{text-decoration:underline} | |
| 10 | +h1,h2,h3{margin:0 0 .4em;line-height:1.2}h1{font-size:1.55rem;color:var(--bleu)}h2{font-size:1.15rem;color:var(--bleu)}h3{font-size:1rem;color:var(--gris)} | |
| 11 | +.muted{color:var(--gris2)}.small{font-size:.85em}.mono{font-family:var(--mono);font-size:.9em} | |
| 12 | +.shell{display:grid;grid-template-columns:250px 1fr;min-height:100vh} | |
| 13 | +.side{background:var(--bleu);color:#fff;display:flex;flex-direction:column;position:sticky;top:0;height:100vh} | |
| 14 | +.brand{display:flex;gap:10px;align-items:center;padding:18px 18px 14px;color:#fff;border-bottom:1px solid rgba(255,255,255,.12)} | |
| 15 | +.brand:hover{text-decoration:none} | |
| 16 | +.brand-mark{background:var(--or);color:var(--bleu);font-weight:800;border-radius:8px;padding:6px 8px;font-size:.95rem;letter-spacing:.5px} | |
| 17 | +.brand-text{display:flex;flex-direction:column;line-height:1.15}.brand-text small{opacity:.7;font-size:.75rem} | |
| 18 | +.nav{display:flex;flex-direction:column;padding:10px 10px;gap:2px;overflow:auto;flex:1} | |
| 19 | +.nav a{color:rgba(255,255,255,.85);padding:8px 12px;border-radius:7px;font-size:.93rem} | |
| 20 | +.nav a:hover{background:rgba(255,255,255,.08);text-decoration:none}.nav a.active{background:rgba(255,255,255,.16);color:#fff;font-weight:600} | |
| 21 | +.nav-sep{font-size:.7rem;text-transform:uppercase;letter-spacing:.12em;opacity:.55;padding:14px 12px 4px} | |
| 22 | +.side-foot{padding:14px 18px;font-size:.75rem;opacity:.75;border-top:1px solid rgba(255,255,255,.12);line-height:1.4} | |
| 23 | +.main{display:flex;flex-direction:column;min-width:0} | |
| 24 | +.topbar{display:flex;align-items:center;gap:14px;padding:12px 24px;background:#fff;border-bottom:1px solid var(--ligne);position:sticky;top:0;z-index:5} | |
| 25 | +.burger{display:none;background:none;border:1px solid var(--ligne);border-radius:6px;padding:4px 9px;font-size:1.1rem} | |
| 26 | +.search{flex:1;max-width:640px}.search input{width:100%;padding:9px 14px;border:1px solid var(--ligne);border-radius:9px;background:var(--bg);font:inherit} | |
| 27 | +.search input:focus{outline:2px solid var(--bleu2);background:#fff} | |
| 28 | +.topbar-right{margin-left:auto;display:flex;gap:10px;align-items:center} | |
| 29 | +.view{padding:22px 24px 30px;flex:1} | |
| 30 | +.foot{display:flex;justify-content:space-between;gap:20px;padding:12px 24px;border-top:1px solid var(--ligne);font-size:.78rem;color:var(--gris2);flex-wrap:wrap} | |
| 31 | +.loading{padding:40px;text-align:center;color:var(--gris2)} | |
| 32 | +.pill{display:inline-block;padding:3px 9px;border-radius:999px;font-size:.78rem;font-weight:600;background:var(--bg);color:var(--gris);white-space:nowrap} | |
| 33 | +.pill-blue{background:#e6eef8;color:var(--bleu)}.pill-or{background:#fbf3d5;color:#7a6300}.pill-vert{background:#e2f3ea;color:var(--vert)} | |
| 34 | +.pill-gris{background:#eceef1;color:var(--gris)}.pill-rouge{background:#fbe6e3;color:var(--rouge)} | |
| 35 | +.btn{display:inline-flex;align-items:center;gap:6px;padding:8px 14px;border-radius:8px;border:1px solid var(--bleu);background:var(--bleu);color:#fff;font:inherit;font-weight:600;cursor:pointer;font-size:.9rem} | |
| 36 | +.btn:hover{background:var(--bleu2);border-color:var(--bleu2);text-decoration:none} | |
| 37 | +.btn-ghost{background:#fff;color:var(--bleu);border-color:var(--ligne)}.btn-ghost:hover{background:var(--bg);color:var(--bleu)} | |
| 38 | +.btn-sm{padding:5px 10px;font-size:.82rem}.btn-or{background:var(--or);border-color:var(--or);color:var(--bleu)} | |
| 39 | +.btn:disabled{opacity:.5;cursor:default} | |
| 40 | +.grid{display:grid;gap:16px}.g2{grid-template-columns:repeat(2,minmax(0,1fr))}.g3{grid-template-columns:repeat(3,minmax(0,1fr))}.g4{grid-template-columns:repeat(4,minmax(0,1fr))} | |
| 41 | +.card{background:var(--card);border:1px solid var(--ligne);border-radius:12px;padding:16px 18px} | |
| 42 | +.card h2,.card h3{margin-bottom:10px} | |
| 43 | +.kpi{display:flex;flex-direction:column;gap:2px}.kpi b{font-size:1.5rem;color:var(--bleu);font-weight:700}.kpi span{font-size:.8rem;color:var(--gris2)} | |
| 44 | +.kpis{display:grid;grid-template-columns:repeat(auto-fit,minmax(150px,1fr));gap:12px;margin-bottom:16px} | |
| 45 | +.chart{position:relative;height:290px}.chart.tall{height:380px}.chart.short{height:220px} | |
| 46 | +.page-head{display:flex;align-items:flex-end;justify-content:space-between;gap:16px;margin-bottom:16px;flex-wrap:wrap} | |
| 47 | +.page-head p{margin:.2em 0 0;color:var(--gris)} | |
| 48 | +.filters{display:grid;grid-template-columns:repeat(auto-fill,minmax(170px,1fr));gap:10px;margin-bottom:14px} | |
| 49 | +.filters label{display:flex;flex-direction:column;gap:3px;font-size:.75rem;color:var(--gris2);text-transform:uppercase;letter-spacing:.04em} | |
| 50 | +.filters input,.filters select,.inp{padding:7px 9px;border:1px solid var(--ligne);border-radius:7px;font:inherit;font-size:.9rem;background:#fff;color:var(--encre)} | |
| 51 | +.filters input:focus,.filters select:focus,.inp:focus,textarea:focus{outline:2px solid var(--bleu2)} | |
| 52 | +.filters .span2{grid-column:span 2} | |
| 53 | +.table-wrap{overflow:auto;border:1px solid var(--ligne);border-radius:12px;background:#fff} | |
| 54 | +table{border-collapse:collapse;width:100%;font-size:.88rem} | |
| 55 | +thead th{position:sticky;top:0;background:var(--bleu);color:#fff;text-align:left;padding:9px 11px;font-weight:600;white-space:nowrap;font-size:.8rem;cursor:pointer;user-select:none} | |
| 56 | +thead th.sorted::after{content:" ▾";opacity:.8}thead th.sorted.asc::after{content:" ▴"} | |
| 57 | +tbody td{padding:8px 11px;border-top:1px solid var(--ligne);vertical-align:top} | |
| 58 | +tbody tr:hover{background:#f7f9fd}tbody tr.click{cursor:pointer} | |
| 59 | +td.num,th.num{text-align:right;font-variant-numeric:tabular-nums} | |
| 60 | +td.trunc{max-width:420px;overflow:hidden;text-overflow:ellipsis;white-space:nowrap} | |
| 61 | +.pager{display:flex;align-items:center;gap:10px;padding:10px 4px;font-size:.85rem;color:var(--gris)} | |
| 62 | +.pager .sp{flex:1} | |
| 63 | +.tags{display:flex;flex-wrap:wrap;gap:5px}.tag{background:var(--bg);border:1px solid var(--ligne);border-radius:6px;padding:2px 7px;font-size:.78rem} | |
| 64 | +.tag.t-tech{background:#e6f2f2;border-color:#bfe0e0;color:var(--teal)}.tag.t-loc{background:#fdf1e4;border-color:#f3d5b5;color:#8a4b00} | |
| 65 | +.tag.t-part{background:#efe9fb;border-color:#d9cdf5;color:var(--violet)}.tag.t-risk{background:#fbe6e3;border-color:#f3c5c0;color:var(--rouge)}.tag.t-ben{background:#e2f3ea;border-color:#bfe3cf;color:var(--vert)} | |
| 66 | +.st{font-weight:600}.st-planned{color:#7a6300}.st-in_progress{color:var(--bleu2)}.st-completed{color:var(--vert)}.st-mentioned{color:var(--gris2)} | |
| 67 | +.dl{display:grid;grid-template-columns:150px 1fr;gap:6px 14px;font-size:.9rem}.dl dt{color:var(--gris2)}.dl dd{margin:0} | |
| 68 | +.timeline{list-style:none;margin:0;padding:0 0 0 14px;border-left:2px solid var(--ligne)} | |
| 69 | +.timeline li{position:relative;padding:0 0 14px 16px} | |
| 70 | +.timeline li::before{content:"";position:absolute;left:-20px;top:5px;width:10px;height:10px;border-radius:50%;background:var(--bleu2);border:2px solid #fff} | |
| 71 | +.timeline .when{font-size:.8rem;color:var(--gris2)}.timeline .snip{color:var(--gris);font-size:.88rem;margin-top:3px} | |
| 72 | +.mention{border:1px solid var(--ligne);border-radius:10px;padding:12px 14px;margin-bottom:10px;background:#fff} | |
| 73 | +.mention .mh{display:flex;gap:10px;align-items:center;flex-wrap:wrap;margin-bottom:6px} | |
| 74 | +.mention p{margin:.3em 0} | |
| 75 | +textarea.sql{width:100%;min-height:150px;font-family:var(--mono);font-size:.88rem;padding:10px 12px;border:1px solid var(--ligne);border-radius:9px;background:#fff;resize:vertical;tab-size:2} | |
| 76 | +.sql-layout{display:grid;grid-template-columns:260px 1fr;gap:16px} | |
| 77 | +.schema{font-size:.82rem;max-height:70vh;overflow:auto} | |
| 78 | +.schema details{margin-bottom:6px}.schema summary{cursor:pointer;font-weight:600;color:var(--bleu)} | |
| 79 | +.schema ul{list-style:none;margin:4px 0 0;padding:0 0 0 10px}.schema li{font-family:var(--mono);font-size:.76rem;color:var(--gris);padding:1px 0} | |
| 80 | +.schema li span{color:var(--gris2)} | |
| 81 | +.examples{display:flex;flex-wrap:wrap;gap:6px;margin:8px 0} | |
| 82 | +.examples button{background:#fff;border:1px solid var(--ligne);border-radius:6px;padding:4px 9px;font:inherit;font-size:.8rem;cursor:pointer;color:var(--bleu)} | |
| 83 | +.examples button:hover{background:var(--bg)} | |
| 84 | +pre.code{background:#0f172a;color:#e2e8f0;border-radius:10px;padding:14px 16px;overflow:auto;font-family:var(--mono);font-size:.82rem;line-height:1.5;margin:0} | |
| 85 | +pre.code .k{color:#7dd3fc}pre.code .s{color:#fcd34d} | |
| 86 | +.err{background:#fbe6e3;color:var(--rouge);border:1px solid #f3c5c0;border-radius:8px;padding:10px 12px;margin:8px 0} | |
| 87 | +.ok{background:#e2f3ea;color:var(--vert);border:1px solid #bfe3cf;border-radius:8px;padding:8px 12px} | |
| 88 | +.api-layout{display:grid;grid-template-columns:300px 1fr;gap:16px} | |
| 89 | +.endpoints{list-style:none;margin:0;padding:0;max-height:70vh;overflow:auto} | |
| 90 | +.endpoints li{padding:7px 10px;border-radius:7px;cursor:pointer;font-size:.86rem;display:flex;gap:8px;align-items:baseline} | |
| 91 | +.endpoints li:hover,.endpoints li.active{background:var(--bg)}.endpoints .m{font-family:var(--mono);font-size:.7rem;font-weight:700;color:var(--vert);min-width:34px} | |
| 92 | +.endpoints .m.post{color:var(--orange)} | |
| 93 | +.params{display:grid;grid-template-columns:repeat(auto-fill,minmax(180px,1fr));gap:10px;margin:10px 0} | |
| 94 | +.params label{display:flex;flex-direction:column;gap:3px;font-size:.75rem;color:var(--gris2)}.params label b{color:var(--encre);font-weight:600;font-family:var(--mono);font-size:.78rem} | |
| 95 | +.tabs{display:flex;gap:4px;border-bottom:1px solid var(--ligne);margin-bottom:10px} | |
| 96 | +.tabs button{background:none;border:none;border-bottom:2px solid transparent;padding:8px 12px;font:inherit;cursor:pointer;color:var(--gris)} | |
| 97 | +.tabs button.active{color:var(--bleu);border-bottom-color:var(--bleu);font-weight:600} | |
| 98 | +.toast{position:fixed;bottom:22px;left:50%;transform:translateX(-50%);background:var(--encre);color:#fff;padding:10px 16px;border-radius:9px;font-size:.88rem;box-shadow:0 8px 30px rgba(0,0,0,.25);z-index:50} | |
| 99 | +.node-chip{display:inline-flex;align-items:center;gap:6px;padding:5px 10px;border-radius:8px;border:1px solid var(--ligne);background:#fff;margin:3px;cursor:pointer;font-size:.85rem} | |
| 100 | +.node-chip:hover{background:var(--bg)}.node-chip .nt{font-size:.68rem;text-transform:uppercase;letter-spacing:.06em;color:var(--gris2)} | |
| 101 | +.nt-Company{border-left:4px solid var(--bleu)}.nt-Project{border-left:4px solid var(--violet)}.nt-Location{border-left:4px solid var(--orange)}.nt-Technology{border-left:4px solid var(--teal)}.nt-Partner{border-left:4px solid var(--or)} | |
| 102 | +.svg-graph{width:100%;height:520px;background:#fff;border:1px solid var(--ligne);border-radius:12px} | |
| 103 | +.svg-graph text{font-family:var(--sans);font-size:11px;fill:var(--encre)} | |
| 104 | +.svg-graph line{stroke:#c9d2e0;stroke-width:1.2} | |
| 105 | +.prose{max-width:860px}.prose p{margin:.5em 0 .9em}.prose li{margin:.25em 0} | |
| 106 | +.hero{background:linear-gradient(120deg,var(--bleu),#0b4f9c 60%,#0f66b8);color:#fff;border-radius:14px;padding:22px 26px;margin-bottom:18px;display:flex;justify-content:space-between;gap:20px;align-items:center;flex-wrap:wrap} | |
| 107 | +.hero h1{color:#fff;margin-bottom:6px}.hero p{margin:0;opacity:.9;max-width:700px} | |
| 108 | +.hero .btn{background:var(--or);border-color:var(--or);color:var(--bleu)} | |
| 109 | +.score{font-variant-numeric:tabular-nums;color:var(--teal);font-weight:600} | |
| 110 | +.warn{background:#fbf3d5;border:1px solid #efdc9a;color:#6b5600;border-radius:8px;padding:8px 12px;font-size:.86rem} | |
| 111 | +.sec-text{white-space:pre-wrap;font-size:.86rem;line-height:1.55;max-height:60vh;overflow:auto;background:#fff;border:1px solid var(--ligne);border-radius:10px;padding:14px} | |
| 112 | +mark{background:#fff1a8;padding:0 2px;border-radius:2px} | |
| 113 | +@media (max-width:1100px){.g4{grid-template-columns:repeat(2,1fr)}.g3{grid-template-columns:repeat(2,1fr)}.sql-layout,.api-layout{grid-template-columns:1fr}} | |
| 114 | +@media (max-width:860px){ | |
| 115 | + .shell{grid-template-columns:1fr}.side{position:fixed;left:-260px;width:250px;transition:left .2s;z-index:20}.side.open{left:0} | |
| 116 | + .burger{display:block}.g2,.g3,.g4{grid-template-columns:1fr}.view{padding:14px}.topbar{padding:10px 14px}.dl{grid-template-columns:1fr} | |
| 117 | +} | |
| 118 | + | |
| 119 | +/* --- playground v2 --- */ | |
| 120 | +.langsel{display:flex;gap:4px;align-items:center;font-size:.8rem;color:var(--gris2)} | |
| 121 | +.langsel button{background:#fff;border:1px solid var(--ligne);border-radius:6px;padding:4px 10px;font:inherit;font-size:.8rem;cursor:pointer;color:var(--gris)} | |
| 122 | +.langsel button.active{background:var(--bleu);color:#fff;border-color:var(--bleu)} | |
| 123 | +.tabs.big{margin-bottom:16px}.tabs.big button{font-size:.95rem;padding:9px 16px} | |
| 124 | +.keybox{display:flex;gap:10px;align-items:flex-end;flex-wrap:wrap}.keybox label{display:flex;flex-direction:column;gap:3px;font-size:.78rem;color:var(--gris2);flex:1;min-width:260px} | |
| 125 | +.keybox label b{color:var(--encre)}.keybox .inp{font-family:var(--mono)} | |
| 126 | +.step{display:inline-flex;width:24px;height:24px;border-radius:50%;background:var(--or);color:var(--bleu);align-items:center;justify-content:center;font-size:.8rem;margin-right:6px;font-weight:800} | |
| 127 | +.codewrap{position:relative}.codewrap pre.code{padding-right:80px}.codewrap .copy{position:absolute;top:8px;right:8px;background:rgba(255,255,255,.12);color:#fff;border:1px solid rgba(255,255,255,.2);border-radius:6px;padding:3px 9px;font-size:.72rem;cursor:pointer} | |
| 128 | +.codewrap .copy:hover{background:rgba(255,255,255,.22)} | |
| 129 | +pre.code.json .k{color:#7dd3fc}pre.code.json .s{color:#fcd34d}pre.code.json .n{color:#a5f3a5}pre.code.json .b{color:#f9a8d4} | |
| 130 | +.endpoints li.grp{font-size:.68rem;text-transform:uppercase;letter-spacing:.1em;color:var(--gris2);cursor:default;padding-top:10px} | |
| 131 | +.endpoints li.grp:hover{background:none} | |
| 132 | +.params label span{font-weight:400;color:var(--gris2);font-family:var(--sans);font-size:.72rem} | |
| 133 | +.params textarea.sql{min-height:110px} | |
| 134 | +table.ref{width:100%;font-size:.85rem}table.ref td,table.ref th{padding:6px 9px;border-top:1px solid var(--ligne);vertical-align:top;text-align:left} | |
| 135 | +table.ref thead th{background:var(--bg);color:var(--gris);position:static;cursor:default} | |
| 136 | +.hist{display:flex;flex-direction:column;gap:6px;max-height:40vh;overflow:auto}.hist-it{border:1px solid var(--ligne);border-radius:7px;padding:6px 8px;cursor:pointer;font-size:.8rem}.hist-it:hover{background:var(--bg)} | |
| 137 | +.recipe pre.code{max-height:360px} | |
| 138 | ||