SPB Git forge

spb/pdb-api

Public
1commits 1branches 0releases
436.0 KBsize
maindefault branch
1 h agolast push
JavaScript 58.4% Python 26.8% CSS 7.5% Objective-C 3% R 2.6% HTML 1.7%

adopt: état déployé sur M4M64a adopté comme source de vérité (remote-first, 2026-10-02)

mld committed 1 h ago (Oct 2, 2026)

16 changed files +2,339 −0

added .gitignore +6 −0
@@ -0,0 +1,6 @@
1 +.venv/
2 +data/
3 +__pycache__/
4 +*.pyc
5 +.DS_Store
6 +.env
added requirements-semantic.txt +3 −0
@@ -0,0 +1,3 @@
1 +# Optionnel : recherche sémantique en texte libre (encodage des requêtes).
2 +# La similarité entre projets (vecteurs stockés) fonctionne sans ce paquet.
3 +sentence-transformers>=2.5
added requirements.txt +4 −0
@@ -0,0 +1,4 @@
1 +fastapi>=0.115
2 +uvicorn[standard]>=0.30
3 +duckdb>=1.1
4 +numpy>=1.26
added server/__init__.py +0 −0
added server/db.py +223 −0
@@ -0,0 +1,223 @@
1 +"""Accès DuckDB en lecture seule + garde-fous pour le bac à sable SQL."""
2 +from __future__ import annotations
3 +
4 +import os
5 +import re
6 +import threading
7 +import time
8 +from pathlib import Path
9 +
10 +import duckdb
11 +import numpy as np
12 +
13 +ROOT = Path(__file__).resolve().parent.parent
14 +DB_PATH = os.environ.get("SPID_DB", str(ROOT / "data" / "spid.duckdb"))
15 +QUERY_TIMEOUT_S = float(os.environ.get("SQL_TIMEOUT_S", "20"))
16 +MAX_ROWS = int(os.environ.get("SQL_MAX_ROWS", "5000"))
17 +
18 +TABLES = ["projects", "project_mentions", "project_timeline", "kg_nodes", "kg_edges", "project_embeddings", "spid_sections"]
19 +
20 +_ALIAS = {"yr": "year", "loc": "location", "lbl": "label"}
21 +_lock = threading.Lock()
22 +_conn: duckdb.DuckDBPyConnection | None = None
23 +_schema_cache: list[dict] | None = None
24 +_emb_cache: dict | None = None
25 +
26 +
27 +def connect() -> duckdb.DuckDBPyConnection:
28 + global _conn
29 + with _lock:
30 + if _conn is None:
31 + if not os.path.exists(DB_PATH):
32 + raise RuntimeError(f"Base introuvable : {DB_PATH}")
33 + _conn = duckdb.connect(
34 + DB_PATH,
35 + read_only=True,
36 + config={
37 + "enable_external_access": "false", # pas de read_parquet/read_csv sur le disque
38 + "threads": os.environ.get("DUCKDB_THREADS", "4"),
39 + "memory_limit": os.environ.get("DUCKDB_MEM", "4GB"),
40 + },
41 + )
42 + try:
43 + _conn.execute("SET lock_configuration = true")
44 + except Exception:
45 + pass
46 + return _conn
47 +
48 +
49 +def cursor() -> duckdb.DuckDBPyConnection:
50 + """Connexion dupliquée (thread-safe) sur la connexion principale."""
51 + return connect().cursor()
52 +
53 +
54 +def rows(sql: str, params: list | tuple = ()) -> list[dict]:
55 + cur = cursor()
56 + try:
57 + res = cur.execute(sql, list(params))
58 + cols = [_ALIAS.get(d[0], d[0]) for d in res.description] # alias courts : year/location/label sont réservés en DuckDB
59 + return [dict(zip(cols, _jsonable(r))) for r in res.fetchall()]
60 + finally:
61 + cur.close()
62 +
63 +
64 +def one(sql: str, params: list | tuple = ()) -> dict | None:
65 + r = rows(sql, params)
66 + return r[0] if r else None
67 +
68 +
69 +def scalar(sql: str, params: list | tuple = ()):
70 + cur = cursor()
71 + try:
72 + return cur.execute(sql, list(params)).fetchone()[0]
73 + finally:
74 + cur.close()
75 +
76 +
77 +def _jsonable(row):
78 + out = []
79 + for v in row:
80 + if hasattr(v, "isoformat"):
81 + v = v.isoformat()
82 + elif isinstance(v, (np.floating,)):
83 + v = float(v)
84 + elif isinstance(v, (np.integer,)):
85 + v = int(v)
86 + elif isinstance(v, float) and (v != v): # NaN
87 + v = None
88 + out.append(v)
89 + return out
90 +
91 +
92 +# ---------------------------------------------------------------- schéma
93 +def schema() -> list[dict]:
94 + global _schema_cache
95 + if _schema_cache is None:
96 + cur = cursor()
97 + try:
98 + out = []
99 + for t in TABLES:
100 + cols = cur.execute(
101 + "select column_name, data_type from information_schema.columns where table_name=? order by ordinal_position", [t]
102 + ).fetchall()
103 + n = cur.execute(f'select count(*) from "{t}"').fetchone()[0]
104 + out.append({"table": t, "rows": n, "columns": [{"name": c, "type": d} for c, d in cols]})
105 + _schema_cache = out
106 + finally:
107 + cur.close()
108 + return _schema_cache
109 +
110 +
111 +# ---------------------------------------------------------------- bac à sable SQL
112 +_FORBIDDEN = re.compile(
113 + r"\b(attach|detach|copy|export|import|install|load|pragma|create|insert|update|delete|drop|alter|call|set|reset|"
114 + r"checkpoint|vacuum|force|begin|commit|rollback|grant|revoke|use|read_\w+|glob|getenv|current_setting|"
115 + r"sqlite_\w+|postgres_\w+|http\w*|write_\w+|list_files|duckdb_secrets|secrets?|parquet_\w+|iceberg_\w+|delta_\w+|"
116 + r"json_extract_path_text|read_json\w*|read_text|read_blob)\b",
117 + re.IGNORECASE,
118 +)
119 +
120 +
121 +def _strip_comments(sql: str) -> str:
122 + sql = re.sub(r"/\*.*?\*/", " ", sql, flags=re.S)
123 + sql = re.sub(r"--[^\n]*", " ", sql)
124 + return sql.strip()
125 +
126 +
127 +def guard_sql(sql: str) -> str:
128 + s = _strip_comments(sql).rstrip(";").strip()
129 + if not s:
130 + raise ValueError("Requête vide.")
131 + if ";" in s:
132 + raise ValueError("Une seule instruction à la fois.")
133 + if not re.match(r"^(select|with|from|describe|show|summarize|explain)\b", s, re.IGNORECASE):
134 + raise ValueError("Seules les requêtes de lecture (SELECT / WITH / DESCRIBE / SUMMARIZE / EXPLAIN) sont autorisées.")
135 + m = _FORBIDDEN.search(s)
136 + if m:
137 + raise ValueError(f"Mot-clé interdit dans le bac à sable : {m.group(0)}")
138 + return s
139 +
140 +
141 +def run_sql(sql: str, limit: int = 500) -> dict:
142 + """Exécute une requête de lecture avec délai maximal et plafond de lignes."""
143 + s = guard_sql(sql)
144 + limit = max(1, min(int(limit), MAX_ROWS))
145 + wrapped = s if re.match(r"^(describe|show|summarize|explain)\b", s, re.IGNORECASE) else f"select * from ({s}) as _q limit {limit + 1}"
146 + cur = cursor()
147 + result: dict = {}
148 + err: list[Exception] = []
149 +
150 + def work():
151 + try:
152 + t0 = time.perf_counter()
153 + res = cur.execute(wrapped)
154 + cols = [d[0] for d in res.description]
155 + data = res.fetchall()
156 + result["columns"] = cols
157 + result["truncated"] = len(data) > limit
158 + result["rows"] = [_jsonable(r) for r in data[:limit]]
159 + result["elapsed_ms"] = round((time.perf_counter() - t0) * 1000, 1)
160 + except Exception as e: # noqa: BLE001
161 + err.append(e)
162 +
163 + th = threading.Thread(target=work, daemon=True)
164 + th.start()
165 + th.join(QUERY_TIMEOUT_S)
166 + if th.is_alive():
167 + try:
168 + cur.interrupt()
169 + except Exception:
170 + pass
171 + th.join(2)
172 + cur.close()
173 + raise TimeoutError(f"Requête interrompue après {QUERY_TIMEOUT_S:.0f} s.")
174 + cur.close()
175 + if err:
176 + raise ValueError(str(err[0]).split("\n")[0][:400])
177 + result["sql"] = s
178 + result["limit"] = limit
179 + return result
180 +
181 +
182 +# ---------------------------------------------------------------- embeddings (similarité entre projets)
183 +def embeddings() -> dict:
184 + global _emb_cache
185 + if _emb_cache is None:
186 + cur = cursor()
187 + try:
188 + data = cur.execute("select project_id, vector from project_embeddings").fetchall()
189 + finally:
190 + cur.close()
191 + ids = [d[0] for d in data]
192 + mat = np.asarray([d[1] for d in data], dtype=np.float32)
193 + norms = np.linalg.norm(mat, axis=1, keepdims=True)
194 + norms[norms == 0] = 1
195 + mat = mat / norms
196 + _emb_cache = {"ids": ids, "index": {pid: i for i, pid in enumerate(ids)}, "mat": mat}
197 + return _emb_cache
198 +
199 +
200 +def similar_ids(project_id: str, k: int = 10) -> list[tuple[str, float]]:
201 + e = embeddings()
202 + i = e["index"].get(project_id)
203 + if i is None:
204 + return []
205 + sims = e["mat"] @ e["mat"][i]
206 + order = np.argsort(-sims)
207 + out = []
208 + for j in order:
209 + if j == i:
210 + continue
211 + out.append((e["ids"][j], float(sims[j])))
212 + if len(out) >= k:
213 + break
214 + return out
215 +
216 +
217 +def nearest_to_vector(vec: np.ndarray, k: int = 20) -> list[tuple[str, float]]:
218 + e = embeddings()
219 + v = np.asarray(vec, dtype=np.float32)
220 + v = v / (np.linalg.norm(v) or 1)
221 + sims = e["mat"] @ v
222 + order = np.argsort(-sims)[:k]
223 + return [(e["ids"][j], float(sims[j])) for j in order]
added server/main.py +521 −0
@@ -0,0 +1,521 @@
1 +"""PDB API — explorateur web + API REST de la base SPID (SEC Project Intelligence Database).
2 +
3 +- /v1/* : API publique, clé requise (en-tête `X-API-Key` ou `Authorization: Bearer …`).
4 +- /app/* : mêmes ressources pour l'interface web (même origine, sans clé — la clé n'est jamais envoyée au navigateur).
5 +- / : interface web (SPA statique dans web/).
6 +- /docs : documentation OpenAPI interactive.
7 +"""
8 +from __future__ import annotations
9 +
10 +import csv
11 +import io
12 +import os
13 +import threading
14 +import time
15 +from collections import defaultdict, deque
16 +from pathlib import Path
17 +from typing import Optional
18 +
19 +from fastapi import APIRouter, Depends, FastAPI, HTTPException, Query, Request
20 +from fastapi.responses import FileResponse, JSONResponse, StreamingResponse
21 +from fastapi.staticfiles import StaticFiles
22 +from pydantic import BaseModel, Field
23 +
24 +from . import db
25 +
26 +ROOT = Path(__file__).resolve().parent.parent
27 +WEB = ROOT / "web"
28 +API_KEY = os.environ.get("API_KEY", "").strip()
29 +PUBLIC_URL = os.environ.get("PUBLIC_URL", "https://www.pdb-api.co").rstrip("/")
30 +RATE_PER_MIN = int(os.environ.get("RATE_PER_MIN", "240"))
31 +
32 +TYPE_LABELS = {
33 + "ma_integration": "Intégration M&A", "plant_construction": "Construction d'usine", "rd_program": "Programme de R&D",
34 + "product_launch": "Lancement de produit", "manufacturing_expansion": "Expansion manufacturière", "partnership": "Partenariat",
35 + "technology_deployment": "Déploiement technologique", "digital_transformation": "Transformation numérique",
36 + "infrastructure": "Infrastructure", "capex_program": "Programme de capex", "ai_initiative": "Initiative IA",
37 + "geographic_expansion": "Expansion géographique", "cost_reduction": "Réduction des coûts", "sustainability": "Durabilité",
38 + "energy_transition": "Transition énergétique", "supply_chain": "Chaîne d'approvisionnement", "drug_pipeline": "Pipeline de médicaments",
39 + "data_center": "Centre de données", "cloud_migration": "Migration infonuagique", "automation": "Automatisation",
40 + "erp_implementation": "Implantation ERP",
41 +}
42 +STATUSES = ["planned", "in_progress", "completed", "mentioned"]
43 +PROJECT_SORTS = {
44 + "amount": "coalesce(total_amount_usd,0)", "mentions": "n_mentions", "filings": "n_filings", "first_seen": "first_seen",
45 + "last_seen": "last_seen", "confidence": "avg_confidence", "name": "project_name", "ticker": "ticker", "type": "project_type",
46 +}
47 +MENTION_SORTS = {"date": "filing_date", "amount": "coalesce(amount_usd,0)", "confidence": "confidence", "ticker": "ticker"}
48 +
49 +app = FastAPI(
50 + title="PDB API — SEC Project Intelligence Database",
51 + version="1.0.0",
52 + description=(
53 + "API REST de la base SPID : 19 227 projets stratégiques et 39 930 mentions extraits des filings SEC "
54 + "(10-K, 10-Q, 8-K) de 500 sociétés du S&P 500, 2010–2026. Toutes les routes `/v1/*` exigent une clé d'API "
55 + "dans l'en-tête `X-API-Key` (ou `Authorization: Bearer <clé>`). Les réponses paginées renvoient "
56 + "`{total, limit, offset, items}`."
57 + ),
58 + docs_url="/docs", redoc_url="/redoc", openapi_url="/openapi.json",
59 +)
60 +
61 +# ---------------------------------------------------------------- sécurité
62 +_buckets: dict[str, deque] = defaultdict(deque)
63 +_bl = threading.Lock()
64 +
65 +
66 +def _rate_limit(request: Request, scope: str):
67 + ip = request.headers.get("x-forwarded-for", request.client.host if request.client else "?").split(",")[0].strip()
68 + key = f"{scope}:{ip}"
69 + now = time.time()
70 + with _bl:
71 + q = _buckets[key]
72 + while q and q[0] < now - 60:
73 + q.popleft()
74 + if len(q) >= RATE_PER_MIN:
75 + raise HTTPException(429, "Trop de requêtes : limite de %d par minute." % RATE_PER_MIN)
76 + q.append(now)
77 +
78 +
79 +def require_key(request: Request):
80 + _rate_limit(request, "v1")
81 + if not API_KEY:
82 + raise HTTPException(503, "API non configurée (clé absente côté serveur).")
83 + key = request.headers.get("x-api-key") or ""
84 + auth = request.headers.get("authorization") or ""
85 + if not key and auth.lower().startswith("bearer "):
86 + key = auth[7:].strip()
87 + if not key:
88 + key = request.query_params.get("api_key", "")
89 + if key != API_KEY:
90 + raise HTTPException(401, "Clé d'API manquante ou invalide (en-tête X-API-Key).")
91 +
92 +
93 +def require_same_origin(request: Request):
94 + _rate_limit(request, "app")
95 + sfs = request.headers.get("sec-fetch-site")
96 + if sfs and sfs not in ("same-origin", "none"):
97 + raise HTTPException(403, "Accès réservé à l'interface web.")
98 + origin = request.headers.get("origin") or request.headers.get("referer")
99 + host = request.headers.get("x-forwarded-host") or request.headers.get("host") or ""
100 + if origin:
101 + from urllib.parse import urlparse
102 + if urlparse(origin).netloc.split(":")[0] != host.split(":")[0]:
103 + raise HTTPException(403, "Accès réservé à l'interface web.")
104 +
105 +
106 +# ---------------------------------------------------------------- helpers
107 +def _like(v: str) -> str:
108 + return f"%{v.lower().strip()}%"
109 +
110 +
111 +def _page(limit: int, offset: int) -> tuple[int, int]:
112 + return max(1, min(limit, 500)), max(0, offset)
113 +
114 +
115 +def _project_filters(q, type_, sector, status, ticker, cik, location, tech, partner, min_amount, max_amount,
116 + year_from, year_to, min_conf, has_amount):
117 + where, params = [], []
118 + if q:
119 + where.append("(lower(project_name) like ? or lower(description) like ? or lower(company_name) like ? or lower(ticker) = ?)")
120 + params += [_like(q), _like(q), _like(q), q.lower().strip()]
121 + if type_:
122 + ts = [t.strip() for t in type_.split(",") if t.strip()]
123 + where.append("project_type in (" + ",".join("?" * len(ts)) + ")"); params += ts
124 + if sector:
125 + where.append("lower(sector) like ?"); params.append(_like(sector))
126 + if status:
127 + ss = [s.strip() for s in status.split(",") if s.strip()]
128 + where.append("status in (" + ",".join("?" * len(ss)) + ")"); params += ss
129 + if ticker:
130 + where.append("upper(ticker) = ?"); params.append(ticker.upper().strip())
131 + if cik:
132 + where.append("cik = ?"); params.append(cik.zfill(10))
133 + if location:
134 + where.append("lower(canonical_location) like ?"); params.append(_like(location))
135 + if tech:
136 + where.append("exists (select 1 from unnest(technologies) as u(t) where lower(t) like ?)"); params.append(_like(tech))
137 + if partner:
138 + where.append("project_id in (select md5(cik||'|'||project_type||'|'||coalesce(nullif(lower(coalesce(try(locations[1]),'')),''), regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{4,})',1), 'general')) from project_mentions, unnest(partners) as u(p) where lower(p) like ?)")
139 + params.append(_like(partner))
140 + if min_amount is not None:
141 + where.append("total_amount_usd >= ?"); params.append(min_amount)
142 + if max_amount is not None:
143 + where.append("total_amount_usd <= ?"); params.append(max_amount)
144 + if year_from is not None:
145 + where.append("year(last_seen) >= ?"); params.append(year_from)
146 + if year_to is not None:
147 + where.append("year(first_seen) <= ?"); params.append(year_to)
148 + if min_conf is not None:
149 + where.append("avg_confidence >= ?"); params.append(min_conf)
150 + if has_amount:
151 + where.append("total_amount_usd > 0")
152 + return (" where " + " and ".join(where)) if where else "", params
153 +
154 +
155 +def _mention_filters(q, type_, sector, status, ticker, cik, form, date_from, date_to, min_amount, min_conf, accession, section):
156 + where, params = [], []
157 + if q:
158 + where.append("(lower(project_name) like ? or lower(description) like ? or lower(company_name) like ?)")
159 + params += [_like(q)] * 3
160 + if type_:
161 + ts = [t.strip() for t in type_.split(",") if t.strip()]
162 + where.append("project_type in (" + ",".join("?" * len(ts)) + ")"); params += ts
163 + if sector:
164 + where.append("lower(sector) like ?"); params.append(_like(sector))
165 + if status:
166 + where.append("status = ?"); params.append(status)
167 + if ticker:
168 + where.append("upper(ticker) = ?"); params.append(ticker.upper().strip())
169 + if cik:
170 + where.append("cik = ?"); params.append(cik.zfill(10))
171 + if form:
172 + fs = [f.strip() for f in form.split(",") if f.strip()]
173 + where.append("form_type in (" + ",".join("?" * len(fs)) + ")"); params += fs
174 + if date_from:
175 + where.append("filing_date >= ?"); params.append(date_from)
176 + if date_to:
177 + where.append("filing_date <= ?"); params.append(date_to)
178 + if min_amount is not None:
179 + where.append("amount_usd >= ?"); params.append(min_amount)
180 + if min_conf is not None:
181 + where.append("confidence >= ?"); params.append(min_conf)
182 + if accession:
183 + where.append("accession_number = ?"); params.append(accession)
184 + if section:
185 + where.append("section_id = ?"); params.append(section)
186 + return (" where " + " and ".join(where)) if where else "", params
187 +
188 +
189 +class SqlBody(BaseModel):
190 + sql: str = Field(..., description="Requête SQL DuckDB de lecture (SELECT / WITH / DESCRIBE / SUMMARIZE).")
191 + limit: int = Field(500, ge=1, le=5000, description="Nombre maximal de lignes renvoyées.")
192 +
193 +
194 +# ---------------------------------------------------------------- routes (fabrique, montée deux fois)
195 +def make_router(tag: str) -> APIRouter:
196 + r = APIRouter(tags=[tag])
197 +
198 + @r.get("/stats", summary="Vue d'ensemble : compteurs et répartitions")
199 + def stats():
200 + ov = db.one("""select count(*) projects, sum(n_mentions) mentions, count(distinct cik) companies,
201 + count(distinct sector) sectors, count(*) filter (where total_amount_usd>0) with_amount,
202 + median(total_amount_usd) median_amount_usd, sum(total_amount_usd) total_amount_usd,
203 + min(first_seen) first_seen, max(last_seen) last_seen from projects""")
204 + ov["mentions"] = db.scalar("select count(*) from project_mentions")
205 + ov["filings"] = db.scalar("select count(distinct accession_number) from project_mentions")
206 + ov["kg_nodes"] = db.scalar("select count(*) from kg_nodes")
207 + ov["kg_edges"] = db.scalar("select count(*) from kg_edges")
208 + ov["sections_10k"] = db.scalar("select count(*) from spid_sections")
209 + by_type = db.rows("""select project_type, count(*) n, sum(total_amount_usd) amount_usd, median(total_amount_usd) median_usd,
210 + round(avg(avg_confidence),3) confidence from projects group by 1 order by n desc""")
211 + for t in by_type:
212 + t["label"] = TYPE_LABELS.get(t["project_type"], t["project_type"])
213 + return {
214 + "overview": ov,
215 + "by_type": by_type,
216 + "by_sector": db.rows("select sector, count(*) n, count(distinct cik) companies, sum(total_amount_usd) amount_usd from projects where sector is not null group by 1 order by n desc"),
217 + "by_status": db.rows("select status, count(*) n from projects group by 1 order by n desc"),
218 + "by_form": db.rows("select form_type, count(*) n from project_mentions group by 1 order by n desc"),
219 + "mentions_by_year": db.rows("""select year(filing_date) as yr, form_type, count(*) n from project_mentions
220 + where filing_date >= '2010-01-01' group by 1,2 order by 1,2"""),
221 + "projects_by_year": db.rows("select year(first_seen) as yr, count(*) n, sum(total_amount_usd) amount_usd from projects where first_seen >= '2010-01-01' group by 1 order by 1"),
222 + "themes_by_year": db.rows("""select year(filing_date) as yr, project_type, count(*) n from project_mentions
223 + where project_type in ('ai_initiative','data_center','cloud_migration','digital_transformation','sustainability','energy_transition')
224 + and filing_date >= '2010-01-01' group by 1,2 order by 1,2"""),
225 + "sector_type": db.rows("select sector, project_type, count(*) n from projects where sector is not null group by 1,2"),
226 + "top_companies": db.rows("select ticker, any_value(company_name) company_name, any_value(sector) sector, count(*) n, sum(total_amount_usd) amount_usd from projects group by 1 order by n desc limit 20"),
227 + "top_technologies": db.rows("select lower(trim(t)) technology, count(*) n from projects, unnest(technologies) as u(t) where t<>'' group by 1 order by n desc limit 25"),
228 + "top_locations": db.rows("select canonical_location as loc, count(*) n from projects where canonical_location is not null and canonical_location<>'' group by 1 order by n desc limit 25"),
229 + }
230 +
231 + @r.get("/taxonomy", summary="Les 21 types de projets (libellés et effectifs)")
232 + def taxonomy():
233 + counts = {x["project_type"]: x for x in db.rows("select project_type, count(*) n, sum(total_amount_usd) amount_usd from projects group by 1")}
234 + return [{"project_type": k, "label": v, "n": counts.get(k, {}).get("n", 0), "amount_usd": counts.get(k, {}).get("amount_usd")} for k, v in TYPE_LABELS.items()]
235 +
236 + @r.get("/sectors", summary="Secteurs GICS")
237 + def sectors():
238 + return db.rows("select sector, count(*) n, count(distinct cik) companies, sum(total_amount_usd) amount_usd from projects where sector is not null group by 1 order by n desc")
239 +
240 + @r.get("/technologies", summary="Technologies citées")
241 + def technologies(q: Optional[str] = None, limit: int = Query(50, le=500)):
242 + w, p = ("where lower(t) like ?", [_like(q)]) if q else ("", [])
243 + return db.rows(f"select lower(trim(t)) technology, count(*) n from projects, unnest(technologies) as u(t) {w} {'and' if w else 'where'} t<>'' group by 1 order by n desc limit {int(limit)}", p)
244 +
245 + @r.get("/locations", summary="Localisations canoniques")
246 + def locations(q: Optional[str] = None, limit: int = Query(50, le=500)):
247 + w, p = ("and lower(canonical_location) like ?", [_like(q)]) if q else ("", [])
248 + return db.rows(f"select canonical_location as loc, count(*) n, sum(total_amount_usd) amount_usd from projects where canonical_location is not null and canonical_location<>'' {w} group by 1 order by n desc limit {int(limit)}", p)
249 +
250 + @r.get("/partners", summary="Partenaires cités dans les mentions")
251 + def partners(q: Optional[str] = None, limit: int = Query(50, le=500)):
252 + w, p = ("and lower(p) like ?", [_like(q)]) if q else ("", [])
253 + return db.rows(f"select lower(trim(p)) partner, any_value(p) as lbl, count(*) n, count(distinct cik) companies from project_mentions, unnest(partners) as u(p) where p<>'' {w} group by 1 order by n desc limit {int(limit)}", p)
254 +
255 + @r.get("/projects", summary="Rechercher des projets (filtres + pagination)")
256 + def projects(
257 + q: Optional[str] = Query(None, description="Texte libre : nom, description, entreprise, ticker"),
258 + type: Optional[str] = Query(None, description="project_type (liste séparée par des virgules)"),
259 + sector: Optional[str] = None, status: Optional[str] = Query(None, description="planned,in_progress,completed,mentioned"),
260 + ticker: Optional[str] = None, cik: Optional[str] = None, location: Optional[str] = None, tech: Optional[str] = None,
261 + partner: Optional[str] = None,
262 + min_amount: Optional[float] = None, max_amount: Optional[float] = None,
263 + year_from: Optional[int] = None, year_to: Optional[int] = None, min_confidence: Optional[float] = None,
264 + has_amount: bool = False,
265 + sort: str = Query("mentions", description="amount | mentions | filings | first_seen | last_seen | confidence | name | ticker | type"),
266 + order: str = Query("desc", pattern="^(asc|desc)$"),
267 + limit: int = Query(50, ge=1, le=500), offset: int = Query(0, ge=0),
268 + ):
269 + w, p = _project_filters(q, type, sector, status, ticker, cik, location, tech, partner, min_amount, max_amount, year_from, year_to, min_confidence, has_amount)
270 + limit, offset = _page(limit, offset)
271 + col = PROJECT_SORTS.get(sort, "n_mentions")
272 + total = db.scalar(f"select count(*) from projects{w}", p)
273 + items = db.rows(f"select * from projects{w} order by {col} {order} nulls last, project_id limit {limit} offset {offset}", p)
274 + for it in items:
275 + it["type_label"] = TYPE_LABELS.get(it["project_type"], it["project_type"])
276 + return {"total": total, "limit": limit, "offset": offset, "items": items}
277 +
278 + @r.get("/projects/export.csv", summary="Exporter les projets filtrés en CSV (max 50 000 lignes)")
279 + def projects_csv(
280 + q: Optional[str] = None, type: Optional[str] = None, sector: Optional[str] = None, status: Optional[str] = None,
281 + ticker: Optional[str] = None, cik: Optional[str] = None, location: Optional[str] = None, tech: Optional[str] = None,
282 + partner: Optional[str] = None, min_amount: Optional[float] = None, max_amount: Optional[float] = None,
283 + year_from: Optional[int] = None, year_to: Optional[int] = None, min_confidence: Optional[float] = None, has_amount: bool = False,
284 + ):
285 + w, p = _project_filters(q, type, sector, status, ticker, cik, location, tech, partner, min_amount, max_amount, year_from, year_to, min_confidence, has_amount)
286 + items = db.rows(f"select * from projects{w} order by n_mentions desc limit 50000", p)
287 +
288 + def gen():
289 + buf = io.StringIO(); wr = csv.writer(buf)
290 + cols = list(items[0].keys()) if items else ["project_id"]
291 + wr.writerow(cols); yield buf.getvalue(); buf.seek(0); buf.truncate()
292 + for it in items:
293 + wr.writerow(["|".join(map(str, v)) if isinstance(v, list) else v for v in it.values()])
294 + yield buf.getvalue(); buf.seek(0); buf.truncate()
295 + return StreamingResponse(gen(), media_type="text/csv", headers={"Content-Disposition": "attachment; filename=spid_projects.csv"})
296 +
297 + @r.get("/projects/{project_id}", summary="Fiche complète d'un projet (chronologie, mentions, graphe, similaires)")
298 + def project(project_id: str):
299 + pr = db.one("select * from projects where project_id = ?", [project_id])
300 + if not pr:
301 + raise HTTPException(404, "Projet introuvable.")
302 + pr["type_label"] = TYPE_LABELS.get(pr["project_type"], pr["project_type"])
303 + timeline = db.rows("select * from project_timeline where project_id = ? order by filing_date", [project_id])
304 + # mentions rattachées : même clé de résolution que spid/resolve.py
305 + mentions = db.rows("""with m as (select *, lower(coalesce(try(locations[1]),'')) loc1,
306 + regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{4,})',1) name_tok from project_mentions)
307 + select * exclude (loc1, name_tok) from m
308 + where md5(cik||'|'||project_type||'|'||coalesce(nullif(loc1,''), nullif(name_tok,''), 'general')) = ?
309 + order by filing_date""", [project_id])
310 + edges = db.rows("""select e.rel, e.src, e.dst, n.node_type, n.label from kg_edges e join kg_nodes n on n.node_id = case when e.src = ? then e.dst else e.src end
311 + where e.src = ? or e.dst = ?""", ["P:" + project_id] * 3)
312 + sims = db.similar_ids(project_id, 8)
313 + similar = []
314 + if sims:
315 + ids = [s[0] for s in sims]
316 + found = {x["project_id"]: x for x in db.rows("select project_id, ticker, company_name, project_type, project_name, status, total_amount_usd, first_seen from projects where project_id in (" + ",".join("?" * len(ids)) + ")", ids)}
317 + for pid, sc in sims:
318 + if pid in found:
319 + found[pid]["score"] = round(sc, 4); found[pid]["type_label"] = TYPE_LABELS.get(found[pid]["project_type"]); similar.append(found[pid])
320 + return {"project": pr, "timeline": timeline, "mentions": mentions, "graph": edges, "similar": similar}
321 +
322 + @r.get("/projects/{project_id}/similar", summary="Projets sémantiquement proches (vecteurs MiniLM stockés)")
323 + def project_similar(project_id: str, k: int = Query(10, le=50)):
324 + sims = db.similar_ids(project_id, k)
325 + if not sims and not db.one("select 1 from projects where project_id=?", [project_id]):
326 + raise HTTPException(404, "Projet introuvable.")
327 + ids = [s[0] for s in sims]
328 + found = {x["project_id"]: x for x in db.rows("select * from projects where project_id in (" + ",".join("?" * len(ids)) + ")", ids)} if ids else {}
329 + return [dict(found[pid], score=round(sc, 4)) for pid, sc in sims if pid in found]
330 +
331 + @r.get("/mentions", summary="Rechercher des mentions (unité d'extraction, une par section et par projet)")
332 + def mentions(
333 + q: Optional[str] = None, type: Optional[str] = None, sector: Optional[str] = None, status: Optional[str] = None,
334 + ticker: Optional[str] = None, cik: Optional[str] = None, form: Optional[str] = Query(None, description="10-K,10-Q,8-K"),
335 + date_from: Optional[str] = None, date_to: Optional[str] = None, min_amount: Optional[float] = None,
336 + min_confidence: Optional[float] = None, accession: Optional[str] = None, section: Optional[str] = None,
337 + sort: str = Query("date", description="date | amount | confidence | ticker"), order: str = Query("desc", pattern="^(asc|desc)$"),
338 + limit: int = Query(50, ge=1, le=500), offset: int = Query(0, ge=0),
339 + ):
340 + w, p = _mention_filters(q, type, sector, status, ticker, cik, form, date_from, date_to, min_amount, min_confidence, accession, section)
341 + limit, offset = _page(limit, offset)
342 + col = MENTION_SORTS.get(sort, "filing_date")
343 + total = db.scalar(f"select count(*) from project_mentions{w}", p)
344 + items = db.rows(f"""select *, md5(cik||'|'||project_type||'|'||coalesce(nullif(lower(coalesce(try(locations[1]),'')),''),
345 + nullif(regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{{4,}})',1),''), 'general')) project_id
346 + from project_mentions{w} order by {col} {order} nulls last, mention_id limit {limit} offset {offset}""", p)
347 + for it in items:
348 + it["type_label"] = TYPE_LABELS.get(it["project_type"], it["project_type"])
349 + return {"total": total, "limit": limit, "offset": offset, "items": items}
350 +
351 + @r.get("/mentions/{mention_id}", summary="Une mention et sa section source (10-K)")
352 + def mention(mention_id: str):
353 + m = db.one("select * from project_mentions where mention_id = ?", [mention_id])
354 + if not m:
355 + raise HTTPException(404, "Mention introuvable.")
356 + m["type_label"] = TYPE_LABELS.get(m["project_type"], m["project_type"])
357 + m["project_id"] = db.scalar("""select md5(cik||'|'||project_type||'|'||coalesce(nullif(lower(coalesce(try(locations[1]),'')),''),
358 + nullif(regexp_extract(lower(regexp_replace(project_name,'[^A-Za-z ]','','g')),'([a-z]{4,})',1),''), 'general'))
359 + from project_mentions where mention_id = ?""", [mention_id])
360 + sec = db.one("select section_id, accession_number, section_name, word_count from spid_sections where section_id = ?", [m["section_id"]])
361 + return {"mention": m, "section": sec, "section_text_url": f"/sections/{m['section_id']}" if sec else None}
362 +
363 + @r.get("/sections/{section_id}", summary="Texte d'une section 10-K ingérée par SPID")
364 + def section(section_id: str, highlight: Optional[str] = Query(None, description="Mot à repérer (renvoie les positions)")):
365 + s = db.one("select * from spid_sections where section_id = ?", [section_id])
366 + if not s:
367 + raise HTTPException(404, "Section introuvable (seules les sections 10-K sont stockées dans la base ; les sections 10-Q/8-K restent dans le corpus parquet amont).")
368 + if highlight:
369 + import re
370 + s["highlights"] = [m.start() for m in re.finditer(re.escape(highlight), s["text"], re.IGNORECASE)][:200]
371 + return s
372 +
373 + @r.get("/companies", summary="Entreprises et taille de leur portefeuille de projets")
374 + def companies(q: Optional[str] = None, sector: Optional[str] = None, sort: str = Query("projects", description="projects | amount | mentions | ticker"),
375 + order: str = Query("desc", pattern="^(asc|desc)$"), limit: int = Query(100, ge=1, le=500), offset: int = Query(0, ge=0)):
376 + where, p = [], []
377 + if q:
378 + where.append("(lower(company_name) like ? or lower(ticker) like ?)"); p += [_like(q), _like(q)]
379 + if sector:
380 + where.append("lower(sector) like ?"); p.append(_like(sector))
381 + w = (" where " + " and ".join(where)) if where else ""
382 + col = {"projects": "n_projects", "amount": "coalesce(amount_usd,0)", "mentions": "n_mentions", "ticker": "ticker"}.get(sort, "n_projects")
383 + limit, offset = _page(limit, offset)
384 + base = f"""with c as (select cik, any_value(ticker) ticker, any_value(company_name) company_name, any_value(sector) sector,
385 + count(*) n_projects, sum(n_mentions) n_mentions, sum(total_amount_usd) amount_usd, min(first_seen) first_seen, max(last_seen) last_seen,
386 + count(distinct project_type) n_types from projects group by cik) select * from c{w}"""
387 + total = db.scalar(f"select count(*) from ({base})", p)
388 + items = db.rows(f"{base} order by {col} {order} nulls last limit {limit} offset {offset}", p)
389 + return {"total": total, "limit": limit, "offset": offset, "items": items}
390 +
391 + @r.get("/companies/{ident}", summary="Profil d'une entreprise (ticker ou CIK)")
392 + def company(ident: str):
393 + ident = ident.strip()
394 + cond, val = ("cik = ?", ident.zfill(10)) if ident.isdigit() else ("upper(ticker) = ?", ident.upper())
395 + prof = db.one(f"""select cik, any_value(ticker) ticker, any_value(company_name) company_name, any_value(sector) sector,
396 + count(*) n_projects, sum(n_mentions) n_mentions, sum(total_amount_usd) amount_usd, min(first_seen) first_seen, max(last_seen) last_seen
397 + from projects where {cond} group by cik""", [val])
398 + if not prof:
399 + raise HTTPException(404, "Entreprise introuvable.")
400 + cik = prof["cik"]
401 + return {
402 + "company": prof,
403 + "by_type": [dict(x, label=TYPE_LABELS.get(x["project_type"])) for x in db.rows("select project_type, count(*) n, sum(total_amount_usd) amount_usd from projects where cik=? group by 1 order by n desc", [cik])],
404 + "by_status": db.rows("select status, count(*) n from projects where cik=? group by 1", [cik]),
405 + "by_year": db.rows("select year(filing_date) as yr, count(*) n from project_mentions where cik=? and filing_date>='2010-01-01' group by 1 order by 1", [cik]),
406 + "by_form": db.rows("select form_type, count(*) n from project_mentions where cik=? group by 1", [cik]),
407 + "locations": db.rows("select canonical_location as loc, count(*) n from projects where cik=? and canonical_location<>'' group by 1 order by n desc limit 15", [cik]),
408 + "technologies": db.rows("select lower(t) technology, count(*) n from projects, unnest(technologies) as u(t) where cik=? group by 1 order by n desc limit 15", [cik]),
409 + "partners": db.rows("select any_value(p) partner, count(*) n from project_mentions, unnest(partners) as u(p) where cik=? and p<>'' group by lower(p) order by n desc limit 15", [cik]),
410 + "projects": [dict(x, type_label=TYPE_LABELS.get(x["project_type"])) for x in db.rows("select * from projects where cik=? order by n_mentions desc, total_amount_usd desc nulls last limit 500", [cik])],
411 + }
412 +
413 + @r.get("/graph/search", summary="Chercher un nœud du graphe (entreprise, projet, lieu, technologie, partenaire)")
414 + def graph_search(q: str, type: Optional[str] = Query(None, description="Company | Project | Location | Technology | Partner"), limit: int = Query(30, le=200)):
415 + w, p = "where lower(label) like ?", [_like(q)]
416 + if type:
417 + w += " and node_type = ?"; p.append(type)
418 + return db.rows(f"""select n.node_id, n.node_type, n.label, n.props, (select count(*) from kg_edges e where e.src=n.node_id or e.dst=n.node_id) degree
419 + from kg_nodes n {w} order by degree desc limit {int(limit)}""", p)
420 +
421 + @r.get("/graph/node/{node_id:path}", summary="Un nœud et son voisinage")
422 + def graph_node(node_id: str, limit: int = Query(200, le=2000)):
423 + n = db.one("select * from kg_nodes where node_id = ?", [node_id])
424 + if not n:
425 + raise HTTPException(404, "Nœud introuvable.")
426 + edges = db.rows(f"""select e.rel, e.src, e.dst, e.props, case when e.src = ? then 'out' else 'in' end direction,
427 + m.node_type neighbor_type, m.label neighbor_label, m.node_id neighbor_id, m.props neighbor_props
428 + from kg_edges e join kg_nodes m on m.node_id = case when e.src = ? then e.dst else e.src end
429 + where e.src = ? or e.dst = ? limit {int(limit)}""", [node_id] * 4)
430 + degree = db.scalar("select count(*) from kg_edges where src = ? or dst = ?", [node_id, node_id])
431 + return {"node": n, "degree": degree, "edges": edges}
432 +
433 + @r.get("/search/semantic", summary="Recherche sémantique en texte libre (modèle all-MiniLM-L6-v2 côté serveur)")
434 + def semantic(q: str, k: int = Query(20, le=100), type: Optional[str] = None, sector: Optional[str] = None):
435 + from . import semantic as sem
436 + try:
437 + vec = sem.encode(q)
438 + except sem.Unavailable as e:
439 + raise HTTPException(501, str(e))
440 + hits = db.nearest_to_vector(vec, k * 5)
441 + ids = [h[0] for h in hits]
442 + where, p = ["project_id in (" + ",".join("?" * len(ids)) + ")"], list(ids)
443 + if type:
444 + where.append("project_type = ?"); p.append(type)
445 + if sector:
446 + where.append("lower(sector) like ?"); p.append(_like(sector))
447 + found = {x["project_id"]: x for x in db.rows("select * from projects where " + " and ".join(where), p)}
448 + out = []
449 + for pid, sc in hits:
450 + if pid in found:
451 + out.append(dict(found[pid], score=round(sc, 4), type_label=TYPE_LABELS.get(found[pid]["project_type"])))
452 + if len(out) >= k:
453 + break
454 + return {"query": q, "items": out}
455 +
456 + @r.get("/schema", summary="Tables et colonnes de la base")
457 + def schema():
458 + return db.schema()
459 +
460 + @r.post("/sql", summary="Bac à sable SQL en lecture seule (DuckDB)")
461 + def sql(body: SqlBody):
462 + try:
463 + return db.run_sql(body.sql, body.limit)
464 + except TimeoutError as e:
465 + raise HTTPException(408, str(e))
466 + except ValueError as e:
467 + raise HTTPException(400, str(e))
468 +
469 + return r
470 +
471 +
472 +app.include_router(make_router("API v1 (clé requise)"), prefix="/v1", dependencies=[Depends(require_key)])
473 +app.include_router(make_router("Interface web (même origine)"), prefix="/app", dependencies=[Depends(require_same_origin)], include_in_schema=False)
474 +
475 +
476 +@app.get("/health", include_in_schema=False)
477 +@app.get("/v1/health", summary="État du service (sans clé)", tags=["Service"])
478 +def health():
479 + try:
480 + n = db.scalar("select count(*) from projects")
481 + return {"status": "ok", "projects": n, "db": os.path.basename(db.DB_PATH), "semantic": _semantic_state()}
482 + except Exception as e: # noqa: BLE001
483 + return JSONResponse({"status": "error", "detail": str(e)[:200]}, status_code=503)
484 +
485 +
486 +def _semantic_state():
487 + try:
488 + from . import semantic as sem
489 + return sem.state()
490 + except Exception:
491 + return "unavailable"
492 +
493 +
494 +@app.get("/v1/meta", summary="Métadonnées de l'API (routes, limites, exemples)", tags=["Service"])
495 +def meta():
496 + return {
497 + "name": "PDB API", "version": app.version, "base_url": PUBLIC_URL + "/v1",
498 + "auth": "En-tête X-API-Key: <clé> (ou Authorization: Bearer <clé>). La clé est fournie par l'équipe UQO ; elle n'est publiée nulle part sur le site.",
499 + "rate_limit_per_minute": RATE_PER_MIN, "pagination": "limit (≤500) / offset ; réponses {total, limit, offset, items}",
500 + "routes": [f"{r.methods and list(r.methods)[0]} {r.path}" for r in app.routes if getattr(r, "path", "").startswith("/v1")],
501 + "docs": PUBLIC_URL + "/docs",
502 + }
503 +
504 +
505 +@app.on_event("startup")
506 +def _warm():
507 + def w():
508 + try:
509 + db.connect(); db.schema(); db.embeddings()
510 + except Exception as e: # noqa: BLE001
511 + print("warmup:", e)
512 + threading.Thread(target=w, daemon=True).start()
513 +
514 +
515 +# ---------------------------------------------------------------- statique
516 +if (WEB / "report.pdf").exists():
517 + @app.get("/report.pdf", include_in_schema=False)
518 + def report():
519 + return FileResponse(WEB / "report.pdf", media_type="application/pdf")
520 +
521 +app.mount("/", StaticFiles(directory=str(WEB), html=True), name="web")
added server/semantic.py +55 −0
@@ -0,0 +1,55 @@
1 +"""Encodage des requêtes en texte libre avec all-MiniLM-L6-v2 (optionnel, chargé paresseusement)."""
2 +from __future__ import annotations
3 +
4 +import os
5 +import threading
6 +
7 +MODEL_NAME = os.environ.get("EMBED_MODEL", "all-MiniLM-L6-v2")
8 +_model = None
9 +_state = "idle"
10 +_lock = threading.Lock()
11 +
12 +
13 +class Unavailable(Exception):
14 + pass
15 +
16 +
17 +def state() -> str:
18 + return _state
19 +
20 +
21 +def _load():
22 + global _model, _state
23 + with _lock:
24 + if _model is not None:
25 + return _model
26 + try:
27 + _state = "loading"
28 + from sentence_transformers import SentenceTransformer # type: ignore
29 + _model = SentenceTransformer(MODEL_NAME)
30 + _state = "ready"
31 + except Exception as e: # noqa: BLE001
32 + _state = f"unavailable: {type(e).__name__}"
33 + raise Unavailable("Recherche sémantique indisponible sur ce serveur (paquet sentence-transformers absent ou modèle non téléchargé). "
34 + "Utilisez /projects?q= ou /projects/{id}/similar.") from e
35 + return _model
36 +
37 +
38 +def encode(text: str):
39 + m = _load()
40 + return m.encode([text], normalize_embeddings=True)[0]
41 +
42 +
43 +def preload():
44 + threading.Thread(target=lambda: _safe(_load), daemon=True).start()
45 +
46 +
47 +def _safe(f):
48 + try:
49 + f()
50 + except Exception:
51 + pass
52 +
53 +
54 +if os.environ.get("SEMANTIC_PRELOAD", "1") == "1":
55 + preload()
added web/app.js +414 −0
@@ -0,0 +1,414 @@
1 +/* PDB — explorateur SPID. SPA sans dépendance (Chart.js chargé par CDN). */
2 +(() => {
3 + const $ = (s, r = document) => r.querySelector(s);
4 + const view = $("#view");
5 + const API = "/app";
6 + const TYPE = {};
7 + const STATUT = { planned: "Planifié", in_progress: "En cours", completed: "Terminé", mentioned: "Mentionné" };
8 + const COL = { bleu: "#003E7E", bleu2: "#0066B3", or: "#C6A300", teal: "#007979", orange: "#D67A00", vert: "#008046", violet: "#5E35B1", rouge: "#B42318", gris: "#8a8c90" };
9 + const PAL = [COL.bleu, COL.or, COL.teal, COL.orange, COL.vert, COL.violet, COL.rouge, COL.bleu2, COL.gris, "#8DA9C4", "#E0C36B"];
10 + let charts = [];
11 + const fmtN = (n) => n == null ? "—" : Number(n).toLocaleString("fr-CA");
12 + const fmtUSD = (v) => v == null ? "—" : v >= 1e9 ? (v / 1e9).toLocaleString("fr-CA", { maximumFractionDigits: 1 }) + " G$" : v >= 1e6 ? (v / 1e6).toLocaleString("fr-CA", { maximumFractionDigits: 0 }) + " M$" : Math.round(v).toLocaleString("fr-CA") + " $";
13 + const esc = (s) => String(s ?? "").replace(/[&<>"']/g, (c) => ({ "&": "&amp;", "<": "&lt;", ">": "&gt;", '"': "&quot;", "'": "&#39;" }[c]));
14 + const tl = (t) => TYPE[t] || t;
15 + const stPill = (s) => `<span class="st st-${esc(s)}">${esc(STATUT[s] || s)}</span>`;
16 + const qs = (o) => Object.entries(o).filter(([, v]) => v !== "" && v != null && v !== false).map(([k, v]) => `${encodeURIComponent(k)}=${encodeURIComponent(v)}`).join("&");
17 + const toast = (m) => { const t = $("#toast"); t.textContent = m; t.hidden = false; clearTimeout(t._h); t._h = setTimeout(() => (t.hidden = true), 2600); };
18 +
19 + async function api(path, params) {
20 + const url = API + path + (params ? "?" + qs(params) : "");
21 + const r = await fetch(url, { headers: { Accept: "application/json" } });
22 + if (!r.ok) { let d = ""; try { d = (await r.json()).detail; } catch { } throw new Error(d || `${r.status} ${r.statusText}`); }
23 + return r.json();
24 + }
25 + function destroyCharts() { charts.forEach((c) => c.destroy()); charts = []; }
26 + function chart(el, cfg) { if (!window.Chart) return; Chart.defaults.font.family = getComputedStyle(document.body).fontFamily; Chart.defaults.color = "#58595B"; const c = new Chart(el, cfg); charts.push(c); return c; }
27 + const hashParams = () => { const h = location.hash.slice(2); const [p, q] = h.split("?"); return { path: p || "", parts: (p || "").split("/"), q: Object.fromEntries(new URLSearchParams(q || "")) }; };
28 + const setQuery = (obj) => { const { parts } = hashParams(); location.hash = "#/" + parts[0] + (Object.keys(obj).length ? "?" + qs(obj) : ""); };
29 +
30 + /* ------------------------------------------------------------ tableau générique */
31 + function table(cols, rows, opts = {}) {
32 + const th = cols.map((c) => `<th class="${c.num ? "num" : ""} ${opts.sort === c.key ? "sorted " + (opts.order || "") : ""}" data-sort="${c.sortKey || ""}">${esc(c.label)}</th>`).join("");
33 + const tr = rows.map((r) => `<tr class="${opts.href ? "click" : ""}" data-href="${opts.href ? esc(opts.href(r)) : ""}">${cols.map((c) => `<td class="${c.num ? "num" : ""} ${c.trunc ? "trunc" : ""}" title="${c.trunc ? esc(c.title ? c.title(r) : r[c.key]) : ""}">${c.render ? c.render(r) : esc(r[c.key])}</td>`).join("")}</tr>`).join("");
34 + return `<div class="table-wrap"><table><thead><tr>${th}</tr></thead><tbody>${tr || `<tr><td colspan="${cols.length}" class="muted">Aucun résultat.</td></tr>`}</tbody></table></div>`;
35 + }
36 + function bindTable(root, onSort) {
37 + root.querySelectorAll("tr[data-href]").forEach((tr) => tr.addEventListener("click", (e) => { if (e.target.closest("a")) return; if (tr.dataset.href) location.hash = tr.dataset.href; }));
38 + if (onSort) root.querySelectorAll("th[data-sort]").forEach((th) => th.addEventListener("click", () => th.dataset.sort && onSort(th.dataset.sort)));
39 + }
40 + function pager(total, limit, offset) {
41 + const page = Math.floor(offset / limit) + 1, pages = Math.max(1, Math.ceil(total / limit));
42 + return `<div class="pager"><span>${fmtN(total)} résultats · page ${page} / ${fmtN(pages)}</span><span class="sp"></span>
43 + <button class="btn btn-ghost btn-sm" data-off="${Math.max(0, offset - limit)}" ${offset === 0 ? "disabled" : ""}>← Précédent</button>
44 + <button class="btn btn-ghost btn-sm" data-off="${offset + limit}" ${offset + limit >= total ? "disabled" : ""}>Suivant →</button></div>`;
45 + }
46 +
47 + /* ------------------------------------------------------------ tableau de bord */
48 + async function dashboard() {
49 + const s = await api("/stats");
50 + const ov = s.overview;
51 + view.innerHTML = `
52 + <div class="hero"><div><h1>SEC Project Intelligence Database</h1>
53 + <p>${fmtN(ov.projects)} projets stratégiques et ${fmtN(ov.mentions)} mentions extraits par LLM des filings SEC (10-K, 10-Q, 8-K) de ${fmtN(ov.companies)} sociétés du S&amp;P 500, ${ov.first_seen?.slice(0, 4)}–${ov.last_seen?.slice(0, 4)}. Explorez, filtrez, interrogez en SQL, ou branchez-vous sur l'API.</p></div>
54 + <div><a class="btn" href="#/projects">Explorer les projets</a> &nbsp; <a class="btn btn-ghost" href="#/api" style="color:#fff;border-color:rgba(255,255,255,.5);background:transparent">API</a></div></div>
55 + <div class="kpis">
56 + ${kpi(ov.projects, "projets résolus")}${kpi(ov.mentions, "mentions extraites")}${kpi(ov.filings, "filings avec projets")}${kpi(ov.companies, "entreprises")}
57 + ${kpi(ov.with_amount, "projets avec montant")}${kpi(fmtUSD(ov.median_amount_usd), "montant médian")}${kpi(ov.kg_nodes, "nœuds du graphe")}${kpi(ov.sections_10k, "sections 10-K")}
58 + </div>
59 + <div class="grid g2">
60 + <div class="card"><h3>Mentions par année et formulaire</h3><div class="chart"><canvas id="c-year"></canvas></div></div>
61 + <div class="card"><h3>Statut des projets</h3><div class="chart"><canvas id="c-status"></canvas></div></div>
62 + <div class="card" style="grid-column:1/-1"><h3>Les 21 types de projets (nombre et capital divulgué)</h3><div class="chart tall"><canvas id="c-type"></canvas></div></div>
63 + <div class="card"><h3>Projets par secteur GICS</h3><div class="chart tall"><canvas id="c-sector"></canvas></div></div>
64 + <div class="card"><h3>Thèmes émergents — part des mentions annuelles</h3><div class="chart tall"><canvas id="c-themes"></canvas></div></div>
65 + <div class="card"><h3>Entreprises les plus prolifiques</h3>${table([
66 + { key: "ticker", label: "Ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` },
67 + { key: "company_name", label: "Entreprise" }, { key: "sector", label: "Secteur" },
68 + { key: "n", label: "Projets", num: true, render: (r) => fmtN(r.n) }, { key: "amount_usd", label: "Capital", num: true, render: (r) => fmtUSD(r.amount_usd) },
69 + ], s.top_companies.slice(0, 12))}</div>
70 + <div class="card"><h3>Technologies et localisations les plus citées</h3><div class="grid g2">
71 + <div class="tags">${s.top_technologies.slice(0, 20).map((t) => `<a class="tag t-tech" href="#/projects?tech=${encodeURIComponent(t.technology)}">${esc(t.technology)} <b>${fmtN(t.n)}</b></a>`).join("")}</div>
72 + <div class="tags">${s.top_locations.slice(0, 20).map((t) => `<a class="tag t-loc" href="#/projects?location=${encodeURIComponent(t.location)}">${esc(t.location)} <b>${fmtN(t.n)}</b></a>`).join("")}</div></div></div>
73 + </div>`;
74 + // graphiques
75 + const years = [...new Set(s.mentions_by_year.map((r) => r.year))].sort();
76 + const forms = ["10-K", "10-Q", "8-K"];
77 + chart($("#c-year"), { type: "bar", data: { labels: years, datasets: forms.map((f, i) => ({ label: f, data: years.map((y) => s.mentions_by_year.find((r) => r.year === y && r.form_type === f)?.n || 0), backgroundColor: [COL.bleu, COL.or, COL.teal][i] })) }, options: { responsive: true, maintainAspectRatio: false, scales: { x: { stacked: true }, y: { stacked: true } }, plugins: { legend: { position: "bottom" } } } });
78 + chart($("#c-status"), { type: "doughnut", data: { labels: s.by_status.map((r) => STATUT[r.status] || r.status), datasets: [{ data: s.by_status.map((r) => r.n), backgroundColor: [COL.bleu2, COL.vert, COL.or, COL.gris] }] }, options: { responsive: true, maintainAspectRatio: false, cutout: "58%", plugins: { legend: { position: "right" } } } });
79 + chart($("#c-type"), { type: "bar", data: { labels: s.by_type.map((r) => r.label), datasets: [{ label: "Projets", data: s.by_type.map((r) => r.n), backgroundColor: COL.bleu, yAxisID: "y" }, { label: "Capital (G$)", data: s.by_type.map((r) => (r.amount_usd || 0) / 1e9), backgroundColor: COL.or, yAxisID: "y2" }] }, options: { responsive: true, maintainAspectRatio: false, scales: { x: { ticks: { autoSkip: false, maxRotation: 60, minRotation: 45 } }, y: { position: "left", title: { display: true, text: "projets" } }, y2: { position: "right", grid: { drawOnChartArea: false }, title: { display: true, text: "G$" } } }, plugins: { legend: { position: "bottom" } }, onClick: (e, els) => { if (els[0]) location.hash = "#/projects?type=" + s.by_type[els[0].index].project_type; } } });
80 + chart($("#c-sector"), { type: "bar", data: { labels: s.by_sector.map((r) => r.sector), datasets: [{ label: "Projets", data: s.by_sector.map((r) => r.n), backgroundColor: COL.bleu2 }] }, options: { indexAxis: "y", responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } }, onClick: (e, els) => { if (els[0]) location.hash = "#/projects?sector=" + encodeURIComponent(s.by_sector[els[0].index].sector); } } });
81 + const tot = {}; s.mentions_by_year.forEach((r) => (tot[r.year] = (tot[r.year] || 0) + r.n));
82 + const themes = [...new Set(s.themes_by_year.map((r) => r.project_type))];
83 + chart($("#c-themes"), { type: "line", data: { labels: years, datasets: themes.map((t, i) => ({ label: tl(t), data: years.map((y) => { const n = s.themes_by_year.find((r) => r.year === y && r.project_type === t)?.n || 0; return tot[y] ? +(100 * n / tot[y]).toFixed(2) : 0; }), borderColor: [COL.rouge, COL.violet, COL.bleu2, COL.teal, COL.vert, COL.or][i], backgroundColor: "transparent", tension: .25, pointRadius: 2 })) }, options: { responsive: true, maintainAspectRatio: false, scales: { y: { title: { display: true, text: "% des mentions" } } }, plugins: { legend: { position: "bottom" } } } });
84 + }
85 + const kpi = (v, l) => `<div class="card kpi"><b>${typeof v === "number" ? fmtN(v) : v}</b><span>${l}</span></div>`;
86 +
87 + /* ------------------------------------------------------------ projets */
88 + const PCOLS = [
89 + { key: "ticker", label: "Ticker", sortKey: "ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` },
90 + { key: "project_name", label: "Projet", sortKey: "name", trunc: true, render: (r) => `<a href="#/project/${esc(r.project_id)}">${esc(r.project_name)}</a><div class="muted small">${esc(r.company_name)}</div>` },
91 + { key: "project_type", label: "Type", sortKey: "type", render: (r) => `<span class="pill pill-blue">${esc(tl(r.project_type))}</span>` },
92 + { key: "status", label: "Statut", render: (r) => stPill(r.status) },
93 + { key: "canonical_location", label: "Lieu" },
94 + { key: "total_amount_usd", label: "Montant", num: true, sortKey: "amount", render: (r) => fmtUSD(r.total_amount_usd) },
95 + { key: "n_mentions", label: "Mentions", num: true, sortKey: "mentions", render: (r) => fmtN(r.n_mentions) },
96 + { key: "first_seen", label: "Première / dernière", sortKey: "first_seen", render: (r) => `${esc(r.first_seen?.slice(0, 7))} → ${esc(r.last_seen?.slice(0, 7))}` },
97 + { key: "avg_confidence", label: "Conf.", num: true, sortKey: "confidence", render: (r) => (r.avg_confidence ?? 0).toFixed(2) },
98 + ];
99 + async function projects() {
100 + const { q } = hashParams();
101 + const f = { q: q.q || "", type: q.type || "", sector: q.sector || "", status: q.status || "", ticker: q.ticker || "", location: q.location || "", tech: q.tech || "", partner: q.partner || "", min_amount: q.min_amount || "", year_from: q.year_from || "", year_to: q.year_to || "", min_confidence: q.min_confidence || "", has_amount: q.has_amount || "", sort: q.sort || "mentions", order: q.order || "desc", limit: 50, offset: +q.offset || 0 };
102 + const [tax, secs] = await Promise.all([api("/taxonomy"), api("/sectors")]);
103 + view.innerHTML = `<div class="page-head"><div><h1>Projets</h1><p>${fmtN(0)} — chargement…</p></div>
104 + <div><a class="btn btn-ghost btn-sm" id="csv" href="${API}/projects/export.csv?${qs(f)}" download>Exporter CSV</a></div></div>
105 + <form class="filters" id="pf">
106 + <label class="span2">Recherche<input name="q" value="${esc(f.q)}" placeholder="nom, description, entreprise, ticker"></label>
107 + <label>Type<select name="type"><option value="">Tous</option>${tax.map((t) => `<option value="${t.project_type}" ${f.type === t.project_type ? "selected" : ""}>${esc(t.label)} (${fmtN(t.n)})</option>`).join("")}</select></label>
108 + <label>Secteur<select name="sector"><option value="">Tous</option>${secs.map((s) => `<option ${f.sector === s.sector ? "selected" : ""}>${esc(s.sector)}</option>`).join("")}</select></label>
109 + <label>Statut<select name="status"><option value="">Tous</option>${Object.entries(STATUT).map(([k, v]) => `<option value="${k}" ${f.status === k ? "selected" : ""}>${v}</option>`).join("")}</select></label>
110 + <label>Ticker<input name="ticker" value="${esc(f.ticker)}"></label>
111 + <label>Lieu<input name="location" value="${esc(f.location)}"></label>
112 + <label>Technologie<input name="tech" value="${esc(f.tech)}"></label>
113 + <label>Partenaire<input name="partner" value="${esc(f.partner)}"></label>
114 + <label>Montant min (US$)<input name="min_amount" type="number" value="${esc(f.min_amount)}" placeholder="ex. 1e9"></label>
115 + <label>Année de<input name="year_from" type="number" value="${esc(f.year_from)}" min="2010" max="2026"></label>
116 + <label>Année à<input name="year_to" type="number" value="${esc(f.year_to)}" min="2010" max="2026"></label>
117 + <label>Confiance min<input name="min_confidence" type="number" step="0.05" min="0" max="1" value="${esc(f.min_confidence)}"></label>
118 + <label>Avec montant<select name="has_amount"><option value="">Indifférent</option><option value="true" ${f.has_amount ? "selected" : ""}>Oui</option></select></label>
119 + <label>&nbsp;<span><button class="btn btn-sm">Filtrer</button> <button type="button" class="btn btn-ghost btn-sm" id="reset">Effacer</button></span></label>
120 + </form><div id="res"><div class="loading">Chargement…</div></div>`;
121 + $("#pf").addEventListener("submit", (e) => { e.preventDefault(); const d = Object.fromEntries(new FormData(e.target)); setQuery({ ...d, sort: f.sort, order: f.order }); });
122 + $("#reset").addEventListener("click", () => (location.hash = "#/projects"));
123 + const data = await api("/projects", f);
124 + $(".page-head p").textContent = `${fmtN(data.total)} projets correspondent aux filtres.`;
125 + const res = $("#res");
126 + res.innerHTML = table(PCOLS, data.items, { href: (r) => `#/project/${r.project_id}`, sort: f.sort, order: f.order }) + pager(data.total, f.limit, f.offset);
127 + bindTable(res, (k) => setQuery({ ...f, offset: 0, sort: k, order: f.sort === k && f.order === "desc" ? "asc" : "desc" }));
128 + res.querySelectorAll("[data-off]").forEach((b) => b.addEventListener("click", () => setQuery({ ...f, offset: b.dataset.off })));
129 + }
130 +
131 + /* ------------------------------------------------------------ fiche projet */
132 + async function project(id) {
133 + const d = await api("/projects/" + id);
134 + const p = d.project;
135 + const list = (arr, cls) => (arr || []).length ? `<div class="tags">${arr.map((x) => `<span class="tag ${cls}">${esc(x)}</span>`).join("")}</div>` : `<span class="muted">—</span>`;
136 + view.innerHTML = `
137 + <div class="page-head"><div><div class="muted small"><a href="#/projects">Projets</a> › <a href="#/company/${esc(p.ticker)}">${esc(p.company_name)}</a></div>
138 + <h1>${esc(p.project_name)}</h1><p><span class="pill pill-blue">${esc(tl(p.project_type))}</span> &nbsp; ${stPill(p.status)} &nbsp; <span class="muted">${esc(p.sector || "")}</span></p></div>
139 + <div><a class="btn btn-ghost btn-sm" href="#/graph?node=${encodeURIComponent("P:" + p.project_id)}">Voir dans le graphe</a> <a class="btn btn-ghost btn-sm" href="/docs#/API%20v1%20(cl%C3%A9%20requise)/project_v1_projects__project_id__get" target="_blank">API ↗</a></div></div>
140 + <div class="grid g3">
141 + <div class="card" style="grid-column:span 2"><h3>Description</h3><p>${esc(p.description)}</p>
142 + <dl class="dl"><dt>Entreprise</dt><dd><a href="#/company/${esc(p.ticker)}">${esc(p.company_name)} (${esc(p.ticker)})</a> · CIK ${esc(p.cik)}</dd>
143 + <dt>Localisation</dt><dd>${p.canonical_location ? `<a href="#/projects?location=${encodeURIComponent(p.canonical_location)}">${esc(p.canonical_location)}</a>` : "—"}</dd>
144 + <dt>Montant (max divulgué)</dt><dd>${fmtUSD(p.total_amount_usd)}</dd>
145 + <dt>Observé</dt><dd>${esc(p.first_seen)} → ${esc(p.last_seen)} · ${fmtN(p.n_mentions)} mentions dans ${fmtN(p.n_filings)} filings</dd>
146 + <dt>Années citées</dt><dd>${p.first_year ?? "—"} → ${p.last_year ?? "—"}</dd>
147 + <dt>Confiance moyenne</dt><dd>${(p.avg_confidence ?? 0).toFixed(3)}</dd>
148 + <dt>Technologies</dt><dd>${list(p.technologies, "t-tech")}</dd>
149 + <dt>Identifiant</dt><dd class="mono">${esc(p.project_id)}</dd></dl></div>
150 + <div class="card"><h3>Chronologie (${d.timeline.length})</h3><ul class="timeline">${d.timeline.map((t) => `<li><div><b>${esc(t.form_type)}</b> · ${stPill(t.status)} ${t.amount_usd ? `· ${fmtUSD(t.amount_usd)}` : ""}</div><div class="when">${esc(t.filing_date)} · conf. ${(t.confidence ?? 0).toFixed(2)} · <span class="mono">${esc(t.accession_number)}</span></div><div class="snip">${esc(t.snippet)}</div></li>`).join("")}</ul></div>
151 + <div class="card" style="grid-column:span 2"><h3>Mentions extraites (${d.mentions.length})</h3>
152 + ${d.mentions.map((m) => `<div class="mention"><div class="mh"><b>${esc(m.form_type)}</b><span class="muted">${esc(m.filing_date)}</span><span class="pill pill-gris">${esc(m.section_name)}</span>${stPill(m.status)}${m.amount_usd ? `<span class="pill pill-or">${fmtUSD(m.amount_usd)}</span>` : ""}<span class="muted small">conf. ${(m.confidence ?? 0).toFixed(2)}</span><span class="sp" style="flex:1"></span><a class="small" href="#/mention/${esc(m.mention_id)}">détail</a></div>
153 + <p><b>${esc(m.project_name)}</b> — ${esc(m.description)}</p>${m.objective ? `<p class="muted"><i>Objectif :</i> ${esc(m.objective)}</p>` : ""}
154 + <div class="grid g2" style="gap:6px">${m.locations?.length ? `<div>${list(m.locations, "t-loc")}</div>` : ""}${m.technologies?.length ? `<div>${list(m.technologies, "t-tech")}</div>` : ""}${m.partners?.length ? `<div><span class="small muted">Partenaires</span> ${list(m.partners, "t-part")}</div>` : ""}${m.benefits?.length ? `<div><span class="small muted">Bénéfices</span> ${list(m.benefits, "t-ben")}</div>` : ""}${m.risks?.length ? `<div><span class="small muted">Risques</span> ${list(m.risks, "t-risk")}</div>` : ""}${m.years?.length ? `<div><span class="small muted">Années</span> ${list(m.years, "")}</div>` : ""}</div></div>`).join("")}</div>
155 + <div class="card"><h3>Projets similaires (embeddings)</h3>${d.similar.length ? d.similar.map((s) => `<div style="padding:6px 0;border-top:1px solid var(--ligne)"><a href="#/project/${esc(s.project_id)}">${esc(s.project_name)}</a><div class="small muted">${esc(s.ticker)} · ${esc(tl(s.project_type))} · <span class="score">${(s.score * 100).toFixed(0)} %</span></div></div>`).join("") : `<span class="muted">—</span>`}
156 + <h3 style="margin-top:14px">Voisins dans le graphe</h3><div>${d.graph.map((g) => `<a class="node-chip nt-${esc(g.node_type)}" href="#/graph?node=${encodeURIComponent(g.src === "P:" + p.project_id ? g.dst : g.src)}"><span class="nt">${esc(g.rel)}</span>${esc(g.label)}</a>`).join("") || `<span class="muted">—</span>`}</div></div>
157 + </div>`;
158 + }
159 +
160 + /* ------------------------------------------------------------ mention */
161 + async function mention(id) {
162 + const d = await api("/mentions/" + id);
163 + const m = d.mention;
164 + view.innerHTML = `<div class="page-head"><div><div class="muted small"><a href="#/mentions">Mentions</a> › <a href="#/project/${esc(m.project_id)}">projet résolu</a></div><h1>${esc(m.project_name)}</h1>
165 + <p><span class="pill pill-blue">${esc(tl(m.project_type))}</span> ${stPill(m.status)} · <a href="#/company/${esc(m.ticker)}">${esc(m.company_name)}</a> · ${esc(m.form_type)} du ${esc(m.filing_date)} · section <b>${esc(m.section_name)}</b></p></div></div>
166 + <div class="grid g2"><div class="card"><h3>Extraction</h3><dl class="dl">
167 + <dt>Description</dt><dd>${esc(m.description)}</dd><dt>Objectif</dt><dd>${esc(m.objective) || "—"}</dd>
168 + <dt>Montant</dt><dd>${fmtUSD(m.amount_usd)} ${m.amount_raw ? `<span class="muted">(${esc(m.amount_raw)})</span>` : ""}</dd>
169 + <dt>Localisations</dt><dd>${(m.locations || []).join(", ") || "—"}</dd><dt>Technologies</dt><dd>${(m.technologies || []).join(", ") || "—"}</dd>
170 + <dt>Partenaires</dt><dd>${(m.partners || []).join(", ") || "—"}</dd><dt>Fournisseurs</dt><dd>${(m.suppliers || []).join(", ") || "—"}</dd>
171 + <dt>Bénéfices</dt><dd>${(m.benefits || []).join(" · ") || "—"}</dd><dt>Risques</dt><dd>${(m.risks || []).join(" · ") || "—"}</dd>
172 + <dt>Années</dt><dd>${(m.years || []).join(", ") || "—"}</dd><dt>Confiance</dt><dd>${(m.confidence ?? 0).toFixed(3)} (backend ${esc(m.backend)})</dd>
173 + <dt>Accession</dt><dd class="mono">${esc(m.accession_number)}</dd><dt>Section</dt><dd class="mono">${esc(m.section_id)}</dd><dt>Mention</dt><dd class="mono">${esc(m.mention_id)}</dd></dl></div>
174 + <div class="card"><h3>Section source</h3>${d.section ? `<p class="small muted">${esc(d.section.section_name)} · ${fmtN(d.section.word_count)} mots · 10-K</p><button class="btn btn-sm" id="load-sec">Afficher le texte de la section</button><div id="sec"></div>` : `<div class="warn">Le texte source n'est disponible dans la base que pour les sections 10-K ingérées par SPID. Cette mention provient d'un ${esc(m.form_type)} dont la section reste dans le corpus parquet amont.</div>`}
175 + <p class="small muted" style="margin-top:12px">Filing sur EDGAR : <a target="_blank" rel="noopener" href="https://www.sec.gov/cgi-bin/browse-edgar?action=getcompany&CIK=${esc(m.cik)}&type=${esc(m.form_type)}&dateb=&owner=include&count=40">liste des ${esc(m.form_type)} de ${esc(m.ticker)} ↗</a></p></div></div>`;
176 + $("#load-sec")?.addEventListener("click", async (e) => {
177 + e.target.disabled = true;
178 + const s = await api("/sections/" + m.section_id);
179 + const key = (m.project_name || "").split(/\s+/).filter((w) => w.length > 4)[0];
180 + let txt = esc(s.text);
181 + if (key) txt = txt.replace(new RegExp(esc(key).replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "gi"), (x) => `<mark>${x}</mark>`);
182 + $("#sec").innerHTML = `<div class="sec-text">${txt}</div>`;
183 + });
184 + }
185 +
186 + /* ------------------------------------------------------------ mentions */
187 + async function mentions() {
188 + const { q } = hashParams();
189 + const f = { q: q.q || "", type: q.type || "", form: q.form || "", ticker: q.ticker || "", status: q.status || "", date_from: q.date_from || "", date_to: q.date_to || "", min_amount: q.min_amount || "", min_confidence: q.min_confidence || "", sort: q.sort || "date", order: q.order || "desc", limit: 50, offset: +q.offset || 0 };
190 + const tax = await api("/taxonomy");
191 + view.innerHTML = `<div class="page-head"><div><h1>Mentions</h1><p>Unité d'extraction : un projet cité dans une section d'un filing.</p></div></div>
192 + <form class="filters" id="mf">
193 + <label class="span2">Recherche<input name="q" value="${esc(f.q)}"></label>
194 + <label>Type<select name="type"><option value="">Tous</option>${tax.map((t) => `<option value="${t.project_type}" ${f.type === t.project_type ? "selected" : ""}>${esc(t.label)}</option>`).join("")}</select></label>
195 + <label>Formulaire<select name="form"><option value="">Tous</option>${["10-K", "10-Q", "8-K"].map((x) => `<option ${f.form === x ? "selected" : ""}>${x}</option>`).join("")}</select></label>
196 + <label>Statut<select name="status"><option value="">Tous</option>${Object.entries(STATUT).map(([k, v]) => `<option value="${k}" ${f.status === k ? "selected" : ""}>${v}</option>`).join("")}</select></label>
197 + <label>Ticker<input name="ticker" value="${esc(f.ticker)}"></label>
198 + <label>Du<input type="date" name="date_from" value="${esc(f.date_from)}"></label><label>Au<input type="date" name="date_to" value="${esc(f.date_to)}"></label>
199 + <label>Montant min<input type="number" name="min_amount" value="${esc(f.min_amount)}"></label>
200 + <label>Confiance min<input type="number" step="0.05" name="min_confidence" value="${esc(f.min_confidence)}"></label>
201 + <label>&nbsp;<span><button class="btn btn-sm">Filtrer</button> <button type="button" class="btn btn-ghost btn-sm" id="reset">Effacer</button></span></label></form>
202 + <div id="res"><div class="loading">Chargement…</div></div>`;
203 + $("#mf").addEventListener("submit", (e) => { e.preventDefault(); setQuery({ ...Object.fromEntries(new FormData(e.target)), sort: f.sort, order: f.order }); });
204 + $("#reset").addEventListener("click", () => (location.hash = "#/mentions"));
205 + const data = await api("/mentions", f);
206 + const cols = [
207 + { key: "filing_date", label: "Date", sortKey: "date" }, { key: "form_type", label: "Form." },
208 + { key: "ticker", label: "Ticker", sortKey: "ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` },
209 + { key: "project_name", label: "Projet / description", trunc: true, title: (r) => r.description, render: (r) => `<a href="#/mention/${esc(r.mention_id)}">${esc(r.project_name)}</a><div class="muted small">${esc(r.description)}</div>` },
210 + { key: "project_type", label: "Type", render: (r) => `<span class="pill pill-blue">${esc(tl(r.project_type))}</span>` },
211 + { key: "status", label: "Statut", render: (r) => stPill(r.status) }, { key: "section_name", label: "Section" },
212 + { key: "amount_usd", label: "Montant", num: true, sortKey: "amount", render: (r) => fmtUSD(r.amount_usd) },
213 + { key: "confidence", label: "Conf.", num: true, sortKey: "confidence", render: (r) => (r.confidence ?? 0).toFixed(2) },
214 + ];
215 + const res = $("#res");
216 + res.innerHTML = `<p class="muted">${fmtN(data.total)} mentions.</p>` + table(cols, data.items, { href: (r) => `#/mention/${r.mention_id}`, sort: f.sort, order: f.order }) + pager(data.total, f.limit, f.offset);
217 + bindTable(res, (k) => setQuery({ ...f, offset: 0, sort: k, order: f.sort === k && f.order === "desc" ? "asc" : "desc" }));
218 + res.querySelectorAll("[data-off]").forEach((b) => b.addEventListener("click", () => setQuery({ ...f, offset: b.dataset.off })));
219 + }
220 +
221 + /* ------------------------------------------------------------ entreprises */
222 + async function companies() {
223 + const { q } = hashParams();
224 + const f = { q: q.q || "", sector: q.sector || "", sort: q.sort || "projects", order: q.order || "desc", limit: 100, offset: +q.offset || 0 };
225 + const secs = await api("/sectors");
226 + view.innerHTML = `<div class="page-head"><div><h1>Entreprises</h1><p>500 sociétés du S&amp;P 500 et la taille de leur portefeuille de projets divulgués.</p></div></div>
227 + <form class="filters" id="cf"><label class="span2">Recherche<input name="q" value="${esc(f.q)}" placeholder="nom ou ticker"></label>
228 + <label>Secteur<select name="sector"><option value="">Tous</option>${secs.map((s) => `<option ${f.sector === s.sector ? "selected" : ""}>${esc(s.sector)}</option>`).join("")}</select></label>
229 + <label>&nbsp;<span><button class="btn btn-sm">Filtrer</button></span></label></form><div id="res"></div>`;
230 + $("#cf").addEventListener("submit", (e) => { e.preventDefault(); setQuery({ ...Object.fromEntries(new FormData(e.target)), sort: f.sort, order: f.order }); });
231 + const data = await api("/companies", f);
232 + const cols = [
233 + { key: "ticker", label: "Ticker", sortKey: "ticker", render: (r) => `<a href="#/company/${esc(r.ticker)}"><b>${esc(r.ticker)}</b></a>` },
234 + { key: "company_name", label: "Entreprise" }, { key: "sector", label: "Secteur" },
235 + { key: "n_projects", label: "Projets", num: true, sortKey: "projects", render: (r) => fmtN(r.n_projects) },
236 + { key: "n_mentions", label: "Mentions", num: true, sortKey: "mentions", render: (r) => fmtN(r.n_mentions) },
237 + { key: "n_types", label: "Types", num: true }, { key: "amount_usd", label: "Capital divulgué", num: true, sortKey: "amount", render: (r) => fmtUSD(r.amount_usd) },
238 + { key: "first_seen", label: "Période", render: (r) => `${esc(r.first_seen?.slice(0, 4))}–${esc(r.last_seen?.slice(0, 4))}` },
239 + ];
240 + const res = $("#res");
241 + res.innerHTML = table(cols, data.items, { href: (r) => `#/company/${r.ticker}`, sort: f.sort, order: f.order }) + pager(data.total, f.limit, f.offset);
242 + bindTable(res, (k) => setQuery({ ...f, offset: 0, sort: k, order: f.sort === k && f.order === "desc" ? "asc" : "desc" }));
243 + res.querySelectorAll("[data-off]").forEach((b) => b.addEventListener("click", () => setQuery({ ...f, offset: b.dataset.off })));
244 + }
245 + async function company(ident) {
246 + const d = await api("/companies/" + encodeURIComponent(ident));
247 + const c = d.company;
248 + view.innerHTML = `<div class="page-head"><div><div class="muted small"><a href="#/companies">Entreprises</a></div><h1>${esc(c.company_name)} <span class="muted">(${esc(c.ticker)})</span></h1>
249 + <p>${esc(c.sector || "")} · CIK ${esc(c.cik)} · ${fmtN(c.n_projects)} projets · ${fmtN(c.n_mentions)} mentions · ${esc(c.first_seen?.slice(0, 4))}–${esc(c.last_seen?.slice(0, 4))}</p></div>
250 + <div><a class="btn btn-ghost btn-sm" href="#/projects?ticker=${esc(c.ticker)}">Filtrer les projets</a> <a class="btn btn-ghost btn-sm" href="#/graph?node=${encodeURIComponent("C:" + c.cik)}">Graphe</a></div></div>
251 + <div class="kpis">${kpi(c.n_projects, "projets")}${kpi(c.n_mentions, "mentions")}${kpi(fmtUSD(c.amount_usd), "capital divulgué (Σ max)")}${kpi(d.by_type.length, "types de projets")}</div>
252 + <div class="grid g3">
253 + <div class="card"><h3>Par type</h3><div class="chart"><canvas id="c1"></canvas></div></div>
254 + <div class="card"><h3>Mentions par année</h3><div class="chart"><canvas id="c2"></canvas></div></div>
255 + <div class="card"><h3>Lieux, technologies, partenaires</h3>
256 + <div class="tags">${d.locations.map((x) => `<a class="tag t-loc" href="#/projects?ticker=${esc(c.ticker)}&location=${encodeURIComponent(x.location)}">${esc(x.location)} ${x.n}</a>`).join("")}</div><br>
257 + <div class="tags">${d.technologies.map((x) => `<a class="tag t-tech" href="#/projects?ticker=${esc(c.ticker)}&tech=${encodeURIComponent(x.technology)}">${esc(x.technology)} ${x.n}</a>`).join("")}</div><br>
258 + <div class="tags">${d.partners.map((x) => `<span class="tag t-part">${esc(x.partner)} ${x.n}</span>`).join("")}</div></div>
259 + </div>
260 + <h2 style="margin-top:18px">Portefeuille de projets (${fmtN(d.projects.length)})</h2><div id="res"></div>`;
261 + chart($("#c1"), { type: "bar", data: { labels: d.by_type.map((r) => r.label), datasets: [{ data: d.by_type.map((r) => r.n), backgroundColor: COL.bleu }] }, options: { indexAxis: "y", responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } } } });
262 + chart($("#c2"), { type: "bar", data: { labels: d.by_year.map((r) => r.year), datasets: [{ data: d.by_year.map((r) => r.n), backgroundColor: COL.or }] }, options: { responsive: true, maintainAspectRatio: false, plugins: { legend: { display: false } } } });
263 + const res = $("#res"); res.innerHTML = table(PCOLS.filter((c) => c.key !== "ticker"), d.projects, { href: (r) => `#/project/${r.project_id}` }); bindTable(res);
264 + }
265 +
266 + /* ------------------------------------------------------------ graphe */
267 + async function graph() {
268 + const { q } = hashParams();
269 + view.innerHTML = `<div class="page-head"><div><h1>Graphe de connaissances</h1><p>32 226 nœuds (entreprises, projets, lieux, technologies, partenaires) et 33 176 arêtes <span class="mono">owns · located_in · uses</span>. Cherchez un nœud puis naviguez de voisin en voisin.</p></div></div>
270 + <form class="filters" id="gf"><label class="span2">Nœud<input name="q" value="${esc(q.q || "")}" placeholder="Texas, Tesla, batterie, AI…"></label>
271 + <label>Type<select name="type"><option value="">Tous</option>${["Company", "Project", "Location", "Technology", "Partner"].map((t) => `<option ${q.type === t ? "selected" : ""}>${t}</option>`).join("")}</select></label>
272 + <label>&nbsp;<span><button class="btn btn-sm">Chercher</button></span></label></form><div id="hits"></div><div id="node"></div>`;
273 + $("#gf").addEventListener("submit", (e) => { e.preventDefault(); const d = Object.fromEntries(new FormData(e.target)); location.hash = "#/graph?" + qs(d); });
274 + if (q.node) await showNode(q.node);
275 + else if (q.q) {
276 + const hits = await api("/graph/search", { q: q.q, type: q.type || "", limit: 60 });
277 + $("#hits").innerHTML = `<div class="card"><h3>${hits.length} nœuds</h3>${hits.map((h) => `<a class="node-chip nt-${esc(h.node_type)}" href="#/graph?node=${encodeURIComponent(h.node_id)}"><span class="nt">${esc(h.node_type)}</span>${esc(h.label)} <span class="muted small">· ${fmtN(h.degree)}</span></a>`).join("") || "<span class='muted'>Aucun résultat.</span>"}</div>`;
278 + } else {
279 + $("#hits").innerHTML = `<div class="card"><h3>Points d'entrée</h3><div>${[["L:texas", "Location"], ["L:china", "Location"], ["L:united states", "Location"], ["T:AI", "Technology"], ["T:Renewable Energy", "Technology"], ["T:Battery/Storage", "Technology"], ["T:Data Center", "Technology"], ["C:0001318605", "Company"], ["C:0000045012", "Company"]].map(([n, t]) => `<a class="node-chip nt-${t}" href="#/graph?node=${encodeURIComponent(n)}"><span class="nt">${t}</span>${esc(n.slice(2))}</a>`).join("")}</div></div>`;
280 + }
281 + }
282 + async function showNode(nodeId) {
283 + const d = await api("/graph/node/" + nodeId, { limit: 400 });
284 + const n = d.node; const props = safeJson(n.props);
285 + const groups = {}; d.edges.forEach((e) => ((groups[e.neighbor_type] ||= []).push(e)));
286 + const target = n.node_type === "Project" ? `<a class="btn btn-sm" href="#/project/${esc(n.node_id.slice(2))}">Fiche du projet</a>` : n.node_type === "Company" ? `<a class="btn btn-sm" href="#/company/${esc(n.node_id.slice(2))}">Fiche de l'entreprise</a>` : n.node_type === "Location" ? `<a class="btn btn-sm" href="#/projects?location=${encodeURIComponent(n.label)}">Projets à cet endroit</a>` : n.node_type === "Technology" ? `<a class="btn btn-sm" href="#/projects?tech=${encodeURIComponent(n.label)}">Projets avec cette technologie</a>` : `<a class="btn btn-sm" href="#/projects?partner=${encodeURIComponent(n.label)}">Projets avec ce partenaire</a>`;
287 + $("#node").innerHTML = `<div class="card"><div class="page-head"><div><span class="pill pill-blue">${esc(n.node_type)}</span> <h2 style="display:inline">${esc(n.label)}</h2><div class="muted small mono">${esc(n.node_id)} · degré ${fmtN(d.degree)}</div>
288 + ${props ? `<div class="small muted">${Object.entries(props).filter(([, v]) => v != null).map(([k, v]) => `${esc(k)} = <b>${esc(k.includes("amount") ? fmtUSD(v) : v)}</b>`).join(" · ")}</div>` : ""}</div><div>${target}</div></div>
289 + <svg class="svg-graph" id="svg"></svg>
290 + ${Object.entries(groups).map(([t, es]) => `<h3 style="margin-top:12px">${esc(t)} (${es.length})</h3><div>${es.slice(0, 200).map((e) => `<a class="node-chip nt-${esc(t)}" href="#/graph?node=${encodeURIComponent(e.neighbor_id)}"><span class="nt">${esc(e.rel)} ${e.direction === "out" ? "→" : "←"}</span>${esc(e.neighbor_label)}</a>`).join("")}${es.length > 200 ? `<span class="muted small"> … ${es.length - 200} de plus</span>` : ""}</div>`).join("") || "<p class='muted'>Ce nœud n'a pas d'arête (les partenaires ne sont pas encore reliés dans cette version du graphe).</p>"}</div>`;
291 + drawRadial($("#svg"), n, d.edges.slice(0, 48));
292 + }
293 + function safeJson(s) { if (!s) return null; try { return typeof s === "string" ? JSON.parse(s) : s; } catch { return null; } }
294 + function drawRadial(svg, center, edges) {
295 + const W = svg.clientWidth || 900, H = 520, cx = W / 2, cy = H / 2;
296 + const colors = { Company: COL.bleu, Project: COL.violet, Location: COL.orange, Technology: COL.teal, Partner: COL.or };
297 + const n = edges.length; const R = Math.min(W, H) / 2 - 70;
298 + let s = `<g>`;
299 + edges.forEach((e, i) => { const a = (2 * Math.PI * i) / n - Math.PI / 2; const x = cx + R * Math.cos(a), y = cy + R * Math.sin(a); s += `<line x1="${cx}" y1="${cy}" x2="${x}" y2="${y}"></line>`; });
300 + edges.forEach((e, i) => { const a = (2 * Math.PI * i) / n - Math.PI / 2; const x = cx + R * Math.cos(a), y = cy + R * Math.sin(a); const lab = e.neighbor_label.length > 28 ? e.neighbor_label.slice(0, 26) + "…" : e.neighbor_label; const anchor = Math.cos(a) > 0.2 ? "start" : Math.cos(a) < -0.2 ? "end" : "middle"; s += `<a href="#/graph?node=${encodeURIComponent(e.neighbor_id)}"><circle cx="${x}" cy="${y}" r="6" fill="${colors[e.neighbor_type] || COL.gris}"></circle><text x="${x + 10 * Math.cos(a)}" y="${y + 10 * Math.sin(a) + 4}" text-anchor="${anchor}">${esc(lab)}</text></a>`; });
301 + s += `<circle cx="${cx}" cy="${cy}" r="14" fill="${colors[center.node_type] || COL.gris}" stroke="#fff" stroke-width="3"></circle><text x="${cx}" y="${cy + 30}" text-anchor="middle" font-weight="700">${esc(center.label.slice(0, 40))}</text></g>`;
302 + svg.setAttribute("viewBox", `0 0 ${W} ${H}`); svg.innerHTML = s;
303 + }
304 +
305 + /* ------------------------------------------------------------ recherche sémantique */
306 + async function semantic() {
307 + const { q } = hashParams();
308 + view.innerHTML = `<div class="page-head"><div><h1>Recherche sémantique</h1><p>Décrivez un projet en langage naturel (anglais recommandé, langue des filings) : la requête est encodée avec all-MiniLM-L6-v2 et comparée aux 19 227 vecteurs de projets.</p></div></div>
309 + <form class="filters" id="sf"><label class="span2" style="grid-column:span 3">Requête<input name="q" value="${esc(q.q || "")}" placeholder="ex. battery cell factory in the United States, AI platform for banks, LNG export terminal…"></label>
310 + <label>&nbsp;<span><button class="btn btn-sm">Chercher</button></span></label></form><div id="res"></div>`;
311 + $("#sf").addEventListener("submit", (e) => { e.preventDefault(); location.hash = "#/semantic?" + qs(Object.fromEntries(new FormData(e.target))); });
312 + if (!q.q) { $("#res").innerHTML = `<div class="examples">${["gigafactory for battery cells", "generative AI assistant for customers", "offshore wind farm", "ERP migration to SAP S/4HANA", "new hospital or clinical trial for oncology drug", "data center campus in Virginia", "restructuring plan with plant closures"].map((x) => `<button data-q="${esc(x)}">${esc(x)}</button>`).join("")}</div>`; $("#res").querySelectorAll("button").forEach((b) => b.addEventListener("click", () => (location.hash = "#/semantic?q=" + encodeURIComponent(b.dataset.q)))); return; }
313 + $("#res").innerHTML = `<div class="loading">Encodage et recherche…</div>`;
314 + try {
315 + const d = await api("/search/semantic", { q: q.q, k: 40 });
316 + const cols = [{ key: "score", label: "Score", num: true, render: (r) => `<span class="score">${(r.score * 100).toFixed(1)} %</span>` }, ...PCOLS.filter((c) => !["avg_confidence", "first_seen"].includes(c.key))];
317 + const res = $("#res"); res.innerHTML = table(cols, d.items, { href: (r) => `#/project/${r.project_id}` }); bindTable(res);
318 + } catch (e) { $("#res").innerHTML = `<div class="warn">${esc(e.message)}</div><p class="muted">La similarité entre projets reste disponible sur chaque fiche (« Projets similaires »), car elle n'utilise que les vecteurs déjà stockés.</p>`; }
319 + }
320 +
321 + /* ------------------------------------------------------------ SQL */
322 + const SQL_EXAMPLES = [
323 + ["Projets par type", "select project_type, count(*) n, round(sum(total_amount_usd)/1e9,1) capital_gusd\nfrom projects group by 1 order by n desc"],
324 + ["Centres de données > 500 M$", "select ticker, project_name, canonical_location, total_amount_usd, first_seen\nfrom projects where project_type='data_center' and total_amount_usd > 5e8\norder by total_amount_usd desc"],
325 + ["Mentions IA par année", "select year(filing_date) y, count(*) n\nfrom project_mentions where project_type='ai_initiative' group by 1 order by 1"],
326 + ["Technologies les plus citées", "select lower(t) technology, count(*) n\nfrom projects, unnest(technologies) as u(t) group by 1 order by n desc limit 25"],
327 + ["Partenaires de Tesla", "select p partner, count(*) n from project_mentions, unnest(partners) as u(p)\nwhere ticker='TSLA' group by 1 order by n desc"],
328 + ["Durée de suivi médiane par type", "select project_type, median(last_year-first_year) mediane_annees, count(*) n\nfrom projects where last_year>=first_year group by 1 order by 2 desc"],
329 + ["Graphe : lieux les plus connectés", "select n.label, count(*) degre from kg_edges e join kg_nodes n on n.node_id=e.dst\nwhere e.rel='located_in' group by 1 order by 2 desc limit 20"],
330 + ["Texte 10-K : sections les plus longues", "select s.accession_number, s.section_name, s.word_count\nfrom spid_sections s order by word_count desc limit 10"],
331 + ];
332 + async function sqlPage() {
333 + const schema = await api("/schema");
334 + const saved = localStorage.getItem("pdb_sql") || SQL_EXAMPLES[0][1];
335 + view.innerHTML = `<div class="page-head"><div><h1>Bac à sable SQL</h1><p>DuckDB en lecture seule sur les 7 tables de la base. SELECT / WITH / DESCRIBE / SUMMARIZE ; 20 s et 5 000 lignes au maximum. ⌘⏎ pour exécuter.</p></div></div>
336 + <div class="sql-layout"><div class="card schema"><h3>Schéma</h3>${schema.map((t) => `<details ${t.table === "projects" ? "open" : ""}><summary>${esc(t.table)} <span class="muted">(${fmtN(t.rows)})</span></summary><ul>${t.columns.map((c) => `<li>${esc(c.name)} <span>${esc(c.type)}</span></li>`).join("")}</ul></details>`).join("")}</div>
337 + <div><textarea class="sql" id="sql">${esc(saved)}</textarea>
338 + <div class="examples">${SQL_EXAMPLES.map((e, i) => `<button data-i="${i}">${esc(e[0])}</button>`).join("")}</div>
339 + <div style="display:flex;gap:10px;align-items:center;margin:6px 0 12px"><button class="btn" id="run">Exécuter</button><label class="small muted">Limite <input class="inp" id="lim" type="number" value="500" min="1" max="5000" style="width:90px"></label><button class="btn btn-ghost btn-sm" id="csv" disabled>Télécharger CSV</button><span class="muted small" id="info"></span></div>
340 + <div id="out"></div></div></div>`;
341 + const ta = $("#sql");
342 + view.querySelectorAll(".examples button").forEach((b) => b.addEventListener("click", () => { ta.value = SQL_EXAMPLES[b.dataset.i][1]; run(); }));
343 + let last = null;
344 + async function run() {
345 + const sql = ta.value; localStorage.setItem("pdb_sql", sql);
346 + $("#out").innerHTML = `<div class="loading">Exécution…</div>`; $("#info").textContent = "";
347 + try {
348 + const r = await fetch(API + "/sql", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ sql, limit: +$("#lim").value || 500 }) });
349 + const d = await r.json();
350 + if (!r.ok) throw new Error(d.detail || r.statusText);
351 + last = d;
352 + $("#info").textContent = `${fmtN(d.rows.length)} lignes${d.truncated ? " (tronqué)" : ""} · ${d.elapsed_ms} ms`; $("#csv").disabled = false;
353 + $("#out").innerHTML = `<div class="table-wrap"><table><thead><tr>${d.columns.map((c) => `<th>${esc(c)}</th>`).join("")}</tr></thead><tbody>${d.rows.map((row) => `<tr>${row.map((v) => `<td class="${typeof v === "number" ? "num" : ""}">${esc(Array.isArray(v) ? v.join(" | ") : typeof v === "object" && v ? JSON.stringify(v) : v)}</td>`).join("")}</tr>`).join("")}</tbody></table></div>`;
354 + } catch (e) { $("#out").innerHTML = `<div class="err">${esc(e.message)}</div>`; }
355 + }
356 + $("#run").addEventListener("click", run);
357 + ta.addEventListener("keydown", (e) => { if ((e.metaKey || e.ctrlKey) && e.key === "Enter") run(); });
358 + $("#csv").addEventListener("click", () => { if (!last) return; const csv = [last.columns.join(",")].concat(last.rows.map((r) => r.map((v) => `"${String(Array.isArray(v) ? v.join("|") : v ?? "").replace(/"/g, '""')}"`).join(","))).join("\n"); const a = document.createElement("a"); a.href = URL.createObjectURL(new Blob([csv], { type: "text/csv" })); a.download = "pdb_query.csv"; a.click(); });
359 + run();
360 + }
361 +
362 + /* ------------------------------------------------------------ à propos */
363 + async function about() {
364 + view.innerHTML = `<div class="prose"><h1>La base SPID et son pipeline</h1>
365 + <p><b>SPID</b> (<i>SEC Project Intelligence Database</i>) reconstruit le portefeuille de projets stratégiques des grandes sociétés cotées américaines à partir du texte brut de leurs documents réglementaires (10-K annuels, 10-Q trimestriels, 8-K d'événements) déposés auprès de la SEC sur EDGAR. Plutôt que de lire ces filings comme des documents financiers, SPID les traite comme une source d'intelligence sur les <i>projets</i> : usines, centres de données, acquisitions, programmes de R&amp;D, initiatives d'IA, transition énergétique…</p>
366 + <h2>D'où viennent les données</h2>
367 + <ul><li><b>Univers</b> : les 500 sociétés du S&amp;P 500 (instantané 2026), chacune reliée à son identifiant SEC (CIK).</li>
368 + <li><b>Corpus</b> : 134 124 dépôts (103 927 8-K, 22 542 10-Q, 7 655 10-K) téléchargés depuis l'API publique de la SEC, janvier 2010 → mai 2026, soit 88 Go de HTML brut, nettoyés et découpés en sections.</li>
369 + <li><b>Sections analysées</b> : 153 894 sections d'au moins 30 mots (les 10-K sont découpés par SPID en Items 1, 1A, 2, 7 et 7A ; les 8-K sont traités en une seule section).</li></ul>
370 + <h2>Comment fonctionne le pipeline</h2>
371 + <ol><li><b>Ingestion</b> des 10-K bruts (nettoyage HTML, découpage en Items).</li>
372 + <li><b>Extraction</b> par grand modèle de langage (gpt-4o-mini, température 0, JSON strict) : un pré-filtre lexical écarte les sections sans signal de projet, une fenêtre de focalisation limite le texte envoyé ; le modèle renvoie nom, type (taxonomie de 21 catégories), description, objectif, statut, montant, lieux, technologies, partenaires, bénéfices, risques, années et confiance. Les mentions sous 0,45 de confiance sont écartées.</li>
373 + <li><b>Résolution</b> : les mentions sont fusionnées en projets par entreprise × type × ancre (première localisation ou premier mot significatif du nom) ; chaque projet reçoit une chronologie, un statut courant, le montant maximal divulgué et des dates de première et dernière observation.</li>
374 + <li><b>Graphe de connaissances</b> : entreprises, projets, lieux, technologies et partenaires reliés par <span class="mono">owns</span>, <span class="mono">located_in</span>, <span class="mono">uses</span>.</li>
375 + <li><b>Embeddings</b> : chaque projet est encodé en 384 dimensions (all-MiniLM-L6-v2) pour la recherche sémantique et les projets similaires.</li></ol>
376 + <h2>Ce que contient la base</h2>
377 + <p>39 930 mentions, 19 227 projets, 39 930 points de chronologie, 32 226 nœuds et 33 176 arêtes, 19 227 vecteurs, 27 164 sections 10-K en texte intégral. Un fichier DuckDB de 2,9 Go, interrogeable ici en SQL.</p>
378 + <h2>Limites à garder en tête</h2>
379 + <ul><li>SPID mesure la <i>divulgation</i> de projets, non les projets eux-mêmes ; les incitations à divulguer varient selon les secteurs.</li>
380 + <li>Les montants sont auto-déclarés, hétérogènes en portée et absents pour la moitié des projets ; quelques valeurs aberrantes subsistent.</li>
381 + <li>L'extraction par LLM comporte des faux positifs et négatifs ; la résolution est heuristique (elle peut scinder ou fusionner à tort).</li>
382 + <li>Les années citées (first_year / last_year) peuvent inclure des horizons projetés ; préférez first_seen / last_seen pour dater l'observation.</li></ul>
383 + <p>Le <a href="/report.pdf" target="_blank">rapport technique complet (PDF)</a> détaille la provenance, chaque étape du pipeline, le schéma et le portrait statistique. Équipe : Manel Kammoun, Charli Tandja Mbianda, Simon-Pierre Boucher — Département des sciences administratives, Université du Québec en Outaouais.</p></div>`;
384 + }
385 +
386 + /* ------------------------------------------------------------ routeur */
387 + const routes = { "": dashboard, projects, companies, mentions, graph, semantic, sql: sqlPage, api: () => window.PDBPlayground.render({ view, api, esc, fmtN, fmtUSD, qs, toast, tl }), about };
388 + async function render() {
389 + destroyCharts();
390 + const { parts } = hashParams();
391 + const r = parts[0];
392 + document.querySelectorAll("#nav a[data-route]").forEach((a) => a.classList.toggle("active", a.dataset.route === r));
393 + $("#burger") && $(".side").classList.remove("open");
394 + window.scrollTo(0, 0);
395 + try {
396 + if (r === "project" && parts[1]) await project(parts[1]);
397 + else if (r === "company" && parts[1]) await company(parts[1]);
398 + else if (r === "mention" && parts[1]) await mention(parts[1]);
399 + else if (routes[r]) await routes[r]();
400 + else view.innerHTML = `<div class="card"><h2>Page introuvable</h2><a href="#/">Retour au tableau de bord</a></div>`;
401 + } catch (e) { view.innerHTML = `<div class="err">Erreur : ${esc(e.message)}</div>`; }
402 + }
403 + async function init() {
404 + try { const tax = await api("/taxonomy"); tax.forEach((t) => (TYPE[t.project_type] = t.label)); } catch { }
405 + try { const h = await fetch("/health").then((r) => r.json()); $("#kpi-projects").textContent = fmtN(h.projects) + " projets"; $("#side-status").textContent = `service ${h.status} · sémantique : ${h.semantic}`; } catch { $("#side-status").textContent = "service indisponible"; }
406 + fetch("/report.pdf", { method: "HEAD" }).then((r) => { if (!r.ok) $("#report-link").remove(); }).catch(() => { });
407 + $("#global-search").addEventListener("submit", (e) => { e.preventDefault(); const v = $("#global-q").value.trim(); if (v) location.hash = "#/projects?q=" + encodeURIComponent(v); });
408 + document.addEventListener("keydown", (e) => { if ((e.metaKey || e.ctrlKey) && e.key.toLowerCase() === "k") { e.preventDefault(); $("#global-q").focus(); } });
409 + $("#burger").addEventListener("click", () => $(".side").classList.toggle("open"));
410 + window.addEventListener("hashchange", render);
411 + render();
412 + }
413 + init();
414 +})();
added web/index.html +62 −0
@@ -0,0 +1,62 @@
1 +<!doctype html>
2 +<html lang="fr">
3 +<head>
4 +<meta charset="utf-8">
5 +<meta name="viewport" content="width=device-width, initial-scale=1">
6 +<title>PDB — SEC Project Intelligence Database · Explorateur</title>
7 +<meta name="description" content="Explorateur et API de la base SPID : 19 227 projets stratégiques extraits des filings SEC de 500 sociétés du S&P 500 (2010–2026). UQO — Département des sciences administratives.">
8 +<link rel="icon" href="data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 32 32'%3E%3Crect width='32' height='32' rx='7' fill='%23003E7E'/%3E%3Ctext x='16' y='22' font-family='Helvetica,Arial' font-size='15' font-weight='700' fill='%23C6A300' text-anchor='middle'%3EPDB%3C/text%3E%3C/svg%3E">
9 +<link rel="preconnect" href="https://cdn.jsdelivr.net">
10 +<link rel="stylesheet" href="/style.css">
11 +<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.4/dist/chart.umd.min.js" defer></script>
12 +<script src="/playground.js" defer></script>
13 +<script src="/app.js" defer></script>
14 +</head>
15 +<body>
16 +<div class="shell">
17 + <aside class="side">
18 + <a class="brand" href="#/">
19 + <span class="brand-mark">PDB</span>
20 + <span class="brand-text"><b>SEC Project Intelligence</b><small>Database · UQO</small></span>
21 + </a>
22 + <nav class="nav" id="nav">
23 + <a href="#/" data-route="">Tableau de bord</a>
24 + <a href="#/projects" data-route="projects">Projets</a>
25 + <a href="#/companies" data-route="companies">Entreprises</a>
26 + <a href="#/mentions" data-route="mentions">Mentions</a>
27 + <a href="#/graph" data-route="graph">Graphe</a>
28 + <a href="#/semantic" data-route="semantic">Recherche sémantique</a>
29 + <div class="nav-sep">Développeurs</div>
30 + <a href="#/sql" data-route="sql">Bac à sable SQL</a>
31 + <a href="#/api" data-route="api">API &amp; playground</a>
32 + <a href="/docs" target="_blank" rel="noopener">Documentation OpenAPI ↗</a>
33 + <div class="nav-sep">À propos</div>
34 + <a href="#/about" data-route="about">La base et le pipeline</a>
35 + <a href="/report.pdf" target="_blank" rel="noopener" id="report-link">Rapport technique (PDF) ↗</a>
36 + </nav>
37 + <div class="side-foot">
38 + <div>Université du Québec en Outaouais<br>Département des sciences administratives</div>
39 + <div class="muted" id="side-status">connexion…</div>
40 + </div>
41 + </aside>
42 + <main class="main">
43 + <header class="topbar">
44 + <button class="burger" id="burger" aria-label="Menu">☰</button>
45 + <form class="search" id="global-search" autocomplete="off">
46 + <input id="global-q" type="search" placeholder="Rechercher un projet, une entreprise, un ticker… (⌘K)">
47 + </form>
48 + <div class="topbar-right">
49 + <span class="pill pill-blue" id="kpi-projects">…</span>
50 + <a class="btn btn-ghost" href="#/api">Obtenir l'API</a>
51 + </div>
52 + </header>
53 + <section id="view" class="view"><div class="loading">Chargement…</div></section>
54 + <footer class="foot">
55 + <span>PDB API · données SPID (filings SEC 10-K · 10-Q · 8-K, 500 sociétés, 2010–2026) · extraction par LLM, résolution, graphe, embeddings.</span>
56 + <span>Kammoun · Tandja Mbianda · Boucher — UQO</span>
57 + </footer>
58 + </main>
59 +</div>
60 +<div id="toast" class="toast" hidden></div>
61 +</body>
62 +</html>
added web/playground.js +337 −0
@@ -0,0 +1,337 @@
1 +/* PDB — page API & playground (démarrage rapide, playground, recettes, référence, SDK). */
2 +window.PDBPlayground = (() => {
3 + const LANGS = [["curl", "curl"], ["py", "Python"], ["js", "JavaScript"], ["r", "R"]];
4 + const KEY_PH = "VOTRE_CLE";
5 + const key = () => localStorage.getItem("pdb_key") || "";
6 + const lang = () => localStorage.getItem("pdb_lang") || "py";
7 + const esc = (s) => String(s ?? "").replace(/[&<>"']/g, (c) => ({ "&": "&amp;", "<": "&lt;", ">": "&gt;", '"': "&quot;", "'": "&#39;" }[c]));
8 + const qs = (o) => Object.entries(o).filter(([, v]) => v !== "" && v != null).map(([k, v]) => `${encodeURIComponent(k)}=${encodeURIComponent(v)}`).join("&");
9 + const fmtN = (n) => n == null ? "—" : Number(n).toLocaleString("fr-CA");
10 +
11 + /* ------------------------------------------------------------------ catalogue des routes */
12 + const P = (name, desc, opts = {}) => ({ name, desc, ...opts });
13 + const PROJECT_FILTERS = [
14 + P("q", "texte libre : nom, description, entreprise, ticker"), P("type", "project_type, liste séparée par des virgules", { enum: "types" }),
15 + P("sector", "secteur GICS (contient)", { enum: "sectors" }), P("status", "planned | in_progress | completed | mentioned (liste possible)"),
16 + P("ticker", "ticker exact"), P("cik", "CIK SEC (10 chiffres ou moins)"), P("location", "localisation canonique (contient)"),
17 + P("tech", "technologie (contient)"), P("partner", "partenaire cité dans une mention (contient)"),
18 + P("min_amount", "montant divulgué minimal, US$", { type: "number" }), P("max_amount", "montant maximal, US$", { type: "number" }),
19 + P("year_from", "dernière observation ≥ année", { type: "number" }), P("year_to", "première observation ≤ année", { type: "number" }),
20 + P("min_confidence", "confiance moyenne minimale (0–1)", { type: "number" }), P("has_amount", "true : seulement les projets avec montant"),
21 + ];
22 + const PAGE = [P("sort", "clé de tri"), P("order", "asc | desc"), P("limit", "≤ 500 (défaut 50)", { type: "number" }), P("offset", "décalage", { type: "number" })];
23 + const ENDPOINTS = [
24 + { g: "Découverte", m: "GET", p: "/v1/health", key: false, d: "État du service : compteur de projets, état du modèle sémantique. Sans clé.", params: [], ex: [{ l: "État", v: {} }] },
25 + { g: "Découverte", m: "GET", p: "/v1/stats", d: "Vue d'ensemble : compteurs, répartitions par type, secteur, statut, formulaire, année, thèmes émergents, top entreprises/technologies/lieux. C'est la source du tableau de bord.", params: [], ex: [{ l: "Tout", v: {} }] },
26 + { g: "Découverte", m: "GET", p: "/v1/taxonomy", d: "Les 21 types de projets (clé, libellé, nombre de projets, capital divulgué).", params: [], ex: [{ l: "Liste", v: {} }] },
27 + { g: "Découverte", m: "GET", p: "/v1/sectors", d: "Les 11 secteurs GICS avec nombre de projets, d'entreprises et capital.", params: [], ex: [{ l: "Liste", v: {} }] },
28 + { g: "Découverte", m: "GET", p: "/v1/technologies", d: "Technologies citées dans les projets, par fréquence.", params: [P("q", "filtre texte"), P("limit", "≤ 500", { type: "number" })], ex: [{ l: "Top 25", v: { limit: 25 } }, { l: "Batteries", v: { q: "batter" } }] },
29 + { g: "Découverte", m: "GET", p: "/v1/locations", d: "Localisations canoniques, par fréquence, avec capital.", params: [P("q", "filtre texte"), P("limit", "≤ 500", { type: "number" })], ex: [{ l: "Top 25", v: { limit: 25 } }, { l: "Texas", v: { q: "texas" } }] },
30 + { g: "Découverte", m: "GET", p: "/v1/partners", d: "Partenaires cités dans les mentions (coentreprises, clients, fournisseurs nommés).", params: [P("q", "filtre texte"), P("limit", "≤ 500", { type: "number" })], ex: [{ l: "Top 25", v: { limit: 25 } }, { l: "Microsoft", v: { q: "microsoft" } }] },
31 + { g: "Projets", m: "GET", p: "/v1/projects", d: "Recherche de projets avec filtres combinables. Réponse paginée {total, limit, offset, items}. Tri : amount, mentions, filings, first_seen, last_seen, confidence, name, ticker, type.", params: [...PROJECT_FILTERS, ...PAGE], ex: [
32 + { l: "Centres de données > 500 M$", v: { type: "data_center", min_amount: "5e8", sort: "amount", limit: 10 } },
33 + { l: "IA depuis 2024", v: { type: "ai_initiative", year_from: 2024, sort: "last_seen", limit: 20 } },
34 + { l: "Usines au Texas", v: { type: "plant_construction,manufacturing_expansion", location: "texas", limit: 20 } },
35 + { l: "Batteries, terminées", v: { tech: "battery", status: "completed", limit: 20 } },
36 + { l: "Tesla", v: { ticker: "TSLA", sort: "mentions", limit: 20 } },
37 + { l: "Partenaire NVIDIA", v: { partner: "nvidia", limit: 20 } },
38 + ] },
39 + { g: "Projets", m: "GET", p: "/v1/projects/export.csv", d: "Export CSV des projets filtrés (mêmes filtres que /v1/projects, ≤ 50 000 lignes, listes jointes par |).", params: PROJECT_FILTERS, raw: true, ex: [{ l: "Tous les centres de données", v: { type: "data_center" } }, { l: "Services publics avec montant", v: { sector: "Utilities", has_amount: "true" } }] },
40 + { g: "Projets", m: "GET", p: "/v1/projects/{project_id}", d: "Fiche complète d'un projet : attributs, chronologie (une ligne par mention datée), mentions détaillées (lieux, technologies, partenaires, bénéfices, risques), voisins du graphe et projets similaires par embeddings.", params: [P("project_id", "identifiant MD5 (voir /v1/projects)", { path: true })], ex: [{ l: "Tesla — Energy Storage Products", v: { project_id: "TSLA_ENERGY" } }, { l: "WEC — Data Center Investments", v: { project_id: "46fe2966ba0cbf6228807faca0645b54" } }] },
41 + { g: "Projets", m: "GET", p: "/v1/projects/{project_id}/similar", d: "Projets sémantiquement proches (similarité cosinus des vecteurs MiniLM 384-d stockés).", params: [P("project_id", "identifiant", { path: true }), P("k", "≤ 50", { type: "number" })], ex: [{ l: "10 voisins", v: { project_id: "46fe2966ba0cbf6228807faca0645b54", k: 10 } }] },
42 + { g: "Mentions & sections", m: "GET", p: "/v1/mentions", d: "Mentions individuelles (une par projet et par section de filing) : l'unité d'extraction brute, avec formulaire, date, section, montant, confiance. Chaque mention porte le project_id du projet résolu.", params: [P("q", "texte libre"), P("type", "project_type", { enum: "types" }), P("sector", "", { enum: "sectors" }), P("status", ""), P("ticker", ""), P("cik", ""), P("form", "10-K, 10-Q, 8-K (liste possible)"), P("date_from", "AAAA-MM-JJ"), P("date_to", "AAAA-MM-JJ"), P("min_amount", "US$", { type: "number" }), P("min_confidence", "0–1", { type: "number" }), P("accession", "numéro d'accession SEC"), P("sort", "date | amount | confidence | ticker"), P("order", "asc | desc"), P("limit", "≤ 500", { type: "number" }), P("offset", "", { type: "number" })], ex: [
43 + { l: "8-K de NVIDIA", v: { ticker: "NVDA", form: "8-K", limit: 20 } },
44 + { l: "IA en 2025, confiance ≥ 0,9", v: { type: "ai_initiative", date_from: "2025-01-01", min_confidence: 0.9, limit: 20 } },
45 + { l: "Gros montants 10-K", v: { form: "10-K", min_amount: "1e10", sort: "amount", limit: 20 } },
46 + ] },
47 + { g: "Mentions & sections", m: "GET", p: "/v1/mentions/{mention_id}", d: "Une mention complète et, si elle vient d'un 10-K, la section source (section_text_url).", params: [P("mention_id", "identifiant SHA-1 (20 car.)", { path: true })], ex: [{ l: "Exemple", v: { mention_id: "MENTION_10K" } }] },
48 + { g: "Mentions & sections", m: "GET", p: "/v1/sections/{section_id}", d: "Texte intégral d'une section 10-K ingérée par SPID (Items 1, 1A, 2, 7, 7A). Les sections 10-Q/8-K ne sont pas stockées dans la base.", params: [P("section_id", "UUID de section", { path: true }), P("highlight", "mot à repérer : renvoie les positions")], ex: [{ l: "Exemple", v: { section_id: "SECTION_10K" } }] },
49 + { g: "Entreprises", m: "GET", p: "/v1/companies", d: "Les 500 entreprises avec taille de portefeuille (projets, mentions, types, capital, période). Tri : projects, amount, mentions, ticker.", params: [P("q", "nom ou ticker"), P("sector", "", { enum: "sectors" }), P("sort", ""), P("order", ""), P("limit", "≤ 500", { type: "number" }), P("offset", "", { type: "number" })], ex: [{ l: "Top 20", v: { sort: "projects", limit: 20 } }, { l: "Santé par capital", v: { sector: "Health Care", sort: "amount", limit: 20 } }] },
50 + { g: "Entreprises", m: "GET", p: "/v1/companies/{ident}", d: "Profil d'une entreprise (ticker ou CIK) : répartitions par type, statut, année, formulaire ; lieux, technologies, partenaires ; portefeuille complet (≤ 500 projets).", params: [P("ident", "ticker ou CIK", { path: true })], ex: [{ l: "Tesla", v: { ident: "TSLA" } }, { l: "Duke Energy", v: { ident: "DUK" } }, { l: "Microsoft (CIK)", v: { ident: "789019" } }] },
51 + { g: "Graphe", m: "GET", p: "/v1/graph/search", d: "Chercher un nœud du graphe de connaissances par libellé, avec son degré.", params: [P("q", "texte"), P("type", "Company | Project | Location | Technology | Partner"), P("limit", "≤ 200", { type: "number" })], ex: [{ l: "Texas", v: { q: "texas" } }, { l: "Technologies IA", v: { q: "ai", type: "Technology" } }] },
52 + { g: "Graphe", m: "GET", p: "/v1/graph/node/{node_id}", d: "Un nœud et ses arêtes (owns, located_in, uses) avec les nœuds voisins. Identifiants : C:<cik>, P:<project_id>, L:<lieu en minuscules>, T:<technologie>, PR:<partenaire>.", params: [P("node_id", "ex. T:AI, L:texas, C:0001318605", { path: true }), P("limit", "≤ 2000", { type: "number" })], ex: [{ l: "T:AI", v: { node_id: "T:AI", limit: 50 } }, { l: "L:texas", v: { node_id: "L:texas", limit: 50 } }, { l: "C:Tesla", v: { node_id: "C:0001318605", limit: 50 } }] },
53 + { g: "Recherche", m: "GET", p: "/v1/search/semantic", d: "Recherche en langage naturel : la requête est encodée avec all-MiniLM-L6-v2 côté serveur et comparée aux 19 227 vecteurs de projets. Anglais recommandé (langue des filings).", params: [P("q", "description en langage naturel"), P("k", "≤ 100", { type: "number" }), P("type", "filtre project_type", { enum: "types" }), P("sector", "filtre secteur", { enum: "sectors" })], ex: [{ l: "Usine de cellules de batteries", v: { q: "battery cell factory", k: 10 } }, { l: "IA générative pour clients", v: { q: "generative AI assistant for customers", k: 10 } }, { l: "Terminal GNL", v: { q: "LNG export terminal", k: 10 } }] },
54 + { g: "SQL", m: "GET", p: "/v1/schema", d: "Tables, colonnes et types de la base DuckDB.", params: [], ex: [{ l: "Schéma", v: {} }] },
55 + { g: "SQL", m: "POST", p: "/v1/sql", d: "Bac à sable SQL DuckDB en lecture seule (SELECT, WITH, DESCRIBE, SUMMARIZE). Corps JSON {sql, limit}. 20 s et 5 000 lignes maximum. Réponse {columns, rows, truncated, elapsed_ms}.", params: [P("sql", "requête SQL", { body: true, textarea: true }), P("limit", "≤ 5000", { body: true, type: "number" })], ex: [
56 + { l: "Projets par type", v: { sql: "select project_type, count(*) n, round(sum(total_amount_usd)/1e9,1) capital_gusd\nfrom projects group by 1 order by n desc", limit: 25 } },
57 + { l: "Mentions IA par année", v: { sql: "select year(filing_date) as yr, count(*) n\nfrom project_mentions where project_type='ai_initiative' group by 1 order by 1", limit: 30 } },
58 + { l: "Panel entreprise × année", v: { sql: "select cik, any_value(ticker) ticker, year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital\nfrom projects group by 1,3 order by 1,3", limit: 5000 } },
59 + { l: "Technologies co-citées", v: { sql: "select a.t t1, b.t t2, count(*) n\nfrom (select project_id, unnest(technologies) t from projects) a\njoin (select project_id, unnest(technologies) t from projects) b on a.project_id=b.project_id and a.t<b.t\ngroup by 1,2 order by n desc", limit: 50 } },
60 + ] },
61 + ];
62 + const GROUPS = [...new Set(ENDPOINTS.map((e) => e.g))];
63 +
64 + /* ------------------------------------------------------------------ générateurs de code */
65 + function buildRequest(e, vals) {
66 + let path = e.p; const query = {}; let body = null;
67 + for (const prm of e.params) {
68 + const v = vals[prm.name]; if (v === undefined || v === "") continue;
69 + if (prm.path) path = path.replace(`{${prm.name}}`, encodeURIComponent(v));
70 + else if (prm.body) (body ||= {})[prm.name] = prm.type === "number" ? Number(v) : v;
71 + else query[prm.name] = v;
72 + }
73 + path = path.replace(/\{[^}]+\}/g, "");
74 + return { method: e.m, path, query, body, url: location.origin + path + (Object.keys(query).length ? "?" + qs(query) : "") };
75 + }
76 + const pyDict = (o) => "{" + Object.entries(o).map(([k, v]) => `${JSON.stringify(k)}: ${typeof v === "number" ? v : JSON.stringify(v)}`).join(", ") + "}";
77 + const rList = (o) => "list(" + Object.entries(o).map(([k, v]) => `${/^[a-z_]+$/i.test(k) ? k : "`" + k + "`"} = ${typeof v === "number" ? v : JSON.stringify(v)}`).join(", ") + ")";
78 + function snippet(l, r, k = KEY_PH, nokey = false) {
79 + const base = location.origin + r.path;
80 + if (nokey) return { curl: `curl '${r.url}'`, py: `import requests\n\nr = requests.get("${r.url}")\nr.raise_for_status()\nprint(r.json())`, js: `const r = await fetch("${r.url}");\nconsole.log(await r.json());`, r: `library(httr2)\nrequest("${r.url}") |> req_perform() |> resp_body_json()` }[l];
81 + if (l === "curl") return r.method === "POST"
82 + ? `curl -X POST '${base}' \\\n -H 'X-API-Key: ${k}' -H 'Content-Type: application/json' \\\n -d '${JSON.stringify(r.body || {})}'`
83 + : `curl '${r.url}' -H 'X-API-Key: ${k}'`;
84 + if (l === "py") return r.method === "POST"
85 + ? `import requests\n\nr = requests.post("${base}",\n headers={"X-API-Key": "${k}"},\n json=${pyDict(r.body || {})})\nr.raise_for_status()\ndata = r.json()\nprint(data)`
86 + : `import requests\n\nr = requests.get("${base}",\n headers={"X-API-Key": "${k}"},${Object.keys(r.query).length ? `\n params=${pyDict(r.query)},` : ""}\n)\nr.raise_for_status()\ndata = r.json()\nprint(data)`;
87 + if (l === "js") return `const r = await fetch("${r.url}", {\n method: "${r.method}",\n headers: { "X-API-Key": "${k}"${r.body ? ', "Content-Type": "application/json"' : ""} },${r.body ? `\n body: JSON.stringify(${JSON.stringify(r.body)}),` : ""}\n});\nif (!r.ok) throw new Error(\`HTTP \${r.status}\`);\nconst data = await r.json();\nconsole.log(data);`;
88 + if (l === "r") return `library(httr2)\n\nresp <- request("${base}") |>\n req_headers(\`X-API-Key\` = "${k}") |>${Object.keys(r.query).length ? `\n req_url_query(!!!${rList(r.query)}) |>` : ""}${r.body ? `\n req_body_json(${rList(r.body)}) |>` : ""}\n req_perform()\ndata <- resp_body_json(resp)\nstr(data, max.level = 1)`;
89 + }
90 +
91 + /* ------------------------------------------------------------------ recettes */
92 + const RECIPES = [
93 + { t: "Premier appel : compter les projets", d: "Vérifier la clé et lire les compteurs globaux.", ep: "/v1/stats", v: {}, code: {
94 + curl: `curl https://www.pdb-api.co/v1/stats -H 'X-API-Key: ${KEY_PH}' | python3 -m json.tool | head -30`,
95 + py: `import requests\nBASE, H = "https://www.pdb-api.co/v1", {"X-API-Key": "${KEY_PH}"}\n\nov = requests.get(f"{BASE}/stats", headers=H).json()["overview"]\nprint(f"{ov['projects']:,} projets · {ov['mentions']:,} mentions · {ov['companies']} entreprises")`,
96 + js: `const BASE = "https://www.pdb-api.co/v1", H = { "X-API-Key": "${KEY_PH}" };\nconst { overview } = await (await fetch(\`\${BASE}/stats\`, { headers: H })).json();\nconsole.log(overview.projects, "projets", overview.mentions, "mentions");`,
97 + r: `library(httr2)\nBASE <- "https://www.pdb-api.co/v1"; KEY <- "${KEY_PH}"\nov <- request(paste0(BASE, "/stats")) |> req_headers(\`X-API-Key\` = KEY) |> req_perform() |> resp_body_json()\nov$overview$projects` } },
98 + { t: "Filtrer et trier des projets", d: "Centres de données de plus de 500 M$, du plus gros au plus petit.", ep: "/v1/projects", v: { type: "data_center", min_amount: "5e8", sort: "amount", limit: 10 }, code: {
99 + curl: `curl 'https://www.pdb-api.co/v1/projects?type=data_center&min_amount=5e8&sort=amount&limit=10' -H 'X-API-Key: ${KEY_PH}'`,
100 + py: `r = requests.get(f"{BASE}/projects", headers=H, params={\n "type": "data_center", "min_amount": 5e8, "sort": "amount", "limit": 10})\nfor p in r.json()["items"]:\n print(f"{p['ticker']:6s} {p['project_name'][:45]:45s} {p['total_amount_usd']/1e9:6.1f} G$ {p['canonical_location']}")`,
101 + js: `const q = new URLSearchParams({ type: "data_center", min_amount: 5e8, sort: "amount", limit: 10 });\nconst { total, items } = await (await fetch(\`\${BASE}/projects?\${q}\`, { headers: H })).json();\nitems.forEach(p => console.log(p.ticker, p.project_name, p.total_amount_usd / 1e9, "G$"));`,
102 + r: `res <- request(paste0(BASE, "/projects")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(type = "data_center", min_amount = 5e8, sort = "amount", limit = 10) |>\n req_perform() |> resp_body_json()\ndo.call(rbind, lapply(res$items, \\(p) data.frame(ticker = p$ticker, projet = p$project_name, gusd = p$total_amount_usd / 1e9)))` } },
103 + { t: "Tout récupérer avec la pagination → DataFrame", d: "Boucler sur limit/offset (≤ 500 par page) pour construire un jeu de données complet.", ep: "/v1/projects", v: { sector: "Utilities", limit: 500 }, code: {
104 + curl: `# page 1, 2, 3… : incrémenter offset de 500 jusqu'à atteindre total\nfor off in 0 500 1000 1500; do\n curl -s "https://www.pdb-api.co/v1/projects?sector=Utilities&limit=500&offset=$off" -H 'X-API-Key: ${KEY_PH}' > utilities_$off.json\ndone`,
105 + py: `import pandas as pd\n\ndef fetch_all(path, **filters):\n offset, rows = 0, []\n while True:\n page = requests.get(f"{BASE}/{path}", headers=H, params={**filters, "limit": 500, "offset": offset}).json()\n rows += page["items"]\n offset += len(page["items"])\n if not page["items"] or offset >= page["total"]:\n return rows\n\ndf = pd.DataFrame(fetch_all("projects", sector="Utilities"))\ndf["technologies"] = df["technologies"].str.join(" | ")\nprint(df.shape)\ndf.groupby("project_type")["total_amount_usd"].agg(["count", "median"]).sort_values("count", ascending=False)`,
106 + js: `async function fetchAll(path, filters) {\n const rows = []; let offset = 0;\n for (;;) {\n const q = new URLSearchParams({ ...filters, limit: 500, offset });\n const page = await (await fetch(\`\${BASE}/\${path}?\${q}\`, { headers: H })).json();\n rows.push(...page.items); offset += page.items.length;\n if (!page.items.length || offset >= page.total) return rows;\n }\n}\nconst utilities = await fetchAll("projects", { sector: "Utilities" });\nconsole.log(utilities.length, "projets");`,
107 + r: `fetch_all <- function(path, ...) {\n out <- list(); offset <- 0\n repeat {\n page <- request(paste0(BASE, "/", path)) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(..., limit = 500, offset = offset) |> req_perform() |> resp_body_json()\n out <- c(out, page$items); offset <- offset + length(page$items)\n if (length(page$items) == 0 || offset >= page$total) break\n }\n out\n}\nprojets <- fetch_all("projects", sector = "Utilities")\ndf <- dplyr::bind_rows(lapply(projets, \\(p) as.data.frame(p[c("ticker","project_type","project_name","status","total_amount_usd","first_seen")])))` } },
108 + { t: "Export CSV direct", d: "Le plus simple pour Excel, R ou pandas : un seul appel, jusqu'à 50 000 lignes.", ep: "/v1/projects/export.csv", v: { type: "data_center" }, code: {
109 + curl: `curl 'https://www.pdb-api.co/v1/projects/export.csv?type=data_center' -H 'X-API-Key: ${KEY_PH}' -o data_centers.csv`,
110 + py: `import io, pandas as pd\ncsv_text = requests.get(f"{BASE}/projects/export.csv", headers=H, params={"type": "data_center"}).text\ndf = pd.read_csv(io.StringIO(csv_text))\ndf.head()`,
111 + js: `const csv = await (await fetch(\`\${BASE}/projects/export.csv?type=data_center\`, { headers: H })).text();\nrequire("fs").writeFileSync("data_centers.csv", csv);`,
112 + r: `csv <- request(paste0(BASE, "/projects/export.csv")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(type = "data_center") |> req_perform() |> resp_body_string()\ndf <- read.csv(text = csv)` } },
113 + { t: "Chronologie d'un projet", d: "Suivre un projet de son annonce (8-K) à ses mentions ultérieures (10-K, 10-Q).", ep: "/v1/projects/{project_id}", v: { project_id: "46fe2966ba0cbf6228807faca0645b54" }, code: {
114 + curl: `PID=$(curl -s 'https://www.pdb-api.co/v1/projects?ticker=TSLA&q=energy%20storage&limit=1' -H 'X-API-Key: ${KEY_PH}' | python3 -c 'import json,sys; print(json.load(sys.stdin)["items"][0]["project_id"])')\ncurl "https://www.pdb-api.co/v1/projects/$PID" -H 'X-API-Key: ${KEY_PH}'`,
115 + py: `pid = requests.get(f"{BASE}/projects", headers=H, params={"ticker": "TSLA", "q": "energy storage", "limit": 1}).json()["items"][0]["project_id"]\nfiche = requests.get(f"{BASE}/projects/{pid}", headers=H).json()\nprint(fiche["project"]["project_name"], fiche["project"]["status"])\nfor t in fiche["timeline"]:\n print(t["filing_date"], t["form_type"], t["status"], t["amount_usd"], "—", t["snippet"][:80])\nprint("similaires :", [s["project_name"] for s in fiche["similar"][:5]])`,
116 + js: `const first = (await (await fetch(\`\${BASE}/projects?ticker=TSLA&q=energy%20storage&limit=1\`, { headers: H })).json()).items[0];\nconst fiche = await (await fetch(\`\${BASE}/projects/\${first.project_id}\`, { headers: H })).json();\nfiche.timeline.forEach(t => console.log(t.filing_date, t.form_type, t.status, t.amount_usd));`,
117 + r: `pid <- (request(paste0(BASE, "/projects")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(ticker = "TSLA", q = "energy storage", limit = 1) |> req_perform() |> resp_body_json())$items[[1]]$project_id\nfiche <- request(paste0(BASE, "/projects/", pid)) |> req_headers(\`X-API-Key\` = KEY) |> req_perform() |> resp_body_json()\nsapply(fiche$timeline, \\(t) paste(t$filing_date, t$form_type, t$status))` } },
118 + { t: "Profil d'une entreprise", d: "Répartitions par type, année et formulaire, plus le portefeuille complet.", ep: "/v1/companies/{ident}", v: { ident: "DUK" }, code: {
119 + curl: `curl https://www.pdb-api.co/v1/companies/DUK -H 'X-API-Key: ${KEY_PH}'`,
120 + py: `duk = requests.get(f"{BASE}/companies/DUK", headers=H).json()\nprint(duk["company"])\npd.DataFrame(duk["by_type"]).set_index("label")["n"].plot.barh(title="Duke Energy — projets par type")`,
121 + js: `const duk = await (await fetch(\`\${BASE}/companies/DUK\`, { headers: H })).json();\nconsole.table(duk.by_type.map(x => ({ type: x.label, n: x.n })));`,
122 + r: `duk <- request(paste0(BASE, "/companies/DUK")) |> req_headers(\`X-API-Key\` = KEY) |> req_perform() |> resp_body_json()\nbarplot(sapply(duk$by_type, \\(x) x$n), names.arg = sapply(duk$by_type, \\(x) x$project_type), las = 2)` } },
123 + { t: "Recherche sémantique", d: "Décrire un projet en langage naturel et obtenir les projets les plus proches.", ep: "/v1/search/semantic", v: { q: "battery cell factory", k: 10 }, code: {
124 + curl: `curl 'https://www.pdb-api.co/v1/search/semantic?q=battery+cell+factory&k=10' -H 'X-API-Key: ${KEY_PH}'`,
125 + py: `hits = requests.get(f"{BASE}/search/semantic", headers=H, params={"q": "battery cell factory", "k": 10}).json()["items"]\nfor h in hits:\n print(f"{h['score']:.3f} {h['ticker']:6s} {h['project_name']}")`,
126 + js: `const { items } = await (await fetch(\`\${BASE}/search/semantic?\${new URLSearchParams({ q: "battery cell factory", k: 10 })}\`, { headers: H })).json();\nitems.forEach(h => console.log(h.score.toFixed(3), h.ticker, h.project_name));`,
127 + r: `hits <- request(paste0(BASE, "/search/semantic")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_url_query(q = "battery cell factory", k = 10) |> req_perform() |> resp_body_json()\nsapply(hits$items, \\(h) sprintf("%.3f %s %s", h$score, h$ticker, h$project_name))` } },
128 + { t: "SQL : agrégations libres", d: "Quand les filtres ne suffisent pas : une requête DuckDB de lecture, résultat en colonnes/lignes.", ep: "/v1/sql", v: { sql: "select year(filing_date) as yr, project_type, count(*) n\nfrom project_mentions\nwhere project_type in ('ai_initiative','data_center','cloud_migration')\ngroup by 1,2 order by 1,2", limit: 200 }, code: {
129 + curl: `curl -X POST https://www.pdb-api.co/v1/sql -H 'X-API-Key: ${KEY_PH}' -H 'Content-Type: application/json' \\\n -d '{"sql":"select year(filing_date) as yr, project_type, count(*) n from project_mentions where project_type in (\\'ai_initiative\\',\\'data_center\\') group by 1,2 order by 1,2","limit":200}'`,
130 + py: `sql = """\nselect year(filing_date) as yr, project_type, count(*) n\nfrom project_mentions\nwhere project_type in ('ai_initiative','data_center','cloud_migration')\ngroup by 1,2 order by 1,2\n"""\nres = requests.post(f"{BASE}/sql", headers=H, json={"sql": sql, "limit": 500}).json()\ndf = pd.DataFrame(res["rows"], columns=res["columns"])\ndf.pivot(index="yr", columns="project_type", values="n").plot(title="Mentions par année")`,
131 + js: `const res = await (await fetch(\`\${BASE}/sql\`, { method: "POST", headers: { ...H, "Content-Type": "application/json" },\n body: JSON.stringify({ sql: "select project_type, count(*) n from projects group by 1 order by n desc", limit: 25 }) })).json();\nconsole.table(res.rows.map(r => Object.fromEntries(res.columns.map((c, i) => [c, r[i]]))));`,
132 + r: `res <- request(paste0(BASE, "/sql")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_body_json(list(sql = "select project_type, count(*) n from projects group by 1 order by n desc", limit = 25)) |>\n req_perform() |> resp_body_json()\ndf <- as.data.frame(do.call(rbind, lapply(res$rows, unlist))); names(df) <- unlist(res$columns); df` } },
133 + { t: "Graphe : qui construit au Texas avec des batteries ?", d: "Partir d'un nœud Lieu, remonter aux projets, croiser avec une technologie.", ep: "/v1/graph/node/{node_id}", v: { node_id: "L:texas", limit: 100 }, code: {
134 + curl: `curl 'https://www.pdb-api.co/v1/graph/node/L:texas?limit=100' -H 'X-API-Key: ${KEY_PH}'`,
135 + py: `node = requests.get(f"{BASE}/graph/node/L:texas", headers=H, params={"limit": 500}).json()\nprojets_texas = {e["neighbor_id"][2:] for e in node["edges"] if e["neighbor_type"] == "Project"}\nbatt = requests.get(f"{BASE}/projects", headers=H, params={"location": "texas", "tech": "batter", "limit": 100}).json()["items"]\nprint(len(projets_texas), "projets au Texas ;", len(batt), "avec batteries :", [(p["ticker"], p["project_name"]) for p in batt][:5])`,
136 + js: `const node = await (await fetch(\`\${BASE}/graph/node/L:texas?limit=500\`, { headers: H })).json();\nconsole.log(node.degree, "projets rattachés à texas");`,
137 + r: `node <- request(paste0(BASE, "/graph/node/L:texas")) |> req_headers(\`X-API-Key\` = KEY) |> req_url_query(limit = 500) |> req_perform() |> resp_body_json()\nnode$degree` } },
138 + { t: "Gérer les erreurs et la limite de débit", d: "401 clé invalide, 404 introuvable, 408 SQL trop long, 429 trop de requêtes (240/min), 501 sémantique indisponible.", ep: "/v1/health", v: {}, code: {
139 + curl: `curl -s -o /dev/null -w '%{http_code}\\n' https://www.pdb-api.co/v1/stats -H 'X-API-Key: mauvaise' # 401`,
140 + py: `import time\n\ndef get(path, **params):\n for attempt in range(4):\n r = requests.get(f"{BASE}/{path}", headers=H, params=params, timeout=60)\n if r.status_code == 429: # limite de débit : attendre puis réessayer\n time.sleep(2 * (attempt + 1)); continue\n if r.status_code >= 400:\n raise RuntimeError(f"HTTP {r.status_code}: {r.json().get('detail')}")\n return r.json()\n raise RuntimeError("trop de tentatives")`,
141 + js: `async function get(path, params = {}) {\n for (let i = 0; i < 4; i++) {\n const r = await fetch(\`\${BASE}/\${path}?\${new URLSearchParams(params)}\`, { headers: H });\n if (r.status === 429) { await new Promise(s => setTimeout(s, 2000 * (i + 1))); continue; }\n if (!r.ok) throw new Error(\`HTTP \${r.status}: \${(await r.json()).detail}\`);\n return r.json();\n }\n}`,
142 + r: `resp <- request(paste0(BASE, "/stats")) |> req_headers(\`X-API-Key\` = KEY) |>\n req_retry(max_tries = 4, is_transient = \\(r) resp_status(r) == 429) |> req_perform()` } },
143 + ];
144 +
145 + /* ------------------------------------------------------------------ référence : dictionnaire des champs */
146 + const FIELDS = {
147 + projects: [["project_id", "MD5 de cik|type|ancre ; identifiant stable"], ["cik / ticker / company_name / sector", "entreprise (GICS)"], ["project_type", "un des 21 types (voir /v1/taxonomy)"], ["project_name / description", "nom et description de la mention la plus confiante"], ["canonical_location", "première localisation (minuscules) ayant servi d'ancre"], ["technologies[]", "union des technologies citées"], ["status", "statut du dépôt le plus récent : planned, in_progress, completed, mentioned"], ["total_amount_usd", "montant MAXIMAL divulgué parmi les mentions (pas une somme)"], ["n_mentions / n_filings", "nombre de mentions et de dépôts distincts"], ["first_seen / last_seen", "dates de dépôt extrêmes (fiables pour dater)"], ["first_year / last_year", "années citées dans le texte (peuvent être projetées)"], ["avg_confidence", "confiance moyenne du modèle (0,45–1)"]],
148 + mentions: [["mention_id", "SHA-1[0:20] de accession|section|type|ancre"], ["accession_number / section_id / section_name", "dépôt et section source"], ["form_type / filing_date / fiscal_year", "10-K, 10-Q ou 8-K ; date de dépôt (fiscal_year non renseigné)"], ["project_name / project_type / description / objective", "extraction du LLM"], ["amount_usd / amount_raw", "montant en US$ et chaîne d'origine"], ["locations[] / technologies[] / partners[] / suppliers[]", "listes extraites (≤ 8 éléments)"], ["benefits[] / risks[] / years[]", "bénéfices, risques, années citées"], ["status / confidence / backend", "statut, confiance 0–1, backend (llm)"], ["project_id", "projet résolu auquel la mention est rattachée (ajouté par l'API)"]],
149 + timeline: [["project_id / accession_number / filing_date / form_type", "un point par mention datée"], ["status / amount_usd / confidence / snippet", "état à cette date, montant, confiance, extrait de 300 caractères"]],
150 + };
151 + const ERRORS = [["200", "OK"], ["400", "Requête SQL refusée (mot-clé interdit, syntaxe) ou paramètre invalide"], ["401", "Clé absente ou invalide (en-tête X-API-Key)"], ["403", "Route /app réservée à l'interface web"], ["404", "Projet, mention, section, entreprise ou nœud introuvable"], ["408", "Requête SQL interrompue après 20 s"], ["422", "Paramètre mal typé (voir detail)"], ["429", "Plus de 240 requêtes par minute pour votre adresse"], ["501", "Recherche sémantique indisponible sur le serveur"], ["503", "Service ou clé non configurés"]];
152 +
153 + /* ------------------------------------------------------------------ rendu */
154 + let ctx, meta = { types: [], sectors: [] }, cur = 7, tab = "start", exampleIds = {};
155 + function render(c) {
156 + ctx = c; const view = c.view;
157 + const hp = new URLSearchParams((location.hash.split("?")[1] || ""));
158 + if (hp.get("tab")) tab = hp.get("tab");
159 + if (hp.get("ep")) { const i = ENDPOINTS.findIndex((e) => e.p === hp.get("ep")); if (i >= 0) { cur = i; tab = "play"; } }
160 + view.innerHTML = `<div class="page-head"><div><h1>API &amp; playground</h1><p>API REST JSON de la base SPID. Base : <span class="mono">${esc(location.origin)}/v1</span> · authentification par en-tête <span class="mono">X-API-Key</span> · 240 requêtes/min · pagination <span class="mono">limit/offset</span> (≤ 500).</p></div>
161 + <div class="langsel">Langage des exemples : ${LANGS.map(([k, l]) => `<button class="${lang() === k ? "active" : ""}" data-lang="${k}">${l}</button>`).join("")}</div></div>
162 + <div class="tabs big" id="api-tabs">${[["start", "Démarrage rapide"], ["play", "Playground"], ["recipes", "Recettes"], ["ref", "Référence"], ["sdk", "SDK Python"]].map(([k, l]) => `<button class="${tab === k ? "active" : ""}" data-tab="${k}">${l}</button>`).join("")}</div>
163 + <div id="api-body"></div>`;
164 + view.querySelectorAll("[data-lang]").forEach((b) => b.addEventListener("click", () => { localStorage.setItem("pdb_lang", b.dataset.lang); render(ctx); }));
165 + view.querySelector("#api-tabs").addEventListener("click", (e) => { if (e.target.dataset.tab) { tab = e.target.dataset.tab; render(ctx); } });
166 + Promise.all([ctx.api("/taxonomy"), ctx.api("/sectors")]).then(([t, s]) => { meta = { types: t, sectors: s }; if (tab === "play") renderPlay(); }).catch(() => { });
167 + ({ start: renderStart, play: renderPlay, recipes: renderRecipes, ref: renderRef, sdk: renderSdk }[tab] || renderStart)();
168 + resolveExampleIds();
169 + }
170 + async function resolveExampleIds() {
171 + if (exampleIds.TSLA_ENERGY) return;
172 + try {
173 + const p = await ctx.api("/projects", { ticker: "TSLA", q: "energy storage", limit: 1 }); exampleIds.TSLA_ENERGY = p.items[0]?.project_id;
174 + const m = await ctx.api("/mentions", { form: "10-K", min_confidence: 0.95, limit: 1 }); exampleIds.MENTION_10K = m.items[0]?.mention_id; exampleIds.SECTION_10K = m.items[0]?.section_id;
175 + } catch { }
176 + }
177 + const sub = (v) => exampleIds[v] || v;
178 +
179 + const keyBox = () => `<div class="keybox"><label><b>Votre clé d'API</b><input class="inp" id="key" type="password" value="${esc(key())}" placeholder="saisir la clé (reste dans ce navigateur)" autocomplete="off"></label>
180 + <button class="btn btn-ghost btn-sm" id="key-show">Afficher</button><button class="btn btn-sm" id="key-test">Tester la clé</button><span id="key-st" class="small muted"></span>
181 + <div class="small muted" style="flex-basis:100%">La clé n'est pas publiée sur ce site. Elle est fournie par l'équipe UQO (<a href="mailto:simon-pierre.boucher@uqo.ca">simon-pierre.boucher@uqo.ca</a>). Elle est conservée en localStorage et envoyée uniquement à cette API.</div></div>`;
182 + function bindKey(root) {
183 + const inp = root.querySelector("#key"); if (!inp) return;
184 + inp.addEventListener("input", () => { localStorage.setItem("pdb_key", inp.value); root.querySelectorAll("pre.code[data-req]").forEach(() => { }); });
185 + root.querySelector("#key-show").addEventListener("click", () => (inp.type = inp.type === "password" ? "text" : "password"));
186 + root.querySelector("#key-test").addEventListener("click", async () => {
187 + const st = root.querySelector("#key-st"); st.textContent = "…";
188 + const r = await fetch("/v1/taxonomy", { headers: { "X-API-Key": key() } });
189 + st.innerHTML = r.ok ? `<span class="ok" style="padding:3px 8px">✓ clé valide (HTTP 200)</span>` : `<span class="err" style="padding:3px 8px;display:inline-block;margin:0">✗ HTTP ${r.status} — ${esc((await r.json()).detail || "")}</span>`;
190 + });
191 + }
192 + const codeBlock = (code, id = "") => `<div class="codewrap"><button class="copy" data-copy>Copier</button><pre class="code" ${id ? `id="${id}"` : ""}>${esc(code)}</pre></div>`;
193 + function bindCopy(root) { root.querySelectorAll("[data-copy]").forEach((b) => b.addEventListener("click", () => { navigator.clipboard.writeText(b.nextElementSibling.textContent); b.textContent = "Copié ✓"; setTimeout(() => (b.textContent = "Copier"), 1500); })); }
194 +
195 + /* ---- Démarrage rapide */
196 + function renderStart() {
197 + const l = lang(); const body = ctx.view.querySelector("#api-body");
198 + const r1 = buildRequest(ENDPOINTS[0], {}); const r2 = buildRequest(ENDPOINTS[1], {}); const r3 = buildRequest(ENDPOINTS.find((e) => e.p === "/v1/projects"), { type: "ai_initiative", year_from: 2024, sort: "last_seen", limit: 5 });
199 + const install = { curl: "# curl est déjà installé sur macOS / Linux / Windows 10+", py: "pip install requests pandas # ou : téléchargez pdb_client.py (onglet SDK Python)", js: "# Node ≥ 18 (fetch natif) ou navigateur — aucune dépendance", r: 'install.packages(c("httr2", "dplyr"))' }[l];
200 + body.innerHTML = `<div class="grid g2">
201 + <div class="card" style="grid-column:1/-1">${keyBox()}</div>
202 + <div class="card"><h3><span class="step">1</span> Installer</h3>${codeBlock(install)}</div>
203 + <div class="card"><h3><span class="step">2</span> Vérifier le service (sans clé)</h3>${codeBlock(snippet(l, r1, KEY_PH, true))}</div>
204 + <div class="card"><h3><span class="step">3</span> Premier appel authentifié</h3>${codeBlock(snippet(l, r2, key() || KEY_PH))}<p class="small muted">Réponse : <span class="mono">{overview:{projects, mentions, companies…}, by_type:[…], by_sector:[…], mentions_by_year:[…]}</span></p></div>
205 + <div class="card"><h3><span class="step">4</span> Filtrer des projets</h3>${codeBlock(snippet(l, r3, key() || KEY_PH))}<p class="small muted">Réponse paginée : <span class="mono">{total, limit, offset, items:[{project_id, ticker, project_name, project_type, status, total_amount_usd, …}]}</span>. Pour tout récupérer, incrémentez <span class="mono">offset</span> de <span class="mono">limit</span> jusqu'à <span class="mono">total</span> (recette « pagination »).</p></div>
206 + <div class="card" style="grid-column:1/-1"><h3>Modèle de données en 30 secondes</h3><div class="grid g3">
207 + <div><b>Projet</b> (19 227) — unité principale. Un projet = mentions fusionnées par entreprise × type × ancre (lieu ou premier mot du nom). <span class="mono">/v1/projects</span>, <span class="mono">/v1/projects/{id}</span>.</div>
208 + <div><b>Mention</b> (39 930) — l'extraction brute : un projet cité dans une section d'un filing, avec date, formulaire, montant, confiance. <span class="mono">/v1/mentions</span>. La chronologie d'un projet = ses mentions triées par date.</div>
209 + <div><b>Graphe</b> (32 226 nœuds) et <b>embeddings</b> (384-d) — navigation par lieu/technologie/partenaire et recherche par le sens. <span class="mono">/v1/graph/*</span>, <span class="mono">/v1/search/semantic</span>, <span class="mono">/v1/projects/{id}/similar</span>.</div></div>
210 + <p class="small muted" style="margin-top:8px">Point d'attention : <span class="mono">total_amount_usd</span> est le montant <i>maximal</i> divulgué pour le projet, auto-déclaré et hétérogène ; <span class="mono">first_seen/last_seen</span> datent l'observation, <span class="mono">first_year/last_year</span> sont les années citées dans le texte (parfois projetées).</p></div>
211 + <div class="card" style="grid-column:1/-1"><h3>Codes de réponse</h3><table class="ref"><tbody>${ERRORS.map(([c, d]) => `<tr><td class="mono"><b>${c}</b></td><td>${esc(d)}</td></tr>`).join("")}</tbody></table></div>
212 + </div>`;
213 + bindKey(body); bindCopy(body);
214 + }
215 +
216 + /* ---- Playground */
217 + const HIST_KEY = "pdb_hist";
218 + function renderPlay() {
219 + const body = ctx.view.querySelector("#api-body"); const e = ENDPOINTS[cur];
220 + body.innerHTML = `<div class="api-layout"><div class="card"><h3>Points d'accès</h3><ul class="endpoints">${GROUPS.map((g) => `<li class="grp">${esc(g)}</li>` + ENDPOINTS.map((x, i) => x.g === g ? `<li data-i="${i}" class="${i === cur ? "active" : ""}"><span class="m ${x.m === "POST" ? "post" : ""}">${x.m}</span><span class="mono">${esc(x.p.replace("/v1", ""))}</span></li>` : "").join("")).join("")}</ul>
221 + <h3 style="margin-top:14px">Historique</h3><div id="hist" class="hist"></div></div>
222 + <div><div class="card">${keyBox()}</div><div class="card" id="pg"></div></div></div>`;
223 + body.querySelectorAll(".endpoints li[data-i]").forEach((li) => li.addEventListener("click", () => { cur = +li.dataset.i; renderPlay(); }));
224 + bindKey(body); renderHist(body);
225 + const pg = body.querySelector("#pg");
226 + const ex0 = e.ex?.[0]?.v || {};
227 + pg.innerHTML = `<h3><span class="m ${e.m === "POST" ? "post" : ""}" style="font-family:var(--mono);color:${e.m === "POST" ? "var(--orange)" : "var(--vert)"}">${e.m}</span> <span class="mono">${esc(e.p)}</span>${e.key === false ? ' <span class="pill pill-vert">sans clé</span>' : ""}</h3><p class="muted">${esc(e.d)}</p>
228 + ${e.ex?.length ? `<div class="examples"><span class="small muted" style="align-self:center">Exemples :</span>${e.ex.map((x, i) => `<button data-ex="${i}">${esc(x.l)}</button>`).join("")}</div>` : ""}
229 + <div class="params">${e.params.map((p) => p.textarea ? `<label style="grid-column:1/-1"><b>${esc(p.name)}</b> <span>${esc(p.desc)}</span><textarea class="sql" data-p="${esc(p.name)}" rows="5">${esc(sub(ex0[p.name] ?? ""))}</textarea></label>` :
230 + `<label><b>${esc(p.name)}${p.path ? " *" : ""}</b><span>${esc(p.desc)}</span>${p.enum ? `<input class="inp" list="dl-${p.enum}" data-p="${esc(p.name)}" value="${esc(sub(ex0[p.name] ?? ""))}">` : `<input class="inp" type="${p.type === "number" ? "text" : "text"}" data-p="${esc(p.name)}" value="${esc(sub(ex0[p.name] ?? ""))}" placeholder="${esc(p.desc.slice(0, 40))}">`}</label>`).join("") || `<span class="muted small">Aucun paramètre.</span>`}</div>
231 + <datalist id="dl-types">${meta.types.map((t) => `<option value="${t.project_type}">${esc(t.label)}</option>`).join("")}</datalist><datalist id="dl-sectors">${meta.sectors.map((s) => `<option value="${esc(s.sector)}">`).join("")}</datalist>
232 + <div style="display:flex;gap:10px;align-items:center;flex-wrap:wrap"><button class="btn" id="send">▶ Envoyer</button><span class="muted small" id="st"></span><span class="sp" style="flex:1"></span><a id="explore" class="btn btn-ghost btn-sm" hidden>Ouvrir dans l'explorateur</a></div>
233 + <div class="tabs" style="margin-top:14px" id="reqtabs">${LANGS.map(([k, l]) => `<button class="${lang() === k ? "active" : ""}" data-lang="${k}">${l}</button>`).join("")}</div>
234 + ${codeBlock("", "req")}
235 + <div style="display:flex;gap:10px;align-items:center;margin-top:14px"><h3 style="margin:0">Réponse</h3><div class="tabs" style="margin:0;border:0" id="resptabs"><button class="active" data-v="json">JSON</button><button data-v="table">Tableau</button></div><span class="sp" style="flex:1"></span><button class="btn btn-ghost btn-sm" id="dl" hidden>Télécharger</button></div>
236 + <div id="resp"><pre class="code">—</pre></div>`;
237 + const vals = () => { const o = {}; pg.querySelectorAll("[data-p]").forEach((i) => { if (i.value !== "") o[i.dataset.p] = i.value; }); return o; };
238 + const showReq = () => { const r = buildRequest(e, vals()); pg.querySelector("#req").textContent = snippet(lang(), r, key() || KEY_PH); const ex = exploreLink(e, vals(), r); const a = pg.querySelector("#explore"); a.hidden = !ex; if (ex) a.href = ex; };
239 + pg.querySelectorAll("[data-p]").forEach((i) => i.addEventListener("input", showReq)); showReq();
240 + pg.querySelectorAll("[data-ex]").forEach((b) => b.addEventListener("click", () => { const v = e.ex[b.dataset.ex].v; pg.querySelectorAll("[data-p]").forEach((i) => (i.value = sub(v[i.dataset.p] ?? ""))); showReq(); send(); }));
241 + pg.querySelector("#reqtabs").addEventListener("click", (ev) => { if (ev.target.dataset.lang) { localStorage.setItem("pdb_lang", ev.target.dataset.lang); pg.querySelectorAll("#reqtabs button").forEach((b) => b.classList.toggle("active", b.dataset.lang === ev.target.dataset.lang)); ctx.view.querySelectorAll(".langsel button").forEach((b) => b.classList.toggle("active", b.dataset.lang === ev.target.dataset.lang)); showReq(); } });
242 + bindCopy(pg);
243 + let last = null, lastRaw = "";
244 + const renderResp = () => {
245 + const v = pg.querySelector("#resptabs .active").dataset.v; const box = pg.querySelector("#resp");
246 + if (!last) { box.innerHTML = `<pre class="code">${esc(lastRaw.slice(0, 60000))}</pre>`; return; }
247 + if (v === "table") { const rows = tabular(last); box.innerHTML = rows ? rows : `<div class="warn">Réponse non tabulaire — voir JSON.</div>`; }
248 + else { const txt = JSON.stringify(last, null, 2); box.innerHTML = `<div class="codewrap"><button class="copy" data-copy>Copier</button><pre class="code json">${hl(txt.length > 80000 ? txt.slice(0, 80000) + "\n… (tronqué)" : txt)}</pre></div>`; bindCopy(box); }
249 + };
250 + pg.querySelector("#resptabs").addEventListener("click", (ev) => { if (ev.target.dataset.v) { pg.querySelectorAll("#resptabs button").forEach((b) => b.classList.toggle("active", b === ev.target)); renderResp(); } });
251 + async function send() {
252 + const r = buildRequest(e, vals()); const st = pg.querySelector("#st"); st.textContent = "…"; const t0 = performance.now();
253 + try {
254 + const resp = await fetch(r.url, { method: r.method, headers: { ...(e.key === false ? {} : { "X-API-Key": key() }), ...(r.body ? { "Content-Type": "application/json" } : {}) }, body: r.body ? JSON.stringify(r.body) : undefined });
255 + lastRaw = await resp.text(); last = null; try { last = JSON.parse(lastRaw); } catch { }
256 + const ms = Math.round(performance.now() - t0);
257 + st.innerHTML = `<span class="${resp.ok ? "pill pill-vert" : "pill pill-rouge"}">HTTP ${resp.status}</span> ${ms} ms · ${(lastRaw.length / 1024).toFixed(1)} ko${last?.total != null ? ` · total ${fmtN(last.total)}` : ""}`;
258 + const dl = pg.querySelector("#dl"); dl.hidden = false; dl.onclick = () => { const a = document.createElement("a"); a.href = URL.createObjectURL(new Blob([lastRaw], { type: e.raw ? "text/csv" : "application/json" })); a.download = e.raw ? "export.csv" : "response.json"; a.click(); };
259 + pushHist({ ep: e.p, vals: vals(), status: resp.status, ms, t: Date.now() }); renderHist(body); renderResp();
260 + } catch (err) { st.textContent = ""; pg.querySelector("#resp").innerHTML = `<div class="err">${esc(String(err))}</div>`; }
261 + }
262 + pg.querySelector("#send").addEventListener("click", send);
263 + }
264 + function exploreLink(e, vals, r) {
265 + if (e.p === "/v1/projects" || e.p === "/v1/projects/export.csv") return "#/projects?" + qs(Object.fromEntries(Object.entries(r.query).filter(([k]) => !["limit", "offset"].includes(k))));
266 + if (e.p.startsWith("/v1/projects/{") && vals.project_id) return "#/project/" + encodeURIComponent(vals.project_id);
267 + if (e.p === "/v1/companies/{ident}" && vals.ident) return "#/company/" + encodeURIComponent(vals.ident);
268 + if (e.p === "/v1/mentions") return "#/mentions?" + qs(r.query);
269 + if (e.p === "/v1/mentions/{mention_id}" && vals.mention_id) return "#/mention/" + encodeURIComponent(vals.mention_id);
270 + if (e.p === "/v1/graph/node/{node_id}" && vals.node_id) return "#/graph?node=" + encodeURIComponent(vals.node_id);
271 + if (e.p === "/v1/graph/search" && vals.q) return "#/graph?q=" + encodeURIComponent(vals.q);
272 + if (e.p === "/v1/search/semantic" && vals.q) return "#/semantic?q=" + encodeURIComponent(vals.q);
273 + if (e.p === "/v1/sql") return "#/sql";
274 + return null;
275 + }
276 + function tabular(d) {
277 + let rows = Array.isArray(d) ? d : d?.items || d?.edges || d?.projects || d?.timeline || null;
278 + if (d?.columns && d?.rows) rows = d.rows.map((r) => Object.fromEntries(d.columns.map((c, i) => [c, r[i]])));
279 + if (!rows || !rows.length || typeof rows[0] !== "object") return null;
280 + const cols = Object.keys(rows[0]).filter((c) => !["description", "props", "neighbor_props", "vector"].includes(c)).slice(0, 14);
281 + return `<div class="table-wrap" style="max-height:60vh"><table><thead><tr>${cols.map((c) => `<th>${esc(c)}</th>`).join("")}</tr></thead><tbody>${rows.slice(0, 500).map((r) => `<tr>${cols.map((c) => { const v = r[c]; return `<td class="${typeof v === "number" ? "num" : ""}">${esc(Array.isArray(v) ? v.join(" | ") : v && typeof v === "object" ? JSON.stringify(v) : v)}</td>`; }).join("")}</tr>`).join("")}</tbody></table></div><div class="small muted" style="margin-top:6px">${fmtN(rows.length)} lignes${rows.length > 500 ? " (500 affichées)" : ""}</div>`;
282 + }
283 + const hl = (txt) => esc(txt).replace(/("(?:\\.|[^"\\])*")(\s*:)?/g, (m, s, c) => c ? `<span class="k">${s}</span>${c}` : `<span class="s">${s}</span>`).replace(/\b(-?\d+(?:\.\d+)?(?:e[+-]?\d+)?)\b/g, `<span class="n">$1</span>`).replace(/\b(true|false|null)\b/g, `<span class="b">$1</span>`);
284 + function pushHist(h) { const a = JSON.parse(localStorage.getItem(HIST_KEY) || "[]"); a.unshift(h); localStorage.setItem(HIST_KEY, JSON.stringify(a.slice(0, 12))); }
285 + function renderHist(root) {
286 + const a = JSON.parse(localStorage.getItem(HIST_KEY) || "[]"); const box = root.querySelector("#hist"); if (!box) return;
287 + box.innerHTML = a.length ? a.map((h, i) => `<div class="hist-it" data-h="${i}"><span class="pill ${h.status < 400 ? "pill-vert" : "pill-rouge"}">${h.status}</span> <span class="mono small">${esc(h.ep.replace("/v1", ""))}</span><div class="small muted">${esc(Object.entries(h.vals).map(([k, v]) => `${k}=${String(v).slice(0, 18)}`).join(" ")) || "—"} · ${h.ms} ms</div></div>`).join("") + `<button class="btn btn-ghost btn-sm" id="hist-clear">Effacer</button>` : `<span class="small muted">Aucune requête envoyée.</span>`;
288 + box.querySelectorAll(".hist-it").forEach((d) => d.addEventListener("click", () => { const h = a[d.dataset.h]; cur = ENDPOINTS.findIndex((e) => e.p === h.ep); renderPlay(); const pg = root.querySelector("#pg"); pg.querySelectorAll("[data-p]").forEach((i) => (i.value = h.vals[i.dataset.p] ?? "")); pg.querySelector("[data-p]")?.dispatchEvent(new Event("input")); }));
289 + box.querySelector("#hist-clear")?.addEventListener("click", () => { localStorage.removeItem(HIST_KEY); renderHist(root); });
290 + }
291 +
292 + /* ---- Recettes */
293 + function renderRecipes() {
294 + const l = lang(); const body = ctx.view.querySelector("#api-body");
295 + const pre = { py: `# Préambule commun aux recettes Python\nimport requests, pandas as pd\nBASE = "https://www.pdb-api.co/v1"\nH = {"X-API-Key": "${key() || KEY_PH}"}`, js: `// Préambule commun (Node ≥ 18 ou navigateur)\nconst BASE = "https://www.pdb-api.co/v1";\nconst H = { "X-API-Key": "${key() || KEY_PH}" };`, r: `# Préambule commun aux recettes R\nlibrary(httr2); library(dplyr)\nBASE <- "https://www.pdb-api.co/v1"; KEY <- "${key() || KEY_PH}"`, curl: `# Remplacez ${KEY_PH} par votre clé. Ajoutez « | python3 -m json.tool » pour lire le JSON.` }[l];
296 + body.innerHTML = `<div class="card">${codeBlock(pre)}</div><div class="grid g2">${RECIPES.map((r, i) => `<div class="card recipe"><h3>${i + 1}. ${esc(r.t)}</h3><p class="muted small">${esc(r.d)}</p>${codeBlock(r.code[l].replace(new RegExp(KEY_PH, "g"), key() || KEY_PH))}<div style="margin-top:8px"><button class="btn btn-ghost btn-sm" data-try="${i}">Essayer dans le playground</button></div></div>`).join("")}</div>`;
297 + bindCopy(body);
298 + body.querySelectorAll("[data-try]").forEach((b) => b.addEventListener("click", () => { const r = RECIPES[b.dataset.try]; cur = ENDPOINTS.findIndex((e) => e.p === r.ep); tab = "play"; render(ctx); const pg = ctx.view.querySelector("#pg"); pg.querySelectorAll("[data-p]").forEach((i) => (i.value = sub(r.v[i.dataset.p] ?? ""))); pg.querySelector("[data-p]")?.dispatchEvent(new Event("input")); window.scrollTo(0, 0); }));
299 + }
300 +
301 + /* ---- Référence */
302 + function renderRef() {
303 + const body = ctx.view.querySelector("#api-body");
304 + body.innerHTML = `<div class="card"><h3>Conventions</h3><ul class="small"><li><b>Authentification</b> : en-tête <span class="mono">X-API-Key: &lt;clé&gt;</span>, ou <span class="mono">Authorization: Bearer &lt;clé&gt;</span>, ou paramètre <span class="mono">?api_key=</span> (déconseillé). <span class="mono">/v1/health</span> est libre.</li>
305 + <li><b>Pagination</b> : <span class="mono">limit</span> (≤ 500) et <span class="mono">offset</span> ; réponse <span class="mono">{total, limit, offset, items}</span>.</li>
306 + <li><b>Filtres texte</b> : insensibles à la casse, correspondance « contient » (sauf <span class="mono">ticker</span> et <span class="mono">cik</span>, exacts). Les paramètres <span class="mono">type</span>, <span class="mono">status</span>, <span class="mono">form</span> acceptent des listes séparées par des virgules.</li>
307 + <li><b>Montants</b> en US$ (<span class="mono">5e8</span> accepté). <b>Dates</b> au format ISO <span class="mono">AAAA-MM-JJ</span>. <b>Listes</b> renvoyées comme tableaux JSON ; en CSV, jointes par <span class="mono">|</span>.</li>
308 + <li><b>Limite</b> : 240 requêtes par minute et par adresse IP (429 au-delà). Bac à sable SQL : 20 s, 5 000 lignes, lecture seule, pas d'accès aux fichiers.</li>
309 + <li><b>OpenAPI</b> : <a href="/openapi.json" target="_blank">openapi.json</a> · <a href="/docs" target="_blank">Swagger UI</a> · <a href="/redoc" target="_blank">ReDoc</a>.</li></ul></div>
310 + <div class="card"><h3>Routes</h3><div class="table-wrap"><table class="ref"><thead><tr><th>Méthode</th><th>Route</th><th>Description</th><th>Paramètres</th></tr></thead><tbody>${ENDPOINTS.map((e) => `<tr><td class="mono"><b>${e.m}</b></td><td class="mono"><a href="#/api?tab=play&ep=${encodeURIComponent(e.p)}">${esc(e.p)}</a></td><td>${esc(e.d)}</td><td class="small">${e.params.map((p) => `<span class="mono">${esc(p.name)}</span>`).join(", ") || "—"}</td></tr>`).join("")}</tbody></table></div></div>
311 + ${Object.entries(FIELDS).map(([t, f]) => `<div class="card"><h3>Champs — ${t}</h3><table class="ref"><tbody>${f.map(([k, d]) => `<tr><td class="mono">${esc(k)}</td><td>${esc(d)}</td></tr>`).join("")}</tbody></table></div>`).join("")}
312 + <div class="card"><h3>Types de projets</h3><table class="ref"><thead><tr><th>project_type</th><th>Libellé</th><th class="num">Projets</th></tr></thead><tbody>${meta.types.map((t) => `<tr><td class="mono">${t.project_type}</td><td>${esc(t.label)}</td><td class="num">${fmtN(t.n)}</td></tr>`).join("") || "<tr><td colspan=3 class='muted'>chargement…</td></tr>"}</tbody></table></div>
313 + <div class="card"><h3>Codes de réponse</h3><table class="ref"><tbody>${ERRORS.map(([c, d]) => `<tr><td class="mono"><b>${c}</b></td><td>${esc(d)}</td></tr>`).join("")}</tbody></table></div>`;
314 + if (!meta.types.length) ctx.api("/taxonomy").then((t) => { meta.types = t; if (tab === "ref") renderRef(); }).catch(() => { });
315 + }
316 +
317 + /* ---- SDK */
318 + function renderSdk() {
319 + const body = ctx.view.querySelector("#api-body"); const k = key() || KEY_PH;
320 + body.innerHTML = `<div class="grid g2">
321 + <div class="card" style="grid-column:1/-1"><h3>pdb_client.py — client Python sans dépendance</h3><p class="muted">Un fichier à déposer à côté de votre script ou notebook. Pagination automatique, réessais sur 429, conversion pandas/CSV, toutes les routes.</p>
322 + <a class="btn" href="/sdk/pdb_client.py" download>⬇ Télécharger pdb_client.py</a> <a class="btn btn-ghost" href="/sdk/pdb_client.py" target="_blank">Voir le source</a> &nbsp; <a class="btn btn-or" href="/sdk/pdb_api.R" download>⬇ Script R prêt à l'emploi (pdb_api.R)</a> &nbsp; <a class="btn btn-ghost" href="/sdk/pdb_api.m" download>⬇ MATLAB (pdb_api.m)</a> <a class="btn btn-ghost" href="/sdk/pdb_api.do" download>⬇ Stata (pdb_api.do)</a></div>
323 + <div class="card" style="grid-column:1/-1"><h3>R — pdb_api.R</h3><p class="muted">Fichier R complet (httr2 + dplyr) : fonctions <span class="mono">pdb_get()</span>, <span class="mono">pdb_sql()</span>, <span class="mono">pdb_all()</span> (pagination) et dix exemples exécutables (filtres, secteur entier, CSV, chronologie, entreprise, sémantique, SQL, panel entreprise × année, graphe). Testé avec R 4.6.</p>${codeBlock(`install.packages(c("httr2", "dplyr"))\nSys.setenv(PDB_API_KEY = "${k}")\nsource("pdb_api.R") # définit pdb_get / pdb_sql / pdb_all\n\ndc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items\nutil <- pdb_all("projects", sector = "Utilities") # data.frame complet\nia <- pdb_sql("select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1")`)}</div>
324 + <div class="card"><h3>MATLAB — pdb_api.m</h3><p class="muted">webread / webwrite natifs (R2018a+), fonctions <span class="mono">pdb_get</span>, <span class="mono">pdb_sql</span>, <span class="mono">pdb_all</span> et dix sections exécutables. Non testé sur MATLAB ici : signalez toute erreur.</p>${codeBlock(`BASE = "${location.origin}/v1"; KEY = "${k}";\nopts = weboptions("HeaderFields", ["X-API-Key", KEY], "ContentType", "json");\nr = webread(BASE + "/projects", "type", "data_center", "min_amount", "5e8", "sort", "amount", "limit", "10", opts);\ndc = struct2table(r.items);\n\nres = webwrite(BASE + "/sql", struct("sql", "select project_type, count(*) n from projects group by 1 order by n desc", "limit", 25), ...\n weboptions("HeaderFields", ["X-API-Key", KEY], "MediaType", "application/json"));\nT = cell2table(res.rows, "VariableNames", res.columns);`)}</div>
325 + <div class="card"><h3>Stata — pdb_api.do</h3><p class="muted">Stata ne peut pas envoyer d'en-tête HTTP : le .do définit <span class="mono">pdb_projects</span>, <span class="mono">pdb_mentions</span>, <span class="mono">pdb_sql</span> via Python intégré (Stata 16+) et une voie <span class="mono">shell curl</span> + <span class="mono">import delimited</span> pour toute version. Non testé sur Stata ici.</p>${codeBlock(`global PDB_KEY "${k}"\ndo pdb_api.do // définit les programmes\npdb_projects, filters(type=data_center min_amount=5e8 sort=amount)\npdb_sql "select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1"\n\n* sans Python :\nshell curl -s "${location.origin}/v1/projects/export.csv?type=plant_construction" -H "X-API-Key: $PDB_KEY" -o usines.csv\nimport delimited using "usines.csv", clear varnames(1) encoding(utf8)`)}</div>
326 + <div class="card"><h3>Prise en main</h3>${codeBlock(`from pdb_client import PDB\n\npdb = PDB("${k}") # ou export PDB_API_KEY=...\nprint(pdb.health())\n\nov = pdb.stats()["overview"]\nprint(ov["projects"], "projets")\n\n# une page\npage = pdb.projects(type="data_center", min_amount=5e8, sort="amount", limit=10)\nfor p in page["items"]:\n print(p["ticker"], p["project_name"], p["total_amount_usd"])`)}</div>
327 + <div class="card"><h3>Tout un jeu de données en DataFrame</h3>${codeBlock(`# pagination automatique (500 par page)\nrows = pdb.iter_projects(sector="Utilities")\ndf = PDB.to_dataframe(rows)\nprint(df.shape)\n\n# mentions 8-K de 2025 sur l'IA\nm = PDB.to_dataframe(pdb.iter_mentions(type="ai_initiative", form="8-K", date_from="2025-01-01"))\nm.groupby("ticker").size().sort_values(ascending=False).head(10)\n\n# export CSV\nPDB.to_csv(pdb.iter_projects(type="plant_construction"), "usines.csv")`)}</div>
328 + <div class="card"><h3>Fiche, similaires, entreprise, graphe</h3>${codeBlock(`fiche = pdb.project(page["items"][0]["project_id"])\nfiche["project"], fiche["timeline"], fiche["mentions"], fiche["similar"]\n\npdb.similar(fiche["project"]["project_id"], k=5)\n\nduk = pdb.company("DUK") # ticker ou CIK\nduk["by_type"], duk["projects"][:3]\n\nnode = pdb.graph_node("L:texas", limit=500)\n[e["neighbor_label"] for e in node["edges"]][:10]`)}</div>
329 + <div class="card"><h3>Sémantique et SQL</h3>${codeBlock(`for h in pdb.semantic("battery cell factory", k=5):\n print(round(h["score"], 3), h["ticker"], h["project_name"])\n\nres = pdb.sql("""\n select year(filing_date) as yr, count(*) n\n from project_mentions where project_type = 'data_center'\n group by 1 order by 1\n""")\nimport pandas as pd\npd.DataFrame(res["rows"], columns=res["columns"]).set_index("yr").plot()\n\n# ou directement en enregistrements\npdb.sql_records("select ticker, count(*) n from projects group by 1 order by n desc limit 5")`)}</div>
330 + <div class="card"><h3>Dans un notebook Jupyter / Colab</h3>${codeBlock(`!curl -sO ${location.origin}/sdk/pdb_client.py\nimport os; os.environ["PDB_API_KEY"] = "${k}"\nfrom pdb_client import PDB\npdb = PDB()`)}</div>
331 + <div class="card"><h3>JavaScript / TypeScript (sans SDK)</h3>${codeBlock(`const BASE = "${location.origin}/v1";\nconst H = { "X-API-Key": process.env.PDB_API_KEY };\n\nexport async function pdb(path, params = {}) {\n const r = await fetch(\`\${BASE}/\${path}?\${new URLSearchParams(params)}\`, { headers: H });\n if (!r.ok) throw new Error(\`HTTP \${r.status}: \${(await r.json()).detail}\`);\n return r.json();\n}\nexport async function* iterate(path, params = {}) {\n for (let offset = 0; ; ) {\n const page = await pdb(path, { ...params, limit: 500, offset });\n yield* page.items; offset += page.items.length;\n if (!page.items.length || offset >= page.total) return;\n }\n}`)}</div>
332 + </div>`;
333 + bindCopy(body);
334 + }
335 +
336 + return { render };
337 +})();
added web/report.pdf +0 −0

Binary file not shown.

added web/sdk/pdb_api.R +113 −0
@@ -0,0 +1,113 @@
1 +# =============================================================================
2 +# pdb_api.R — utiliser la PDB API (SEC Project Intelligence Database, UQO) en R
3 +# install.packages(c("httr2", "dplyr", "ggplot2")) # une seule fois
4 +# source("pdb_api.R") ou copier-coller ce fichier dans RStudio
5 +# =============================================================================
6 +library(httr2)
7 +library(dplyr)
8 +
9 +BASE <- "https://www.pdb-api.co/v1"
10 +KEY <- Sys.getenv("PDB_API_KEY", unset = "VOTRE_CLE") # ou : KEY <- "VOTRE_CLE"
11 +
12 +# ---- 1. Deux fonctions génériques ------------------------------------------
13 +pdb_get <- function(path, ...) {
14 + q <- Filter(Negate(is.null), list(...))
15 + req <- request(paste0(BASE, "/", path)) |>
16 + req_headers(`X-API-Key` = KEY) |>
17 + req_retry(max_tries = 4, is_transient = \(r) resp_status(r) == 429) |>
18 + req_error(body = \(r) tryCatch(resp_body_json(r)$detail, error = \(e) NULL))
19 + if (length(q)) req <- req |> req_url_query(!!!q)
20 + req |> req_perform() |> resp_body_json(simplifyVector = TRUE)
21 +}
22 +
23 +pdb_sql <- function(sql, limit = 500) {
24 + res <- request(paste0(BASE, "/sql")) |>
25 + req_headers(`X-API-Key` = KEY) |>
26 + req_body_json(list(sql = sql, limit = limit)) |>
27 + req_perform() |> resp_body_json(simplifyVector = TRUE)
28 + df <- as.data.frame(res$rows, stringsAsFactors = FALSE)
29 + if (nrow(df)) names(df) <- res$columns
30 + df
31 +}
32 +
33 +# Tout récupérer (pagination automatique, 500 par page) -> data.frame
34 +pdb_all <- function(path, ...) {
35 + pages <- list(); offset <- 0
36 + repeat {
37 + page <- pdb_get(path, ..., limit = 500, offset = offset)
38 + items <- as.data.frame(page$items)
39 + pages[[length(pages) + 1]] <- items
40 + offset <- offset + nrow(items)
41 + if (nrow(items) == 0 || offset >= page$total) break
42 + }
43 + bind_rows(pages)
44 +}
45 +
46 +# ---- 2. Exemples -----------------------------------------------------------
47 +if (sys.nframe() == 0) { # exécuté seulement si on lance le fichier directement
48 +
49 + # a) vérifier le service et la clé
50 + print(pdb_get("health"))
51 + ov <- pdb_get("stats")$overview
52 + cat(sprintf("%s projets · %s mentions · %s entreprises\n",
53 + format(ov$projects, big.mark = " "), format(ov$mentions, big.mark = " "), ov$companies))
54 +
55 + # b) une page de projets filtrés (centres de données > 500 M$)
56 + dc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items
57 + print(dc[, c("ticker", "project_name", "canonical_location", "total_amount_usd", "first_seen")])
58 +
59 + # c) tout un secteur en data.frame, puis agrégation dplyr
60 + util <- pdb_all("projects", sector = "Utilities")
61 + util |>
62 + group_by(project_type) |>
63 + summarise(n = n(), capital_gusd = sum(total_amount_usd, na.rm = TRUE) / 1e9, .groups = "drop") |>
64 + arrange(desc(n)) |>
65 + print(n = 10)
66 +
67 + # d) export CSV direct (le plus simple pour un jeu de données complet)
68 + csv <- request(paste0(BASE, "/projects/export.csv")) |>
69 + req_headers(`X-API-Key` = KEY) |>
70 + req_url_query(type = "plant_construction") |>
71 + req_perform() |> resp_body_string()
72 + usines <- read.csv(text = csv)
73 + cat("usines :", nrow(usines), "lignes\n")
74 +
75 + # e) fiche d'un projet : chronologie et projets similaires
76 + pid <- pdb_get("projects", ticker = "TSLA", q = "energy storage", limit = 1)$items$project_id[1]
77 + fiche <- pdb_get(paste0("projects/", pid))
78 + cat(fiche$project$project_name, "-", fiche$project$status, "\n")
79 + print(fiche$timeline[, c("filing_date", "form_type", "status", "amount_usd")])
80 + print(fiche$similar[, c("score", "ticker", "project_name")])
81 +
82 + # f) profil d'une entreprise
83 + duk <- pdb_get("companies/DUK")
84 + print(duk$by_type[, c("label", "n", "amount_usd")])
85 + barplot(duk$by_year$n, names.arg = duk$by_year$year, las = 2, main = "Duke Energy — mentions par année")
86 +
87 + # g) recherche sémantique (langage naturel, anglais recommandé)
88 + sem <- pdb_get("search/semantic", q = "battery cell factory", k = 5)$items
89 + print(sem[, c("score", "ticker", "project_name")])
90 +
91 + # h) SQL libre (lecture seule, DuckDB)
92 + ia <- pdb_sql("
93 + select year(filing_date) as yr, count(*) n
94 + from project_mentions
95 + where project_type = 'ai_initiative'
96 + group by 1 order by 1")
97 + print(ia)
98 + plot(ia$yr, ia$n, type = "b", xlab = "année", ylab = "mentions", main = "Initiatives IA dans les filings")
99 +
100 + # i) panel entreprise × année pour l'économétrie
101 + panel <- pdb_sql("
102 + select cik, any_value(ticker) ticker, any_value(sector) sector,
103 + year(first_seen) as yr, count(*) n_projets,
104 + sum(total_amount_usd) capital_usd
105 + from projects group by 1, 4 order by 1, 4", limit = 5000)
106 + cat("panel :", nrow(panel), "lignes (entreprise × année)\n")
107 + # summary(lm(log1p(n_projets) ~ factor(yr) + factor(sector), data = panel))
108 +
109 + # j) graphe : projets rattachés au Texas
110 + tx <- pdb_get("graph/node/L:texas", limit = 500)
111 + cat("degré de L:texas :", tx$degree, "\n")
112 + print(head(tx$edges[, c("rel", "neighbor_type", "neighbor_label")]))
113 +}
added web/sdk/pdb_api.do +150 −0
@@ -0,0 +1,150 @@
1 +* =============================================================================
2 +* pdb_api.do — utiliser la PDB API (SEC Project Intelligence Database, UQO) dans Stata
3 +*
4 +* Stata ne peut pas envoyer d'en-tête HTTP avec `copy` ou `import delimited`.
5 +* Deux voies :
6 +* A) Stata 16+ avec Python intégré (recommandé) : `python:` appelle l'API et écrit un CSV.
7 +* B) Toute version : `shell curl` télécharge le CSV / JSON, puis `import delimited`.
8 +* Remplacer VOTRE_CLE ci-dessous (ou définir la variable d'environnement PDB_API_KEY).
9 +* =============================================================================
10 +clear all
11 +set more off
12 +global PDB_BASE "https://www.pdb-api.co/v1"
13 +global PDB_KEY "VOTRE_CLE"
14 +
15 +* -----------------------------------------------------------------------------
16 +* A) Voie Python (Stata 16+) — définit trois programmes : pdb_projects, pdb_mentions, pdb_sql
17 +* -----------------------------------------------------------------------------
18 +python:
19 +import csv, json, os, urllib.parse, urllib.request
20 +from sfi import Macro
21 +
22 +BASE = Macro.getGlobal("PDB_BASE"); KEY = os.environ.get("PDB_API_KEY") or Macro.getGlobal("PDB_KEY")
23 +
24 +def _req(path, params=None, body=None):
25 + url = f"{BASE}/{path}" + (f"?{urllib.parse.urlencode(params)}" if params else "")
26 + data = json.dumps(body).encode() if body is not None else None
27 + req = urllib.request.Request(url, data=data, headers={"X-API-Key": KEY, "Content-Type": "application/json"})
28 + with urllib.request.urlopen(req, timeout=120) as r:
29 + return json.loads(r.read())
30 +
31 +def _flat(rec):
32 + return {k: (" | ".join(map(str, v)) if isinstance(v, list) else v) for k, v in rec.items()}
33 +
34 +def _write(rows, path):
35 + if not rows:
36 + open(path, "w").close(); return 0
37 + with open(path, "w", newline="", encoding="utf-8") as f:
38 + w = csv.DictWriter(f, fieldnames=list(rows[0].keys())); w.writeheader()
39 + for r in rows: w.writerow(_flat(r))
40 + return len(rows)
41 +
42 +def pdb_pages(path, csvfile, **filters):
43 + """Tous les enregistrements paginés (500 par page) -> CSV."""
44 + rows, offset = [], 0
45 + while True:
46 + page = _req(path, {**filters, "limit": 500, "offset": offset})
47 + rows += page["items"]; offset += len(page["items"])
48 + if not page["items"] or offset >= page["total"]: break
49 + n = _write(rows, csvfile); Macro.setLocal("pdb_n", str(n))
50 +
51 +def pdb_sql(sql, csvfile, limit=5000):
52 + res = _req("sql", body={"sql": sql, "limit": limit})
53 + rows = [dict(zip(res["columns"], r)) for r in res["rows"]]
54 + n = _write(rows, csvfile); Macro.setLocal("pdb_n", str(n))
55 +end
56 +
57 +* --- Programmes Stata enveloppant les fonctions Python ------------------------
58 +capture program drop pdb_projects
59 +program define pdb_projects
60 + * usage : pdb_projects, filters(type=data_center min_amount=5e8) [file(x.csv)]
61 + syntax , [FILters(string) FILE(string)]
62 + if "`file'" == "" local file "pdb_projects.csv"
63 + local kw ""
64 + foreach f of local filters {
65 + gettoken k v : f, parse("=")
66 + local v = subinstr("`v'", "=", "", 1)
67 + local kw `"`kw' `k'="`v'","'
68 + }
69 + python: pdb_pages("projects", "`file'" `kw')
70 + import delimited using "`file'", clear varnames(1) encoding(utf8) stringcols(_all)
71 + destring total_amount_usd n_mentions n_filings first_year last_year avg_confidence, replace force
72 + gen date_first = date(first_seen, "YMD"); format date_first %td
73 + gen date_last = date(last_seen, "YMD"); format date_last %td
74 + di as txt "`pdb_n' projets importés"
75 +end
76 +
77 +capture program drop pdb_mentions
78 +program define pdb_mentions
79 + syntax , [FILters(string) FILE(string)]
80 + if "`file'" == "" local file "pdb_mentions.csv"
81 + local kw ""
82 + foreach f of local filters {
83 + gettoken k v : f, parse("=")
84 + local v = subinstr("`v'", "=", "", 1)
85 + local kw `"`kw' `k'="`v'","'
86 + }
87 + python: pdb_pages("mentions", "`file'" `kw')
88 + import delimited using "`file'", clear varnames(1) encoding(utf8) stringcols(_all)
89 + destring amount_usd confidence, replace force
90 + gen date_filing = date(filing_date, "YMD"); format date_filing %td
91 + di as txt "`pdb_n' mentions importées"
92 +end
93 +
94 +capture program drop pdb_sql
95 +program define pdb_sql
96 + * usage : pdb_sql "select ... " [, file(x.csv) limit(5000)]
97 + syntax anything(everything name=sql) [, FILE(string) LIMit(integer 5000)]
98 + if "`file'" == "" local file "pdb_sql.csv"
99 + local sql = subinstr(`"`sql'"', `"""', "", .)
100 + python: pdb_sql("""`sql'""", "`file'", `limit')
101 + import delimited using "`file'", clear varnames(1) encoding(utf8)
102 + di as txt "`pdb_n' lignes importées"
103 +end
104 +
105 +* -----------------------------------------------------------------------------
106 +* Exemples (voie A)
107 +* -----------------------------------------------------------------------------
108 +
109 +* 1. Centres de données > 500 M$
110 +pdb_projects, filters(type=data_center min_amount=5e8 sort=amount)
111 +list ticker project_name canonical_location total_amount_usd first_seen in 1/10, clean
112 +
113 +* 2. Tout un secteur, puis agrégation
114 +pdb_projects, filters(sector=Utilities) file(utilities.csv)
115 +tab project_type, sort
116 +collapse (count) n=project_id (sum) capital=total_amount_usd, by(project_type)
117 +gsort -n
118 +list, clean
119 +
120 +* 3. Mentions 8-K sur l'IA depuis 2024
121 +pdb_mentions, filters(type=ai_initiative form=8-K date_from=2024-01-01)
122 +gen yr = year(date_filing)
123 +tab yr
124 +
125 +* 4. SQL libre : mentions IA par année
126 +pdb_sql "select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1"
127 +twoway connected n yr, title("Initiatives IA dans les filings") ytitle("mentions")
128 +
129 +* 5. Panel entreprise x année pour l'économétrie
130 +pdb_sql "select cik, any_value(ticker) ticker, any_value(sector) sector, year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital_usd from projects group by 1, 4 order by 1, 4", file(panel.csv)
131 +encode cik, gen(id)
132 +xtset id yr
133 +xtpoisson n_projets i.yr, fe // intensité d'annonce de projets, effets fixes entreprise
134 +* xtreg ln_capital i.yr, fe après : gen ln_capital = ln(capital_usd)
135 +
136 +* -----------------------------------------------------------------------------
137 +* B) Voie curl (toute version de Stata) : CSV d'export puis import delimited
138 +* -----------------------------------------------------------------------------
139 +* Sur macOS / Linux : `shell` ; sur Windows : remplacer par `winexec` ou `shell curl.exe ...`
140 +shell curl -s "$PDB_BASE/projects/export.csv?type=plant_construction" -H "X-API-Key: $PDB_KEY" -o usines.csv
141 +import delimited using "usines.csv", clear varnames(1) encoding(utf8) stringcols(_all)
142 +destring total_amount_usd n_mentions n_filings avg_confidence, replace force
143 +gen date_first = date(first_seen, "YMD"); format date_first %td
144 +describe, short
145 +summarize total_amount_usd, detail
146 +
147 +* SQL via curl : la réponse est en JSON ; convertir en CSV avec python3 du système
148 +shell curl -s -X POST "$PDB_BASE/sql" -H "X-API-Key: $PDB_KEY" -H "Content-Type: application/json" -d "{\"sql\":\"select project_type, count(*) n, sum(total_amount_usd) capital from projects group by 1 order by n desc\",\"limit\":100}" | python3 -c "import csv,json,sys; d=json.load(sys.stdin); w=csv.writer(sys.stdout); w.writerow(d['columns']); w.writerows(d['rows'])" > types.csv
149 +import delimited using "types.csv", clear varnames(1)
150 +list, clean
added web/sdk/pdb_api.m +111 −0
@@ -0,0 +1,111 @@
1 +%% pdb_api.m — utiliser la PDB API (SEC Project Intelligence Database, UQO) dans MATLAB
2 +% MATLAB R2018a ou plus récent (webread / webwrite / jsondecode). Aucun toolbox requis.
3 +% Usage : ouvrir ce fichier, remplacer KEY, exécuter section par section (Ctrl+Entrée).
4 +% Les fonctions pdb_get / pdb_sql / pdb_all sont définies en fin de fichier (fonctions locales de script).
5 +
6 +BASE = "https://www.pdb-api.co/v1";
7 +KEY = getenv("PDB_API_KEY"); if isempty(KEY), KEY = "VOTRE_CLE"; end
8 +
9 +%% 1. Vérifier le service et la clé
10 +disp(pdb_get(BASE, KEY, "health"))
11 +ov = pdb_get(BASE, KEY, "stats").overview;
12 +fprintf("%d projets, %d mentions, %d entreprises\n", ov.projects, ov.mentions, ov.companies);
13 +
14 +%% 2. Centres de données > 500 M$, du plus gros au plus petit
15 +r = pdb_get(BASE, KEY, "projects", type="data_center", min_amount=5e8, sort="amount", limit=10);
16 +dc = struct2table(r.items); % tableau MATLAB
17 +disp(dc(:, {'ticker','project_name','canonical_location','total_amount_usd','first_seen'}))
18 +
19 +%% 3. Tout un secteur (pagination automatique) puis agrégation
20 +util = pdb_all(BASE, KEY, "projects", sector="Utilities"); % ~1 956 lignes
21 +G = groupsummary(util, "project_type", "sum", "total_amount_usd");
22 +G = sortrows(G, "GroupCount", "descend");
23 +disp(G(1:10, :))
24 +
25 +%% 4. Export CSV direct (le plus simple pour un jeu de données complet)
26 +opts = weboptions("HeaderFields", ["X-API-Key", KEY], "ContentType", "text", "Timeout", 120);
27 +csvText = webread(BASE + "/projects/export.csv", "type", "plant_construction", opts);
28 +fid = fopen("usines.csv", "w"); fwrite(fid, csvText); fclose(fid);
29 +usines = readtable("usines.csv", "TextType", "string");
30 +fprintf("usines : %d lignes\n", height(usines));
31 +
32 +%% 5. Fiche d'un projet : chronologie et projets similaires
33 +p1 = pdb_get(BASE, KEY, "projects", ticker="TSLA", q="energy storage", limit=1);
34 +fiche = pdb_get(BASE, KEY, "projects/" + string(p1.items(1).project_id));
35 +fprintf("%s — %s\n", fiche.project.project_name, fiche.project.status);
36 +disp(struct2table(fiche.timeline)(:, {'filing_date','form_type','status','amount_usd'}))
37 +disp(struct2table(fiche.similar)(:, {'score','ticker','project_name'}))
38 +
39 +%% 6. Profil d'une entreprise (ticker ou CIK)
40 +duk = pdb_get(BASE, KEY, "companies/DUK");
41 +disp(struct2table(duk.by_type)(:, {'label','n','amount_usd'}))
42 +figure; bar([duk.by_year.n]); xticklabels(string([duk.by_year.year])); xticks(1:numel(duk.by_year));
43 +title("Duke Energy — mentions par année"); ylabel("mentions");
44 +
45 +%% 7. Recherche sémantique (langage naturel, anglais recommandé)
46 +sem = pdb_get(BASE, KEY, "search/semantic", q="battery cell factory", k=5);
47 +disp(struct2table(sem.items)(:, {'score','ticker','project_name'}))
48 +
49 +%% 8. SQL libre (DuckDB, lecture seule) -> table
50 +ia = pdb_sql(BASE, KEY, join([ ...
51 + "select year(filing_date) as yr, count(*) n", ...
52 + "from project_mentions where project_type = 'ai_initiative'", ...
53 + "group by 1 order by 1"], " "));
54 +figure; plot(ia.yr, ia.n, "-o"); xlabel("année"); ylabel("mentions"); title("Initiatives IA dans les filings");
55 +
56 +%% 9. Panel entreprise × année pour l'économétrie
57 +panel = pdb_sql(BASE, KEY, join([ ...
58 + "select cik, any_value(ticker) ticker, any_value(sector) sector,", ...
59 + " year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital_usd", ...
60 + "from projects group by 1, 4 order by 1, 4"], " "), 5000);
61 +fprintf("panel : %d lignes\n", height(panel));
62 +% mdl = fitlm(panel, "n_projets ~ yr + sector"); % Statistics and Machine Learning Toolbox
63 +
64 +%% 10. Graphe : projets rattachés au Texas
65 +tx = pdb_get(BASE, KEY, "graph/node/L:texas", limit=500);
66 +fprintf("degré de L:texas : %d\n", tx.degree);
67 +disp(struct2table(tx.edges)(1:6, {'rel','neighbor_type','neighbor_label'}))
68 +
69 +%% ------------------------------------------------------------------ fonctions locales
70 +function out = pdb_get(base, key, path, varargin)
71 +% GET générique : pdb_get(BASE, KEY, "projects", type="data_center", limit=10)
72 + opts = weboptions("HeaderFields", ["X-API-Key", key], "ContentType", "json", "Timeout", 120);
73 + args = {};
74 + for i = 1:2:numel(varargin)
75 + v = varargin{i+1};
76 + if isnumeric(v), v = num2str(v, "%.10g"); end
77 + args(end+1:end+2) = {char(varargin{i}), char(string(v))};
78 + end
79 + try
80 + out = webread(base + "/" + path, args{:}, opts);
81 + catch e
82 + error("PDB API : %s", e.message); % 401 = clé invalide, 404 = introuvable, 429 = trop de requêtes
83 + end
84 +end
85 +
86 +function T = pdb_sql(base, key, sql, limit)
87 +% Requête SQL de lecture -> table MATLAB
88 + if nargin < 4, limit = 500; end
89 + opts = weboptions("HeaderFields", ["X-API-Key", key], "MediaType", "application/json", "ContentType", "json", "Timeout", 120);
90 + res = webwrite(base + "/sql", struct("sql", sql, "limit", limit), opts);
91 + if isempty(res.rows), T = table(); return; end
92 + rows = res.rows;
93 + if iscell(rows), rows = vertcat(rows{:}); end % cellules si types mixtes
94 + T = cell2table(num2cell(rows), "VariableNames", matlab.lang.makeValidName(string(res.columns)));
95 + if iscell(rows) || ~isnumeric(rows)
96 + T = cell2table(rows, "VariableNames", matlab.lang.makeValidName(string(res.columns)));
97 + end
98 +end
99 +
100 +function T = pdb_all(base, key, path, varargin)
101 +% Pagination automatique (500 par page) -> table complète
102 + offset = 0; parts = {};
103 + while true
104 + page = pdb_get(base, key, path, varargin{:}, "limit", 500, "offset", offset);
105 + if isempty(page.items), break; end
106 + parts{end+1} = struct2table(page.items, "AsArray", true); %#ok<AGROW>
107 + offset = offset + numel(page.items);
108 + if offset >= page.total, break; end
109 + end
110 + T = vertcat(parts{:});
111 +end
added web/sdk/pdb_client.py +203 −0
@@ -0,0 +1,203 @@
1 +"""pdb_client — client Python minimal pour la PDB API (SEC Project Intelligence Database, UQO).
2 +
3 +Aucune dépendance obligatoire (urllib). pandas est optionnel pour `to_dataframe`.
4 +
5 + from pdb_client import PDB
6 + pdb = PDB("VOTRE_CLE") # ou variable d'environnement PDB_API_KEY
7 + pdb.stats()["overview"]
8 + for p in pdb.iter_projects(type="data_center", min_amount=5e8):
9 + print(p["ticker"], p["project_name"], p["total_amount_usd"])
10 + df = pdb.to_dataframe(pdb.iter_projects(sector="Utilities"))
11 + pdb.sql("select project_type, count(*) n from projects group by 1 order by n desc")
12 +"""
13 +from __future__ import annotations
14 +
15 +import csv
16 +import io
17 +import json
18 +import os
19 +import time
20 +import urllib.error
21 +import urllib.parse
22 +import urllib.request
23 +from typing import Any, Iterator
24 +
25 +__version__ = "1.0.0"
26 +DEFAULT_BASE = "https://www.pdb-api.co/v1"
27 +
28 +
29 +class PDBError(RuntimeError):
30 + def __init__(self, status: int, detail: str, url: str):
31 + super().__init__(f"HTTP {status} — {detail} ({url})")
32 + self.status, self.detail, self.url = status, detail, url
33 +
34 +
35 +class PDB:
36 + def __init__(self, api_key: str | None = None, base_url: str = DEFAULT_BASE, timeout: float = 60.0, retries: int = 3):
37 + self.api_key = api_key or os.environ.get("PDB_API_KEY", "")
38 + if not self.api_key:
39 + raise ValueError("Clé d'API manquante : PDB(api_key=...) ou variable PDB_API_KEY.")
40 + self.base_url = base_url.rstrip("/")
41 + self.timeout = timeout
42 + self.retries = retries
43 +
44 + # ------------------------------------------------------------ transport
45 + def request(self, method: str, path: str, params: dict | None = None, body: dict | None = None, raw: bool = False) -> Any:
46 + q = {k: (str(v).lower() if isinstance(v, bool) else v) for k, v in (params or {}).items() if v is not None and v != ""}
47 + url = f"{self.base_url}/{path.lstrip('/')}" + (f"?{urllib.parse.urlencode(q)}" if q else "")
48 + data = json.dumps(body).encode() if body is not None else None
49 + headers = {"X-API-Key": self.api_key, "Accept": "application/json", "User-Agent": f"pdb_client/{__version__}"}
50 + if data is not None:
51 + headers["Content-Type"] = "application/json"
52 + for attempt in range(self.retries + 1):
53 + req = urllib.request.Request(url, data=data, method=method, headers=headers)
54 + try:
55 + with urllib.request.urlopen(req, timeout=self.timeout) as r:
56 + payload = r.read()
57 + return payload.decode() if raw else json.loads(payload)
58 + except urllib.error.HTTPError as e:
59 + detail = e.read().decode(errors="replace")
60 + try:
61 + detail = json.loads(detail).get("detail", detail)
62 + except Exception:
63 + pass
64 + if e.code in (429, 502, 503, 504) and attempt < self.retries:
65 + time.sleep(1.5 * (attempt + 1))
66 + continue
67 + raise PDBError(e.code, detail, url) from None
68 +
69 + def get(self, path: str, **params) -> Any:
70 + return self.request("GET", path, params)
71 +
72 + # ------------------------------------------------------------ découverte
73 + def health(self) -> dict:
74 + return self.get("health")
75 +
76 + def stats(self) -> dict:
77 + return self.get("stats")
78 +
79 + def taxonomy(self) -> list[dict]:
80 + return self.get("taxonomy")
81 +
82 + def sectors(self) -> list[dict]:
83 + return self.get("sectors")
84 +
85 + def technologies(self, q: str | None = None, limit: int = 50) -> list[dict]:
86 + return self.get("technologies", q=q, limit=limit)
87 +
88 + def locations(self, q: str | None = None, limit: int = 50) -> list[dict]:
89 + return self.get("locations", q=q, limit=limit)
90 +
91 + def partners(self, q: str | None = None, limit: int = 50) -> list[dict]:
92 + return self.get("partners", q=q, limit=limit)
93 +
94 + def schema(self) -> list[dict]:
95 + return self.get("schema")
96 +
97 + # ------------------------------------------------------------ projets
98 + def projects(self, **filters) -> dict:
99 + """Une page : {total, limit, offset, items}. Filtres : q, type, sector, status, ticker, cik, location, tech,
100 + partner, min_amount, max_amount, year_from, year_to, min_confidence, has_amount, sort, order, limit, offset."""
101 + return self.get("projects", **filters)
102 +
103 + def iter_projects(self, page_size: int = 500, max_items: int | None = None, **filters) -> Iterator[dict]:
104 + """Itère sur tous les projets correspondant aux filtres (pagination automatique)."""
105 + yield from self._paginate("projects", page_size, max_items, **filters)
106 +
107 + def project(self, project_id: str) -> dict:
108 + """Fiche complète : {project, timeline, mentions, graph, similar}."""
109 + return self.get(f"projects/{project_id}")
110 +
111 + def similar(self, project_id: str, k: int = 10) -> list[dict]:
112 + return self.get(f"projects/{project_id}/similar", k=k)
113 +
114 + def projects_csv(self, **filters) -> str:
115 + return self.request("GET", "projects/export.csv", filters, raw=True)
116 +
117 + # ------------------------------------------------------------ mentions / sections
118 + def mentions(self, **filters) -> dict:
119 + return self.get("mentions", **filters)
120 +
121 + def iter_mentions(self, page_size: int = 500, max_items: int | None = None, **filters) -> Iterator[dict]:
122 + yield from self._paginate("mentions", page_size, max_items, **filters)
123 +
124 + def mention(self, mention_id: str) -> dict:
125 + return self.get(f"mentions/{mention_id}")
126 +
127 + def section(self, section_id: str, highlight: str | None = None) -> dict:
128 + return self.get(f"sections/{section_id}", highlight=highlight)
129 +
130 + # ------------------------------------------------------------ entreprises
131 + def companies(self, **filters) -> dict:
132 + return self.get("companies", **filters)
133 +
134 + def iter_companies(self, page_size: int = 500, **filters) -> Iterator[dict]:
135 + yield from self._paginate("companies", page_size, None, **filters)
136 +
137 + def company(self, ticker_or_cik: str) -> dict:
138 + return self.get(f"companies/{ticker_or_cik}")
139 +
140 + # ------------------------------------------------------------ graphe / recherche / SQL
141 + def graph_search(self, q: str, type: str | None = None, limit: int = 30) -> list[dict]:
142 + return self.get("graph/search", q=q, type=type, limit=limit)
143 +
144 + def graph_node(self, node_id: str, limit: int = 200) -> dict:
145 + return self.get(f"graph/node/{urllib.parse.quote(node_id, safe=':')}", limit=limit)
146 +
147 + def semantic(self, q: str, k: int = 20, type: str | None = None, sector: str | None = None) -> list[dict]:
148 + return self.get("search/semantic", q=q, k=k, type=type, sector=sector)["items"]
149 +
150 + def sql(self, query: str, limit: int = 500) -> dict:
151 + """Requête de lecture DuckDB. Retour : {columns, rows, truncated, elapsed_ms}."""
152 + return self.request("POST", "sql", body={"sql": query, "limit": limit})
153 +
154 + def sql_records(self, query: str, limit: int = 500) -> list[dict]:
155 + r = self.sql(query, limit)
156 + return [dict(zip(r["columns"], row)) for row in r["rows"]]
157 +
158 + # ------------------------------------------------------------ utilitaires
159 + def _paginate(self, path: str, page_size: int, max_items: int | None, **filters) -> Iterator[dict]:
160 + offset, n = int(filters.pop("offset", 0) or 0), 0
161 + page_size = max(1, min(page_size, 500))
162 + while True:
163 + page = self.get(path, limit=page_size, offset=offset, **filters)
164 + items = page.get("items", [])
165 + for it in items:
166 + yield it
167 + n += 1
168 + if max_items and n >= max_items:
169 + return
170 + offset += len(items)
171 + if not items or offset >= page.get("total", 0):
172 + return
173 +
174 + @staticmethod
175 + def to_dataframe(records):
176 + """Convertit une liste/itérateur de dicts en DataFrame pandas (listes jointes par ' | ')."""
177 + import pandas as pd # optionnel
178 + rows = []
179 + for r in records:
180 + rows.append({k: (" | ".join(map(str, v)) if isinstance(v, list) else v) for k, v in r.items()})
181 + return pd.DataFrame(rows)
182 +
183 + @staticmethod
184 + def to_csv(records, path: str) -> int:
185 + rows = list(records)
186 + if not rows:
187 + return 0
188 + with open(path, "w", newline="", encoding="utf-8") as f:
189 + w = csv.DictWriter(f, fieldnames=list(rows[0].keys()))
190 + w.writeheader()
191 + for r in rows:
192 + w.writerow({k: (" | ".join(map(str, v)) if isinstance(v, list) else v) for k, v in r.items()})
193 + return len(rows)
194 +
195 +
196 +if __name__ == "__main__": # petit test : python pdb_client.py VOTRE_CLE
197 + import sys
198 + c = PDB(sys.argv[1] if len(sys.argv) > 1 else None)
199 + print(json.dumps(c.health(), indent=2))
200 + ov = c.stats()["overview"]
201 + print(f"{ov['projects']:,} projets, {ov['mentions']:,} mentions, {ov['companies']} entreprises")
202 + for p in c.iter_projects(type="data_center", min_amount=5e8, sort="amount", max_items=5):
203 + print(f" {p['ticker']:6s} {p['project_name'][:50]:50s} {p['total_amount_usd']/1e9:6.1f} G$")
added web/style.css +137 −0
@@ -0,0 +1,137 @@
1 +:root{
2 + --bleu:#003E7E;--bleu2:#0066B3;--or:#C6A300;--gris:#58595B;--gris2:#8a8c90;--ligne:#e3e7ee;--bg:#f4f6fa;--card:#fff;
3 + --vert:#008046;--rouge:#B42318;--orange:#D67A00;--teal:#007979;--violet:#5E35B1;--encre:#14213d;
4 + --mono:ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
5 + --sans:-apple-system,BlinkMacSystemFont,"Segoe UI",Inter,Roboto,Helvetica,Arial,sans-serif;
6 +}
7 +*{box-sizing:border-box}
8 +html,body{margin:0;height:100%;font-family:var(--sans);color:var(--encre);background:var(--bg);font-size:14.5px;line-height:1.45}
9 +a{color:var(--bleu2);text-decoration:none}a:hover{text-decoration:underline}
10 +h1,h2,h3{margin:0 0 .4em;line-height:1.2}h1{font-size:1.55rem;color:var(--bleu)}h2{font-size:1.15rem;color:var(--bleu)}h3{font-size:1rem;color:var(--gris)}
11 +.muted{color:var(--gris2)}.small{font-size:.85em}.mono{font-family:var(--mono);font-size:.9em}
12 +.shell{display:grid;grid-template-columns:250px 1fr;min-height:100vh}
13 +.side{background:var(--bleu);color:#fff;display:flex;flex-direction:column;position:sticky;top:0;height:100vh}
14 +.brand{display:flex;gap:10px;align-items:center;padding:18px 18px 14px;color:#fff;border-bottom:1px solid rgba(255,255,255,.12)}
15 +.brand:hover{text-decoration:none}
16 +.brand-mark{background:var(--or);color:var(--bleu);font-weight:800;border-radius:8px;padding:6px 8px;font-size:.95rem;letter-spacing:.5px}
17 +.brand-text{display:flex;flex-direction:column;line-height:1.15}.brand-text small{opacity:.7;font-size:.75rem}
18 +.nav{display:flex;flex-direction:column;padding:10px 10px;gap:2px;overflow:auto;flex:1}
19 +.nav a{color:rgba(255,255,255,.85);padding:8px 12px;border-radius:7px;font-size:.93rem}
20 +.nav a:hover{background:rgba(255,255,255,.08);text-decoration:none}.nav a.active{background:rgba(255,255,255,.16);color:#fff;font-weight:600}
21 +.nav-sep{font-size:.7rem;text-transform:uppercase;letter-spacing:.12em;opacity:.55;padding:14px 12px 4px}
22 +.side-foot{padding:14px 18px;font-size:.75rem;opacity:.75;border-top:1px solid rgba(255,255,255,.12);line-height:1.4}
23 +.main{display:flex;flex-direction:column;min-width:0}
24 +.topbar{display:flex;align-items:center;gap:14px;padding:12px 24px;background:#fff;border-bottom:1px solid var(--ligne);position:sticky;top:0;z-index:5}
25 +.burger{display:none;background:none;border:1px solid var(--ligne);border-radius:6px;padding:4px 9px;font-size:1.1rem}
26 +.search{flex:1;max-width:640px}.search input{width:100%;padding:9px 14px;border:1px solid var(--ligne);border-radius:9px;background:var(--bg);font:inherit}
27 +.search input:focus{outline:2px solid var(--bleu2);background:#fff}
28 +.topbar-right{margin-left:auto;display:flex;gap:10px;align-items:center}
29 +.view{padding:22px 24px 30px;flex:1}
30 +.foot{display:flex;justify-content:space-between;gap:20px;padding:12px 24px;border-top:1px solid var(--ligne);font-size:.78rem;color:var(--gris2);flex-wrap:wrap}
31 +.loading{padding:40px;text-align:center;color:var(--gris2)}
32 +.pill{display:inline-block;padding:3px 9px;border-radius:999px;font-size:.78rem;font-weight:600;background:var(--bg);color:var(--gris);white-space:nowrap}
33 +.pill-blue{background:#e6eef8;color:var(--bleu)}.pill-or{background:#fbf3d5;color:#7a6300}.pill-vert{background:#e2f3ea;color:var(--vert)}
34 +.pill-gris{background:#eceef1;color:var(--gris)}.pill-rouge{background:#fbe6e3;color:var(--rouge)}
35 +.btn{display:inline-flex;align-items:center;gap:6px;padding:8px 14px;border-radius:8px;border:1px solid var(--bleu);background:var(--bleu);color:#fff;font:inherit;font-weight:600;cursor:pointer;font-size:.9rem}
36 +.btn:hover{background:var(--bleu2);border-color:var(--bleu2);text-decoration:none}
37 +.btn-ghost{background:#fff;color:var(--bleu);border-color:var(--ligne)}.btn-ghost:hover{background:var(--bg);color:var(--bleu)}
38 +.btn-sm{padding:5px 10px;font-size:.82rem}.btn-or{background:var(--or);border-color:var(--or);color:var(--bleu)}
39 +.btn:disabled{opacity:.5;cursor:default}
40 +.grid{display:grid;gap:16px}.g2{grid-template-columns:repeat(2,minmax(0,1fr))}.g3{grid-template-columns:repeat(3,minmax(0,1fr))}.g4{grid-template-columns:repeat(4,minmax(0,1fr))}
41 +.card{background:var(--card);border:1px solid var(--ligne);border-radius:12px;padding:16px 18px}
42 +.card h2,.card h3{margin-bottom:10px}
43 +.kpi{display:flex;flex-direction:column;gap:2px}.kpi b{font-size:1.5rem;color:var(--bleu);font-weight:700}.kpi span{font-size:.8rem;color:var(--gris2)}
44 +.kpis{display:grid;grid-template-columns:repeat(auto-fit,minmax(150px,1fr));gap:12px;margin-bottom:16px}
45 +.chart{position:relative;height:290px}.chart.tall{height:380px}.chart.short{height:220px}
46 +.page-head{display:flex;align-items:flex-end;justify-content:space-between;gap:16px;margin-bottom:16px;flex-wrap:wrap}
47 +.page-head p{margin:.2em 0 0;color:var(--gris)}
48 +.filters{display:grid;grid-template-columns:repeat(auto-fill,minmax(170px,1fr));gap:10px;margin-bottom:14px}
49 +.filters label{display:flex;flex-direction:column;gap:3px;font-size:.75rem;color:var(--gris2);text-transform:uppercase;letter-spacing:.04em}
50 +.filters input,.filters select,.inp{padding:7px 9px;border:1px solid var(--ligne);border-radius:7px;font:inherit;font-size:.9rem;background:#fff;color:var(--encre)}
51 +.filters input:focus,.filters select:focus,.inp:focus,textarea:focus{outline:2px solid var(--bleu2)}
52 +.filters .span2{grid-column:span 2}
53 +.table-wrap{overflow:auto;border:1px solid var(--ligne);border-radius:12px;background:#fff}
54 +table{border-collapse:collapse;width:100%;font-size:.88rem}
55 +thead th{position:sticky;top:0;background:var(--bleu);color:#fff;text-align:left;padding:9px 11px;font-weight:600;white-space:nowrap;font-size:.8rem;cursor:pointer;user-select:none}
56 +thead th.sorted::after{content:" ▾";opacity:.8}thead th.sorted.asc::after{content:" ▴"}
57 +tbody td{padding:8px 11px;border-top:1px solid var(--ligne);vertical-align:top}
58 +tbody tr:hover{background:#f7f9fd}tbody tr.click{cursor:pointer}
59 +td.num,th.num{text-align:right;font-variant-numeric:tabular-nums}
60 +td.trunc{max-width:420px;overflow:hidden;text-overflow:ellipsis;white-space:nowrap}
61 +.pager{display:flex;align-items:center;gap:10px;padding:10px 4px;font-size:.85rem;color:var(--gris)}
62 +.pager .sp{flex:1}
63 +.tags{display:flex;flex-wrap:wrap;gap:5px}.tag{background:var(--bg);border:1px solid var(--ligne);border-radius:6px;padding:2px 7px;font-size:.78rem}
64 +.tag.t-tech{background:#e6f2f2;border-color:#bfe0e0;color:var(--teal)}.tag.t-loc{background:#fdf1e4;border-color:#f3d5b5;color:#8a4b00}
65 +.tag.t-part{background:#efe9fb;border-color:#d9cdf5;color:var(--violet)}.tag.t-risk{background:#fbe6e3;border-color:#f3c5c0;color:var(--rouge)}.tag.t-ben{background:#e2f3ea;border-color:#bfe3cf;color:var(--vert)}
66 +.st{font-weight:600}.st-planned{color:#7a6300}.st-in_progress{color:var(--bleu2)}.st-completed{color:var(--vert)}.st-mentioned{color:var(--gris2)}
67 +.dl{display:grid;grid-template-columns:150px 1fr;gap:6px 14px;font-size:.9rem}.dl dt{color:var(--gris2)}.dl dd{margin:0}
68 +.timeline{list-style:none;margin:0;padding:0 0 0 14px;border-left:2px solid var(--ligne)}
69 +.timeline li{position:relative;padding:0 0 14px 16px}
70 +.timeline li::before{content:"";position:absolute;left:-20px;top:5px;width:10px;height:10px;border-radius:50%;background:var(--bleu2);border:2px solid #fff}
71 +.timeline .when{font-size:.8rem;color:var(--gris2)}.timeline .snip{color:var(--gris);font-size:.88rem;margin-top:3px}
72 +.mention{border:1px solid var(--ligne);border-radius:10px;padding:12px 14px;margin-bottom:10px;background:#fff}
73 +.mention .mh{display:flex;gap:10px;align-items:center;flex-wrap:wrap;margin-bottom:6px}
74 +.mention p{margin:.3em 0}
75 +textarea.sql{width:100%;min-height:150px;font-family:var(--mono);font-size:.88rem;padding:10px 12px;border:1px solid var(--ligne);border-radius:9px;background:#fff;resize:vertical;tab-size:2}
76 +.sql-layout{display:grid;grid-template-columns:260px 1fr;gap:16px}
77 +.schema{font-size:.82rem;max-height:70vh;overflow:auto}
78 +.schema details{margin-bottom:6px}.schema summary{cursor:pointer;font-weight:600;color:var(--bleu)}
79 +.schema ul{list-style:none;margin:4px 0 0;padding:0 0 0 10px}.schema li{font-family:var(--mono);font-size:.76rem;color:var(--gris);padding:1px 0}
80 +.schema li span{color:var(--gris2)}
81 +.examples{display:flex;flex-wrap:wrap;gap:6px;margin:8px 0}
82 +.examples button{background:#fff;border:1px solid var(--ligne);border-radius:6px;padding:4px 9px;font:inherit;font-size:.8rem;cursor:pointer;color:var(--bleu)}
83 +.examples button:hover{background:var(--bg)}
84 +pre.code{background:#0f172a;color:#e2e8f0;border-radius:10px;padding:14px 16px;overflow:auto;font-family:var(--mono);font-size:.82rem;line-height:1.5;margin:0}
85 +pre.code .k{color:#7dd3fc}pre.code .s{color:#fcd34d}
86 +.err{background:#fbe6e3;color:var(--rouge);border:1px solid #f3c5c0;border-radius:8px;padding:10px 12px;margin:8px 0}
87 +.ok{background:#e2f3ea;color:var(--vert);border:1px solid #bfe3cf;border-radius:8px;padding:8px 12px}
88 +.api-layout{display:grid;grid-template-columns:300px 1fr;gap:16px}
89 +.endpoints{list-style:none;margin:0;padding:0;max-height:70vh;overflow:auto}
90 +.endpoints li{padding:7px 10px;border-radius:7px;cursor:pointer;font-size:.86rem;display:flex;gap:8px;align-items:baseline}
91 +.endpoints li:hover,.endpoints li.active{background:var(--bg)}.endpoints .m{font-family:var(--mono);font-size:.7rem;font-weight:700;color:var(--vert);min-width:34px}
92 +.endpoints .m.post{color:var(--orange)}
93 +.params{display:grid;grid-template-columns:repeat(auto-fill,minmax(180px,1fr));gap:10px;margin:10px 0}
94 +.params label{display:flex;flex-direction:column;gap:3px;font-size:.75rem;color:var(--gris2)}.params label b{color:var(--encre);font-weight:600;font-family:var(--mono);font-size:.78rem}
95 +.tabs{display:flex;gap:4px;border-bottom:1px solid var(--ligne);margin-bottom:10px}
96 +.tabs button{background:none;border:none;border-bottom:2px solid transparent;padding:8px 12px;font:inherit;cursor:pointer;color:var(--gris)}
97 +.tabs button.active{color:var(--bleu);border-bottom-color:var(--bleu);font-weight:600}
98 +.toast{position:fixed;bottom:22px;left:50%;transform:translateX(-50%);background:var(--encre);color:#fff;padding:10px 16px;border-radius:9px;font-size:.88rem;box-shadow:0 8px 30px rgba(0,0,0,.25);z-index:50}
99 +.node-chip{display:inline-flex;align-items:center;gap:6px;padding:5px 10px;border-radius:8px;border:1px solid var(--ligne);background:#fff;margin:3px;cursor:pointer;font-size:.85rem}
100 +.node-chip:hover{background:var(--bg)}.node-chip .nt{font-size:.68rem;text-transform:uppercase;letter-spacing:.06em;color:var(--gris2)}
101 +.nt-Company{border-left:4px solid var(--bleu)}.nt-Project{border-left:4px solid var(--violet)}.nt-Location{border-left:4px solid var(--orange)}.nt-Technology{border-left:4px solid var(--teal)}.nt-Partner{border-left:4px solid var(--or)}
102 +.svg-graph{width:100%;height:520px;background:#fff;border:1px solid var(--ligne);border-radius:12px}
103 +.svg-graph text{font-family:var(--sans);font-size:11px;fill:var(--encre)}
104 +.svg-graph line{stroke:#c9d2e0;stroke-width:1.2}
105 +.prose{max-width:860px}.prose p{margin:.5em 0 .9em}.prose li{margin:.25em 0}
106 +.hero{background:linear-gradient(120deg,var(--bleu),#0b4f9c 60%,#0f66b8);color:#fff;border-radius:14px;padding:22px 26px;margin-bottom:18px;display:flex;justify-content:space-between;gap:20px;align-items:center;flex-wrap:wrap}
107 +.hero h1{color:#fff;margin-bottom:6px}.hero p{margin:0;opacity:.9;max-width:700px}
108 +.hero .btn{background:var(--or);border-color:var(--or);color:var(--bleu)}
109 +.score{font-variant-numeric:tabular-nums;color:var(--teal);font-weight:600}
110 +.warn{background:#fbf3d5;border:1px solid #efdc9a;color:#6b5600;border-radius:8px;padding:8px 12px;font-size:.86rem}
111 +.sec-text{white-space:pre-wrap;font-size:.86rem;line-height:1.55;max-height:60vh;overflow:auto;background:#fff;border:1px solid var(--ligne);border-radius:10px;padding:14px}
112 +mark{background:#fff1a8;padding:0 2px;border-radius:2px}
113 +@media (max-width:1100px){.g4{grid-template-columns:repeat(2,1fr)}.g3{grid-template-columns:repeat(2,1fr)}.sql-layout,.api-layout{grid-template-columns:1fr}}
114 +@media (max-width:860px){
115 + .shell{grid-template-columns:1fr}.side{position:fixed;left:-260px;width:250px;transition:left .2s;z-index:20}.side.open{left:0}
116 + .burger{display:block}.g2,.g3,.g4{grid-template-columns:1fr}.view{padding:14px}.topbar{padding:10px 14px}.dl{grid-template-columns:1fr}
117 +}
118 +
119 +/* --- playground v2 --- */
120 +.langsel{display:flex;gap:4px;align-items:center;font-size:.8rem;color:var(--gris2)}
121 +.langsel button{background:#fff;border:1px solid var(--ligne);border-radius:6px;padding:4px 10px;font:inherit;font-size:.8rem;cursor:pointer;color:var(--gris)}
122 +.langsel button.active{background:var(--bleu);color:#fff;border-color:var(--bleu)}
123 +.tabs.big{margin-bottom:16px}.tabs.big button{font-size:.95rem;padding:9px 16px}
124 +.keybox{display:flex;gap:10px;align-items:flex-end;flex-wrap:wrap}.keybox label{display:flex;flex-direction:column;gap:3px;font-size:.78rem;color:var(--gris2);flex:1;min-width:260px}
125 +.keybox label b{color:var(--encre)}.keybox .inp{font-family:var(--mono)}
126 +.step{display:inline-flex;width:24px;height:24px;border-radius:50%;background:var(--or);color:var(--bleu);align-items:center;justify-content:center;font-size:.8rem;margin-right:6px;font-weight:800}
127 +.codewrap{position:relative}.codewrap pre.code{padding-right:80px}.codewrap .copy{position:absolute;top:8px;right:8px;background:rgba(255,255,255,.12);color:#fff;border:1px solid rgba(255,255,255,.2);border-radius:6px;padding:3px 9px;font-size:.72rem;cursor:pointer}
128 +.codewrap .copy:hover{background:rgba(255,255,255,.22)}
129 +pre.code.json .k{color:#7dd3fc}pre.code.json .s{color:#fcd34d}pre.code.json .n{color:#a5f3a5}pre.code.json .b{color:#f9a8d4}
130 +.endpoints li.grp{font-size:.68rem;text-transform:uppercase;letter-spacing:.1em;color:var(--gris2);cursor:default;padding-top:10px}
131 +.endpoints li.grp:hover{background:none}
132 +.params label span{font-weight:400;color:var(--gris2);font-family:var(--sans);font-size:.72rem}
133 +.params textarea.sql{min-height:110px}
134 +table.ref{width:100%;font-size:.85rem}table.ref td,table.ref th{padding:6px 9px;border-top:1px solid var(--ligne);vertical-align:top;text-align:left}
135 +table.ref thead th{background:var(--bg);color:var(--gris);position:static;cursor:default}
136 +.hist{display:flex;flex-direction:column;gap:6px;max-height:40vh;overflow:auto}.hist-it{border:1px solid var(--ligne);border-radius:7px;padding:6px 8px;cursor:pointer;font-size:.8rem}.hist-it:hover{background:var(--bg)}
137 +.recipe pre.code{max-height:360px}
138