JavaScript 58.4%
Python 26.8%
CSS 7.5%
Objective-C 3%
R 2.6%
HTML 1.7%
1# =============================================================================2# pdb_api.R — utiliser la PDB API (SEC Project Intelligence Database, UQO) en R3# install.packages(c("httr2", "dplyr", "ggplot2")) # une seule fois4# source("pdb_api.R") ou copier-coller ce fichier dans RStudio5# =============================================================================6library(httr2)7library(dplyr)89BASE <- "https://www.pdb-api.co/v1"10KEY <- Sys.getenv("PDB_API_KEY", unset = "VOTRE_CLE") # ou : KEY <- "VOTRE_CLE"1112# ---- 1. Deux fonctions génériques ------------------------------------------13pdb_get <- function(path, ...) {14 q <- Filter(Negate(is.null), list(...))15 req <- request(paste0(BASE, "/", path)) |>16 req_headers(`X-API-Key` = KEY) |>17 req_retry(max_tries = 4, is_transient = \(r) resp_status(r) == 429) |>18 req_error(body = \(r) tryCatch(resp_body_json(r)$detail, error = \(e) NULL))19 if (length(q)) req <- req |> req_url_query(!!!q)20 req |> req_perform() |> resp_body_json(simplifyVector = TRUE)21}2223pdb_sql <- function(sql, limit = 500) {24 res <- request(paste0(BASE, "/sql")) |>25 req_headers(`X-API-Key` = KEY) |>26 req_body_json(list(sql = sql, limit = limit)) |>27 req_perform() |> resp_body_json(simplifyVector = TRUE)28 df <- as.data.frame(res$rows, stringsAsFactors = FALSE)29 if (nrow(df)) names(df) <- res$columns30 df31}3233# Tout récupérer (pagination automatique, 500 par page) -> data.frame34pdb_all <- function(path, ...) {35 pages <- list(); offset <- 036 repeat {37 page <- pdb_get(path, ..., limit = 500, offset = offset)38 items <- as.data.frame(page$items)39 pages[[length(pages) + 1]] <- items40 offset <- offset + nrow(items)41 if (nrow(items) == 0 || offset >= page$total) break42 }43 bind_rows(pages)44}4546# ---- 2. Exemples -----------------------------------------------------------47if (sys.nframe() == 0) { # exécuté seulement si on lance le fichier directement4849 # a) vérifier le service et la clé50 print(pdb_get("health"))51 ov <- pdb_get("stats")$overview52 cat(sprintf("%s projets · %s mentions · %s entreprises\n",53 format(ov$projects, big.mark = " "), format(ov$mentions, big.mark = " "), ov$companies))5455 # b) une page de projets filtrés (centres de données > 500 M$)56 dc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items57 print(dc[, c("ticker", "project_name", "canonical_location", "total_amount_usd", "first_seen")])5859 # c) tout un secteur en data.frame, puis agrégation dplyr60 util <- pdb_all("projects", sector = "Utilities")61 util |>62 group_by(project_type) |>63 summarise(n = n(), capital_gusd = sum(total_amount_usd, na.rm = TRUE) / 1e9, .groups = "drop") |>64 arrange(desc(n)) |>65 print(n = 10)6667 # d) export CSV direct (le plus simple pour un jeu de données complet)68 csv <- request(paste0(BASE, "/projects/export.csv")) |>69 req_headers(`X-API-Key` = KEY) |>70 req_url_query(type = "plant_construction") |>71 req_perform() |> resp_body_string()72 usines <- read.csv(text = csv)73 cat("usines :", nrow(usines), "lignes\n")7475 # e) fiche d'un projet : chronologie et projets similaires76 pid <- pdb_get("projects", ticker = "TSLA", q = "energy storage", limit = 1)$items$project_id[1]77 fiche <- pdb_get(paste0("projects/", pid))78 cat(fiche$project$project_name, "-", fiche$project$status, "\n")79 print(fiche$timeline[, c("filing_date", "form_type", "status", "amount_usd")])80 print(fiche$similar[, c("score", "ticker", "project_name")])8182 # f) profil d'une entreprise83 duk <- pdb_get("companies/DUK")84 print(duk$by_type[, c("label", "n", "amount_usd")])85 barplot(duk$by_year$n, names.arg = duk$by_year$year, las = 2, main = "Duke Energy — mentions par année")8687 # g) recherche sémantique (langage naturel, anglais recommandé)88 sem <- pdb_get("search/semantic", q = "battery cell factory", k = 5)$items89 print(sem[, c("score", "ticker", "project_name")])9091 # h) SQL libre (lecture seule, DuckDB)92 ia <- pdb_sql("93 select year(filing_date) as yr, count(*) n94 from project_mentions95 where project_type = 'ai_initiative'96 group by 1 order by 1")97 print(ia)98 plot(ia$yr, ia$n, type = "b", xlab = "année", ylab = "mentions", main = "Initiatives IA dans les filings")99100 # i) panel entreprise × année pour l'économétrie101 panel <- pdb_sql("102 select cik, any_value(ticker) ticker, any_value(sector) sector,103 year(first_seen) as yr, count(*) n_projets,104 sum(total_amount_usd) capital_usd105 from projects group by 1, 4 order by 1, 4", limit = 5000)106 cat("panel :", nrow(panel), "lignes (entreprise × année)\n")107 # summary(lm(log1p(n_projets) ~ factor(yr) + factor(sector), data = panel))108109 # j) graphe : projets rattachés au Texas110 tx <- pdb_get("graph/node/L:texas", limit = 500)111 cat("degré de L:texas :", tx$degree, "\n")112 print(head(tx$edges[, c("rel", "neighbor_type", "neighbor_label")]))113}114