# ============================================================================= # pdb_api.R — utiliser la PDB API (SEC Project Intelligence Database, UQO) en R # install.packages(c("httr2", "dplyr", "ggplot2")) # une seule fois # source("pdb_api.R") ou copier-coller ce fichier dans RStudio # ============================================================================= library(httr2) library(dplyr) BASE <- "https://www.pdb-api.co/v1" KEY <- Sys.getenv("PDB_API_KEY", unset = "VOTRE_CLE") # ou : KEY <- "VOTRE_CLE" # ---- 1. Deux fonctions génériques ------------------------------------------ pdb_get <- function(path, ...) { q <- Filter(Negate(is.null), list(...)) req <- request(paste0(BASE, "/", path)) |> req_headers(`X-API-Key` = KEY) |> req_retry(max_tries = 4, is_transient = \(r) resp_status(r) == 429) |> req_error(body = \(r) tryCatch(resp_body_json(r)$detail, error = \(e) NULL)) if (length(q)) req <- req |> req_url_query(!!!q) req |> req_perform() |> resp_body_json(simplifyVector = TRUE) } pdb_sql <- function(sql, limit = 500) { res <- request(paste0(BASE, "/sql")) |> req_headers(`X-API-Key` = KEY) |> req_body_json(list(sql = sql, limit = limit)) |> req_perform() |> resp_body_json(simplifyVector = TRUE) df <- as.data.frame(res$rows, stringsAsFactors = FALSE) if (nrow(df)) names(df) <- res$columns df } # Tout récupérer (pagination automatique, 500 par page) -> data.frame pdb_all <- function(path, ...) { pages <- list(); offset <- 0 repeat { page <- pdb_get(path, ..., limit = 500, offset = offset) items <- as.data.frame(page$items) pages[[length(pages) + 1]] <- items offset <- offset + nrow(items) if (nrow(items) == 0 || offset >= page$total) break } bind_rows(pages) } # ---- 2. Exemples ----------------------------------------------------------- if (sys.nframe() == 0) { # exécuté seulement si on lance le fichier directement # a) vérifier le service et la clé print(pdb_get("health")) ov <- pdb_get("stats")$overview cat(sprintf("%s projets · %s mentions · %s entreprises\n", format(ov$projects, big.mark = " "), format(ov$mentions, big.mark = " "), ov$companies)) # b) une page de projets filtrés (centres de données > 500 M$) dc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items print(dc[, c("ticker", "project_name", "canonical_location", "total_amount_usd", "first_seen")]) # c) tout un secteur en data.frame, puis agrégation dplyr util <- pdb_all("projects", sector = "Utilities") util |> group_by(project_type) |> summarise(n = n(), capital_gusd = sum(total_amount_usd, na.rm = TRUE) / 1e9, .groups = "drop") |> arrange(desc(n)) |> print(n = 10) # d) export CSV direct (le plus simple pour un jeu de données complet) csv <- request(paste0(BASE, "/projects/export.csv")) |> req_headers(`X-API-Key` = KEY) |> req_url_query(type = "plant_construction") |> req_perform() |> resp_body_string() usines <- read.csv(text = csv) cat("usines :", nrow(usines), "lignes\n") # e) fiche d'un projet : chronologie et projets similaires pid <- pdb_get("projects", ticker = "TSLA", q = "energy storage", limit = 1)$items$project_id[1] fiche <- pdb_get(paste0("projects/", pid)) cat(fiche$project$project_name, "-", fiche$project$status, "\n") print(fiche$timeline[, c("filing_date", "form_type", "status", "amount_usd")]) print(fiche$similar[, c("score", "ticker", "project_name")]) # f) profil d'une entreprise duk <- pdb_get("companies/DUK") print(duk$by_type[, c("label", "n", "amount_usd")]) barplot(duk$by_year$n, names.arg = duk$by_year$year, las = 2, main = "Duke Energy — mentions par année") # g) recherche sémantique (langage naturel, anglais recommandé) sem <- pdb_get("search/semantic", q = "battery cell factory", k = 5)$items print(sem[, c("score", "ticker", "project_name")]) # h) SQL libre (lecture seule, DuckDB) ia <- pdb_sql(" select year(filing_date) as yr, count(*) n from project_mentions where project_type = 'ai_initiative' group by 1 order by 1") print(ia) plot(ia$yr, ia$n, type = "b", xlab = "année", ylab = "mentions", main = "Initiatives IA dans les filings") # i) panel entreprise × année pour l'économétrie panel <- pdb_sql(" select cik, any_value(ticker) ticker, any_value(sector) sector, year(first_seen) as yr, count(*) n_projets, sum(total_amount_usd) capital_usd from projects group by 1, 4 order by 1, 4", limit = 5000) cat("panel :", nrow(panel), "lignes (entreprise × année)\n") # summary(lm(log1p(n_projets) ~ factor(yr) + factor(sector), data = panel)) # j) graphe : projets rattachés au Texas tx <- pdb_get("graph/node/L:texas", limit = 500) cat("degré de L:texas :", tx$degree, "\n") print(head(tx$edges[, c("rel", "neighbor_type", "neighbor_label")])) }