SPB Git forge

spb/pdb-api

Public
1commits 1branches 0releases
436.0 KBsize
maindefault branch
2 h agolast push
JavaScript 58.4% Python 26.8% CSS 7.5% Objective-C 3% R 2.6% HTML 1.7%
4.8 KB · 114 lines r
Raw Blame History
1# =============================================================================2# pdb_api.R — utiliser la PDB API (SEC Project Intelligence Database, UQO) en R3#   install.packages(c("httr2", "dplyr", "ggplot2"))   # une seule fois4#   source("pdb_api.R")  ou  copier-coller ce fichier dans RStudio5# =============================================================================6library(httr2)7library(dplyr)89BASE <- "https://www.pdb-api.co/v1"10KEY  <- Sys.getenv("PDB_API_KEY", unset = "VOTRE_CLE")   # ou : KEY <- "VOTRE_CLE"1112# ---- 1. Deux fonctions génériques ------------------------------------------13pdb_get <- function(path, ...) {14  q <- Filter(Negate(is.null), list(...))15  req <- request(paste0(BASE, "/", path)) |>16    req_headers(`X-API-Key` = KEY) |>17    req_retry(max_tries = 4, is_transient = \(r) resp_status(r) == 429) |>18    req_error(body = \(r) tryCatch(resp_body_json(r)$detail, error = \(e) NULL))19  if (length(q)) req <- req |> req_url_query(!!!q)20  req |> req_perform() |> resp_body_json(simplifyVector = TRUE)21}2223pdb_sql <- function(sql, limit = 500) {24  res <- request(paste0(BASE, "/sql")) |>25    req_headers(`X-API-Key` = KEY) |>26    req_body_json(list(sql = sql, limit = limit)) |>27    req_perform() |> resp_body_json(simplifyVector = TRUE)28  df <- as.data.frame(res$rows, stringsAsFactors = FALSE)29  if (nrow(df)) names(df) <- res$columns30  df31}3233# Tout récupérer (pagination automatique, 500 par page) -> data.frame34pdb_all <- function(path, ...) {35  pages <- list(); offset <- 036  repeat {37    page <- pdb_get(path, ..., limit = 500, offset = offset)38    items <- as.data.frame(page$items)39    pages[[length(pages) + 1]] <- items40    offset <- offset + nrow(items)41    if (nrow(items) == 0 || offset >= page$total) break42  }43  bind_rows(pages)44}4546# ---- 2. Exemples -----------------------------------------------------------47if (sys.nframe() == 0) {   # exécuté seulement si on lance le fichier directement4849  # a) vérifier le service et la clé50  print(pdb_get("health"))51  ov <- pdb_get("stats")$overview52  cat(sprintf("%s projets · %s mentions · %s entreprises\n",53              format(ov$projects, big.mark = " "), format(ov$mentions, big.mark = " "), ov$companies))5455  # b) une page de projets filtrés (centres de données > 500 M$)56  dc <- pdb_get("projects", type = "data_center", min_amount = 5e8, sort = "amount", limit = 10)$items57  print(dc[, c("ticker", "project_name", "canonical_location", "total_amount_usd", "first_seen")])5859  # c) tout un secteur en data.frame, puis agrégation dplyr60  util <- pdb_all("projects", sector = "Utilities")61  util |>62    group_by(project_type) |>63    summarise(n = n(), capital_gusd = sum(total_amount_usd, na.rm = TRUE) / 1e9, .groups = "drop") |>64    arrange(desc(n)) |>65    print(n = 10)6667  # d) export CSV direct (le plus simple pour un jeu de données complet)68  csv <- request(paste0(BASE, "/projects/export.csv")) |>69    req_headers(`X-API-Key` = KEY) |>70    req_url_query(type = "plant_construction") |>71    req_perform() |> resp_body_string()72  usines <- read.csv(text = csv)73  cat("usines :", nrow(usines), "lignes\n")7475  # e) fiche d'un projet : chronologie et projets similaires76  pid <- pdb_get("projects", ticker = "TSLA", q = "energy storage", limit = 1)$items$project_id[1]77  fiche <- pdb_get(paste0("projects/", pid))78  cat(fiche$project$project_name, "-", fiche$project$status, "\n")79  print(fiche$timeline[, c("filing_date", "form_type", "status", "amount_usd")])80  print(fiche$similar[, c("score", "ticker", "project_name")])8182  # f) profil d'une entreprise83  duk <- pdb_get("companies/DUK")84  print(duk$by_type[, c("label", "n", "amount_usd")])85  barplot(duk$by_year$n, names.arg = duk$by_year$year, las = 2, main = "Duke Energy — mentions par année")8687  # g) recherche sémantique (langage naturel, anglais recommandé)88  sem <- pdb_get("search/semantic", q = "battery cell factory", k = 5)$items89  print(sem[, c("score", "ticker", "project_name")])9091  # h) SQL libre (lecture seule, DuckDB)92  ia <- pdb_sql("93    select year(filing_date) as yr, count(*) n94    from project_mentions95    where project_type = 'ai_initiative'96    group by 1 order by 1")97  print(ia)98  plot(ia$yr, ia$n, type = "b", xlab = "année", ylab = "mentions", main = "Initiatives IA dans les filings")99100  # i) panel entreprise × année pour l'économétrie101  panel <- pdb_sql("102    select cik, any_value(ticker) ticker, any_value(sector) sector,103           year(first_seen) as yr, count(*) n_projets,104           sum(total_amount_usd) capital_usd105    from projects group by 1, 4 order by 1, 4", limit = 5000)106  cat("panel :", nrow(panel), "lignes (entreprise × année)\n")107  # summary(lm(log1p(n_projets) ~ factor(yr) + factor(sector), data = panel))108109  # j) graphe : projets rattachés au Texas110  tx <- pdb_get("graph/node/L:texas", limit = 500)111  cat("degré de L:texas :", tx$degree, "\n")112  print(head(tx$edges[, c("rel", "neighbor_type", "neighbor_label")]))113}114