Python 75.9%
TeX 24%
1#!/usr/bin/env python32"""3================================================================4Auteur : Simon-Pierre Boucher5Contact : contact@spboucher.ai6Projet : Prévision de volatilité réalisée multi-actifs7 (HAR-RV vs GARCH vs Machine Learning)8Fichier : 07_make_figures.py9Description : Étape 07 — Figures publication-ready du papier10 (séries RV, mesures, ACF, ratios QLIKE, MCS,11 sous-périodes, SHAP, Giacomini-Rossi, timing).12Usage : python scripts/07_make_figures.py13================================================================14"""1516from __future__ import annotations1718import logging19import sys20from pathlib import Path2122import matplotlib.pyplot as plt23import numpy as np24import pandas as pd2526sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))2728from wp12 import backtest as bt, config # noqa: E40229from wp12.plots import ( # noqa: E40230 BLUE, BLUES, GREY, INK, LIGHT, RED, TEXTWIDTH,31 annualized_vol, apply_style, panel_label,32)3334logger = logging.getLogger("07_figures")3536FIG = config.path("figures")37OUT = config.path("reproduced")38REPRESENTATIVE = {"equity": "SPY", "fx": "EURUSD", "crypto": "BTC", "futures": "ES"}39CLS_LABEL = {"equity": "Equities/ETFs", "fx": "FX", "crypto": "Crypto",40 "futures": "Futures"}4142# headline models shown in model-comparison figures, display order43SHOW_MODELS = [44 "GARCH", "GJR", "EGARCH", "RealGARCH",45 "HAR-J", "HAR-CJ", "SHAR", "HARQ", "LogHAR",46 "Ridge-H", "LASSO-H", "RF-H", "XGBoost-H", "LightGBM-H",47 "Ridge-X", "LASSO-X", "RF-X", "XGBoost-X", "LightGBM-X",48 "LSTM", "Transformer", "Pooled-LGBM", "Comb-Mean", "Comb-InvMSE",49]505152def _panel() -> pd.DataFrame:53 return pd.read_parquet(config.path("processed") / "panel.parquet")545556def fig_rv_series() -> None:57 """Annualized RV time series, one panel per asset class representative."""58 panel = _panel()59 fig, axes = plt.subplots(4, 1, figsize=(TEXTWIDTH, 6.6), sharex=True)60 for ax, (cls, tk) in zip(axes, REPRESENTATIVE.items()):61 df = panel[panel["ticker"] == tk].sort_index()62 vol = annualized_vol(df["rv5ss"])63 ax.plot(df.index, vol, lw=0.4, color=BLUE)64 panel_label(ax, f"Panel {'ABCD'[list(REPRESENTATIVE).index(cls)]}. "65 f"{tk} ({CLS_LABEL[cls]})")66 ax.set_ylabel("Ann. vol. (%)")67 # cap the axis at the 99.5th percentile so isolated early-sample68 # noise spikes (e.g. 2013 BTC) do not compress the whole panel69 ax.set_ylim(0, float(vol.quantile(0.995)) * 1.1)70 axes[-1].set_xlabel("")71 fig.tight_layout()72 fig.savefig(FIG / "fig_rv_series.png")73 plt.close(fig)747576def fig_measures() -> None:77 """Yearly mean annualized vol by RV measure — microstructure noise check."""78 panel = _panel()79 fig, axes = plt.subplots(1, 2, figsize=(TEXTWIDTH, 2.6))80 for ax, tk, letter in zip(axes, ["SPY", "BTC"], "AB"):81 df = panel[panel["ticker"] == tk]82 yearly = df.groupby(df.index.year)[["rv1", "rv5ss", "rk"]].mean()83 for col, color, ls, lab in [84 ("rv1", RED, "-", "RV 1-min"),85 ("rv5ss", BLUE, "-", "RV 5-min (subsampled)"),86 ("rk", INK, "--", "Realized kernel"),87 ]:88 ax.plot(yearly.index, annualized_vol(yearly[col]), color=color,89 ls=ls, lw=1.0, label=lab)90 panel_label(ax, f"Panel {letter}. {tk}")91 ax.set_ylabel("Mean ann. vol. (%)")92 if letter == "A":93 ax.legend(loc="upper right", fontsize=7)94 fig.tight_layout()95 fig.savefig(FIG / "fig_measures.png")96 plt.close(fig)979899def fig_acf() -> None:100 """Autocorrelation of log RV by class representative (long memory)."""101 panel = _panel()102 fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.8))103 colors = {"equity": BLUE, "fx": INK, "crypto": RED, "futures": GREY}104 for cls, tk in REPRESENTATIVE.items():105 s = np.log(panel[panel["ticker"] == tk]["rv5ss"].clip(lower=1e-12))106 acf = [s.autocorr(k) for k in range(1, 101)]107 ax.plot(range(1, 101), acf, lw=1.0, color=colors[cls],108 ls="--" if cls in ("fx", "futures") else "-",109 label=f"{tk} ({CLS_LABEL[cls]})")110 ax.axhline(0, color=GREY, lw=0.5)111 ax.set_xlabel("Lag (days)")112 ax.set_ylabel("ACF of log RV")113 ax.legend(fontsize=7)114 fig.tight_layout()115 fig.savefig(FIG / "fig_acf.png")116 plt.close(fig)117118119def fig_qlike_ratio() -> None:120 """QLIKE ratio to HAR by model and horizon (dot plot)."""121 tab = pd.read_csv(OUT / "losses_overall.csv")122 models = [m for m in SHOW_MODELS if m in set(tab["model"])]123 fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True)124 for ax, h in zip(axes, (1, 5, 22)):125 sub = tab[tab["h"] == h].set_index("model").reindex(models)126 y = np.arange(len(models))127 better = sub["qlike_ratio"] < 1128 ax.scatter(sub["qlike_ratio"], y, s=14,129 c=[BLUE if b else RED for b in better], zorder=3)130 ax.axvline(1.0, color=GREY, lw=0.7, ls="--")131 ax.set_title(f"h = {h}", fontsize=9)132 ax.set_xlabel("QLIKE ratio vs HAR")133 ax.set_yticks(y)134 ax.set_yticklabels(models, fontsize=7)135 ax.invert_yaxis()136 fig.tight_layout()137 fig.savefig(FIG / "fig_qlike_ratio.png")138 plt.close(fig)139140141def fig_mcs() -> None:142 """MCS inclusion frequency across assets, per horizon."""143 tab = pd.read_csv(OUT / "mcs_inclusion.csv")144 models = ["HAR"] + [m for m in SHOW_MODELS if m in set(tab["model"])]145 fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True)146 for ax, h in zip(axes, (1, 5, 22)):147 sub = tab[tab["h"] == h].set_index("model").reindex(models)148 y = np.arange(len(models))149 ax.barh(y, sub["mcs_inclusion"], color=BLUE, height=0.65)150 ax.set_yticks(y)151 ax.set_yticklabels(models, fontsize=7)152 ax.invert_yaxis()153 ax.set_xlim(0, 1)154 ax.set_title(f"h = {h}", fontsize=9)155 ax.set_xlabel("Share of assets in 90% MCS")156 fig.tight_layout()157 fig.savefig(FIG / "fig_mcs.png")158 plt.close(fig)159160161def fig_subperiods() -> None:162 """QLIKE ratio vs HAR across sub-periods for selected models (h=1)."""163 tab = pd.read_csv(OUT / "subperiod_losses.csv")164 tab = tab[tab["h"] == 1]165 bench = tab[tab["model"] == "HAR"].set_index("period")["qlike"]166 show = ["LogHAR", "RealGARCH", "LightGBM-X", "Pooled-LGBM", "Comb-InvMSE", "LSTM"]167 periods = ["pre2020", "covid", "inflation", "recent"]168 plabel = {"pre2020": "2013–19", "covid": "2020", "inflation": "2021–22",169 "recent": "2024–26"}170 fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.9))171 colors = [BLUE, INK, RED, BLUES[5], GREY, BLUES[2]]172 markers = ["o", "s", "^", "D", "v", "P"]173 for m, c, mk in zip(show, colors, markers):174 sub = tab[tab["model"] == m].set_index("period")175 ratio = (sub["qlike"] / bench).reindex(periods)176 ax.plot(range(len(periods)), ratio, marker=mk, ms=4, lw=1.0,177 color=c, label=m)178 ax.axhline(1.0, color=GREY, lw=0.7, ls="--")179 ax.set_xticks(range(len(periods)))180 ax.set_xticklabels([plabel[p] for p in periods])181 ax.set_ylabel("QLIKE ratio vs HAR (h=1)")182 ax.legend(fontsize=7, ncol=2)183 fig.tight_layout()184 fig.savefig(FIG / "fig_subperiods.png")185 plt.close(fig)186187188def fig_shap() -> None:189 """Mean |SHAP| of the pooled LightGBM (h=1)."""190 tab = pd.read_csv(OUT / "shap_h1.csv").head(14)191 fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.7, 3.2))192 y = np.arange(len(tab))193 ax.barh(y, tab["mean_abs_shap"], color=BLUE, height=0.65)194 ax.set_yticks(y)195 ax.set_yticklabels(tab["feature"], fontsize=7)196 ax.invert_yaxis()197 ax.set_xlabel(r"Mean $|$SHAP$|$ (log-RV units)")198 fig.tight_layout()199 fig.savefig(FIG / "fig_shap.png")200 plt.close(fig)201202203def fig_gr() -> None:204 """Giacomini-Rossi fluctuation paths, LightGBM-X vs HAR."""205 tab = pd.read_csv(OUT / "gr_fluctuation.csv", parse_dates=["date"])206 tab = tab[tab["model"] == "LightGBM-X"]207 fig, axes = plt.subplots(2, 2, figsize=(TEXTWIDTH, 4.4))208 reps = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"}209 for ax, (cls, tk), letter in zip(axes.ravel(), reps.items(), "ABCD"):210 sub = tab[tab["ticker"] == tk]211 if sub.empty:212 continue213 ax.plot(sub["date"], sub["stat"], lw=0.7, color=BLUE)214 crit = sub["crit"].iloc[0]215 ax.axhline(crit, color=GREY, lw=0.6, ls="--")216 ax.axhline(-crit, color=GREY, lw=0.6, ls="--")217 ax.axhline(0, color=GREY, lw=0.4)218 panel_label(ax, f"Panel {letter}. {tk} ({CLS_LABEL[cls]})")219 ax.set_ylabel("Fluctuation stat.")220 fig.tight_layout()221 fig.savefig(FIG / "fig_gr.png")222 plt.close(fig)223224225def fig_timing() -> None:226 """Cumulative net log returns of volatility timing on SPY."""227 merged = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet")228 merged["date"] = pd.to_datetime(merged["date"])229 panel = _panel()230 ret = panel[panel["ticker"] == "SPY"].sort_index()["ret_cc"]231 cfg = config.load_config()["backtest"]232 fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.85, 3.0))233 series = [("HAR", INK, "-"), ("GARCH", GREY, "-"),234 ("LightGBM-X", BLUE, "-"), ("Comb-InvMSE", RED, "--")]235 for m, color, ls in series:236 sub = merged[(merged["ticker"] == "SPY") & (merged["h"] == 1)237 & (merged["model"] == m)]238 if sub.empty:239 continue240 pv = sub.set_index("date")["forecast"]241 led = bt.volatility_timing(ret, pv, cfg["vol_target_ann"],242 cfg["leverage_cap"], cfg["tc_bps"])243 ax.plot(led.index, led["net"].cumsum() * 100, lw=0.9, color=color,244 ls=ls, label=m)245 ax.set_ylabel("Cumulative net return (%)")246 ax.legend(fontsize=7)247 fig.tight_layout()248 fig.savefig(FIG / "fig_timing.png")249 plt.close(fig)250251252def main() -> None:253 """Generate every figure of the paper."""254 config.setup_logging()255 apply_style()256 for fn in [fig_rv_series, fig_measures, fig_acf, fig_qlike_ratio,257 fig_mcs, fig_subperiods, fig_shap, fig_gr, fig_timing]:258 try:259 fn()260 logger.info("figure %s done", fn.__name__)261 except Exception as exc: # noqa: BLE001262 logger.error("figure %s FAILED: %s", fn.__name__, exc)263264265if __name__ == "__main__":266 main()267