#!/usr/bin/env python3 """ ================================================================ Auteur : Simon-Pierre Boucher Contact : contact@spboucher.ai Projet : Prévision de volatilité réalisée multi-actifs (HAR-RV vs GARCH vs Machine Learning) Fichier : 07_make_figures.py Description : Étape 07 — Figures publication-ready du papier (séries RV, mesures, ACF, ratios QLIKE, MCS, sous-périodes, SHAP, Giacomini-Rossi, timing). Usage : python scripts/07_make_figures.py ================================================================ """ from __future__ import annotations import logging import sys from pathlib import Path import matplotlib.pyplot as plt import numpy as np import pandas as pd sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from wp12 import backtest as bt, config # noqa: E402 from wp12.plots import ( # noqa: E402 BLUE, BLUES, GREY, INK, LIGHT, RED, TEXTWIDTH, annualized_vol, apply_style, panel_label, ) logger = logging.getLogger("07_figures") FIG = config.path("figures") OUT = config.path("reproduced") REPRESENTATIVE = {"equity": "SPY", "fx": "EURUSD", "crypto": "BTC", "futures": "ES"} CLS_LABEL = {"equity": "Equities/ETFs", "fx": "FX", "crypto": "Crypto", "futures": "Futures"} # headline models shown in model-comparison figures, display order SHOW_MODELS = [ "GARCH", "GJR", "EGARCH", "RealGARCH", "HAR-J", "HAR-CJ", "SHAR", "HARQ", "LogHAR", "Ridge-H", "LASSO-H", "RF-H", "XGBoost-H", "LightGBM-H", "Ridge-X", "LASSO-X", "RF-X", "XGBoost-X", "LightGBM-X", "LSTM", "Transformer", "Pooled-LGBM", "Comb-Mean", "Comb-InvMSE", ] def _panel() -> pd.DataFrame: return pd.read_parquet(config.path("processed") / "panel.parquet") def fig_rv_series() -> None: """Annualized RV time series, one panel per asset class representative.""" panel = _panel() fig, axes = plt.subplots(4, 1, figsize=(TEXTWIDTH, 6.6), sharex=True) for ax, (cls, tk) in zip(axes, REPRESENTATIVE.items()): df = panel[panel["ticker"] == tk].sort_index() vol = annualized_vol(df["rv5ss"]) ax.plot(df.index, vol, lw=0.4, color=BLUE) panel_label(ax, f"Panel {'ABCD'[list(REPRESENTATIVE).index(cls)]}. " f"{tk} ({CLS_LABEL[cls]})") ax.set_ylabel("Ann. vol. (%)") # cap the axis at the 99.5th percentile so isolated early-sample # noise spikes (e.g. 2013 BTC) do not compress the whole panel ax.set_ylim(0, float(vol.quantile(0.995)) * 1.1) axes[-1].set_xlabel("") fig.tight_layout() fig.savefig(FIG / "fig_rv_series.png") plt.close(fig) def fig_measures() -> None: """Yearly mean annualized vol by RV measure — microstructure noise check.""" panel = _panel() fig, axes = plt.subplots(1, 2, figsize=(TEXTWIDTH, 2.6)) for ax, tk, letter in zip(axes, ["SPY", "BTC"], "AB"): df = panel[panel["ticker"] == tk] yearly = df.groupby(df.index.year)[["rv1", "rv5ss", "rk"]].mean() for col, color, ls, lab in [ ("rv1", RED, "-", "RV 1-min"), ("rv5ss", BLUE, "-", "RV 5-min (subsampled)"), ("rk", INK, "--", "Realized kernel"), ]: ax.plot(yearly.index, annualized_vol(yearly[col]), color=color, ls=ls, lw=1.0, label=lab) panel_label(ax, f"Panel {letter}. {tk}") ax.set_ylabel("Mean ann. vol. (%)") if letter == "A": ax.legend(loc="upper right", fontsize=7) fig.tight_layout() fig.savefig(FIG / "fig_measures.png") plt.close(fig) def fig_acf() -> None: """Autocorrelation of log RV by class representative (long memory).""" panel = _panel() fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.8)) colors = {"equity": BLUE, "fx": INK, "crypto": RED, "futures": GREY} for cls, tk in REPRESENTATIVE.items(): s = np.log(panel[panel["ticker"] == tk]["rv5ss"].clip(lower=1e-12)) acf = [s.autocorr(k) for k in range(1, 101)] ax.plot(range(1, 101), acf, lw=1.0, color=colors[cls], ls="--" if cls in ("fx", "futures") else "-", label=f"{tk} ({CLS_LABEL[cls]})") ax.axhline(0, color=GREY, lw=0.5) ax.set_xlabel("Lag (days)") ax.set_ylabel("ACF of log RV") ax.legend(fontsize=7) fig.tight_layout() fig.savefig(FIG / "fig_acf.png") plt.close(fig) def fig_qlike_ratio() -> None: """QLIKE ratio to HAR by model and horizon (dot plot).""" tab = pd.read_csv(OUT / "losses_overall.csv") models = [m for m in SHOW_MODELS if m in set(tab["model"])] fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True) for ax, h in zip(axes, (1, 5, 22)): sub = tab[tab["h"] == h].set_index("model").reindex(models) y = np.arange(len(models)) better = sub["qlike_ratio"] < 1 ax.scatter(sub["qlike_ratio"], y, s=14, c=[BLUE if b else RED for b in better], zorder=3) ax.axvline(1.0, color=GREY, lw=0.7, ls="--") ax.set_title(f"h = {h}", fontsize=9) ax.set_xlabel("QLIKE ratio vs HAR") ax.set_yticks(y) ax.set_yticklabels(models, fontsize=7) ax.invert_yaxis() fig.tight_layout() fig.savefig(FIG / "fig_qlike_ratio.png") plt.close(fig) def fig_mcs() -> None: """MCS inclusion frequency across assets, per horizon.""" tab = pd.read_csv(OUT / "mcs_inclusion.csv") models = ["HAR"] + [m for m in SHOW_MODELS if m in set(tab["model"])] fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True) for ax, h in zip(axes, (1, 5, 22)): sub = tab[tab["h"] == h].set_index("model").reindex(models) y = np.arange(len(models)) ax.barh(y, sub["mcs_inclusion"], color=BLUE, height=0.65) ax.set_yticks(y) ax.set_yticklabels(models, fontsize=7) ax.invert_yaxis() ax.set_xlim(0, 1) ax.set_title(f"h = {h}", fontsize=9) ax.set_xlabel("Share of assets in 90% MCS") fig.tight_layout() fig.savefig(FIG / "fig_mcs.png") plt.close(fig) def fig_subperiods() -> None: """QLIKE ratio vs HAR across sub-periods for selected models (h=1).""" tab = pd.read_csv(OUT / "subperiod_losses.csv") tab = tab[tab["h"] == 1] bench = tab[tab["model"] == "HAR"].set_index("period")["qlike"] show = ["LogHAR", "RealGARCH", "LightGBM-X", "Pooled-LGBM", "Comb-InvMSE", "LSTM"] periods = ["pre2020", "covid", "inflation", "recent"] plabel = {"pre2020": "2013–19", "covid": "2020", "inflation": "2021–22", "recent": "2024–26"} fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.9)) colors = [BLUE, INK, RED, BLUES[5], GREY, BLUES[2]] markers = ["o", "s", "^", "D", "v", "P"] for m, c, mk in zip(show, colors, markers): sub = tab[tab["model"] == m].set_index("period") ratio = (sub["qlike"] / bench).reindex(periods) ax.plot(range(len(periods)), ratio, marker=mk, ms=4, lw=1.0, color=c, label=m) ax.axhline(1.0, color=GREY, lw=0.7, ls="--") ax.set_xticks(range(len(periods))) ax.set_xticklabels([plabel[p] for p in periods]) ax.set_ylabel("QLIKE ratio vs HAR (h=1)") ax.legend(fontsize=7, ncol=2) fig.tight_layout() fig.savefig(FIG / "fig_subperiods.png") plt.close(fig) def fig_shap() -> None: """Mean |SHAP| of the pooled LightGBM (h=1).""" tab = pd.read_csv(OUT / "shap_h1.csv").head(14) fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.7, 3.2)) y = np.arange(len(tab)) ax.barh(y, tab["mean_abs_shap"], color=BLUE, height=0.65) ax.set_yticks(y) ax.set_yticklabels(tab["feature"], fontsize=7) ax.invert_yaxis() ax.set_xlabel(r"Mean $|$SHAP$|$ (log-RV units)") fig.tight_layout() fig.savefig(FIG / "fig_shap.png") plt.close(fig) def fig_gr() -> None: """Giacomini-Rossi fluctuation paths, LightGBM-X vs HAR.""" tab = pd.read_csv(OUT / "gr_fluctuation.csv", parse_dates=["date"]) tab = tab[tab["model"] == "LightGBM-X"] fig, axes = plt.subplots(2, 2, figsize=(TEXTWIDTH, 4.4)) reps = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"} for ax, (cls, tk), letter in zip(axes.ravel(), reps.items(), "ABCD"): sub = tab[tab["ticker"] == tk] if sub.empty: continue ax.plot(sub["date"], sub["stat"], lw=0.7, color=BLUE) crit = sub["crit"].iloc[0] ax.axhline(crit, color=GREY, lw=0.6, ls="--") ax.axhline(-crit, color=GREY, lw=0.6, ls="--") ax.axhline(0, color=GREY, lw=0.4) panel_label(ax, f"Panel {letter}. {tk} ({CLS_LABEL[cls]})") ax.set_ylabel("Fluctuation stat.") fig.tight_layout() fig.savefig(FIG / "fig_gr.png") plt.close(fig) def fig_timing() -> None: """Cumulative net log returns of volatility timing on SPY.""" merged = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet") merged["date"] = pd.to_datetime(merged["date"]) panel = _panel() ret = panel[panel["ticker"] == "SPY"].sort_index()["ret_cc"] cfg = config.load_config()["backtest"] fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.85, 3.0)) series = [("HAR", INK, "-"), ("GARCH", GREY, "-"), ("LightGBM-X", BLUE, "-"), ("Comb-InvMSE", RED, "--")] for m, color, ls in series: sub = merged[(merged["ticker"] == "SPY") & (merged["h"] == 1) & (merged["model"] == m)] if sub.empty: continue pv = sub.set_index("date")["forecast"] led = bt.volatility_timing(ret, pv, cfg["vol_target_ann"], cfg["leverage_cap"], cfg["tc_bps"]) ax.plot(led.index, led["net"].cumsum() * 100, lw=0.9, color=color, ls=ls, label=m) ax.set_ylabel("Cumulative net return (%)") ax.legend(fontsize=7) fig.tight_layout() fig.savefig(FIG / "fig_timing.png") plt.close(fig) def main() -> None: """Generate every figure of the paper.""" config.setup_logging() apply_style() for fn in [fig_rv_series, fig_measures, fig_acf, fig_qlike_ratio, fig_mcs, fig_subperiods, fig_shap, fig_gr, fig_timing]: try: fn() logger.info("figure %s done", fn.__name__) except Exception as exc: # noqa: BLE001 logger.error("figure %s FAILED: %s", fn.__name__, exc) if __name__ == "__main__": main()