#!/usr/bin/env python3 """ ================================================================ Auteur : Simon-Pierre Boucher Contact : contact@spboucher.ai Projet : Prévision de volatilité réalisée multi-actifs (HAR-RV vs GARCH vs Machine Learning) Fichier : 05_robustness.py Description : Étape 05 — Robustesse : sous-périodes, sensibilité à la fréquence d'échantillonnage et à la fenêtre d'estimation, transférabilité cross-asset (pooled leave-one-out) et fluctuation test Giacomini-Rossi. Usage : python scripts/05_robustness.py [--part all|subperiods|freq|window|loo|gr] ================================================================ """ from __future__ import annotations import argparse import logging import sys from concurrent.futures import ProcessPoolExecutor, as_completed from pathlib import Path import numpy as np import pandas as pd sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from wp12 import config, evaluation as ev, robustness as rb # noqa: E402 from wp12.models_econ import horizon_target # noqa: E402 logger = logging.getLogger("05_robustness") OUT = config.path("reproduced") HORIZONS = tuple(config.load_config()["evaluation"]["horizons"]) WINDOW = int(config.load_config()["evaluation"]["estimation_window_days"]) REFIT = int(config.load_config()["evaluation"]["reestimation_freq_days"]) # representative asset per class for the expensive experiments REPRESENTATIVE = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"} # models re-run in the sensitivity experiments SENS_MODELS = ["HAR", "LogHAR", "LightGBM"] def _merged() -> pd.DataFrame: df = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet") df["date"] = pd.to_datetime(df["date"]) return df def part_subperiods() -> None: """Mean QLIKE per model within each sub-period of the config.""" sp = {k: tuple(v) for k, v in config.load_config()["robustness"]["subperiods"].items()} merged = _merged() tab = rb.loss_by_period(merged, sp) tab.to_csv(OUT / "subperiod_losses.csv", index=False) logger.info("subperiod table written (%d rows)", len(tab)) def _sens_one(job: tuple[str, str, str, int]) -> pd.DataFrame: """Worker: re-run one (ticker, model, measure, window) variant.""" from wp12 import models_econ as me, models_ml as ml ticker, model, measure, window = job df = pd.read_parquet(config.path("processed") / f"rv_{ticker}.parquet") if model in ("HAR", "LogHAR"): fc = me.rolling_har_forecasts( df, model, horizons=HORIZONS, measure=measure, window=window, refit_every=REFIT) else: # LightGBM extended panel = pd.read_parquet(config.path("processed") / "panel.parquet") frames = ml.ml_features(panel, measure=measure) fc = ml.rolling_ml_forecasts( frames[ticker], "LightGBM", feature_set="extended", horizons=HORIZONS, window=window, refit_every=REFIT) fc["model"] = "LightGBM" fc["ticker"] = ticker # attach actuals in the SAME measure rv = df[measure] parts = [] for h in HORIZONS: t = horizon_target(rv, h).rename("actual").reset_index() t["h"] = h parts.append(t) tgt = pd.concat(parts, ignore_index=True) fc["date"] = pd.to_datetime(fc["date"]) tgt["date"] = pd.to_datetime(tgt["date"]) out = fc.merge(tgt, on=["date", "h"]).dropna(subset=["actual", "forecast"]) return out[out["actual"] > 0] def _run_jobs(jobs: list[tuple], workers: int) -> list[pd.DataFrame]: res = [] with ProcessPoolExecutor(max_workers=workers) as pool: futs = {pool.submit(_sens_one, j): j for j in jobs} for fut in as_completed(futs): j = futs[fut] try: res.append(fut.result()) logger.info("done %s", j) except Exception as exc: # noqa: BLE001 logger.error("FAILED %s: %s", j, exc) return res def part_frequency(workers: int) -> None: """Sensitivity to the RV sampling scheme (1-min, 5-min ss, kernel).""" tickers = [i["ticker"] for i in config.universe()] freqs = {"rv1min": "rv1", "rv5min_ss": "rv5ss", "rkernel": "rk"} results = {} for label, measure in freqs.items(): jobs = [(tk, m, measure, WINDOW) for tk in tickers for m in SENS_MODELS] frames = _run_jobs(jobs, workers) results[label] = pd.concat(frames, ignore_index=True) tab = rb.sensitivity_table(results, "frequency") tab.to_csv(OUT / "frequency_sensitivity.csv", index=False) logger.info("frequency sensitivity written") def part_window(workers: int) -> None: """Sensitivity to the estimation window (500 / 1000 / 2000 days).""" tickers = [i["ticker"] for i in config.universe()] results = {} for w in config.load_config()["robustness"]["window_sizes"]: jobs = [(tk, m, "rv5ss", int(w)) for tk in tickers for m in SENS_MODELS] frames = _run_jobs(jobs, workers) results[str(w)] = pd.concat(frames, ignore_index=True) tab = rb.sensitivity_table(results, "window") tab.to_csv(OUT / "window_sensitivity.csv", index=False) logger.info("window sensitivity written") def part_loo() -> None: """Cross-asset transferability: pooled model with one class representative excluded from training, predicted out-of-pool.""" from wp12 import models_ml as ml panel = pd.read_parquet(config.path("processed") / "panel.parquet") frames = ml.ml_features(panel) parts = [] for cls, tk in REPRESENTATIVE.items(): fc = ml.rolling_pooled_lgbm( frames, horizons=HORIZONS, window=WINDOW, refit_every=REFIT, exclude=tk) fc["cls"] = cls parts.append(fc) loo = pd.concat(parts, ignore_index=True) loo.to_parquet(config.path("processed") / "forecasts" / "pooled_loo.parquet", index=False) # compare in-pool vs out-of-pool QLIKE on the same dates merged = _merged() rows = [] for cls, tk in REPRESENTATIVE.items(): sub_loo = loo[loo["ticker"] == tk].copy() sub_loo["date"] = pd.to_datetime(sub_loo["date"]) base = merged[(merged["ticker"] == tk) & (merged["model"].isin(["Pooled-LGBM", "HAR", "LightGBM-X"]))] act = base.drop_duplicates(["date", "h"])[["date", "h", "actual"]] sub = sub_loo.merge(act, on=["date", "h"]).dropna(subset=["actual"]) for h in HORIZONS: d = sub[sub["h"] == h] q_loo = float(np.mean(ev.qlike(d["actual"].to_numpy(), d["forecast"].to_numpy()))) for m in ["Pooled-LGBM", "HAR", "LightGBM-X"]: b = base[(base["h"] == h) & (base["model"] == m)] b = b.merge(d[["date"]], on="date") q = float(np.mean(ev.qlike(b["actual"].to_numpy(), b["forecast"].to_numpy()))) rows.append((cls, tk, h, m, q)) rows.append((cls, tk, h, "Pooled-LGBM-LOO", q_loo)) tab = pd.DataFrame(rows, columns=["cls", "ticker", "h", "model", "qlike"]) tab.to_csv(OUT / "transferability.csv", index=False) logger.info("transferability written") def part_gr() -> None: """Giacomini-Rossi fluctuation paths: best ML vs HAR, h=1, per class rep.""" merged = _merged() paths = [] for cls, tk in REPRESENTATIVE.items(): sub = merged[(merged["ticker"] == tk) & (merged["h"] == 1)] wide = sub.pivot_table(index="date", columns="model", values="forecast") actual = sub.drop_duplicates("date").set_index("date")["actual"] for m in ["LightGBM-X", "Pooled-LGBM"]: if m not in wide.columns: continue both = wide[[m, "HAR"]].dropna() la = pd.Series(ev.qlike(actual[both.index].to_numpy(), both[m].to_numpy()), index=both.index) lb = pd.Series(ev.qlike(actual[both.index].to_numpy(), both["HAR"].to_numpy()), index=both.index) path = rb.giacomini_rossi_fluctuation(la, lb, h=1, mu=0.3) path["cls"] = cls path["ticker"] = tk path["model"] = m paths.append(path) pd.concat(paths, ignore_index=True).to_csv(OUT / "gr_fluctuation.csv", index=False) logger.info("GR fluctuation paths written") def main() -> None: """Run the requested robustness parts.""" ap = argparse.ArgumentParser() ap.add_argument("--part", default="all", choices=["all", "subperiods", "freq", "window", "loo", "gr"]) ap.add_argument("--workers", type=int, default=8) args = ap.parse_args() config.setup_logging() if args.part in ("all", "subperiods"): part_subperiods() if args.part in ("all", "freq"): part_frequency(args.workers) if args.part in ("all", "window"): part_window(args.workers) if args.part in ("all", "loo"): part_loo() if args.part in ("all", "gr"): part_gr() if __name__ == "__main__": main()