#!/usr/bin/env python3 """ ================================================================ Auteur : Simon-Pierre Boucher Contact : contact@spboucher.ai Projet : Prévision de volatilité réalisée multi-actifs (HAR-RV vs GARCH vs Machine Learning) Fichier : 04_evaluate.py Description : Étape 04 — Évaluation out-of-sample : assemblage des prévisions, combinaisons, pertes QLIKE/MSE, Diebold-Mariano vs HAR, Model Confidence Set, Mincer-Zarnowitz et forecast encompassing. Usage : python scripts/04_evaluate.py ================================================================ """ from __future__ import annotations import logging import sys from pathlib import Path import numpy as np import pandas as pd sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from wp12 import config, evaluation as ev # noqa: E402 from wp12.models_econ import horizon_target # noqa: E402 logger = logging.getLogger("04_evaluate") FC_DIR = config.path("processed") / "forecasts" OUT = config.path("reproduced") HORIZONS = tuple(config.load_config()["evaluation"]["horizons"]) BENCH = "HAR" def load_forecasts() -> pd.DataFrame: """Concatenate every stored forecast file into one long table.""" frames = [] for f in sorted(FC_DIR.glob("*.parquet")): df = pd.read_parquet(f) frames.append(df) out = pd.concat(frames, ignore_index=True) out["date"] = pd.to_datetime(out["date"]) logger.info("loaded %d forecasts, %d models, %d tickers", len(out), out["model"].nunique(), out["ticker"].nunique()) return out def attach_actuals(fc: pd.DataFrame, measure: str = "rv5ss") -> pd.DataFrame: """Join the realized horizon-average RV target onto the forecast table.""" panel = pd.read_parquet(config.path("processed") / "panel.parquet") targets = [] for tk, df in panel.groupby("ticker"): rv = df.sort_index()[measure] for h in HORIZONS: t = horizon_target(rv, h).rename("actual").reset_index() t["ticker"] = tk t["h"] = h targets.append(t) tgt = pd.concat(targets, ignore_index=True) merged = fc.merge(tgt, on=["date", "ticker", "h"], how="inner") merged = merged.dropna(subset=["actual", "forecast"]) merged = merged[merged["actual"] > 0] logger.info("merged: %d forecast-actual pairs", len(merged)) return merged def add_combinations(merged: pd.DataFrame) -> pd.DataFrame: """Append simple-mean and inverse-MSE combination forecasts.""" combos = [] for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): wide = sub.pivot_table(index="date", columns="model", values="forecast") actual = sub.drop_duplicates("date").set_index("date")["actual"] c = ev.combine_forecasts(wide, actual.reindex(wide.index)) c = c.stack().rename("forecast").reset_index() c.columns = ["date", "model", "forecast"] c["ticker"] = tk c["h"] = h c = c.merge(actual.rename("actual").reset_index(), on="date") combos.append(c.dropna(subset=["forecast", "actual"])) out = pd.concat([merged] + combos, ignore_index=True) logger.info("with combinations: %d rows", len(out)) return out def loss_tables(merged: pd.DataFrame) -> None: """Mean losses overall / by class / by ticker, with ratios to HAR.""" panel_cls = ( pd.read_parquet(config.path("processed") / "panel.parquet") .groupby("ticker")["cls"].first() ) merged = merged.assign(cls=merged["ticker"].map(panel_cls)) for loss in ("qlike", "mse"): lf = ev.LOSSES[loss] merged[loss] = lf(merged["actual"].to_numpy(), merged["forecast"].to_numpy()) per_asset = merged.groupby(["ticker", "cls", "h", "model"], observed=True)[ ["qlike", "mse"]].mean().reset_index() per_asset.to_csv(OUT / "losses_by_ticker.csv", index=False) def _with_ratio(df: pd.DataFrame, keys: list[str]) -> pd.DataFrame: bench = df[df["model"] == BENCH].set_index(keys)["qlike"] df = df.set_index(keys) df["qlike_ratio"] = df["qlike"] / bench return df.reset_index() overall = per_asset.groupby(["h", "model"], observed=True)[ ["qlike", "mse"]].mean().reset_index() _with_ratio(overall, ["h"]).to_csv(OUT / "losses_overall.csv", index=False) by_class = per_asset.groupby(["cls", "h", "model"], observed=True)[ ["qlike", "mse"]].mean().reset_index() _with_ratio(by_class, ["cls", "h"]).to_csv(OUT / "losses_by_class.csv", index=False) logger.info("loss tables written") def dm_tables(merged: pd.DataFrame) -> None: """Diebold-Mariano tests of every model against the HAR benchmark.""" rows = [] for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): wide_f = sub.pivot_table(index="date", columns="model", values="forecast") actual = sub.drop_duplicates("date").set_index("date")["actual"] if BENCH not in wide_f.columns: continue lb = ev.qlike(actual.to_numpy(), wide_f[BENCH].to_numpy()) for m in wide_f.columns: if m == BENCH: continue both = np.isfinite(wide_f[m].to_numpy()) & np.isfinite(wide_f[BENCH].to_numpy()) la = ev.qlike(actual.to_numpy()[both], wide_f[m].to_numpy()[both]) stat, p = ev.diebold_mariano(la, lb[both], h=h) rows.append((tk, h, m, stat, p)) dm = pd.DataFrame(rows, columns=["ticker", "h", "model", "dm_stat", "dm_p"]) dm.to_csv(OUT / "dm_by_ticker.csv", index=False) summary = dm.groupby(["h", "model"], observed=True).apply( lambda d: pd.Series({ "n_assets": len(d), "median_dm": d["dm_stat"].median(), "pct_better": float((d["dm_stat"] < 0).mean()), "pct_sig_better": float(((d["dm_stat"] < 0) & (d["dm_p"] < 0.05)).mean()), "pct_sig_worse": float(((d["dm_stat"] > 0) & (d["dm_p"] < 0.05)).mean()), }), include_groups=False).reset_index() summary.to_csv(OUT / "dm_summary.csv", index=False) logger.info("DM tables written") def mcs_tables(merged: pd.DataFrame) -> None: """90% Model Confidence Set per (ticker, horizon); inclusion shares.""" cfg = config.load_config()["evaluation"] rows = [] for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): wide_f = sub.pivot_table(index="date", columns="model", values="forecast") actual = sub.drop_duplicates("date").set_index("date")["actual"] losses = pd.DataFrame({ m: ev.qlike(actual.to_numpy(), wide_f[m].to_numpy()) for m in wide_f.columns }, index=wide_f.index).dropna() if len(losses) < 200: continue mcs = ev.model_confidence_set( losses, alpha=cfg["mcs_alpha"], n_boot=int(cfg["mcs_bootstrap"])) mcs["ticker"] = tk mcs["h"] = h rows.append(mcs) allmcs = pd.concat(rows, ignore_index=True) allmcs.to_csv(OUT / "mcs_by_ticker.csv", index=False) inc = allmcs.groupby(["h", "model"], observed=True)["in_mcs"].mean().rename( "mcs_inclusion").reset_index() inc.to_csv(OUT / "mcs_inclusion.csv", index=False) logger.info("MCS tables written") def mz_and_encompassing(merged: pd.DataFrame) -> None: """Mincer-Zarnowitz per model and HAR-vs-ML encompassing tests.""" rows = [] for (tk, h, m), sub in merged.groupby(["ticker", "h", "model"], observed=True): res = ev.mincer_zarnowitz(sub["actual"].to_numpy(), sub["forecast"].to_numpy(), h=h) rows.append((tk, h, m, res["alpha"], res["beta"], res["r2"], res["p_joint"])) mz = pd.DataFrame(rows, columns=["ticker", "h", "model", "alpha", "beta", "r2", "p_joint"]) mz.to_csv(OUT / "mz_by_ticker.csv", index=False) mz_sum = mz.groupby(["h", "model"], observed=True).apply( lambda d: pd.Series({ "med_beta": d["beta"].median(), "med_r2": d["r2"].median(), "pct_reject": float((d["p_joint"] < 0.05).mean()), }), include_groups=False).reset_index() mz_sum.to_csv(OUT / "mz_summary.csv", index=False) rows = [] rivals = [m for m in merged["model"].unique() if m != BENCH] for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): wide_f = sub.pivot_table(index="date", columns="model", values="forecast") actual = sub.drop_duplicates("date").set_index("date")["actual"] if BENCH not in wide_f.columns: continue for m in rivals: if m not in wide_f.columns: continue res = ev.forecast_encompassing( actual.to_numpy(), wide_f[BENCH].to_numpy(), wide_f[m].to_numpy(), h=h) rows.append((tk, h, m, res["b1"], res["b2"], res["p_b2"])) enc = pd.DataFrame(rows, columns=["ticker", "h", "model", "b1", "b2", "p_b2"]) enc.to_csv(OUT / "encompassing_by_ticker.csv", index=False) enc_sum = enc.groupby(["h", "model"], observed=True).apply( lambda d: pd.Series({ "med_b2": d["b2"].median(), "pct_adds_info": float(((d["b2"] > 0) & (d["p_b2"] < 0.05)).mean()), }), include_groups=False).reset_index() enc_sum.to_csv(OUT / "encompassing_summary.csv", index=False) logger.info("MZ and encompassing tables written") def main() -> None: """Run the full evaluation battery and persist intermediate tables.""" config.setup_logging() fc = load_forecasts() merged = attach_actuals(fc) merged = add_combinations(merged) merged.to_parquet(config.path("processed") / "merged_forecasts.parquet", index=False) loss_tables(merged) dm_tables(merged) mcs_tables(merged) mz_and_encompassing(merged) logger.info("evaluation complete") if __name__ == "__main__": main()