""" ================================================================ Auteur : Simon-Pierre Boucher Contact : contact@spboucher.ai Projet : Prévision de volatilité réalisée multi-actifs (HAR-RV vs GARCH vs Machine Learning) Fichier : test_evaluation.py Description : Tests pytest de l'évaluation — propriétés des pertes, taille/puissance du test DM, MCS, Mincer- Zarnowitz, encompassing et combinaisons. ================================================================ """ from __future__ import annotations import sys from pathlib import Path import numpy as np import pandas as pd import pytest sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from wp12 import evaluation as ev # noqa: E402 RNG = np.random.default_rng(20260810) class TestLosses: def test_qlike_zero_at_perfect_forecast(self) -> None: a = RNG.uniform(0.5, 2.0, 100) assert np.allclose(ev.qlike(a, a), 0.0) def test_qlike_positive_and_asymmetric(self) -> None: a = np.ones(10) under = ev.qlike(a, 0.5 * a).mean() over = ev.qlike(a, 2.0 * a).mean() assert under > 0 and over > 0 assert under > over # QLIKE punishes under-prediction more def test_mse_basic(self) -> None: assert ev.mse(np.array([2.0]), np.array([1.0]))[0] == 1.0 class TestDieboldMariano: def test_dm_detects_better_model(self) -> None: n = 800 actual = np.exp(RNG.normal(0, 0.5, n)) good = actual * np.exp(RNG.normal(0, 0.1, n)) bad = actual * np.exp(RNG.normal(0.5, 0.5, n)) la, lb = ev.qlike(actual, good), ev.qlike(actual, bad) stat, p = ev.diebold_mariano(la, lb, h=1) assert stat < -2 and p < 0.05 def test_dm_size_under_null(self) -> None: """Two equally noisy forecasts: rejection rate ~ nominal.""" rejects = 0 for _ in range(200): actual = np.exp(RNG.normal(0, 0.5, 300)) f1 = actual * np.exp(RNG.normal(0, 0.3, 300)) f2 = actual * np.exp(RNG.normal(0, 0.3, 300)) _, p = ev.diebold_mariano(ev.qlike(actual, f1), ev.qlike(actual, f2)) rejects += p < 0.05 assert rejects / 200 < 0.12 class TestMCS: def test_mcs_keeps_good_removes_bad(self) -> None: n = 600 actual = np.exp(RNG.normal(0, 0.5, n)) losses = pd.DataFrame({ "good1": ev.qlike(actual, actual * np.exp(RNG.normal(0, 0.1, n))), "good2": ev.qlike(actual, actual * np.exp(RNG.normal(0, 0.1, n))), "bad": ev.qlike(actual, actual * np.exp(RNG.normal(1.0, 0.5, n))), }) out = ev.model_confidence_set(losses, alpha=0.10, n_boot=800) assert not out.loc[out["model"] == "bad", "in_mcs"].iloc[0] assert out["in_mcs"].sum() >= 1 good_kept = out.loc[out["model"].str.startswith("good"), "in_mcs"] assert good_kept.any() class TestMZAndEncompassing: def test_mz_unbiased_forecast(self) -> None: n = 1000 f = np.exp(RNG.normal(0, 0.5, n)) actual = f * np.exp(RNG.normal(-0.005, 0.1, n)) res = ev.mincer_zarnowitz(actual, f) assert res["beta"] == pytest.approx(1.0, abs=0.1) assert res["r2"] > 0.8 def test_mz_biased_forecast_rejected(self) -> None: n = 1000 f = np.exp(RNG.normal(0, 0.5, n)) actual = 2.0 * f * np.exp(RNG.normal(0, 0.1, n)) res = ev.mincer_zarnowitz(actual, f) assert res["p_joint"] < 0.01 def test_encompassing_detects_added_info(self) -> None: n = 1000 true = np.exp(RNG.normal(0, 0.5, n)) actual = true * np.exp(RNG.normal(0, 0.2, n)) f_base = true * np.exp(RNG.normal(0, 0.4, n)) # noisy version f_rival = 0.5 * f_base + 0.5 * true # contains extra info res = ev.forecast_encompassing(actual, f_base, f_rival) assert res["p_b2"] < 0.05 assert res["b2"] > 0 class TestCombination: def test_combined_beats_worst(self) -> None: n = 700 idx = pd.date_range("2020-01-01", periods=n, freq="B") actual = pd.Series(np.exp(RNG.normal(0, 0.5, n)), index=idx) f = pd.DataFrame({ "m1": actual * np.exp(RNG.normal(0, 0.2, n)), "m2": actual * np.exp(RNG.normal(0.3, 0.4, n)), }, index=idx) combs = ev.combine_forecasts(f, actual, window=60) q_mean = ev.qlike(actual.to_numpy(), combs["Comb-Mean"].to_numpy()).mean() q_inv = np.nanmean(ev.qlike(actual.to_numpy(), combs["Comb-InvMSE"].to_numpy())) q_worst = ev.qlike(actual.to_numpy(), f["m2"].to_numpy()).mean() assert q_mean < q_worst assert q_inv < q_worst