SPB Git forge

spb/wp12_uqo

Public
5commits 1branches 0releases
1.2 MBsize
maindefault branch
1 mo agolast push
Python 75.9% TeX 24%

Scripts 03-09, tests, figures/tables generators, methodology & literature sections, references.bib

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Simon-Pierre Boucher committed 1 mo ago (Aug 11, 2026) parent 85ba409

12 changed files +2,442 −3

modified paper/references.bib +569 −0
@@ -0,0 +1,569 @@
1 +% ================================================================
2 +% Auteur : Simon-Pierre Boucher
3 +% Contact : contact@spboucher.ai
4 +% Projet : Prévision de volatilité réalisée multi-actifs
5 +% (HAR-RV vs GARCH vs Machine Learning)
6 +% Fichier : references.bib
7 +% Description : Références bibliographiques du papier (BibTeX).
8 +% ================================================================
9 +
10 +@article{AndersenBollerslev1998,
11 + author = {Andersen, Torben G. and Bollerslev, Tim},
12 + title = {Answering the Skeptics: Yes, Standard Volatility Models Do Provide Accurate Forecasts},
13 + journal = {International Economic Review},
14 + year = {1998},
15 + volume = {39},
16 + number = {4},
17 + pages = {885--905}
18 +}
19 +
20 +@article{ABDL2001,
21 + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X. and Labys, Paul},
22 + title = {The Distribution of Realized Exchange Rate Volatility},
23 + journal = {Journal of the American Statistical Association},
24 + year = {2001},
25 + volume = {96},
26 + number = {453},
27 + pages = {42--55}
28 +}
29 +
30 +@article{ABDL2003,
31 + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X. and Labys, Paul},
32 + title = {Modeling and Forecasting Realized Volatility},
33 + journal = {Econometrica},
34 + year = {2003},
35 + volume = {71},
36 + number = {2},
37 + pages = {579--625}
38 +}
39 +
40 +@article{ABDE2001,
41 + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X. and Ebens, Heiko},
42 + title = {The Distribution of Realized Stock Return Volatility},
43 + journal = {Journal of Financial Economics},
44 + year = {2001},
45 + volume = {61},
46 + number = {1},
47 + pages = {43--76}
48 +}
49 +
50 +@article{ABD2007,
51 + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X.},
52 + title = {Roughing It Up: Including Jump Components in the Measurement, Modeling, and Forecasting of Return Volatility},
53 + journal = {Review of Economics and Statistics},
54 + year = {2007},
55 + volume = {89},
56 + number = {4},
57 + pages = {701--720}
58 +}
59 +
60 +@article{BNS2002,
61 + author = {Barndorff-Nielsen, Ole E. and Shephard, Neil},
62 + title = {Econometric Analysis of Realized Volatility and Its Use in Estimating Stochastic Volatility Models},
63 + journal = {Journal of the Royal Statistical Society: Series B},
64 + year = {2002},
65 + volume = {64},
66 + number = {2},
67 + pages = {253--280}
68 +}
69 +
70 +@article{BNS2004,
71 + author = {Barndorff-Nielsen, Ole E. and Shephard, Neil},
72 + title = {Power and Bipower Variation with Stochastic Volatility and Jumps},
73 + journal = {Journal of Financial Econometrics},
74 + year = {2004},
75 + volume = {2},
76 + number = {1},
77 + pages = {1--37}
78 +}
79 +
80 +@article{BNS2006,
81 + author = {Barndorff-Nielsen, Ole E. and Shephard, Neil},
82 + title = {Econometrics of Testing for Jumps in Financial Economics Using Bipower Variation},
83 + journal = {Journal of Financial Econometrics},
84 + year = {2006},
85 + volume = {4},
86 + number = {1},
87 + pages = {1--30}
88 +}
89 +
90 +@article{BNHLS2008,
91 + author = {Barndorff-Nielsen, Ole E. and Hansen, Peter Reinhard and Lunde, Asger and Shephard, Neil},
92 + title = {Designing Realized Kernels to Measure the Ex Post Variation of Equity Prices in the Presence of Noise},
93 + journal = {Econometrica},
94 + year = {2008},
95 + volume = {76},
96 + number = {6},
97 + pages = {1481--1536}
98 +}
99 +
100 +@article{BNHLS2009,
101 + author = {Barndorff-Nielsen, Ole E. and Hansen, Peter Reinhard and Lunde, Asger and Shephard, Neil},
102 + title = {Realized Kernels in Practice: Trades and Quotes},
103 + journal = {Econometrics Journal},
104 + year = {2009},
105 + volume = {12},
106 + number = {3},
107 + pages = {C1--C32}
108 +}
109 +
110 +@incollection{BNKS2010,
111 + author = {Barndorff-Nielsen, Ole E. and Kinnebrock, Silja and Shephard, Neil},
112 + title = {Measuring Downside Risk: Realised Semivariance},
113 + booktitle = {Volatility and Time Series Econometrics: Essays in Honor of Robert F. Engle},
114 + editor = {Bollerslev, Tim and Russell, Jeffrey R. and Watson, Mark W.},
115 + publisher = {Oxford University Press},
116 + year = {2010},
117 + pages = {117--136}
118 +}
119 +
120 +@article{ZMA2005,
121 + author = {Zhang, Lan and Mykland, Per A. and A{\"i}t-Sahalia, Yacine},
122 + title = {A Tale of Two Time Scales: Determining Integrated Volatility with Noisy High-Frequency Data},
123 + journal = {Journal of the American Statistical Association},
124 + year = {2005},
125 + volume = {100},
126 + number = {472},
127 + pages = {1394--1411}
128 +}
129 +
130 +@article{HansenLunde2006,
131 + author = {Hansen, Peter R. and Lunde, Asger},
132 + title = {Realized Variance and Market Microstructure Noise},
133 + journal = {Journal of Business \& Economic Statistics},
134 + year = {2006},
135 + volume = {24},
136 + number = {2},
137 + pages = {127--161}
138 +}
139 +
140 +@article{BandiRussell2008,
141 + author = {Bandi, Federico M. and Russell, Jeffrey R.},
142 + title = {Microstructure Noise, Realized Variance, and Optimal Sampling},
143 + journal = {Review of Economic Studies},
144 + year = {2008},
145 + volume = {75},
146 + number = {2},
147 + pages = {339--369}
148 +}
149 +
150 +@article{LiuPattonSheppard2015,
151 + author = {Liu, Lily Y. and Patton, Andrew J. and Sheppard, Kevin},
152 + title = {Does Anything Beat 5-Minute {RV}? A Comparison of Realized Measures Across Multiple Asset Classes},
153 + journal = {Journal of Econometrics},
154 + year = {2015},
155 + volume = {187},
156 + number = {1},
157 + pages = {293--311}
158 +}
159 +
160 +@article{Corsi2009,
161 + author = {Corsi, Fulvio},
162 + title = {A Simple Approximate Long-Memory Model of Realized Volatility},
163 + journal = {Journal of Financial Econometrics},
164 + year = {2009},
165 + volume = {7},
166 + number = {2},
167 + pages = {174--196}
168 +}
169 +
170 +@article{CorsiPirinoReno2010,
171 + author = {Corsi, Fulvio and Pirino, Davide and Ren{\`o}, Roberto},
172 + title = {Threshold Bipower Variation and the Impact of Jumps on Volatility Forecasting},
173 + journal = {Journal of Econometrics},
174 + year = {2010},
175 + volume = {159},
176 + number = {2},
177 + pages = {276--288}
178 +}
179 +
180 +@article{PattonSheppard2015,
181 + author = {Patton, Andrew J. and Sheppard, Kevin},
182 + title = {Good Volatility, Bad Volatility: Signed Jumps and the Persistence of Volatility},
183 + journal = {Review of Economics and Statistics},
184 + year = {2015},
185 + volume = {97},
186 + number = {3},
187 + pages = {683--697}
188 +}
189 +
190 +@article{BPQ2016,
191 + author = {Bollerslev, Tim and Patton, Andrew J. and Quaedvlieg, Rogier},
192 + title = {Exploiting the Errors: A Simple Approach for Improved Volatility Forecasting},
193 + journal = {Journal of Econometrics},
194 + year = {2016},
195 + volume = {192},
196 + number = {1},
197 + pages = {1--18}
198 +}
199 +
200 +@article{Engle1982,
201 + author = {Engle, Robert F.},
202 + title = {Autoregressive Conditional Heteroscedasticity with Estimates of the Variance of United Kingdom Inflation},
203 + journal = {Econometrica},
204 + year = {1982},
205 + volume = {50},
206 + number = {4},
207 + pages = {987--1007}
208 +}
209 +
210 +@article{Bollerslev1986,
211 + author = {Bollerslev, Tim},
212 + title = {Generalized Autoregressive Conditional Heteroskedasticity},
213 + journal = {Journal of Econometrics},
214 + year = {1986},
215 + volume = {31},
216 + number = {3},
217 + pages = {307--327}
218 +}
219 +
220 +@article{Nelson1991,
221 + author = {Nelson, Daniel B.},
222 + title = {Conditional Heteroskedasticity in Asset Returns: A New Approach},
223 + journal = {Econometrica},
224 + year = {1991},
225 + volume = {59},
226 + number = {2},
227 + pages = {347--370}
228 +}
229 +
230 +@article{GJR1993,
231 + author = {Glosten, Lawrence R. and Jagannathan, Ravi and Runkle, David E.},
232 + title = {On the Relation Between the Expected Value and the Volatility of the Nominal Excess Return on Stocks},
233 + journal = {Journal of Finance},
234 + year = {1993},
235 + volume = {48},
236 + number = {5},
237 + pages = {1779--1801}
238 +}
239 +
240 +@article{HansenLunde2005,
241 + author = {Hansen, Peter R. and Lunde, Asger},
242 + title = {A Forecast Comparison of Volatility Models: Does Anything Beat a {GARCH}(1,1)?},
243 + journal = {Journal of Applied Econometrics},
244 + year = {2005},
245 + volume = {20},
246 + number = {7},
247 + pages = {873--889}
248 +}
249 +
250 +@article{HansenHuangShek2012,
251 + author = {Hansen, Peter Reinhard and Huang, Zhuo and Shek, Howard Howan},
252 + title = {Realized {GARCH}: A Joint Model for Returns and Realized Measures of Volatility},
253 + journal = {Journal of Applied Econometrics},
254 + year = {2012},
255 + volume = {27},
256 + number = {6},
257 + pages = {877--906}
258 +}
259 +
260 +@article{Patton2011,
261 + author = {Patton, Andrew J.},
262 + title = {Volatility Forecast Comparison Using Imperfect Volatility Proxies},
263 + journal = {Journal of Econometrics},
264 + year = {2011},
265 + volume = {160},
266 + number = {1},
267 + pages = {246--256}
268 +}
269 +
270 +@article{DieboldMariano1995,
271 + author = {Diebold, Francis X. and Mariano, Roberto S.},
272 + title = {Comparing Predictive Accuracy},
273 + journal = {Journal of Business \& Economic Statistics},
274 + year = {1995},
275 + volume = {13},
276 + number = {3},
277 + pages = {253--263}
278 +}
279 +
280 +@article{GiacominiWhite2006,
281 + author = {Giacomini, Raffaella and White, Halbert},
282 + title = {Tests of Conditional Predictive Ability},
283 + journal = {Econometrica},
284 + year = {2006},
285 + volume = {74},
286 + number = {6},
287 + pages = {1545--1578}
288 +}
289 +
290 +@article{HLN2011,
291 + author = {Hansen, Peter R. and Lunde, Asger and Nason, James M.},
292 + title = {The Model Confidence Set},
293 + journal = {Econometrica},
294 + year = {2011},
295 + volume = {79},
296 + number = {2},
297 + pages = {453--497}
298 +}
299 +
300 +@article{GiacominiRossi2010,
301 + author = {Giacomini, Raffaella and Rossi, Barbara},
302 + title = {Forecast Comparisons in Unstable Environments},
303 + journal = {Journal of Applied Econometrics},
304 + year = {2010},
305 + volume = {25},
306 + number = {4},
307 + pages = {595--620}
308 +}
309 +
310 +@incollection{MincerZarnowitz1969,
311 + author = {Mincer, Jacob A. and Zarnowitz, Victor},
312 + title = {The Evaluation of Economic Forecasts},
313 + booktitle = {Economic Forecasts and Expectations: Analysis of Forecasting Behavior and Performance},
314 + publisher = {NBER},
315 + year = {1969},
316 + pages = {3--46}
317 +}
318 +
319 +@article{HarveyLeybourneNewbold1998,
320 + author = {Harvey, David I. and Leybourne, Stephen J. and Newbold, Paul},
321 + title = {Tests for Forecast Encompassing},
322 + journal = {Journal of Business \& Economic Statistics},
323 + year = {1998},
324 + volume = {16},
325 + number = {2},
326 + pages = {254--259}
327 +}
328 +
329 +@article{BatesGranger1969,
330 + author = {Bates, John M. and Granger, Clive W. J.},
331 + title = {The Combination of Forecasts},
332 + journal = {Journal of the Operational Research Society},
333 + year = {1969},
334 + volume = {20},
335 + number = {4},
336 + pages = {451--468}
337 +}
338 +
339 +@incollection{Timmermann2006,
340 + author = {Timmermann, Allan},
341 + title = {Forecast Combinations},
342 + booktitle = {Handbook of Economic Forecasting},
343 + editor = {Elliott, Graham and Granger, Clive W. J. and Timmermann, Allan},
344 + publisher = {Elsevier},
345 + year = {2006},
346 + volume = {1},
347 + pages = {135--196}
348 +}
349 +
350 +@article{PolitisRomano1994,
351 + author = {Politis, Dimitris N. and Romano, Joseph P.},
352 + title = {The Stationary Bootstrap},
353 + journal = {Journal of the American Statistical Association},
354 + year = {1994},
355 + volume = {89},
356 + number = {428},
357 + pages = {1303--1313}
358 +}
359 +
360 +@article{FKO2001,
361 + author = {Fleming, Jeff and Kirby, Chris and Ostdiek, Barbara},
362 + title = {The Economic Value of Volatility Timing},
363 + journal = {Journal of Finance},
364 + year = {2001},
365 + volume = {56},
366 + number = {1},
367 + pages = {329--352}
368 +}
369 +
370 +@article{FKO2003,
371 + author = {Fleming, Jeff and Kirby, Chris and Ostdiek, Barbara},
372 + title = {The Economic Value of Volatility Timing Using ``Realized'' Volatility},
373 + journal = {Journal of Financial Economics},
374 + year = {2003},
375 + volume = {67},
376 + number = {3},
377 + pages = {473--509}
378 +}
379 +
380 +@article{Kupiec1995,
381 + author = {Kupiec, Paul H.},
382 + title = {Techniques for Verifying the Accuracy of Risk Measurement Models},
383 + journal = {Journal of Derivatives},
384 + year = {1995},
385 + volume = {3},
386 + number = {2},
387 + pages = {73--84}
388 +}
389 +
390 +@article{Christoffersen1998,
391 + author = {Christoffersen, Peter F.},
392 + title = {Evaluating Interval Forecasts},
393 + journal = {International Economic Review},
394 + year = {1998},
395 + volume = {39},
396 + number = {4},
397 + pages = {841--862}
398 +}
399 +
400 +@article{Bucci2020,
401 + author = {Bucci, Andrea},
402 + title = {Realized Volatility Forecasting with Neural Networks},
403 + journal = {Journal of Financial Econometrics},
404 + year = {2020},
405 + volume = {18},
406 + number = {3},
407 + pages = {502--531}
408 +}
409 +
410 +@article{ChristensenSiggaardVeliyev2023,
411 + author = {Christensen, Kim and Siggaard, Mathias and Veliyev, Bezirgen},
412 + title = {A Machine Learning Approach to Volatility Forecasting},
413 + journal = {Journal of Financial Econometrics},
414 + year = {2023},
415 + volume = {21},
416 + number = {5},
417 + pages = {1680--1727}
418 +}
419 +
420 +@article{GuKellyXiu2020,
421 + author = {Gu, Shihao and Kelly, Bryan and Xiu, Dacheng},
422 + title = {Empirical Asset Pricing via Machine Learning},
423 + journal = {Review of Financial Studies},
424 + year = {2020},
425 + volume = {33},
426 + number = {5},
427 + pages = {2223--2273}
428 +}
429 +
430 +@article{AudrinoKnaus2016,
431 + author = {Audrino, Francesco and Knaus, Simon D.},
432 + title = {Lassoing the {HAR} Model: A Model Selection Perspective on Realized Volatility Dynamics},
433 + journal = {Econometric Reviews},
434 + year = {2016},
435 + volume = {35},
436 + number = {8--10},
437 + pages = {1485--1521}
438 +}
439 +
440 +@article{RahimikiaPoon2020,
441 + author = {Rahimikia, Eghbal and Poon, Ser-Huang},
442 + title = {Machine Learning for Realised Volatility Forecasting},
443 + journal = {SSRN Working Paper No. 3707796},
444 + year = {2020}
445 +}
446 +
447 +@article{Katsiampa2017,
448 + author = {Katsiampa, Paraskevi},
449 + title = {Volatility Estimation for {Bitcoin}: A Comparison of {GARCH} Models},
450 + journal = {Economics Letters},
451 + year = {2017},
452 + volume = {158},
453 + pages = {3--6}
454 +}
455 +
456 +@article{Duan1983,
457 + author = {Duan, Naihua},
458 + title = {Smearing Estimate: A Nonparametric Retransformation Method},
459 + journal = {Journal of the American Statistical Association},
460 + year = {1983},
461 + volume = {78},
462 + number = {383},
463 + pages = {605--610}
464 +}
465 +
466 +@article{HochreiterSchmidhuber1997,
467 + author = {Hochreiter, Sepp and Schmidhuber, J{\"u}rgen},
468 + title = {Long Short-Term Memory},
469 + journal = {Neural Computation},
470 + year = {1997},
471 + volume = {9},
472 + number = {8},
473 + pages = {1735--1780}
474 +}
475 +
476 +@inproceedings{Vaswani2017,
477 + author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N. and Kaiser, {\L}ukasz and Polosukhin, Illia},
478 + title = {Attention Is All You Need},
479 + booktitle = {Advances in Neural Information Processing Systems},
480 + year = {2017},
481 + volume = {30},
482 + pages = {5998--6008}
483 +}
484 +
485 +@inproceedings{LundbergLee2017,
486 + author = {Lundberg, Scott M. and Lee, Su-In},
487 + title = {A Unified Approach to Interpreting Model Predictions},
488 + booktitle = {Advances in Neural Information Processing Systems},
489 + year = {2017},
490 + volume = {30},
491 + pages = {4765--4774}
492 +}
493 +
494 +@article{Breiman2001,
495 + author = {Breiman, Leo},
496 + title = {Random Forests},
497 + journal = {Machine Learning},
498 + year = {2001},
499 + volume = {45},
500 + number = {1},
501 + pages = {5--32}
502 +}
503 +
504 +@inproceedings{ChenGuestrin2016,
505 + author = {Chen, Tianqi and Guestrin, Carlos},
506 + title = {{XGBoost}: A Scalable Tree Boosting System},
507 + booktitle = {Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining},
508 + year = {2016},
509 + pages = {785--794}
510 +}
511 +
512 +@inproceedings{Ke2017,
513 + author = {Ke, Guolin and Meng, Qi and Finley, Thomas and Wang, Taifeng and Chen, Wei and Ma, Weidong and Ye, Qiwei and Liu, Tie-Yan},
514 + title = {{LightGBM}: A Highly Efficient Gradient Boosting Decision Tree},
515 + booktitle = {Advances in Neural Information Processing Systems},
516 + year = {2017},
517 + volume = {30},
518 + pages = {3146--3154}
519 +}
520 +
521 +@article{Tibshirani1996,
522 + author = {Tibshirani, Robert},
523 + title = {Regression Shrinkage and Selection via the Lasso},
524 + journal = {Journal of the Royal Statistical Society: Series B},
525 + year = {1996},
526 + volume = {58},
527 + number = {1},
528 + pages = {267--288}
529 +}
530 +
531 +@article{ZouHastie2005,
532 + author = {Zou, Hui and Hastie, Trevor},
533 + title = {Regularization and Variable Selection via the Elastic Net},
534 + journal = {Journal of the Royal Statistical Society: Series B},
535 + year = {2005},
536 + volume = {67},
537 + number = {2},
538 + pages = {301--320}
539 +}
540 +
541 +@article{NeweyWest1987,
542 + author = {Newey, Whitney K. and West, Kenneth D.},
543 + title = {A Simple, Positive Semi-Definite, Heteroskedasticity and Autocorrelation Consistent Covariance Matrix},
544 + journal = {Econometrica},
545 + year = {1987},
546 + volume = {55},
547 + number = {3},
548 + pages = {703--708}
549 +}
550 +
551 +@article{Mueller1997,
552 + author = {M{\"u}ller, Ulrich A. and Dacorogna, Michel M. and Dav{\'e}, Rakhal D. and Olsen, Richard B. and Pictet, Olivier V. and von Weizs{\"a}cker, Jacob E.},
553 + title = {Volatilities of Different Time Resolutions --- Analyzing the Dynamics of Market Components},
554 + journal = {Journal of Empirical Finance},
555 + year = {1997},
556 + volume = {4},
557 + number = {2--3},
558 + pages = {213--239}
559 +}
560 +
561 +@article{PoonGranger2003,
562 + author = {Poon, Ser-Huang and Granger, Clive W. J.},
563 + title = {Forecasting Volatility in Financial Markets: A Review},
564 + journal = {Journal of Economic Literature},
565 + year = {2003},
566 + volume = {41},
567 + number = {2},
568 + pages = {478--539}
569 +}
modified paper/sections/literature.tex +107 −1
@@ -4,5 +4,111 @@
4 4 % Projet : Prévision de volatilité réalisée multi-actifs
5 5 % (HAR-RV vs GARCH vs Machine Learning)
6 6 % Fichier : literature.tex
7 −% Description : Section « literature » — à rédiger après exécution du pipeline.
7 +% Description : Section 2 — Revue de littérature et gap.
8 8 % ================================================================
9 +\section{Related literature}
10 +\label{sec:literature}
11 +
12 +\subsection{Realized volatility and its measurement}
13 +
14 +The modern volatility-forecasting literature begins with the insight
15 +that intraday returns turn latent variance into an effectively
16 +observable quantity. \citet{AndersenBollerslev1998} showed that
17 +standard models forecast well once evaluated against realized rather
18 +than squared daily returns; \citet{ABDL2001,ABDE2001} documented the
19 +distributional properties of realized variance for exchange rates and
20 +equities; and \citet{BNS2002} developed its asymptotic theory.
21 +\citet{ABDL2003} demonstrated that simple reduced-form models of
22 +realized volatility outperform latent-variable GARCH and stochastic
23 +volatility models --- the observation that motivates the entire
24 +measure-then-model paradigm this paper operates in.
25 +
26 +At very high sampling frequencies, market microstructure noise breaks
27 +the consistency of plain realized variance
28 +\citep{HansenLunde2006,BandiRussell2008}. The literature responded with
29 +noise-robust estimators: subsampled and two-scale estimators
30 +\citep{ZMA2005}, and realized kernels \citep{BNHLS2008,BNHLS2009}.
31 +\citet{LiuPattonSheppard2015} compared some 400 realized measures
32 +across asset classes and found that 5-minute RV is remarkably hard to
33 +beat --- a finding that guides our choice of subsampled 5-minute RV as
34 +the headline measure, with 1-minute RV and realized kernels as
35 +robustness checks. A parallel branch separates the continuous and jump
36 +components of quadratic variation: bipower variation \citep{BNS2004},
37 +formal jump tests \citep{BNS2006}, and threshold refinements
38 +\citep{CorsiPirinoReno2010}. \citet{BNKS2010} introduced realized
39 +semivariances, whose signed decomposition \citet{PattonSheppard2015}
40 +showed to matter for persistence: negative semivariance predicts future
41 +volatility far more strongly than positive.
42 +
43 +\subsection{HAR models and extensions}
44 +
45 +\citet{Corsi2009} proposed the heterogeneous autoregressive (HAR)
46 +model, whose daily--weekly--monthly cascade approximates long memory
47 +with three OLS coefficients; it has become the workhorse benchmark of
48 +the field. Extensions incorporate the continuous/jump split
49 +\citep{ABD2007}, signed semivariances \citep{PattonSheppard2015}, and
50 +attenuation-bias corrections: \citet{BPQ2016} let the daily coefficient
51 +shrink on days when realized quarticity signals a noisy RV estimate
52 +(HARQ). Logarithmic specifications are common because log RV is close
53 +to Gaussian \citep{ABDL2001} and the log transform tames heteroskedastic
54 +measurement error. On the GARCH side, \citet{HansenLunde2005} showed
55 +that little beats a GARCH(1,1) for daily returns, while
56 +\citet{HansenHuangShek2012} tied the latent-variance recursion to a
57 +realized measure (Realized GARCH), typically closing most of the gap to
58 +HAR-type models. Our benchmark suite spans exactly this spectrum ---
59 +five HAR variants, three daily GARCH variants, and Realized GARCH ---
60 +so that the machine-learning comparison is anchored against the
61 +strongest available econometrics rather than a single straw man.
62 +
63 +\subsection{Machine learning for volatility forecasting}
64 +
65 +Applications of statistical learning to realized volatility are more
66 +recent. \citet{AudrinoKnaus2016} used the LASSO to select lag
67 +structures, finding it recovers HAR-like patterns.
68 +\citet{Bucci2020} documented gains from recurrent neural networks for
69 +monthly equity-index volatility. \citet{RahimikiaPoon2020} trained
70 +learners on large limit-order-book and news datasets for intraday
71 +volatility. Closest to our design, \citet{ChristensenSiggaardVeliyev2023}
72 +ran a systematic horse race of regularized regressions, trees and
73 +neural networks against HAR on 29 Dow Jones stocks, finding double-digit
74 +QLIKE improvements concentrated in periods of market stress and in
75 +longer horizons. In the broader return-prediction literature,
76 +\citet{GuKellyXiu2020} established the now-standard result that trees
77 +and shallow networks dominate linear methods and that pooling across
78 +assets is crucial for regularization --- a theme our pooled multi-asset
79 +model imports into the volatility context. For cryptocurrencies,
80 +GARCH-type comparisons dominate \citep{Katsiampa2017}, with
81 +high-frequency ML applications still scarce.
82 +
83 +\subsection{Forecast evaluation and economic value}
84 +
85 +Because volatility is observed only through noisy proxies, loss
86 +functions must be chosen carefully: \citet{Patton2011} characterizes
87 +the class of robust losses --- QLIKE and MSE --- for which rankings
88 +under a proxy coincide with rankings under the truth. Statistical
89 +comparisons rely on \citet{DieboldMariano1995} and
90 +\citet{GiacominiWhite2006} tests, the model confidence set of
91 +\citet{HLN2011}, and, in unstable environments, the fluctuation
92 +analysis of \citet{GiacominiRossi2010}. Forecast combination is a
93 +persistent empirical winner \citep{BatesGranger1969,Timmermann2006}.
94 +The economic-value strand converts statistical accuracy into portfolio
95 +outcomes: \citet{FKO2001,FKO2003} priced volatility timing with
96 +mean-variance utility, and risk-management evaluation backtests VaR
97 +coverage \citep{Kupiec1995,Christoffersen1998}.
98 +
99 +\subsection{Contribution relative to the literature}
100 +
101 +Three gaps motivate this paper. First, most ML-versus-HAR evidence
102 +comes from single asset classes --- typically US equities
103 +\citep{ChristensenSiggaardVeliyev2023} --- leaving open whether the
104 +documented gains generalize to FX, futures and crypto, whose
105 +volatility dynamics, trading sessions and noise properties differ
106 +sharply. Second, the pooling and \emph{transferability} question ---
107 +whether one model trained on many assets can forecast an asset it has
108 +never seen --- has been studied for returns \citep{GuKellyXiu2020} but
109 +barely for realized volatility. Third, statistical gains are rarely
110 +pushed through to economic value and risk management within the same
111 +controlled design. We address all three within a single, fully
112 +reproducible pipeline: identical rolling protocol, identical robust
113 +losses, identical inference, from raw 1-minute bars to utility gains
114 +and VaR coverage, across 25 assets in four classes.
modified paper/sections/methodology.tex +276 −1
@@ -4,5 +4,280 @@
4 4 % Projet : Prévision de volatilité réalisée multi-actifs
5 5 % (HAR-RV vs GARCH vs Machine Learning)
6 6 % Fichier : methodology.tex
7 −% Description : Section « methodology » — à rédiger après exécution du pipeline.
7 +% Description : Section 4 — Mesures de volatilité réalisée,
8 +% modèles, protocole d'évaluation et tests.
8 9 % ================================================================
10 +\section{Methodology}
11 +\label{sec:methodology}
12 +
13 +\subsection{Realized measures}
14 +\label{sec:rv_measures}
15 +
16 +Let $p_{t,i}$ denote the $i$-th intraday log price of a given asset on
17 +trading day $t$, sampled on an equidistant grid of $n$ intraday returns
18 +$r_{t,i} = p_{t,i} - p_{t,i-1}$. Under the standard jump--diffusion
19 +semimartingale
20 +\begin{equation}
21 + dp_s = \mu_s \, ds + \sigma_s \, dW_s + \kappa_s \, dq_s ,
22 + \label{eq:sde}
23 +\end{equation}
24 +where $W$ is a Brownian motion, $q$ a counting process, and $\kappa$ the
25 +jump size, the realized variance
26 +\begin{equation}
27 + RV_t \;=\; \sum_{i=1}^{n} r_{t,i}^2
28 + \;\xrightarrow{\;p\;}\;
29 + \underbrace{\int_{t-1}^{t} \sigma^2_s \, ds}_{\text{integrated variance}}
30 + \;+\; \underbrace{\sum_{t-1 < s \le t} \kappa_s^2}_{\text{jump variation}}
31 + \label{eq:rv}
32 +\end{equation}
33 +consistently estimates total quadratic variation as $n \to \infty$
34 +\citep{AndersenBollerslev1998,BNS2002}. In practice market microstructure
35 +noise contaminates returns at the very highest frequencies
36 +\citep{HansenLunde2006,BandiRussell2008}, and we therefore compute several
37 +noise-robust variants.
38 +
39 +\paragraph{Subsampled 5-minute RV.}
40 +Our headline measure, denoted $RV^{ss}_t$, averages the realized variance
41 +computed on five staggered 5-minute grids offset by one minute each
42 +\citep{ZMA2005}. Following \citet{LiuPattonSheppard2015}, 5-minute
43 +sampling is a demanding benchmark that is difficult to beat across asset
44 +classes; subsampling recovers part of the efficiency lost to sparse
45 +sampling.
46 +
47 +\paragraph{Realized kernel.}
48 +We also compute the flat-top-free Parzen realized kernel of
49 +\citet{BNHLS2008},
50 +\begin{equation}
51 + RK_t \;=\; \gamma_0 + \sum_{h=1}^{H} k\!\left(\tfrac{h}{H+1}\right)
52 + \left(\gamma_h + \gamma_{-h}\right), \qquad
53 + \gamma_h = \sum_i r_{t,i}\, r_{t,i-h},
54 + \label{eq:rk}
55 +\end{equation}
56 +on 1-minute returns, where $k(\cdot)$ is the Parzen kernel and the
57 +bandwidth follows the feasible rule $H^{*} = c^{*} \xi^{4/5} n^{3/5}$
58 +with $c^{*} = 3.5134$, noise-to-signal ratio
59 +$\xi^2 = \hat\omega^2 / \widehat{IV}$, noise variance estimated as
60 +$\hat\omega^2 = RV^{(1\mathrm{min})}_t / (2n)$, and $\widehat{IV}$ proxied
61 +by $RV^{ss}_t$ \citep{BNHLS2009}. The non-flat-top Parzen weights
62 +guarantee non-negativity.
63 +
64 +\paragraph{Bipower variation and jumps.}
65 +The realized bipower variation of \citet{BNS2004},
66 +\begin{equation}
67 + BV_t \;=\; \mu_1^{-2} \, \frac{n}{n-1} \sum_{i=2}^{n}
68 + |r_{t,i}|\,|r_{t,i-1}|, \qquad \mu_1 = \sqrt{2/\pi},
69 + \label{eq:bv}
70 +\end{equation}
71 +is consistent for integrated variance in the presence of jumps. Jump days
72 +are identified with the ratio-form test of \citet{BNS2006},
73 +\begin{equation}
74 + Z_t \;=\; \frac{1 - BV_t / RV_t}
75 + {\sqrt{\theta \, \max\!\left(1, \, TQ_t / BV_t^2\right) / n}}
76 + \;\xrightarrow{d}\; N(0,1), \qquad
77 + \theta = \tfrac{\pi^2}{4} + \pi - 5,
78 + \label{eq:bns}
79 +\end{equation}
80 +where $TQ_t$ is the tripower quarticity. The jump and continuous
81 +components are then
82 +\begin{equation}
83 + J_t = \mathbb{1}\{Z_t > \Phi^{-1}_{1-\alpha}\} \cdot
84 + \max(RV_t - BV_t, 0), \qquad
85 + C_t = RV_t - J_t,
86 + \label{eq:cj}
87 +\end{equation}
88 +with $\alpha = 0.001$ \citep{ABD2007}. To limit noise distortions,
89 +$BV_t$, $TQ_t$ and the test are computed on the sparse 5-minute grid.
90 +
91 +\paragraph{Realized semivariances.}
92 +Following \citet{BNKS2010}, the positive and negative semivariances
93 +\begin{equation}
94 + RS^{+}_t = \sum_i r_{t,i}^2 \, \mathbb{1}\{r_{t,i} > 0\}, \qquad
95 + RS^{-}_t = \sum_i r_{t,i}^2 \, \mathbb{1}\{r_{t,i} < 0\},
96 + \label{eq:semivar}
97 +\end{equation}
98 +decompose $RV_t$ by the sign of the underlying returns; the signed jump
99 +variation is $SJ_t = RS^{+}_t - RS^{-}_t$ \citep{PattonSheppard2015}.
100 +Finally, the realized quarticity $RQ_t = \tfrac{n}{3}\sum_i r_{t,i}^4$
101 +measures the sampling variability of $RV_t$ and feeds the HARQ model
102 +below.
103 +
104 +\paragraph{Trading sessions.}
105 +Equity and ETF measures use regular trading hours only (09:30--16:00,
106 +exchange time); FX and futures use the full trading day on a calendar-day
107 +partition, with holidays and thin days (fewer than 700 one-minute bars,
108 +250 for equities) discarded; crypto trades continuously and its day is
109 +the calendar day in UTC. Daily close-to-close returns $r_t$ (which
110 +include the overnight component for session-limited markets) feed the
111 +GARCH-family models and the economic-value exercises.
112 +
113 +\subsection{Econometric benchmarks}
114 +\label{sec:econ_models}
115 +
116 +\paragraph{HAR family.}
117 +The heterogeneous autoregressive model of \citet{Corsi2009} regresses
118 +future RV on daily, weekly and monthly averages,
119 +\begin{equation}
120 + \overline{RV}_{t+1:t+h} \;=\; \beta_0 + \beta_d \, RV_t
121 + + \beta_w \, \overline{RV}_{t-4:t}
122 + + \beta_m \, \overline{RV}_{t-21:t} + \varepsilon_{t+h},
123 + \label{eq:har}
124 +\end{equation}
125 +where $\overline{RV}_{t+1:t+h} = h^{-1}\sum_{j=1}^h RV_{t+j}$ and all
126 +horizons are estimated by direct projection. We consider five
127 +extensions: HAR-J adds the jump component $J_t$; HAR-CJ replaces the RV
128 +averages with continuous-component averages and adds $J_t$
129 +\citep{ABD2007}; SHAR splits the daily lag into $RS^{+}_t$ and
130 +$RS^{-}_t$ \citep{PattonSheppard2015}; HARQ interacts the daily lag with
131 +$\sqrt{RQ_t}$ to discount noisy RV observations \citep{BPQ2016}; and
132 +log-HAR estimates \eqref{eq:har} in logarithms, with forecasts
133 +retransformed as $\exp(\hat y + \hat\sigma^2_{\varepsilon}/2)$. Following
134 +\citet{BPQ2016}, all forecasts pass through an ``insanity filter'' that
135 +clips them to the range of the target observed in the estimation window.
136 +
137 +\paragraph{GARCH family.}
138 +On daily returns we estimate GARCH(1,1) \citep{Bollerslev1986},
139 +GJR-GARCH \citep{GJR1993} and EGARCH \citep{Nelson1991} with Gaussian
140 +quasi-likelihood. Because these models forecast the conditional variance
141 +of close-to-close returns --- which includes overnight variation absent
142 +from intraday RV --- their forecasts are mapped into RV units with the
143 +in-sample proportionality factor $\hat c = \overline{RV} /
144 +\overline{\hat\varepsilon^2}$ re-estimated in each rolling window, in the
145 +spirit of \citet{HansenLunde2005}. Multi-step variance paths are
146 +analytic for GARCH and GJR and simulated (500 paths) for EGARCH.
147 +
148 +\paragraph{Realized GARCH.}
149 +The log-linear Realized GARCH(1,1) of \citet{HansenHuangShek2012} couples
150 +the latent variance $h_t$ with the realized measure $x_t$:
151 +\begin{align}
152 + r_t &= \sqrt{h_t}\, z_t, \qquad z_t \sim N(0,1), \notag \\
153 + \log h_t &= \omega + \beta \log h_{t-1} + \gamma \log x_{t-1},
154 + \label{eq:rgarch} \\
155 + \log x_t &= \xi + \varphi \log h_t + \tau_1 z_t + \tau_2(z_t^2 - 1)
156 + + u_t, \qquad u_t \sim N(0, \sigma_u^2). \notag
157 +\end{align}
158 +We estimate \eqref{eq:rgarch} by quasi-maximum likelihood and forecast
159 +the realized measure directly, using the analytic recursion for
160 +$\mathbb{E}[\log h_{t+k}]$ with persistence $\pi = \beta +
161 +\gamma\varphi$ and a lognormal correction for the accumulated forecast
162 +variance.
163 +
164 +\subsection{Machine-learning models}
165 +\label{sec:ml_models}
166 +
167 +\paragraph{Information sets.}
168 +All learners forecast $\log \overline{RV}_{t+1:t+h}$ and are evaluated
169 +after retransformation with the nonparametric smearing factor of
170 +\citet{Duan1983} and the same insanity filter as the HAR models. Two
171 +nested predictor sets are used. The \emph{HAR set} contains exactly the
172 +HAR information: $\log RV_t$, $\log \overline{RV}_{t-4:t}$,
173 +$\log \overline{RV}_{t-21:t}$. The \emph{extended set} adds the jump
174 +share $J_t / RV_t$, log semivariances, normalized signed jumps, log
175 +realized quarticity, daily/weekly/monthly returns, log volume growth,
176 +the day of the week, the (lagged) log VIX level, and cross-asset
177 +averages of log RV within each asset class --- fourteen additional
178 +predictors, all known at the end of day $t$.
179 +
180 +\paragraph{Learners.}
181 +We consider penalized linear models --- ridge, LASSO
182 +\citep{Tibshirani1996} and elastic net \citep{ZouHastie2005} on
183 +standardized predictors --- and tree ensembles: random forests
184 +\citep{Breiman2001}, XGBoost \citep{ChenGuestrin2016} and LightGBM
185 +\citep{Ke2017}. Hyperparameters are tuned by walk-forward validation
186 +inside the estimation window (the last 20\% of the window, in time
187 +order, serves as the validation split), re-tuned once a year, and the
188 +winning configuration is refit on the full window; no future information
189 +enters the tuning loop at any point.
190 +
191 +\paragraph{Deep learners.}
192 +Two sequence models map the last 22 days of extended features to all
193 +three horizons jointly: a single-layer LSTM
194 +\citep{HochreiterSchmidhuber1997} with 64 hidden units, and a
195 +two-block Transformer encoder \citep{Vaswani2017} with model dimension
196 +32 and 4 attention heads. Both are trained with Adam, early stopping on
197 +a temporal validation split, and quarterly re-estimation.
198 +
199 +\paragraph{Pooled multi-asset model.}
200 +Beyond the asset-by-asset models, a single pooled LightGBM is trained on
201 +the stacked panel of all 25 assets with the ticker and asset-class
202 +identifiers as categorical features, re-estimated monthly on the
203 +trailing window. The pooled design tests whether volatility dynamics are
204 +sufficiently common across assets for cross-sectional information to
205 +sharpen asset-level forecasts \citep{GuKellyXiu2020,
206 +ChristensenSiggaardVeliyev2023}. A leave-one-out variant excludes a
207 +target asset from the training pool entirely and predicts it
208 +out-of-pool, measuring the transferability of pooled dynamics to unseen
209 +assets. Model interpretation relies on SHAP values \citep{LundbergLee2017}
210 +and permutation importance.
211 +
212 +\subsection{Forecast evaluation}
213 +\label{sec:evaluation}
214 +
215 +\paragraph{Rolling protocol.}
216 +All models are estimated on a rolling window of 1{,}000 trading days and
217 +re-estimated every 22 days; forecasts are produced daily for horizons
218 +$h \in \{1, 5, 22\}$, defined as the \emph{average} daily variance over
219 +the next $h$ days (cumulative variance equals $h$ times the average).
220 +The evaluation sample therefore starts after the first
221 +1{,}000-plus-$h$ observations of each series and runs through July
222 +2026.
223 +
224 +\paragraph{Loss functions.}
225 +Our primary loss is the QLIKE,
226 +\begin{equation}
227 + L^{Q}(RV, F) \;=\; \frac{RV}{F} - \log\frac{RV}{F} - 1,
228 + \label{eq:qlike}
229 +\end{equation}
230 +which together with the MSE is robust to noise in the volatility proxy
231 +\citep{Patton2011}: rankings based on \eqref{eq:qlike} are consistent
232 +for the rankings that would obtain under the true conditional variance.
233 +QLIKE is also scale-free, which permits averaging across assets; MSE
234 +results are reported as a complement.
235 +
236 +\paragraph{Statistical tests.}
237 +Pairwise forecast comparisons use \citet{DieboldMariano1995} tests with
238 +Newey--West long-run variances \citep{NeweyWest1987} and bandwidth
239 +$h-1$. Joint comparisons use the 90\% model confidence set of
240 +\citet{HLN2011} with the $T_{\max}$ statistic and 5{,}000 stationary
241 +bootstrap replications \citep{PolitisRomano1994}. Forecast
242 +unbiasedness is assessed with \citet{MincerZarnowitz1969} levels
243 +regressions, $RV_{t+1:t+h} = a + b F_{t,h} + e_{t+h}$, testing
244 +$(a,b) = (0,1)$ with HAC standard errors. Whether machine-learning
245 +forecasts contain information absent from HAR is tested with forecast
246 +encompassing regressions \citep{HarveyLeybourneNewbold1998},
247 +$RV = a + b_1 F^{HAR} + b_2 F^{ML} + e$: rejection of $b_2 = 0$ means
248 +HAR does not encompass the rival. Finally, we report two forecast
249 +combinations \citep{BatesGranger1969,Timmermann2006}: the simple mean
250 +of all individual models and inverse-MSE weights computed on a trailing
251 +60-day window.
252 +
253 +\paragraph{Stability.}
254 +Because relative performance may drift across regimes, we complement
255 +full-sample tests with the fluctuation test of \citet{GiacominiRossi2010},
256 +which tracks the standardized DM statistic over a centred rolling window
257 +of 30\% of the evaluation sample and compares its path to
258 +regime-robust critical values.
259 +
260 +\subsection{Economic value}
261 +\label{sec:econ_value_method}
262 +
263 +\paragraph{Volatility timing.}
264 +For each asset and model we run the volatility-timing strategy of
265 +\citet{FKO2001,FKO2003}: the weight on the risky asset is
266 +$w_t = \min\!\left(\bar w, \; \sigma^{\ast} / \hat\sigma_{t+1}\right)$,
267 +with annualized target $\sigma^{\ast} = 10\%$, leverage cap
268 +$\bar w = 3$, and $\hat\sigma_{t+1}$ the one-day-ahead forecast
269 +volatility. Net returns subtract proportional transaction costs of 5
270 +basis points on turnover (0--20 bps in sensitivity checks). We report
271 +annualized Sharpe ratios and the \citet{FKO2001} utility gain --- the
272 +annualized fee $\Delta$ a mean-variance investor with relative risk
273 +aversion $\gamma = 5$ would pay to switch from HAR-based to model-based
274 +timing.
275 +
276 +\paragraph{Value-at-Risk.}
277 +One-day Gaussian VaR forecasts, $\mathrm{VaR}^{q}_{t+1} = -z_q \,
278 +\hat\sigma^{r}_{t+1}$, are backtested at the 1\% and 5\% levels with the
279 +unconditional-coverage test of \citet{Kupiec1995} and the
280 +conditional-coverage test of \citet{Christoffersen1998}. Because RV
281 +omits overnight variance, forecasts are rescaled into close-to-close
282 +return units with a trailing 250-day variance ratio, so no future
283 +information is used.
added requirements.txt +24 −0
@@ -0,0 +1,24 @@
1 +# ================================================================
2 +# Auteur : Simon-Pierre Boucher
3 +# Contact : contact@spboucher.ai
4 +# Projet : Prévision de volatilité réalisée multi-actifs
5 +# (HAR-RV vs GARCH vs Machine Learning)
6 +# Fichier : requirements.txt
7 +# Description : Dépendances Python (>= 3.11), versions épinglées
8 +# à l'environnement de l'analyse.
9 +# ================================================================
10 +numpy==2.4.4
11 +pandas==3.0.2
12 +pyarrow>=16.0
13 +scipy==1.17.1
14 +scikit-learn==1.6.1
15 +statsmodels==0.14.6
16 +arch==8.0.0
17 +xgboost==3.2.0
18 +lightgbm==4.6.0
19 +torch==2.12.0
20 +shap==0.48.0
21 +matplotlib==3.10.9
22 +requests==2.34.2
23 +PyYAML==6.0.3
24 +pytest==9.1.1
modified scripts/03_run_models.py +16 −1
@@ -100,6 +100,19 @@ def run_deep(ticker: str) -> str:
100 100 return f"{ticker}: {len(out)} deep forecasts"
101 101
102 102
103 +def run_interpret() -> str:
104 + """SHAP values and permutation importance of the pooled LightGBM."""
105 + from wp12 import models_ml as ml
106 +
107 + panel = pd.read_parquet(config.path("processed") / "panel.parquet")
108 + frames = ml.ml_features(panel)
109 + shap_tab, perm_tab = ml.shap_and_permutation(frames, h=1, window=WINDOW)
110 + out = config.path("reproduced")
111 + shap_tab.to_csv(out / "shap_h1.csv", index=False)
112 + perm_tab.to_csv(out / "permutation_h1.csv", index=False)
113 + return f"interpret: {len(shap_tab)} features"
114 +
115 +
103 116 def run_pooled() -> str:
104 117 """Pooled multi-asset LightGBM (single job, all tickers jointly)."""
105 118 from wp12 import models_ml as ml
@@ -117,7 +130,7 @@ def main() -> None:
117 130 """Dispatch model runs across tickers with a process pool."""
118 131 ap = argparse.ArgumentParser()
119 132 ap.add_argument("--family", default="all",
120 − choices=["econ", "ml", "deep", "pooled", "all"])
133 + choices=["econ", "ml", "deep", "pooled", "interpret", "all"])
121 134 ap.add_argument("--tickers", default=None, help="comma-separated subset")
122 135 ap.add_argument("--workers", type=int, default=8)
123 136 args = ap.parse_args()
@@ -147,6 +160,8 @@ def main() -> None:
147 160 logger.error("FAILED %s %s: %s", name, tk, exc)
148 161 if args.family in ("pooled", "all"):
149 162 logger.info(run_pooled())
163 + if args.family in ("interpret", "all"):
164 + logger.info(run_interpret())
150 165
151 166
152 167 if __name__ == "__main__":
added scripts/04_evaluate.py +230 −0
@@ -0,0 +1,230 @@
1 +#!/usr/bin/env python3
2 +"""
3 +================================================================
4 +Auteur : Simon-Pierre Boucher
5 +Contact : contact@spboucher.ai
6 +Projet : Prévision de volatilité réalisée multi-actifs
7 + (HAR-RV vs GARCH vs Machine Learning)
8 +Fichier : 04_evaluate.py
9 +Description : Étape 04 — Évaluation out-of-sample : assemblage
10 + des prévisions, combinaisons, pertes QLIKE/MSE,
11 + Diebold-Mariano vs HAR, Model Confidence Set,
12 + Mincer-Zarnowitz et forecast encompassing.
13 +Usage : python scripts/04_evaluate.py
14 +================================================================
15 +"""
16 +
17 +from __future__ import annotations
18 +
19 +import logging
20 +import sys
21 +from pathlib import Path
22 +
23 +import numpy as np
24 +import pandas as pd
25 +
26 +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
27 +
28 +from wp12 import config, evaluation as ev # noqa: E402
29 +from wp12.models_econ import horizon_target # noqa: E402
30 +
31 +logger = logging.getLogger("04_evaluate")
32 +
33 +FC_DIR = config.path("processed") / "forecasts"
34 +OUT = config.path("reproduced")
35 +HORIZONS = tuple(config.load_config()["evaluation"]["horizons"])
36 +BENCH = "HAR"
37 +
38 +
39 +def load_forecasts() -> pd.DataFrame:
40 + """Concatenate every stored forecast file into one long table."""
41 + frames = []
42 + for f in sorted(FC_DIR.glob("*.parquet")):
43 + df = pd.read_parquet(f)
44 + frames.append(df)
45 + out = pd.concat(frames, ignore_index=True)
46 + out["date"] = pd.to_datetime(out["date"])
47 + logger.info("loaded %d forecasts, %d models, %d tickers",
48 + len(out), out["model"].nunique(), out["ticker"].nunique())
49 + return out
50 +
51 +
52 +def attach_actuals(fc: pd.DataFrame, measure: str = "rv5ss") -> pd.DataFrame:
53 + """Join the realized horizon-average RV target onto the forecast table."""
54 + panel = pd.read_parquet(config.path("processed") / "panel.parquet")
55 + targets = []
56 + for tk, df in panel.groupby("ticker"):
57 + rv = df.sort_index()[measure]
58 + for h in HORIZONS:
59 + t = horizon_target(rv, h).rename("actual").reset_index()
60 + t["ticker"] = tk
61 + t["h"] = h
62 + targets.append(t)
63 + tgt = pd.concat(targets, ignore_index=True)
64 + merged = fc.merge(tgt, on=["date", "ticker", "h"], how="inner")
65 + merged = merged.dropna(subset=["actual", "forecast"])
66 + merged = merged[merged["actual"] > 0]
67 + logger.info("merged: %d forecast-actual pairs", len(merged))
68 + return merged
69 +
70 +
71 +def add_combinations(merged: pd.DataFrame) -> pd.DataFrame:
72 + """Append simple-mean and inverse-MSE combination forecasts."""
73 + combos = []
74 + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True):
75 + wide = sub.pivot_table(index="date", columns="model", values="forecast")
76 + actual = sub.drop_duplicates("date").set_index("date")["actual"]
77 + c = ev.combine_forecasts(wide, actual.reindex(wide.index))
78 + c = c.stack().rename("forecast").reset_index()
79 + c.columns = ["date", "model", "forecast"]
80 + c["ticker"] = tk
81 + c["h"] = h
82 + c = c.merge(actual.rename("actual").reset_index(), on="date")
83 + combos.append(c.dropna(subset=["forecast", "actual"]))
84 + out = pd.concat([merged] + combos, ignore_index=True)
85 + logger.info("with combinations: %d rows", len(out))
86 + return out
87 +
88 +
89 +def loss_tables(merged: pd.DataFrame) -> None:
90 + """Mean losses overall / by class / by ticker, with ratios to HAR."""
91 + panel_cls = (
92 + pd.read_parquet(config.path("processed") / "panel.parquet")
93 + .groupby("ticker")["cls"].first()
94 + )
95 + merged = merged.assign(cls=merged["ticker"].map(panel_cls))
96 + for loss in ("qlike", "mse"):
97 + lf = ev.LOSSES[loss]
98 + merged[loss] = lf(merged["actual"].to_numpy(), merged["forecast"].to_numpy())
99 +
100 + per_asset = merged.groupby(["ticker", "cls", "h", "model"], observed=True)[
101 + ["qlike", "mse"]].mean().reset_index()
102 + per_asset.to_csv(OUT / "losses_by_ticker.csv", index=False)
103 +
104 + def _with_ratio(df: pd.DataFrame, keys: list[str]) -> pd.DataFrame:
105 + bench = df[df["model"] == BENCH].set_index(keys)["qlike"]
106 + df = df.set_index(keys)
107 + df["qlike_ratio"] = df["qlike"] / bench
108 + return df.reset_index()
109 +
110 + overall = per_asset.groupby(["h", "model"], observed=True)[
111 + ["qlike", "mse"]].mean().reset_index()
112 + _with_ratio(overall, ["h"]).to_csv(OUT / "losses_overall.csv", index=False)
113 +
114 + by_class = per_asset.groupby(["cls", "h", "model"], observed=True)[
115 + ["qlike", "mse"]].mean().reset_index()
116 + _with_ratio(by_class, ["cls", "h"]).to_csv(OUT / "losses_by_class.csv", index=False)
117 + logger.info("loss tables written")
118 +
119 +
120 +def dm_tables(merged: pd.DataFrame) -> None:
121 + """Diebold-Mariano tests of every model against the HAR benchmark."""
122 + rows = []
123 + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True):
124 + wide_f = sub.pivot_table(index="date", columns="model", values="forecast")
125 + actual = sub.drop_duplicates("date").set_index("date")["actual"]
126 + if BENCH not in wide_f.columns:
127 + continue
128 + lb = ev.qlike(actual.to_numpy(), wide_f[BENCH].to_numpy())
129 + for m in wide_f.columns:
130 + if m == BENCH:
131 + continue
132 + both = np.isfinite(wide_f[m].to_numpy()) & np.isfinite(wide_f[BENCH].to_numpy())
133 + la = ev.qlike(actual.to_numpy()[both], wide_f[m].to_numpy()[both])
134 + stat, p = ev.diebold_mariano(la, lb[both], h=h)
135 + rows.append((tk, h, m, stat, p))
136 + dm = pd.DataFrame(rows, columns=["ticker", "h", "model", "dm_stat", "dm_p"])
137 + dm.to_csv(OUT / "dm_by_ticker.csv", index=False)
138 + summary = dm.groupby(["h", "model"], observed=True).apply(
139 + lambda d: pd.Series({
140 + "n_assets": len(d),
141 + "median_dm": d["dm_stat"].median(),
142 + "pct_better": float((d["dm_stat"] < 0).mean()),
143 + "pct_sig_better": float(((d["dm_stat"] < 0) & (d["dm_p"] < 0.05)).mean()),
144 + "pct_sig_worse": float(((d["dm_stat"] > 0) & (d["dm_p"] < 0.05)).mean()),
145 + }), include_groups=False).reset_index()
146 + summary.to_csv(OUT / "dm_summary.csv", index=False)
147 + logger.info("DM tables written")
148 +
149 +
150 +def mcs_tables(merged: pd.DataFrame) -> None:
151 + """90% Model Confidence Set per (ticker, horizon); inclusion shares."""
152 + cfg = config.load_config()["evaluation"]
153 + rows = []
154 + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True):
155 + wide_f = sub.pivot_table(index="date", columns="model", values="forecast")
156 + actual = sub.drop_duplicates("date").set_index("date")["actual"]
157 + losses = pd.DataFrame({
158 + m: ev.qlike(actual.to_numpy(), wide_f[m].to_numpy())
159 + for m in wide_f.columns
160 + }, index=wide_f.index).dropna()
161 + if len(losses) < 200:
162 + continue
163 + mcs = ev.model_confidence_set(
164 + losses, alpha=cfg["mcs_alpha"], n_boot=int(cfg["mcs_bootstrap"]))
165 + mcs["ticker"] = tk
166 + mcs["h"] = h
167 + rows.append(mcs)
168 + allmcs = pd.concat(rows, ignore_index=True)
169 + allmcs.to_csv(OUT / "mcs_by_ticker.csv", index=False)
170 + inc = allmcs.groupby(["h", "model"], observed=True)["in_mcs"].mean().rename(
171 + "mcs_inclusion").reset_index()
172 + inc.to_csv(OUT / "mcs_inclusion.csv", index=False)
173 + logger.info("MCS tables written")
174 +
175 +
176 +def mz_and_encompassing(merged: pd.DataFrame) -> None:
177 + """Mincer-Zarnowitz per model and HAR-vs-ML encompassing tests."""
178 + rows = []
179 + for (tk, h, m), sub in merged.groupby(["ticker", "h", "model"], observed=True):
180 + res = ev.mincer_zarnowitz(sub["actual"].to_numpy(), sub["forecast"].to_numpy(), h=h)
181 + rows.append((tk, h, m, res["alpha"], res["beta"], res["r2"], res["p_joint"]))
182 + mz = pd.DataFrame(rows, columns=["ticker", "h", "model", "alpha", "beta", "r2", "p_joint"])
183 + mz.to_csv(OUT / "mz_by_ticker.csv", index=False)
184 + mz_sum = mz.groupby(["h", "model"], observed=True).apply(
185 + lambda d: pd.Series({
186 + "med_beta": d["beta"].median(), "med_r2": d["r2"].median(),
187 + "pct_reject": float((d["p_joint"] < 0.05).mean()),
188 + }), include_groups=False).reset_index()
189 + mz_sum.to_csv(OUT / "mz_summary.csv", index=False)
190 +
191 + rows = []
192 + rivals = [m for m in merged["model"].unique() if m != BENCH]
193 + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True):
194 + wide_f = sub.pivot_table(index="date", columns="model", values="forecast")
195 + actual = sub.drop_duplicates("date").set_index("date")["actual"]
196 + if BENCH not in wide_f.columns:
197 + continue
198 + for m in rivals:
199 + if m not in wide_f.columns:
200 + continue
201 + res = ev.forecast_encompassing(
202 + actual.to_numpy(), wide_f[BENCH].to_numpy(), wide_f[m].to_numpy(), h=h)
203 + rows.append((tk, h, m, res["b1"], res["b2"], res["p_b2"]))
204 + enc = pd.DataFrame(rows, columns=["ticker", "h", "model", "b1", "b2", "p_b2"])
205 + enc.to_csv(OUT / "encompassing_by_ticker.csv", index=False)
206 + enc_sum = enc.groupby(["h", "model"], observed=True).apply(
207 + lambda d: pd.Series({
208 + "med_b2": d["b2"].median(),
209 + "pct_adds_info": float(((d["b2"] > 0) & (d["p_b2"] < 0.05)).mean()),
210 + }), include_groups=False).reset_index()
211 + enc_sum.to_csv(OUT / "encompassing_summary.csv", index=False)
212 + logger.info("MZ and encompassing tables written")
213 +
214 +
215 +def main() -> None:
216 + """Run the full evaluation battery and persist intermediate tables."""
217 + config.setup_logging()
218 + fc = load_forecasts()
219 + merged = attach_actuals(fc)
220 + merged = add_combinations(merged)
221 + merged.to_parquet(config.path("processed") / "merged_forecasts.parquet", index=False)
222 + loss_tables(merged)
223 + dm_tables(merged)
224 + mcs_tables(merged)
225 + mz_and_encompassing(merged)
226 + logger.info("evaluation complete")
227 +
228 +
229 +if __name__ == "__main__":
230 + main()
added scripts/05_robustness.py +220 −0
@@ -0,0 +1,220 @@
1 +#!/usr/bin/env python3
2 +"""
3 +================================================================
4 +Auteur : Simon-Pierre Boucher
5 +Contact : contact@spboucher.ai
6 +Projet : Prévision de volatilité réalisée multi-actifs
7 + (HAR-RV vs GARCH vs Machine Learning)
8 +Fichier : 05_robustness.py
9 +Description : Étape 05 — Robustesse : sous-périodes, sensibilité
10 + à la fréquence d'échantillonnage et à la fenêtre
11 + d'estimation, transférabilité cross-asset (pooled
12 + leave-one-out) et fluctuation test Giacomini-Rossi.
13 +Usage : python scripts/05_robustness.py [--part all|subperiods|freq|window|loo|gr]
14 +================================================================
15 +"""
16 +
17 +from __future__ import annotations
18 +
19 +import argparse
20 +import logging
21 +import sys
22 +from concurrent.futures import ProcessPoolExecutor, as_completed
23 +from pathlib import Path
24 +
25 +import numpy as np
26 +import pandas as pd
27 +
28 +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
29 +
30 +from wp12 import config, evaluation as ev, robustness as rb # noqa: E402
31 +from wp12.models_econ import horizon_target # noqa: E402
32 +
33 +logger = logging.getLogger("05_robustness")
34 +
35 +OUT = config.path("reproduced")
36 +HORIZONS = tuple(config.load_config()["evaluation"]["horizons"])
37 +WINDOW = int(config.load_config()["evaluation"]["estimation_window_days"])
38 +REFIT = int(config.load_config()["evaluation"]["reestimation_freq_days"])
39 +
40 +# representative asset per class for the expensive experiments
41 +REPRESENTATIVE = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"}
42 +# models re-run in the sensitivity experiments
43 +SENS_MODELS = ["HAR", "LogHAR", "LightGBM"]
44 +
45 +
46 +def _merged() -> pd.DataFrame:
47 + df = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet")
48 + df["date"] = pd.to_datetime(df["date"])
49 + return df
50 +
51 +
52 +def part_subperiods() -> None:
53 + """Mean QLIKE per model within each sub-period of the config."""
54 + sp = {k: tuple(v) for k, v in config.load_config()["robustness"]["subperiods"].items()}
55 + merged = _merged()
56 + tab = rb.loss_by_period(merged, sp)
57 + tab.to_csv(OUT / "subperiod_losses.csv", index=False)
58 + logger.info("subperiod table written (%d rows)", len(tab))
59 +
60 +
61 +def _sens_one(job: tuple[str, str, str, int]) -> pd.DataFrame:
62 + """Worker: re-run one (ticker, model, measure, window) variant."""
63 + from wp12 import models_econ as me, models_ml as ml
64 +
65 + ticker, model, measure, window = job
66 + df = pd.read_parquet(config.path("processed") / f"rv_{ticker}.parquet")
67 + if model in ("HAR", "LogHAR"):
68 + fc = me.rolling_har_forecasts(
69 + df, model, horizons=HORIZONS, measure=measure,
70 + window=window, refit_every=REFIT)
71 + else: # LightGBM extended
72 + panel = pd.read_parquet(config.path("processed") / "panel.parquet")
73 + frames = ml.ml_features(panel, measure=measure)
74 + fc = ml.rolling_ml_forecasts(
75 + frames[ticker], "LightGBM", feature_set="extended",
76 + horizons=HORIZONS, window=window, refit_every=REFIT)
77 + fc["model"] = "LightGBM"
78 + fc["ticker"] = ticker
79 + # attach actuals in the SAME measure
80 + rv = df[measure]
81 + parts = []
82 + for h in HORIZONS:
83 + t = horizon_target(rv, h).rename("actual").reset_index()
84 + t["h"] = h
85 + parts.append(t)
86 + tgt = pd.concat(parts, ignore_index=True)
87 + fc["date"] = pd.to_datetime(fc["date"])
88 + tgt["date"] = pd.to_datetime(tgt["date"])
89 + out = fc.merge(tgt, on=["date", "h"]).dropna(subset=["actual", "forecast"])
90 + return out[out["actual"] > 0]
91 +
92 +
93 +def _run_jobs(jobs: list[tuple], workers: int) -> list[pd.DataFrame]:
94 + res = []
95 + with ProcessPoolExecutor(max_workers=workers) as pool:
96 + futs = {pool.submit(_sens_one, j): j for j in jobs}
97 + for fut in as_completed(futs):
98 + j = futs[fut]
99 + try:
100 + res.append(fut.result())
101 + logger.info("done %s", j)
102 + except Exception as exc: # noqa: BLE001
103 + logger.error("FAILED %s: %s", j, exc)
104 + return res
105 +
106 +
107 +def part_frequency(workers: int) -> None:
108 + """Sensitivity to the RV sampling scheme (1-min, 5-min ss, kernel)."""
109 + tickers = [i["ticker"] for i in config.universe()]
110 + freqs = {"rv1min": "rv1", "rv5min_ss": "rv5ss", "rkernel": "rk"}
111 + results = {}
112 + for label, measure in freqs.items():
113 + jobs = [(tk, m, measure, WINDOW) for tk in tickers for m in SENS_MODELS]
114 + frames = _run_jobs(jobs, workers)
115 + results[label] = pd.concat(frames, ignore_index=True)
116 + tab = rb.sensitivity_table(results, "frequency")
117 + tab.to_csv(OUT / "frequency_sensitivity.csv", index=False)
118 + logger.info("frequency sensitivity written")
119 +
120 +
121 +def part_window(workers: int) -> None:
122 + """Sensitivity to the estimation window (500 / 1000 / 2000 days)."""
123 + tickers = [i["ticker"] for i in config.universe()]
124 + results = {}
125 + for w in config.load_config()["robustness"]["window_sizes"]:
126 + jobs = [(tk, m, "rv5ss", int(w)) for tk in tickers for m in SENS_MODELS]
127 + frames = _run_jobs(jobs, workers)
128 + results[str(w)] = pd.concat(frames, ignore_index=True)
129 + tab = rb.sensitivity_table(results, "window")
130 + tab.to_csv(OUT / "window_sensitivity.csv", index=False)
131 + logger.info("window sensitivity written")
132 +
133 +
134 +def part_loo() -> None:
135 + """Cross-asset transferability: pooled model with one class representative
136 + excluded from training, predicted out-of-pool."""
137 + from wp12 import models_ml as ml
138 +
139 + panel = pd.read_parquet(config.path("processed") / "panel.parquet")
140 + frames = ml.ml_features(panel)
141 + parts = []
142 + for cls, tk in REPRESENTATIVE.items():
143 + fc = ml.rolling_pooled_lgbm(
144 + frames, horizons=HORIZONS, window=WINDOW, refit_every=REFIT, exclude=tk)
145 + fc["cls"] = cls
146 + parts.append(fc)
147 + loo = pd.concat(parts, ignore_index=True)
148 + loo.to_parquet(config.path("processed") / "forecasts" / "pooled_loo.parquet",
149 + index=False)
150 + # compare in-pool vs out-of-pool QLIKE on the same dates
151 + merged = _merged()
152 + rows = []
153 + for cls, tk in REPRESENTATIVE.items():
154 + sub_loo = loo[loo["ticker"] == tk].copy()
155 + sub_loo["date"] = pd.to_datetime(sub_loo["date"])
156 + base = merged[(merged["ticker"] == tk)
157 + & (merged["model"].isin(["Pooled-LGBM", "HAR", "LightGBM-X"]))]
158 + act = base.drop_duplicates(["date", "h"])[["date", "h", "actual"]]
159 + sub = sub_loo.merge(act, on=["date", "h"]).dropna(subset=["actual"])
160 + for h in HORIZONS:
161 + d = sub[sub["h"] == h]
162 + q_loo = float(np.mean(ev.qlike(d["actual"].to_numpy(), d["forecast"].to_numpy())))
163 + for m in ["Pooled-LGBM", "HAR", "LightGBM-X"]:
164 + b = base[(base["h"] == h) & (base["model"] == m)]
165 + b = b.merge(d[["date"]], on="date")
166 + q = float(np.mean(ev.qlike(b["actual"].to_numpy(), b["forecast"].to_numpy())))
167 + rows.append((cls, tk, h, m, q))
168 + rows.append((cls, tk, h, "Pooled-LGBM-LOO", q_loo))
169 + tab = pd.DataFrame(rows, columns=["cls", "ticker", "h", "model", "qlike"])
170 + tab.to_csv(OUT / "transferability.csv", index=False)
171 + logger.info("transferability written")
172 +
173 +
174 +def part_gr() -> None:
175 + """Giacomini-Rossi fluctuation paths: best ML vs HAR, h=1, per class rep."""
176 + merged = _merged()
177 + paths = []
178 + for cls, tk in REPRESENTATIVE.items():
179 + sub = merged[(merged["ticker"] == tk) & (merged["h"] == 1)]
180 + wide = sub.pivot_table(index="date", columns="model", values="forecast")
181 + actual = sub.drop_duplicates("date").set_index("date")["actual"]
182 + for m in ["LightGBM-X", "Pooled-LGBM"]:
183 + if m not in wide.columns:
184 + continue
185 + both = wide[[m, "HAR"]].dropna()
186 + la = pd.Series(ev.qlike(actual[both.index].to_numpy(), both[m].to_numpy()),
187 + index=both.index)
188 + lb = pd.Series(ev.qlike(actual[both.index].to_numpy(), both["HAR"].to_numpy()),
189 + index=both.index)
190 + path = rb.giacomini_rossi_fluctuation(la, lb, h=1, mu=0.3)
191 + path["cls"] = cls
192 + path["ticker"] = tk
193 + path["model"] = m
194 + paths.append(path)
195 + pd.concat(paths, ignore_index=True).to_csv(OUT / "gr_fluctuation.csv", index=False)
196 + logger.info("GR fluctuation paths written")
197 +
198 +
199 +def main() -> None:
200 + """Run the requested robustness parts."""
201 + ap = argparse.ArgumentParser()
202 + ap.add_argument("--part", default="all",
203 + choices=["all", "subperiods", "freq", "window", "loo", "gr"])
204 + ap.add_argument("--workers", type=int, default=8)
205 + args = ap.parse_args()
206 + config.setup_logging()
207 + if args.part in ("all", "subperiods"):
208 + part_subperiods()
209 + if args.part in ("all", "freq"):
210 + part_frequency(args.workers)
211 + if args.part in ("all", "window"):
212 + part_window(args.workers)
213 + if args.part in ("all", "loo"):
214 + part_loo()
215 + if args.part in ("all", "gr"):
216 + part_gr()
217 +
218 +
219 +if __name__ == "__main__":
220 + main()
added scripts/06_economic_value.py +150 −0
@@ -0,0 +1,150 @@
1 +#!/usr/bin/env python3
2 +"""
3 +================================================================
4 +Auteur : Simon-Pierre Boucher
5 +Contact : contact@spboucher.ai
6 +Projet : Prévision de volatilité réalisée multi-actifs
7 + (HAR-RV vs GARCH vs Machine Learning)
8 +Fichier : 06_economic_value.py
9 +Description : Étape 06 — Valeur économique : volatility timing
10 + (Sharpe net, sensibilité aux coûts), gains
11 + d'utilité mean-variance vs HAR, backtests VaR
12 + (Kupiec, Christoffersen).
13 +Usage : python scripts/06_economic_value.py
14 +================================================================
15 +"""
16 +
17 +from __future__ import annotations
18 +
19 +import logging
20 +import sys
21 +from pathlib import Path
22 +
23 +import numpy as np
24 +import pandas as pd
25 +
26 +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
27 +
28 +from wp12 import backtest as bt, config # noqa: E402
29 +
30 +logger = logging.getLogger("06_economic_value")
31 +
32 +OUT = config.path("reproduced")
33 +CFG = config.load_config()["backtest"]
34 +
35 +# models carried into the economic-value exercise (h=1 forecasts)
36 +EV_MODELS = [
37 + "HAR", "LogHAR", "HARQ", "SHAR", "HAR-CJ",
38 + "GARCH", "GJR", "EGARCH", "RealGARCH",
39 + "Ridge-X", "LASSO-X", "RF-X", "XGBoost-X", "LightGBM-X",
40 + "LSTM", "Transformer", "Pooled-LGBM", "Comb-Mean", "Comb-InvMSE",
41 +]
42 +
43 +
44 +def _load() -> tuple[pd.DataFrame, dict[str, pd.Series], dict[str, str]]:
45 + """Merged h=1 forecasts + daily returns per ticker + class map."""
46 + merged = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet")
47 + merged["date"] = pd.to_datetime(merged["date"])
48 + merged = merged[(merged["h"] == 1) & merged["model"].isin(EV_MODELS)]
49 + panel = pd.read_parquet(config.path("processed") / "panel.parquet")
50 + rets = {tk: d.sort_index()["ret_cc"] for tk, d in panel.groupby("ticker")}
51 + cls = panel.groupby("ticker")["cls"].first().to_dict()
52 + return merged, rets, cls
53 +
54 +
55 +def timing_tables(merged: pd.DataFrame, rets: dict, cls: dict) -> None:
56 + """Volatility-timing Sharpe (net) per model, plus cost sensitivity."""
57 + rows = []
58 + for (tk, m), sub in merged.groupby(["ticker", "model"], observed=True):
59 + pv = sub.set_index("date")["forecast"]
60 + ret = rets[tk]
61 + for tc in CFG["tc_grid_bps"]:
62 + led = bt.volatility_timing(
63 + ret, pv, vol_target_ann=CFG["vol_target_ann"],
64 + leverage_cap=CFG["leverage_cap"], tc_bps=float(tc))
65 + st = bt.perf_stats(led["net"])
66 + rows.append((tk, cls[tk], m, tc, st["ann_ret"], st["ann_vol"],
67 + st["sharpe"], st["max_dd"],
68 + float(led["weight"].diff().abs().mean())))
69 + tab = pd.DataFrame(rows, columns=[
70 + "ticker", "cls", "model", "tc_bps", "ann_ret", "ann_vol",
71 + "sharpe", "max_dd", "turnover"])
72 + tab.to_csv(OUT / "timing_by_ticker.csv", index=False)
73 + base = tab[tab["tc_bps"] == CFG["tc_bps"]]
74 + summary = base.groupby(["cls", "model"], observed=True)["sharpe"].mean().reset_index()
75 + overall = base.groupby("model", observed=True)[["sharpe", "ann_ret", "ann_vol",
76 + "turnover"]].mean().reset_index()
77 + overall["cls"] = "all"
78 + summary.to_csv(OUT / "timing_by_class.csv", index=False)
79 + overall.to_csv(OUT / "timing_overall.csv", index=False)
80 + logger.info("timing tables written")
81 +
82 +
83 +def utility_tables(merged: pd.DataFrame, rets: dict, cls: dict) -> None:
84 + """Annualized utility gain (bps) of each model over HAR."""
85 + gamma = CFG["risk_aversion"]
86 + rows = []
87 + for tk, sub in merged.groupby("ticker", observed=True):
88 + ret = rets[tk]
89 + nets = {}
90 + for m, s in sub.groupby("model", observed=True):
91 + pv = s.set_index("date")["forecast"]
92 + led = bt.volatility_timing(
93 + ret, pv, vol_target_ann=CFG["vol_target_ann"],
94 + leverage_cap=CFG["leverage_cap"], tc_bps=float(CFG["tc_bps"]))
95 + nets[m] = led["net"]
96 + if "HAR" not in nets:
97 + continue
98 + for m, net in nets.items():
99 + if m == "HAR":
100 + continue
101 + gain = bt.utility_gain(net, nets["HAR"], risk_aversion=gamma)
102 + rows.append((tk, cls[tk], m, gain * 1e4))
103 + tab = pd.DataFrame(rows, columns=["ticker", "cls", "model", "utility_gain_bps"])
104 + tab.to_csv(OUT / "utility_by_ticker.csv", index=False)
105 + tab.groupby("model", observed=True)["utility_gain_bps"].agg(
106 + ["mean", "median"]).reset_index().to_csv(OUT / "utility_summary.csv", index=False)
107 + logger.info("utility tables written")
108 +
109 +
110 +def var_tables(merged: pd.DataFrame, rets: dict, cls: dict) -> None:
111 + """VaR 1% / 5% backtests; RV forecasts scaled to return variance with a
112 + trailing 250-day factor (no look-ahead)."""
113 + rows = []
114 + for (tk, m), sub in merged.groupby(["ticker", "model"], observed=True):
115 + pv = sub.set_index("date")["forecast"]
116 + ret = rets[tk]
117 + # trailing conversion factor: return variance per unit of RV
118 + rv_al = pv.reindex(ret.index)
119 + c = ((ret**2).rolling(250).mean() / rv_al.rolling(250).mean()).clip(0.2, 5.0)
120 + pv_ret = (rv_al * c).dropna()
121 + for level in CFG["var_levels"]:
122 + res = bt.var_backtest(ret, pv_ret, float(level))
123 + rows.append((tk, cls[tk], m, level, res["n"], res["viol_rate"],
124 + res["kupiec_p"], res["christoffersen_p"]))
125 + tab = pd.DataFrame(rows, columns=[
126 + "ticker", "cls", "model", "level", "n", "viol_rate",
127 + "kupiec_p", "christoffersen_p"])
128 + tab.to_csv(OUT / "var_by_ticker.csv", index=False)
129 + summary = tab.groupby(["model", "level"], observed=True).apply(
130 + lambda d: pd.Series({
131 + "mean_viol_rate": d["viol_rate"].mean(),
132 + "pct_pass_kupiec": float((d["kupiec_p"] > 0.05).mean()),
133 + "pct_pass_cc": float((d["christoffersen_p"] > 0.05).mean()),
134 + }), include_groups=False).reset_index()
135 + summary.to_csv(OUT / "var_summary.csv", index=False)
136 + logger.info("VaR tables written")
137 +
138 +
139 +def main() -> None:
140 + """Run all economic-value blocks."""
141 + config.setup_logging()
142 + merged, rets, cls = _load()
143 + timing_tables(merged, rets, cls)
144 + utility_tables(merged, rets, cls)
145 + var_tables(merged, rets, cls)
146 + logger.info("economic value complete")
147 +
148 +
149 +if __name__ == "__main__":
150 + main()
added scripts/07_make_figures.py +263 −0
@@ -0,0 +1,263 @@
1 +#!/usr/bin/env python3
2 +"""
3 +================================================================
4 +Auteur : Simon-Pierre Boucher
5 +Contact : contact@spboucher.ai
6 +Projet : Prévision de volatilité réalisée multi-actifs
7 + (HAR-RV vs GARCH vs Machine Learning)
8 +Fichier : 07_make_figures.py
9 +Description : Étape 07 — Figures publication-ready du papier
10 + (séries RV, mesures, ACF, ratios QLIKE, MCS,
11 + sous-périodes, SHAP, Giacomini-Rossi, timing).
12 +Usage : python scripts/07_make_figures.py
13 +================================================================
14 +"""
15 +
16 +from __future__ import annotations
17 +
18 +import logging
19 +import sys
20 +from pathlib import Path
21 +
22 +import matplotlib.pyplot as plt
23 +import numpy as np
24 +import pandas as pd
25 +
26 +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
27 +
28 +from wp12 import backtest as bt, config # noqa: E402
29 +from wp12.plots import ( # noqa: E402
30 + BLUE, BLUES, GREY, INK, LIGHT, RED, TEXTWIDTH,
31 + annualized_vol, apply_style, panel_label,
32 +)
33 +
34 +logger = logging.getLogger("07_figures")
35 +
36 +FIG = config.path("figures")
37 +OUT = config.path("reproduced")
38 +REPRESENTATIVE = {"equity": "SPY", "fx": "EURUSD", "crypto": "BTC", "futures": "ES"}
39 +CLS_LABEL = {"equity": "Equities/ETFs", "fx": "FX", "crypto": "Crypto",
40 + "futures": "Futures"}
41 +
42 +# headline models shown in model-comparison figures, display order
43 +SHOW_MODELS = [
44 + "GARCH", "GJR", "EGARCH", "RealGARCH",
45 + "HAR-J", "HAR-CJ", "SHAR", "HARQ", "LogHAR",
46 + "Ridge-H", "LASSO-H", "RF-H", "XGBoost-H", "LightGBM-H",
47 + "Ridge-X", "LASSO-X", "RF-X", "XGBoost-X", "LightGBM-X",
48 + "LSTM", "Transformer", "Pooled-LGBM", "Comb-Mean", "Comb-InvMSE",
49 +]
50 +
51 +
52 +def _panel() -> pd.DataFrame:
53 + return pd.read_parquet(config.path("processed") / "panel.parquet")
54 +
55 +
56 +def fig_rv_series() -> None:
57 + """Annualized RV time series, one panel per asset class representative."""
58 + panel = _panel()
59 + fig, axes = plt.subplots(4, 1, figsize=(TEXTWIDTH, 6.6), sharex=True)
60 + for ax, (cls, tk) in zip(axes, REPRESENTATIVE.items()):
61 + df = panel[panel["ticker"] == tk].sort_index()
62 + ax.plot(df.index, annualized_vol(df["rv5ss"]), lw=0.4, color=BLUE)
63 + panel_label(ax, f"Panel {'ABCD'[list(REPRESENTATIVE).index(cls)]}. "
64 + f"{tk} ({CLS_LABEL[cls]})")
65 + ax.set_ylabel("Ann. vol. (%)")
66 + ax.set_ylim(bottom=0)
67 + axes[-1].set_xlabel("")
68 + fig.tight_layout()
69 + fig.savefig(FIG / "fig_rv_series.png")
70 + plt.close(fig)
71 +
72 +
73 +def fig_measures() -> None:
74 + """Yearly mean annualized vol by RV measure — microstructure noise check."""
75 + panel = _panel()
76 + fig, axes = plt.subplots(1, 2, figsize=(TEXTWIDTH, 2.6))
77 + for ax, tk, letter in zip(axes, ["SPY", "BTC"], "AB"):
78 + df = panel[panel["ticker"] == tk]
79 + yearly = df.groupby(df.index.year)[["rv1", "rv5ss", "rk"]].mean()
80 + for col, color, ls, lab in [
81 + ("rv1", RED, "-", "RV 1-min"),
82 + ("rv5ss", BLUE, "-", "RV 5-min (subsampled)"),
83 + ("rk", INK, "--", "Realized kernel"),
84 + ]:
85 + ax.plot(yearly.index, annualized_vol(yearly[col]), color=color,
86 + ls=ls, lw=1.0, label=lab)
87 + panel_label(ax, f"Panel {letter}. {tk}")
88 + ax.set_ylabel("Mean ann. vol. (%)")
89 + if letter == "A":
90 + ax.legend(loc="upper right", fontsize=7)
91 + fig.tight_layout()
92 + fig.savefig(FIG / "fig_measures.png")
93 + plt.close(fig)
94 +
95 +
96 +def fig_acf() -> None:
97 + """Autocorrelation of log RV by class representative (long memory)."""
98 + panel = _panel()
99 + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.8))
100 + colors = {"equity": BLUE, "fx": INK, "crypto": RED, "futures": GREY}
101 + for cls, tk in REPRESENTATIVE.items():
102 + s = np.log(panel[panel["ticker"] == tk]["rv5ss"].clip(lower=1e-12))
103 + acf = [s.autocorr(k) for k in range(1, 101)]
104 + ax.plot(range(1, 101), acf, lw=1.0, color=colors[cls],
105 + ls="--" if cls in ("fx", "futures") else "-",
106 + label=f"{tk} ({CLS_LABEL[cls]})")
107 + ax.axhline(0, color=GREY, lw=0.5)
108 + ax.set_xlabel("Lag (days)")
109 + ax.set_ylabel("ACF of log RV")
110 + ax.legend(fontsize=7)
111 + fig.tight_layout()
112 + fig.savefig(FIG / "fig_acf.png")
113 + plt.close(fig)
114 +
115 +
116 +def fig_qlike_ratio() -> None:
117 + """QLIKE ratio to HAR by model and horizon (dot plot)."""
118 + tab = pd.read_csv(OUT / "losses_overall.csv")
119 + models = [m for m in SHOW_MODELS if m in set(tab["model"])]
120 + fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True)
121 + for ax, h in zip(axes, (1, 5, 22)):
122 + sub = tab[tab["h"] == h].set_index("model").reindex(models)
123 + y = np.arange(len(models))
124 + better = sub["qlike_ratio"] < 1
125 + ax.scatter(sub["qlike_ratio"], y, s=14,
126 + c=[BLUE if b else RED for b in better], zorder=3)
127 + ax.axvline(1.0, color=GREY, lw=0.7, ls="--")
128 + ax.set_title(f"h = {h}", fontsize=9)
129 + ax.set_xlabel("QLIKE ratio vs HAR")
130 + ax.set_yticks(y)
131 + ax.set_yticklabels(models, fontsize=7)
132 + ax.invert_yaxis()
133 + fig.tight_layout()
134 + fig.savefig(FIG / "fig_qlike_ratio.png")
135 + plt.close(fig)
136 +
137 +
138 +def fig_mcs() -> None:
139 + """MCS inclusion frequency across assets, per horizon."""
140 + tab = pd.read_csv(OUT / "mcs_inclusion.csv")
141 + models = ["HAR"] + [m for m in SHOW_MODELS if m in set(tab["model"])]
142 + fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True)
143 + for ax, h in zip(axes, (1, 5, 22)):
144 + sub = tab[tab["h"] == h].set_index("model").reindex(models)
145 + y = np.arange(len(models))
146 + ax.barh(y, sub["mcs_inclusion"], color=BLUE, height=0.65)
147 + ax.set_yticks(y)
148 + ax.set_yticklabels(models, fontsize=7)
149 + ax.invert_yaxis()
150 + ax.set_xlim(0, 1)
151 + ax.set_title(f"h = {h}", fontsize=9)
152 + ax.set_xlabel("Share of assets in 90% MCS")
153 + fig.tight_layout()
154 + fig.savefig(FIG / "fig_mcs.png")
155 + plt.close(fig)
156 +
157 +
158 +def fig_subperiods() -> None:
159 + """QLIKE ratio vs HAR across sub-periods for selected models (h=1)."""
160 + tab = pd.read_csv(OUT / "subperiod_losses.csv")
161 + tab = tab[tab["h"] == 1]
162 + bench = tab[tab["model"] == "HAR"].set_index("period")["qlike"]
163 + show = ["LogHAR", "RealGARCH", "LightGBM-X", "Pooled-LGBM", "Comb-InvMSE", "LSTM"]
164 + periods = ["pre2020", "covid", "inflation", "recent"]
165 + plabel = {"pre2020": "2013–19", "covid": "2020", "inflation": "2021–22",
166 + "recent": "2024–26"}
167 + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.9))
168 + colors = [BLUE, INK, RED, BLUES[5], GREY, BLUES[2]]
169 + markers = ["o", "s", "^", "D", "v", "P"]
170 + for m, c, mk in zip(show, colors, markers):
171 + sub = tab[tab["model"] == m].set_index("period")
172 + ratio = (sub["qlike"] / bench).reindex(periods)
173 + ax.plot(range(len(periods)), ratio, marker=mk, ms=4, lw=1.0,
174 + color=c, label=m)
175 + ax.axhline(1.0, color=GREY, lw=0.7, ls="--")
176 + ax.set_xticks(range(len(periods)))
177 + ax.set_xticklabels([plabel[p] for p in periods])
178 + ax.set_ylabel("QLIKE ratio vs HAR (h=1)")
179 + ax.legend(fontsize=7, ncol=2)
180 + fig.tight_layout()
181 + fig.savefig(FIG / "fig_subperiods.png")
182 + plt.close(fig)
183 +
184 +
185 +def fig_shap() -> None:
186 + """Mean |SHAP| of the pooled LightGBM (h=1)."""
187 + tab = pd.read_csv(OUT / "shap_h1.csv").head(14)
188 + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.7, 3.2))
189 + y = np.arange(len(tab))
190 + ax.barh(y, tab["mean_abs_shap"], color=BLUE, height=0.65)
191 + ax.set_yticks(y)
192 + ax.set_yticklabels(tab["feature"], fontsize=7)
193 + ax.invert_yaxis()
194 + ax.set_xlabel(r"Mean $|$SHAP$|$ (log-RV units)")
195 + fig.tight_layout()
196 + fig.savefig(FIG / "fig_shap.png")
197 + plt.close(fig)
198 +
199 +
200 +def fig_gr() -> None:
201 + """Giacomini-Rossi fluctuation paths, LightGBM-X vs HAR."""
202 + tab = pd.read_csv(OUT / "gr_fluctuation.csv", parse_dates=["date"])
203 + tab = tab[tab["model"] == "LightGBM-X"]
204 + fig, axes = plt.subplots(2, 2, figsize=(TEXTWIDTH, 4.4))
205 + reps = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"}
206 + for ax, (cls, tk), letter in zip(axes.ravel(), reps.items(), "ABCD"):
207 + sub = tab[tab["ticker"] == tk]
208 + if sub.empty:
209 + continue
210 + ax.plot(sub["date"], sub["stat"], lw=0.7, color=BLUE)
211 + crit = sub["crit"].iloc[0]
212 + ax.axhline(crit, color=GREY, lw=0.6, ls="--")
213 + ax.axhline(-crit, color=GREY, lw=0.6, ls="--")
214 + ax.axhline(0, color=GREY, lw=0.4)
215 + panel_label(ax, f"Panel {letter}. {tk} ({CLS_LABEL[cls]})")
216 + ax.set_ylabel("Fluctuation stat.")
217 + fig.tight_layout()
218 + fig.savefig(FIG / "fig_gr.png")
219 + plt.close(fig)
220 +
221 +
222 +def fig_timing() -> None:
223 + """Cumulative net log returns of volatility timing on SPY."""
224 + merged = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet")
225 + merged["date"] = pd.to_datetime(merged["date"])
226 + panel = _panel()
227 + ret = panel[panel["ticker"] == "SPY"].sort_index()["ret_cc"]
228 + cfg = config.load_config()["backtest"]
229 + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.85, 3.0))
230 + series = [("HAR", INK, "-"), ("GARCH", GREY, "-"),
231 + ("LightGBM-X", BLUE, "-"), ("Comb-InvMSE", RED, "--")]
232 + for m, color, ls in series:
233 + sub = merged[(merged["ticker"] == "SPY") & (merged["h"] == 1)
234 + & (merged["model"] == m)]
235 + if sub.empty:
236 + continue
237 + pv = sub.set_index("date")["forecast"]
238 + led = bt.volatility_timing(ret, pv, cfg["vol_target_ann"],
239 + cfg["leverage_cap"], cfg["tc_bps"])
240 + ax.plot(led.index, led["net"].cumsum() * 100, lw=0.9, color=color,
241 + ls=ls, label=m)
242 + ax.set_ylabel("Cumulative net return (%)")
243 + ax.legend(fontsize=7)
244 + fig.tight_layout()
245 + fig.savefig(FIG / "fig_timing.png")
246 + plt.close(fig)
247 +
248 +
249 +def main() -> None:
250 + """Generate every figure of the paper."""
251 + config.setup_logging()
252 + apply_style()
253 + for fn in [fig_rv_series, fig_measures, fig_acf, fig_qlike_ratio,
254 + fig_mcs, fig_subperiods, fig_shap, fig_gr, fig_timing]:
255 + try:
256 + fn()
257 + logger.info("figure %s done", fn.__name__)
258 + except Exception as exc: # noqa: BLE001
259 + logger.error("figure %s FAILED: %s", fn.__name__, exc)
260 +
261 +
262 +if __name__ == "__main__":
263 + main()
added scripts/08_make_tables.py +404 −0
@@ -0,0 +1,404 @@
1 +#!/usr/bin/env python3
2 +"""
3 +================================================================
4 +Auteur : Simon-Pierre Boucher
5 +Contact : contact@spboucher.ai
6 +Projet : Prévision de volatilité réalisée multi-actifs
7 + (HAR-RV vs GARCH vs Machine Learning)
8 +Fichier : 08_make_tables.py
9 +Description : Étape 08 — Génération des tables LaTeX du papier
10 + (booktabs) à partir des CSV de results/reproduced.
11 + AUCUN chiffre n'est écrit à la main.
12 +Usage : python scripts/08_make_tables.py
13 +================================================================
14 +"""
15 +
16 +from __future__ import annotations
17 +
18 +import logging
19 +import sys
20 +from pathlib import Path
21 +
22 +import numpy as np
23 +import pandas as pd
24 +
25 +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
26 +
27 +from wp12 import config # noqa: E402
28 +
29 +logger = logging.getLogger("08_tables")
30 +
31 +OUT = config.path("reproduced")
32 +TAB = config.path("tables")
33 +
34 +GROUPS: list[tuple[str, list[str]]] = [
35 + ("A. GARCH family", ["GARCH", "GJR", "EGARCH", "RealGARCH"]),
36 + ("B. HAR family", ["HAR", "HAR-J", "HAR-CJ", "SHAR", "HARQ", "LogHAR"]),
37 + ("C. Machine learning --- HAR information set",
38 + ["Ridge-H", "LASSO-H", "ElasticNet-H", "RF-H", "XGBoost-H", "LightGBM-H"]),
39 + ("D. Machine learning --- extended information set",
40 + ["Ridge-X", "LASSO-X", "ElasticNet-X", "RF-X", "XGBoost-X", "LightGBM-X"]),
41 + ("E. Deep learning and pooled", ["LSTM", "Transformer", "Pooled-LGBM"]),
42 + ("F. Forecast combinations", ["Comb-Mean", "Comb-InvMSE"]),
43 +]
44 +
45 +CLS_LABEL = {"equity": "Equities/ETFs", "fx": "FX", "crypto": "Crypto",
46 + "futures": "Futures"}
47 +
48 +
49 +def _fmt(x: float, nd: int = 3, bold: bool = False) -> str:
50 + if not np.isfinite(x):
51 + return "---"
52 + s = f"{x:.{nd}f}"
53 + return f"\\textbf{{{s}}}" if bold else s
54 +
55 +
56 +def _write(name: str, lines: list[str]) -> None:
57 + f = TAB / f"{name}.tex"
58 + f.write_text("\n".join(lines) + "\n", encoding="utf-8")
59 + logger.info("wrote %s", f.name)
60 +
61 +
62 +def table_summary_stats() -> None:
63 + """Descriptive statistics of daily annualized RV by instrument."""
64 + panel = pd.read_parquet(config.path("processed") / "panel.parquet")
65 + lines = [
66 + "\\begin{tabular}{lrrrrrrr}", "\\toprule",
67 + "Ticker & Days & \\makecell{Mean vol.\\\\(\\%)} & \\makecell{Median vol.\\\\(\\%)}"
68 + " & \\makecell{P95 vol.\\\\(\\%)} & \\makecell{Skew.\\\\$\\ln$RV}"
69 + " & \\makecell{AC(1)\\\\$\\ln$RV} & \\makecell{Jump\\\\days (\\%)} \\\\",
70 + "\\midrule",
71 + ]
72 + for cls in ["equity", "fx", "crypto", "futures"]:
73 + sub = panel[panel["cls"] == cls]
74 + if sub.empty:
75 + continue
76 + lines.append(f"\\multicolumn{{8}}{{l}}{{\\itshape {CLS_LABEL[cls]}}}\\\\")
77 + for tk, df in sub.groupby("ticker"):
78 + vol = np.sqrt(df["rv5ss"] * 252) * 100
79 + lrv = np.log(df["rv5ss"].clip(lower=1e-12))
80 + lines.append(
81 + f"\\quad {tk} & {len(df):,} & {vol.mean():.1f} & {vol.median():.1f}"
82 + f" & {vol.quantile(0.95):.1f} & {lrv.skew():.2f}"
83 + f" & {lrv.autocorr(1):.2f} & {100 * (df['jump'] > 0).mean():.1f} \\\\"
84 + )
85 + lines += ["\\bottomrule", "\\end{tabular}"]
86 + _write("summary_stats", lines)
87 +
88 +
89 +def _grouped_metric_table(
90 + name: str, wide: pd.DataFrame, cols: list[tuple[str, str]],
91 + nd: int = 3, note_bench: str | None = "HAR", bold_min_cols: bool = True,
92 +) -> None:
93 + """Generic grouped model table. ``wide`` is indexed by model with the
94 + metric columns; ``cols`` maps (column, header)."""
95 + align = "l" + "c" * len(cols)
96 + lines = [f"\\begin{{tabular}}{{{align}}}", "\\toprule",
97 + "Model & " + " & ".join(h for _, h in cols) + " \\\\", "\\midrule"]
98 + mins = {c: wide[c].min() for c, _ in cols} if bold_min_cols else {}
99 + for gname, models in GROUPS:
100 + avail = [m for m in models if m in wide.index]
101 + if not avail:
102 + continue
103 + lines.append(f"\\multicolumn{{{len(cols) + 1}}}{{l}}{{\\itshape {gname}}}\\\\")
104 + for m in avail:
105 + cells = []
106 + for c, _ in cols:
107 + v = wide.loc[m, c]
108 + bold = bold_min_cols and np.isfinite(v) and v == mins[c]
109 + cells.append(_fmt(v, nd, bold))
110 + lines.append(f"\\quad {m} & " + " & ".join(cells) + " \\\\")
111 + lines += ["\\bottomrule", "\\end{tabular}"]
112 + _write(name, lines)
113 +
114 +
115 +def table_losses_main() -> None:
116 + """Headline QLIKE table: level for HAR, ratio for everything else."""
117 + tab = pd.read_csv(OUT / "losses_overall.csv")
118 + wide = tab.pivot_table(index="model", columns="h", values="qlike_ratio")
119 + wide.columns = [f"r{h}" for h in wide.columns]
120 + ql = tab.pivot_table(index="model", columns="h", values="qlike")
121 + ql.columns = [f"q{h}" for h in ql.columns]
122 + wide = wide.join(ql)
123 + cols = [("q1", "QLIKE $h{=}1$"), ("r1", "Ratio $h{=}1$"),
124 + ("q5", "QLIKE $h{=}5$"), ("r5", "Ratio $h{=}5$"),
125 + ("q22", "QLIKE $h{=}22$"), ("r22", "Ratio $h{=}22$")]
126 + _grouped_metric_table("losses_main", wide, cols)
127 +
128 +
129 +def table_losses_class() -> None:
130 + """QLIKE ratio vs HAR by asset class (h=1 and h=22)."""
131 + tab = pd.read_csv(OUT / "losses_by_class.csv")
132 + parts = {}
133 + for h in (1, 22):
134 + sub = tab[tab["h"] == h].pivot_table(
135 + index="model", columns="cls", values="qlike_ratio")
136 + for cls in ["equity", "fx", "crypto", "futures"]:
137 + if cls in sub.columns:
138 + parts[f"{cls}_{h}"] = sub[cls]
139 + wide = pd.DataFrame(parts)
140 + cols = ([(f"{c}_1", CLS_LABEL[c].split("/")[0]) for c in
141 + ["equity", "fx", "crypto", "futures"]]
142 + + [(f"{c}_22", CLS_LABEL[c].split("/")[0]) for c in
143 + ["equity", "fx", "crypto", "futures"]])
144 + align = "l" + "cccc" + "cccc"
145 + lines = [f"\\begin{{tabular}}{{{align}}}", "\\toprule",
146 + " & \\multicolumn{4}{c}{$h = 1$} & \\multicolumn{4}{c}{$h = 22$} \\\\",
147 + "\\cmidrule(lr){2-5}\\cmidrule(lr){6-9}",
148 + "Model & " + " & ".join(h for _, h in cols) + " \\\\", "\\midrule"]
149 + mins = {c: wide[c].min() for c, _ in cols}
150 + for gname, models in GROUPS:
151 + avail = [m for m in models if m in wide.index and m != "HAR"]
152 + if not avail:
153 + continue
154 + lines.append(f"\\multicolumn{{9}}{{l}}{{\\itshape {gname}}}\\\\")
155 + for m in avail:
156 + cells = [_fmt(wide.loc[m, c], 3, np.isfinite(wide.loc[m, c])
157 + and wide.loc[m, c] == mins[c]) for c, _ in cols]
158 + lines.append(f"\\quad {m} & " + " & ".join(cells) + " \\\\")
159 + lines += ["\\bottomrule", "\\end{tabular}"]
160 + _write("losses_class", lines)
161 +
162 +
163 +def table_dm_mcs() -> None:
164 + """DM outcomes vs HAR and MCS inclusion, side by side."""
165 + dm = pd.read_csv(OUT / "dm_summary.csv")
166 + mcs = pd.read_csv(OUT / "mcs_inclusion.csv")
167 + parts = {}
168 + for h in (1, 5, 22):
169 + d = dm[dm["h"] == h].set_index("model")
170 + parts[f"sb{h}"] = d["pct_sig_better"] * 100
171 + parts[f"sw{h}"] = d["pct_sig_worse"] * 100
172 + parts[f"mcs{h}"] = mcs[mcs["h"] == h].set_index("model")["mcs_inclusion"] * 100
173 + wide = pd.DataFrame(parts)
174 + lines = ["\\begin{tabular}{lrrr rrr rrr}", "\\toprule",
175 + " & \\multicolumn{3}{c}{Sig.\\ better than HAR (\\%)}"
176 + " & \\multicolumn{3}{c}{Sig.\\ worse than HAR (\\%)}"
177 + " & \\multicolumn{3}{c}{In 90\\% MCS (\\%)} \\\\",
178 + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-7}\\cmidrule(lr){8-10}",
179 + "Model & $h{=}1$ & $h{=}5$ & $h{=}22$ & $h{=}1$ & $h{=}5$ & $h{=}22$"
180 + " & $h{=}1$ & $h{=}5$ & $h{=}22$ \\\\", "\\midrule"]
181 + for gname, models in GROUPS:
182 + avail = [m for m in models if m in wide.index]
183 + if gname.startswith("B."):
184 + avail = ["HAR"] + [m for m in avail if m != "HAR"] if "HAR" in wide.index else avail
185 + if not avail:
186 + continue
187 + lines.append(f"\\multicolumn{{10}}{{l}}{{\\itshape {gname}}}\\\\")
188 + for m in avail:
189 + def _c(key: str) -> str:
190 + v = wide.loc[m, key] if key in wide.columns else np.nan
191 + return f"{v:.0f}" if np.isfinite(v) else "---"
192 + row = [_c(f"sb{h}") for h in (1, 5, 22)] + \
193 + [_c(f"sw{h}") for h in (1, 5, 22)] + \
194 + [_c(f"mcs{h}") for h in (1, 5, 22)]
195 + lines.append(f"\\quad {m} & " + " & ".join(row) + " \\\\")
196 + lines += ["\\bottomrule", "\\end{tabular}"]
197 + _write("dm_mcs", lines)
198 +
199 +
200 +def table_mz_encompassing() -> None:
201 + """Mincer-Zarnowitz medians and encompassing outcomes (h=1)."""
202 + mz = pd.read_csv(OUT / "mz_summary.csv")
203 + enc = pd.read_csv(OUT / "encompassing_summary.csv")
204 + m1 = mz[mz["h"] == 1].set_index("model")
205 + e1 = enc[enc["h"] == 1].set_index("model")
206 + wide = pd.DataFrame({
207 + "beta": m1["med_beta"], "r2": m1["med_r2"],
208 + "rej": m1["pct_reject"] * 100,
209 + "b2": e1["med_b2"], "adds": e1["pct_adds_info"] * 100,
210 + })
211 + lines = ["\\begin{tabular}{lccccc}", "\\toprule",
212 + " & \\multicolumn{3}{c}{Mincer--Zarnowitz}"
213 + " & \\multicolumn{2}{c}{Encompassing (vs HAR)} \\\\",
214 + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-6}",
215 + "Model & Med.\\ $\\hat\\beta$ & Med.\\ $R^2$ & Rej.\\ (\\%)"
216 + " & Med.\\ $\\hat b_2$ & Adds info (\\%) \\\\", "\\midrule"]
217 + for gname, models in GROUPS:
218 + avail = [m for m in models if m in wide.index]
219 + if not avail:
220 + continue
221 + lines.append(f"\\multicolumn{{6}}{{l}}{{\\itshape {gname}}}\\\\")
222 + for m in avail:
223 + r = wide.loc[m]
224 + b2 = f"{r['b2']:.2f}" if np.isfinite(r["b2"]) else "---"
225 + adds = f"{r['adds']:.0f}" if np.isfinite(r["adds"]) else "---"
226 + lines.append(
227 + f"\\quad {m} & {r['beta']:.2f} & {r['r2']:.2f} & {r['rej']:.0f}"
228 + f" & {b2} & {adds} \\\\")
229 + lines += ["\\bottomrule", "\\end{tabular}"]
230 + _write("mz_encompassing", lines)
231 +
232 +
233 +def table_subperiods() -> None:
234 + """QLIKE ratio vs HAR per sub-period (h=1)."""
235 + tab = pd.read_csv(OUT / "subperiod_losses.csv")
236 + tab = tab[tab["h"] == 1]
237 + bench = tab[tab["model"] == "HAR"].set_index("period")["qlike"]
238 + wide = tab.pivot_table(index="model", columns="period", values="qlike")
239 + wide = wide.div(bench, axis=1)
240 + periods = ["pre2020", "covid", "inflation", "recent"]
241 + heads = ["2013--19", "2020", "2021--22", "2024--26"]
242 + cols = [(p, h) for p, h in zip(periods, heads) if p in wide.columns]
243 + _grouped_metric_table("subperiods", wide, cols)
244 +
245 +
246 +def table_sensitivity() -> None:
247 + """Frequency and window sensitivity (QLIKE, h=1)."""
248 + lines = ["\\begin{tabular}{lccc c ccc}", "\\toprule",
249 + " & \\multicolumn{3}{c}{RV measure} & &"
250 + " \\multicolumn{3}{c}{Estimation window (days)} \\\\",
251 + "\\cmidrule(lr){2-4}\\cmidrule(lr){6-8}",
252 + "Model & 1-min & 5-min ss & Kernel & & 500 & 1000 & 2000 \\\\",
253 + "\\midrule"]
254 + freq = pd.read_csv(OUT / "frequency_sensitivity.csv")
255 + freq = freq[freq["h"] == 1].pivot_table(index="model", columns="frequency",
256 + values="qlike")
257 + win = pd.read_csv(OUT / "window_sensitivity.csv")
258 + win = win[win["h"] == 1].pivot_table(index="model", columns="window",
259 + values="qlike")
260 + win.columns = [str(c) for c in win.columns]
261 + for m in ["HAR", "LogHAR", "LightGBM"]:
262 + f = freq.loc[m] if m in freq.index else pd.Series(dtype=float)
263 + w = win.loc[m] if m in win.index else pd.Series(dtype=float)
264 + cells = [
265 + _fmt(f.get("rv1min", np.nan)), _fmt(f.get("rv5min_ss", np.nan)),
266 + _fmt(f.get("rkernel", np.nan)), "",
267 + _fmt(w.get("500", np.nan)), _fmt(w.get("1000", np.nan)),
268 + _fmt(w.get("2000", np.nan)),
269 + ]
270 + lines.append(f"{m} & " + " & ".join(cells) + " \\\\")
271 + lines += ["\\bottomrule", "\\end{tabular}"]
272 + _write("sensitivity", lines)
273 +
274 +
275 +def table_transferability() -> None:
276 + """Cross-asset transferability: QLIKE h=1 for the class representatives."""
277 + tab = pd.read_csv(OUT / "transferability.csv")
278 + tab = tab[tab["h"] == 1]
279 + wide = tab.pivot_table(index="model", columns="ticker", values="qlike")
280 + order = ["HAR", "LightGBM-X", "Pooled-LGBM", "Pooled-LGBM-LOO"]
281 + label = {"HAR": "HAR (own asset)", "LightGBM-X": "LightGBM (own asset)",
282 + "Pooled-LGBM": "Pooled LightGBM (asset in pool)",
283 + "Pooled-LGBM-LOO": "Pooled LightGBM (asset excluded)"}
284 + tickers = [t for t in ["NVDA", "GBPUSD", "ETH", "GC"] if t in wide.columns]
285 + lines = ["\\begin{tabular}{l" + "c" * len(tickers) + "}", "\\toprule",
286 + "Model & " + " & ".join(tickers) + " \\\\", "\\midrule"]
287 + for m in order:
288 + if m not in wide.index:
289 + continue
290 + cells = [_fmt(wide.loc[m, t]) for t in tickers]
291 + lines.append(f"{label[m]} & " + " & ".join(cells) + " \\\\")
292 + lines += ["\\bottomrule", "\\end{tabular}"]
293 + _write("transferability", lines)
294 +
295 +
296 +def table_timing() -> None:
297 + """Volatility timing: net Sharpe, cost sensitivity, utility gains."""
298 + overall = pd.read_csv(OUT / "timing_overall.csv").set_index("model")
299 + util = pd.read_csv(OUT / "utility_summary.csv").set_index("model")
300 + byt = pd.read_csv(OUT / "timing_by_ticker.csv")
301 + sens = byt.groupby(["model", "tc_bps"], observed=True)["sharpe"].mean().unstack()
302 + wide = pd.DataFrame({
303 + "sharpe": overall["sharpe"], "ann_ret": overall["ann_ret"] * 100,
304 + "turn": overall["turnover"],
305 + "s0": sens.get(0), "s10": sens.get(10), "s20": sens.get(20),
306 + "ug": util["mean"],
307 + })
308 + lines = ["\\begin{tabular}{lccc ccc c}", "\\toprule",
309 + " & \\multicolumn{3}{c}{Net of 5 bps} &"
310 + " \\multicolumn{3}{c}{Sharpe at cost (bps)} & \\\\",
311 + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-7}",
312 + "Model & Sharpe & \\makecell{Ann.\\ ret.\\\\(\\%)} & Turnover"
313 + " & 0 & 10 & 20 & \\makecell{Utility gain\\\\vs HAR (bps)} \\\\",
314 + "\\midrule"]
315 + smax = wide["sharpe"].max()
316 + for gname, models in GROUPS:
317 + avail = [m for m in models if m in wide.index]
318 + if not avail:
319 + continue
320 + lines.append(f"\\multicolumn{{8}}{{l}}{{\\itshape {gname}}}\\\\")
321 + for m in avail:
322 + r = wide.loc[m]
323 + ug = f"{r['ug']:.0f}" if np.isfinite(r["ug"]) else "---"
324 + sh = _fmt(r["sharpe"], 2, r["sharpe"] == smax)
325 + lines.append(
326 + f"\\quad {m} & {sh} & {r['ann_ret']:.1f} & {r['turn']:.3f}"
327 + f" & {_fmt(r['s0'], 2)} & {_fmt(r['s10'], 2)} & {_fmt(r['s20'], 2)}"
328 + f" & {ug} \\\\")
329 + lines += ["\\bottomrule", "\\end{tabular}"]
330 + _write("timing", lines)
331 +
332 +
333 +def table_var() -> None:
334 + """VaR backtest outcomes at 1% and 5%."""
335 + tab = pd.read_csv(OUT / "var_summary.csv")
336 + parts = {}
337 + for lv in (0.01, 0.05):
338 + s = tab[tab["level"] == lv].set_index("model")
339 + parts[f"vr{lv}"] = s["mean_viol_rate"] * 100
340 + parts[f"ku{lv}"] = s["pct_pass_kupiec"] * 100
341 + parts[f"cc{lv}"] = s["pct_pass_cc"] * 100
342 + wide = pd.DataFrame(parts)
343 + lines = ["\\begin{tabular}{lccc ccc}", "\\toprule",
344 + " & \\multicolumn{3}{c}{VaR 1\\%} & \\multicolumn{3}{c}{VaR 5\\%} \\\\",
345 + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-7}",
346 + "Model & \\makecell{Viol.\\\\(\\%)} & \\makecell{Kupiec\\\\pass (\\%)}"
347 + " & \\makecell{CC\\\\pass (\\%)}"
348 + " & \\makecell{Viol.\\\\(\\%)} & \\makecell{Kupiec\\\\pass (\\%)}"
349 + " & \\makecell{CC\\\\pass (\\%)} \\\\", "\\midrule"]
350 + for gname, models in GROUPS:
351 + avail = [m for m in models if m in wide.index]
352 + if not avail:
353 + continue
354 + lines.append(f"\\multicolumn{{7}}{{l}}{{\\itshape {gname}}}\\\\")
355 + for m in avail:
356 + r = wide.loc[m]
357 + lines.append(
358 + f"\\quad {m} & {r['vr0.01']:.2f} & {r['ku0.01']:.0f} & {r['cc0.01']:.0f}"
359 + f" & {r['vr0.05']:.2f} & {r['ku0.05']:.0f} & {r['cc0.05']:.0f} \\\\")
360 + lines += ["\\bottomrule", "\\end{tabular}"]
361 + _write("var", lines)
362 +
363 +
364 +def table_by_ticker_appendix() -> None:
365 + """Appendix: QLIKE h=1 per ticker for headline models."""
366 + tab = pd.read_csv(OUT / "losses_by_ticker.csv")
367 + tab = tab[tab["h"] == 1]
368 + models = ["HAR", "LogHAR", "RealGARCH", "LightGBM-X", "Pooled-LGBM",
369 + "LSTM", "Comb-InvMSE"]
370 + wide = tab.pivot_table(index=["cls", "ticker"], columns="model",
371 + values="qlike")
372 + avail = [m for m in models if m in wide.columns]
373 + lines = ["\\begin{tabular}{l" + "c" * len(avail) + "}", "\\toprule",
374 + "Ticker & " + " & ".join(avail) + " \\\\", "\\midrule"]
375 + for cls in ["equity", "fx", "crypto", "futures"]:
376 + if cls not in wide.index.get_level_values(0):
377 + continue
378 + lines.append(f"\\multicolumn{{{len(avail) + 1}}}{{l}}"
379 + f"{{\\itshape {CLS_LABEL[cls]}}}\\\\")
380 + sub = wide.loc[cls]
381 + for tk, r in sub.iterrows():
382 + best = r[avail].min()
383 + cells = [_fmt(r[m], 3, np.isfinite(r[m]) and r[m] == best)
384 + for m in avail]
385 + lines.append(f"\\quad {tk} & " + " & ".join(cells) + " \\\\")
386 + lines += ["\\bottomrule", "\\end{tabular}"]
387 + _write("by_ticker_appendix", lines)
388 +
389 +
390 +def main() -> None:
391 + """Generate every LaTeX table from the reproduced CSVs."""
392 + config.setup_logging()
393 + for fn in [table_summary_stats, table_losses_main, table_losses_class,
394 + table_dm_mcs, table_mz_encompassing, table_subperiods,
395 + table_sensitivity, table_transferability, table_timing,
396 + table_var, table_by_ticker_appendix]:
397 + try:
398 + fn()
399 + except Exception as exc: # noqa: BLE001
400 + logger.error("table %s FAILED: %s", fn.__name__, exc)
401 +
402 +
403 +if __name__ == "__main__":
404 + main()
added scripts/09_build_paper.py +106 −0
@@ -0,0 +1,106 @@
1 +#!/usr/bin/env python3
2 +"""
3 +================================================================
4 +Auteur : Simon-Pierre Boucher
5 +Contact : contact@spboucher.ai
6 +Projet : Prévision de volatilité réalisée multi-actifs
7 + (HAR-RV vs GARCH vs Machine Learning)
8 +Fichier : 09_build_paper.py
9 +Description : Étape 09 — Vérification des headers de tous les
10 + fichiers de code, présence des tables/figures
11 + référencées, et compilation du PDF (latexmk).
12 +Usage : python scripts/09_build_paper.py [--no-compile]
13 +================================================================
14 +"""
15 +
16 +from __future__ import annotations
17 +
18 +import argparse
19 +import logging
20 +import re
21 +import subprocess
22 +import sys
23 +from pathlib import Path
24 +
25 +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
26 +
27 +from wp12 import config # noqa: E402
28 +
29 +logger = logging.getLogger("09_build_paper")
30 +
31 +ROOT = config.ROOT
32 +HEADER_MARK = "Auteur : Simon-Pierre Boucher"
33 +
34 +
35 +def check_headers() -> bool:
36 + """Every .py / config / paper source file must carry the project header."""
37 + ok = True
38 + files = (
39 + list((ROOT / "src").rglob("*.py"))
40 + + list((ROOT / "scripts").glob("*.py"))
41 + + list((ROOT / "tests").glob("*.py"))
42 + + [ROOT / "config.yaml", ROOT / "requirements.txt"]
43 + + list((ROOT / "paper").glob("*.tex"))
44 + + list((ROOT / "paper" / "sections").glob("*.tex"))
45 + )
46 + for f in files:
47 + if "__pycache__" in str(f):
48 + continue
49 + head = f.read_text(encoding="utf-8", errors="ignore")[:600]
50 + if HEADER_MARK not in head:
51 + logger.error("MISSING HEADER: %s", f.relative_to(ROOT))
52 + ok = False
53 + logger.info("header check: %d files, %s", len(files), "OK" if ok else "FAILURES")
54 + return ok
55 +
56 +
57 +def check_inputs() -> bool:
58 + """All tables/figures referenced by the paper must exist on disk."""
59 + ok = True
60 + tex = ""
61 + for f in [ROOT / "paper" / "main.tex", *(ROOT / "paper" / "sections").glob("*.tex")]:
62 + tex += f.read_text(encoding="utf-8")
63 + for m in re.finditer(r"\\input\{(\.\./results/tables/[^}]+)\}", tex):
64 + p = (ROOT / "paper" / m.group(1)).resolve()
65 + p = p if p.suffix else p.with_suffix(".tex")
66 + if not p.exists():
67 + logger.error("missing table: %s", m.group(1))
68 + ok = False
69 + for m in re.finditer(r"\\includegraphics\[[^\]]*\]\{([^}]+)\}", tex):
70 + name = m.group(1)
71 + if name.startswith("uq_logo"):
72 + continue
73 + if not (ROOT / "figures" / name).exists():
74 + logger.error("missing figure: %s", name)
75 + ok = False
76 + logger.info("inputs check: %s", "OK" if ok else "FAILURES")
77 + return ok
78 +
79 +
80 +def compile_pdf() -> bool:
81 + """Compile the paper with latexmk (pdflatex + bibtex)."""
82 + res = subprocess.run(
83 + ["latexmk", "-pdf", "-interaction=nonstopmode", "main.tex"],
84 + cwd=ROOT / "paper", capture_output=True, text=True, timeout=600,
85 + )
86 + if res.returncode != 0:
87 + logger.error("latexmk failed:\n%s", res.stdout[-3000:])
88 + return False
89 + logger.info("PDF compiled: paper/main.pdf")
90 + return True
91 +
92 +
93 +def main() -> None:
94 + """Run all pre-flight checks, then compile."""
95 + ap = argparse.ArgumentParser()
96 + ap.add_argument("--no-compile", action="store_true")
97 + args = ap.parse_args()
98 + config.setup_logging()
99 + ok = check_headers() & check_inputs()
100 + if not args.no_compile:
101 + ok &= compile_pdf()
102 + sys.exit(0 if ok else 1)
103 +
104 +
105 +if __name__ == "__main__":
106 + main()
added src/wp12/plots.py +77 −0
@@ -0,0 +1,77 @@
1 +"""
2 +================================================================
3 +Auteur : Simon-Pierre Boucher
4 +Contact : contact@spboucher.ai
5 +Projet : Prévision de volatilité réalisée multi-actifs
6 + (HAR-RV vs GARCH vs Machine Learning)
7 +Fichier : plots.py
8 +Description : Style matplotlib commun et figures publication-
9 + ready du papier (calibre revue) — mêmes conventions
10 + que la série de working papers UQO.
11 +================================================================
12 +"""
13 +
14 +from __future__ import annotations
15 +
16 +import logging
17 +
18 +import matplotlib
19 +
20 +matplotlib.use("Agg")
21 +import matplotlib.pyplot as plt # noqa: E402
22 +import numpy as np # noqa: E402
23 +import pandas as pd # noqa: E402
24 +
25 +logger = logging.getLogger(__name__)
26 +
27 +INK = "#1a1a1a" # primary marks
28 +BLUE = "#2e6da4" # accent series
29 +RED = "#c04848" # contrast series / polarity
30 +GREY = "#8a8a8a" # reference lines
31 +LIGHT = "#d9d9d9" # fills, bands
32 +GRID = "#e3e3e3"
33 +
34 +BLUES = ["#c6d7e8", "#9dbcd8", "#74a1c8", "#4b86b8", "#2e6da4", "#1d4a75"]
35 +
36 +TEXTWIDTH = 6.3 # \textwidth in inches
37 +
38 +
39 +def apply_style() -> None:
40 + """Apply the shared print-journal matplotlib style."""
41 + plt.rcParams.update({
42 + "font.family": "serif",
43 + "font.serif": ["Times New Roman", "Times", "STIXGeneral", "DejaVu Serif"],
44 + "mathtext.fontset": "stix",
45 + "font.size": 9,
46 + "axes.labelsize": 9,
47 + "xtick.labelsize": 8,
48 + "ytick.labelsize": 8,
49 + "legend.fontsize": 8,
50 + "figure.dpi": 150,
51 + "savefig.dpi": 300,
52 + "axes.spines.top": False,
53 + "axes.spines.right": False,
54 + "axes.linewidth": 0.6,
55 + "xtick.major.width": 0.6,
56 + "ytick.major.width": 0.6,
57 + "xtick.major.size": 3,
58 + "ytick.major.size": 3,
59 + "xtick.direction": "out",
60 + "ytick.direction": "out",
61 + "axes.grid": False,
62 + "grid.color": GRID,
63 + "grid.linewidth": 0.5,
64 + "legend.frameon": False,
65 + "savefig.bbox": "tight",
66 + "savefig.pad_inches": 0.02,
67 + })
68 +
69 +
70 +def panel_label(ax, text: str) -> None:
71 + """Bold flush-left panel header above the axes (JF house style)."""
72 + ax.set_title(text, loc="left", fontweight="bold", fontsize=9, pad=6)
73 +
74 +
75 +def annualized_vol(rv: pd.Series) -> pd.Series:
76 + """Convert daily realized variance to annualized percent volatility."""
77 + return np.sqrt(rv * 252) * 100
78