Scripts 03-09, tests, figures/tables generators, methodology & literature sections, references.bib
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
12 changed files +2,442 −3
modified
paper/references.bib
+569 −0
@@ -0,0 +1,569 @@ | ||
| 1 | +% ================================================================ | |
| 2 | +% Auteur : Simon-Pierre Boucher | |
| 3 | +% Contact : contact@spboucher.ai | |
| 4 | +% Projet : Prévision de volatilité réalisée multi-actifs | |
| 5 | +% (HAR-RV vs GARCH vs Machine Learning) | |
| 6 | +% Fichier : references.bib | |
| 7 | +% Description : Références bibliographiques du papier (BibTeX). | |
| 8 | +% ================================================================ | |
| 9 | + | |
| 10 | +@article{AndersenBollerslev1998, | |
| 11 | + author = {Andersen, Torben G. and Bollerslev, Tim}, | |
| 12 | + title = {Answering the Skeptics: Yes, Standard Volatility Models Do Provide Accurate Forecasts}, | |
| 13 | + journal = {International Economic Review}, | |
| 14 | + year = {1998}, | |
| 15 | + volume = {39}, | |
| 16 | + number = {4}, | |
| 17 | + pages = {885--905} | |
| 18 | +} | |
| 19 | + | |
| 20 | +@article{ABDL2001, | |
| 21 | + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X. and Labys, Paul}, | |
| 22 | + title = {The Distribution of Realized Exchange Rate Volatility}, | |
| 23 | + journal = {Journal of the American Statistical Association}, | |
| 24 | + year = {2001}, | |
| 25 | + volume = {96}, | |
| 26 | + number = {453}, | |
| 27 | + pages = {42--55} | |
| 28 | +} | |
| 29 | + | |
| 30 | +@article{ABDL2003, | |
| 31 | + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X. and Labys, Paul}, | |
| 32 | + title = {Modeling and Forecasting Realized Volatility}, | |
| 33 | + journal = {Econometrica}, | |
| 34 | + year = {2003}, | |
| 35 | + volume = {71}, | |
| 36 | + number = {2}, | |
| 37 | + pages = {579--625} | |
| 38 | +} | |
| 39 | + | |
| 40 | +@article{ABDE2001, | |
| 41 | + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X. and Ebens, Heiko}, | |
| 42 | + title = {The Distribution of Realized Stock Return Volatility}, | |
| 43 | + journal = {Journal of Financial Economics}, | |
| 44 | + year = {2001}, | |
| 45 | + volume = {61}, | |
| 46 | + number = {1}, | |
| 47 | + pages = {43--76} | |
| 48 | +} | |
| 49 | + | |
| 50 | +@article{ABD2007, | |
| 51 | + author = {Andersen, Torben G. and Bollerslev, Tim and Diebold, Francis X.}, | |
| 52 | + title = {Roughing It Up: Including Jump Components in the Measurement, Modeling, and Forecasting of Return Volatility}, | |
| 53 | + journal = {Review of Economics and Statistics}, | |
| 54 | + year = {2007}, | |
| 55 | + volume = {89}, | |
| 56 | + number = {4}, | |
| 57 | + pages = {701--720} | |
| 58 | +} | |
| 59 | + | |
| 60 | +@article{BNS2002, | |
| 61 | + author = {Barndorff-Nielsen, Ole E. and Shephard, Neil}, | |
| 62 | + title = {Econometric Analysis of Realized Volatility and Its Use in Estimating Stochastic Volatility Models}, | |
| 63 | + journal = {Journal of the Royal Statistical Society: Series B}, | |
| 64 | + year = {2002}, | |
| 65 | + volume = {64}, | |
| 66 | + number = {2}, | |
| 67 | + pages = {253--280} | |
| 68 | +} | |
| 69 | + | |
| 70 | +@article{BNS2004, | |
| 71 | + author = {Barndorff-Nielsen, Ole E. and Shephard, Neil}, | |
| 72 | + title = {Power and Bipower Variation with Stochastic Volatility and Jumps}, | |
| 73 | + journal = {Journal of Financial Econometrics}, | |
| 74 | + year = {2004}, | |
| 75 | + volume = {2}, | |
| 76 | + number = {1}, | |
| 77 | + pages = {1--37} | |
| 78 | +} | |
| 79 | + | |
| 80 | +@article{BNS2006, | |
| 81 | + author = {Barndorff-Nielsen, Ole E. and Shephard, Neil}, | |
| 82 | + title = {Econometrics of Testing for Jumps in Financial Economics Using Bipower Variation}, | |
| 83 | + journal = {Journal of Financial Econometrics}, | |
| 84 | + year = {2006}, | |
| 85 | + volume = {4}, | |
| 86 | + number = {1}, | |
| 87 | + pages = {1--30} | |
| 88 | +} | |
| 89 | + | |
| 90 | +@article{BNHLS2008, | |
| 91 | + author = {Barndorff-Nielsen, Ole E. and Hansen, Peter Reinhard and Lunde, Asger and Shephard, Neil}, | |
| 92 | + title = {Designing Realized Kernels to Measure the Ex Post Variation of Equity Prices in the Presence of Noise}, | |
| 93 | + journal = {Econometrica}, | |
| 94 | + year = {2008}, | |
| 95 | + volume = {76}, | |
| 96 | + number = {6}, | |
| 97 | + pages = {1481--1536} | |
| 98 | +} | |
| 99 | + | |
| 100 | +@article{BNHLS2009, | |
| 101 | + author = {Barndorff-Nielsen, Ole E. and Hansen, Peter Reinhard and Lunde, Asger and Shephard, Neil}, | |
| 102 | + title = {Realized Kernels in Practice: Trades and Quotes}, | |
| 103 | + journal = {Econometrics Journal}, | |
| 104 | + year = {2009}, | |
| 105 | + volume = {12}, | |
| 106 | + number = {3}, | |
| 107 | + pages = {C1--C32} | |
| 108 | +} | |
| 109 | + | |
| 110 | +@incollection{BNKS2010, | |
| 111 | + author = {Barndorff-Nielsen, Ole E. and Kinnebrock, Silja and Shephard, Neil}, | |
| 112 | + title = {Measuring Downside Risk: Realised Semivariance}, | |
| 113 | + booktitle = {Volatility and Time Series Econometrics: Essays in Honor of Robert F. Engle}, | |
| 114 | + editor = {Bollerslev, Tim and Russell, Jeffrey R. and Watson, Mark W.}, | |
| 115 | + publisher = {Oxford University Press}, | |
| 116 | + year = {2010}, | |
| 117 | + pages = {117--136} | |
| 118 | +} | |
| 119 | + | |
| 120 | +@article{ZMA2005, | |
| 121 | + author = {Zhang, Lan and Mykland, Per A. and A{\"i}t-Sahalia, Yacine}, | |
| 122 | + title = {A Tale of Two Time Scales: Determining Integrated Volatility with Noisy High-Frequency Data}, | |
| 123 | + journal = {Journal of the American Statistical Association}, | |
| 124 | + year = {2005}, | |
| 125 | + volume = {100}, | |
| 126 | + number = {472}, | |
| 127 | + pages = {1394--1411} | |
| 128 | +} | |
| 129 | + | |
| 130 | +@article{HansenLunde2006, | |
| 131 | + author = {Hansen, Peter R. and Lunde, Asger}, | |
| 132 | + title = {Realized Variance and Market Microstructure Noise}, | |
| 133 | + journal = {Journal of Business \& Economic Statistics}, | |
| 134 | + year = {2006}, | |
| 135 | + volume = {24}, | |
| 136 | + number = {2}, | |
| 137 | + pages = {127--161} | |
| 138 | +} | |
| 139 | + | |
| 140 | +@article{BandiRussell2008, | |
| 141 | + author = {Bandi, Federico M. and Russell, Jeffrey R.}, | |
| 142 | + title = {Microstructure Noise, Realized Variance, and Optimal Sampling}, | |
| 143 | + journal = {Review of Economic Studies}, | |
| 144 | + year = {2008}, | |
| 145 | + volume = {75}, | |
| 146 | + number = {2}, | |
| 147 | + pages = {339--369} | |
| 148 | +} | |
| 149 | + | |
| 150 | +@article{LiuPattonSheppard2015, | |
| 151 | + author = {Liu, Lily Y. and Patton, Andrew J. and Sheppard, Kevin}, | |
| 152 | + title = {Does Anything Beat 5-Minute {RV}? A Comparison of Realized Measures Across Multiple Asset Classes}, | |
| 153 | + journal = {Journal of Econometrics}, | |
| 154 | + year = {2015}, | |
| 155 | + volume = {187}, | |
| 156 | + number = {1}, | |
| 157 | + pages = {293--311} | |
| 158 | +} | |
| 159 | + | |
| 160 | +@article{Corsi2009, | |
| 161 | + author = {Corsi, Fulvio}, | |
| 162 | + title = {A Simple Approximate Long-Memory Model of Realized Volatility}, | |
| 163 | + journal = {Journal of Financial Econometrics}, | |
| 164 | + year = {2009}, | |
| 165 | + volume = {7}, | |
| 166 | + number = {2}, | |
| 167 | + pages = {174--196} | |
| 168 | +} | |
| 169 | + | |
| 170 | +@article{CorsiPirinoReno2010, | |
| 171 | + author = {Corsi, Fulvio and Pirino, Davide and Ren{\`o}, Roberto}, | |
| 172 | + title = {Threshold Bipower Variation and the Impact of Jumps on Volatility Forecasting}, | |
| 173 | + journal = {Journal of Econometrics}, | |
| 174 | + year = {2010}, | |
| 175 | + volume = {159}, | |
| 176 | + number = {2}, | |
| 177 | + pages = {276--288} | |
| 178 | +} | |
| 179 | + | |
| 180 | +@article{PattonSheppard2015, | |
| 181 | + author = {Patton, Andrew J. and Sheppard, Kevin}, | |
| 182 | + title = {Good Volatility, Bad Volatility: Signed Jumps and the Persistence of Volatility}, | |
| 183 | + journal = {Review of Economics and Statistics}, | |
| 184 | + year = {2015}, | |
| 185 | + volume = {97}, | |
| 186 | + number = {3}, | |
| 187 | + pages = {683--697} | |
| 188 | +} | |
| 189 | + | |
| 190 | +@article{BPQ2016, | |
| 191 | + author = {Bollerslev, Tim and Patton, Andrew J. and Quaedvlieg, Rogier}, | |
| 192 | + title = {Exploiting the Errors: A Simple Approach for Improved Volatility Forecasting}, | |
| 193 | + journal = {Journal of Econometrics}, | |
| 194 | + year = {2016}, | |
| 195 | + volume = {192}, | |
| 196 | + number = {1}, | |
| 197 | + pages = {1--18} | |
| 198 | +} | |
| 199 | + | |
| 200 | +@article{Engle1982, | |
| 201 | + author = {Engle, Robert F.}, | |
| 202 | + title = {Autoregressive Conditional Heteroscedasticity with Estimates of the Variance of United Kingdom Inflation}, | |
| 203 | + journal = {Econometrica}, | |
| 204 | + year = {1982}, | |
| 205 | + volume = {50}, | |
| 206 | + number = {4}, | |
| 207 | + pages = {987--1007} | |
| 208 | +} | |
| 209 | + | |
| 210 | +@article{Bollerslev1986, | |
| 211 | + author = {Bollerslev, Tim}, | |
| 212 | + title = {Generalized Autoregressive Conditional Heteroskedasticity}, | |
| 213 | + journal = {Journal of Econometrics}, | |
| 214 | + year = {1986}, | |
| 215 | + volume = {31}, | |
| 216 | + number = {3}, | |
| 217 | + pages = {307--327} | |
| 218 | +} | |
| 219 | + | |
| 220 | +@article{Nelson1991, | |
| 221 | + author = {Nelson, Daniel B.}, | |
| 222 | + title = {Conditional Heteroskedasticity in Asset Returns: A New Approach}, | |
| 223 | + journal = {Econometrica}, | |
| 224 | + year = {1991}, | |
| 225 | + volume = {59}, | |
| 226 | + number = {2}, | |
| 227 | + pages = {347--370} | |
| 228 | +} | |
| 229 | + | |
| 230 | +@article{GJR1993, | |
| 231 | + author = {Glosten, Lawrence R. and Jagannathan, Ravi and Runkle, David E.}, | |
| 232 | + title = {On the Relation Between the Expected Value and the Volatility of the Nominal Excess Return on Stocks}, | |
| 233 | + journal = {Journal of Finance}, | |
| 234 | + year = {1993}, | |
| 235 | + volume = {48}, | |
| 236 | + number = {5}, | |
| 237 | + pages = {1779--1801} | |
| 238 | +} | |
| 239 | + | |
| 240 | +@article{HansenLunde2005, | |
| 241 | + author = {Hansen, Peter R. and Lunde, Asger}, | |
| 242 | + title = {A Forecast Comparison of Volatility Models: Does Anything Beat a {GARCH}(1,1)?}, | |
| 243 | + journal = {Journal of Applied Econometrics}, | |
| 244 | + year = {2005}, | |
| 245 | + volume = {20}, | |
| 246 | + number = {7}, | |
| 247 | + pages = {873--889} | |
| 248 | +} | |
| 249 | + | |
| 250 | +@article{HansenHuangShek2012, | |
| 251 | + author = {Hansen, Peter Reinhard and Huang, Zhuo and Shek, Howard Howan}, | |
| 252 | + title = {Realized {GARCH}: A Joint Model for Returns and Realized Measures of Volatility}, | |
| 253 | + journal = {Journal of Applied Econometrics}, | |
| 254 | + year = {2012}, | |
| 255 | + volume = {27}, | |
| 256 | + number = {6}, | |
| 257 | + pages = {877--906} | |
| 258 | +} | |
| 259 | + | |
| 260 | +@article{Patton2011, | |
| 261 | + author = {Patton, Andrew J.}, | |
| 262 | + title = {Volatility Forecast Comparison Using Imperfect Volatility Proxies}, | |
| 263 | + journal = {Journal of Econometrics}, | |
| 264 | + year = {2011}, | |
| 265 | + volume = {160}, | |
| 266 | + number = {1}, | |
| 267 | + pages = {246--256} | |
| 268 | +} | |
| 269 | + | |
| 270 | +@article{DieboldMariano1995, | |
| 271 | + author = {Diebold, Francis X. and Mariano, Roberto S.}, | |
| 272 | + title = {Comparing Predictive Accuracy}, | |
| 273 | + journal = {Journal of Business \& Economic Statistics}, | |
| 274 | + year = {1995}, | |
| 275 | + volume = {13}, | |
| 276 | + number = {3}, | |
| 277 | + pages = {253--263} | |
| 278 | +} | |
| 279 | + | |
| 280 | +@article{GiacominiWhite2006, | |
| 281 | + author = {Giacomini, Raffaella and White, Halbert}, | |
| 282 | + title = {Tests of Conditional Predictive Ability}, | |
| 283 | + journal = {Econometrica}, | |
| 284 | + year = {2006}, | |
| 285 | + volume = {74}, | |
| 286 | + number = {6}, | |
| 287 | + pages = {1545--1578} | |
| 288 | +} | |
| 289 | + | |
| 290 | +@article{HLN2011, | |
| 291 | + author = {Hansen, Peter R. and Lunde, Asger and Nason, James M.}, | |
| 292 | + title = {The Model Confidence Set}, | |
| 293 | + journal = {Econometrica}, | |
| 294 | + year = {2011}, | |
| 295 | + volume = {79}, | |
| 296 | + number = {2}, | |
| 297 | + pages = {453--497} | |
| 298 | +} | |
| 299 | + | |
| 300 | +@article{GiacominiRossi2010, | |
| 301 | + author = {Giacomini, Raffaella and Rossi, Barbara}, | |
| 302 | + title = {Forecast Comparisons in Unstable Environments}, | |
| 303 | + journal = {Journal of Applied Econometrics}, | |
| 304 | + year = {2010}, | |
| 305 | + volume = {25}, | |
| 306 | + number = {4}, | |
| 307 | + pages = {595--620} | |
| 308 | +} | |
| 309 | + | |
| 310 | +@incollection{MincerZarnowitz1969, | |
| 311 | + author = {Mincer, Jacob A. and Zarnowitz, Victor}, | |
| 312 | + title = {The Evaluation of Economic Forecasts}, | |
| 313 | + booktitle = {Economic Forecasts and Expectations: Analysis of Forecasting Behavior and Performance}, | |
| 314 | + publisher = {NBER}, | |
| 315 | + year = {1969}, | |
| 316 | + pages = {3--46} | |
| 317 | +} | |
| 318 | + | |
| 319 | +@article{HarveyLeybourneNewbold1998, | |
| 320 | + author = {Harvey, David I. and Leybourne, Stephen J. and Newbold, Paul}, | |
| 321 | + title = {Tests for Forecast Encompassing}, | |
| 322 | + journal = {Journal of Business \& Economic Statistics}, | |
| 323 | + year = {1998}, | |
| 324 | + volume = {16}, | |
| 325 | + number = {2}, | |
| 326 | + pages = {254--259} | |
| 327 | +} | |
| 328 | + | |
| 329 | +@article{BatesGranger1969, | |
| 330 | + author = {Bates, John M. and Granger, Clive W. J.}, | |
| 331 | + title = {The Combination of Forecasts}, | |
| 332 | + journal = {Journal of the Operational Research Society}, | |
| 333 | + year = {1969}, | |
| 334 | + volume = {20}, | |
| 335 | + number = {4}, | |
| 336 | + pages = {451--468} | |
| 337 | +} | |
| 338 | + | |
| 339 | +@incollection{Timmermann2006, | |
| 340 | + author = {Timmermann, Allan}, | |
| 341 | + title = {Forecast Combinations}, | |
| 342 | + booktitle = {Handbook of Economic Forecasting}, | |
| 343 | + editor = {Elliott, Graham and Granger, Clive W. J. and Timmermann, Allan}, | |
| 344 | + publisher = {Elsevier}, | |
| 345 | + year = {2006}, | |
| 346 | + volume = {1}, | |
| 347 | + pages = {135--196} | |
| 348 | +} | |
| 349 | + | |
| 350 | +@article{PolitisRomano1994, | |
| 351 | + author = {Politis, Dimitris N. and Romano, Joseph P.}, | |
| 352 | + title = {The Stationary Bootstrap}, | |
| 353 | + journal = {Journal of the American Statistical Association}, | |
| 354 | + year = {1994}, | |
| 355 | + volume = {89}, | |
| 356 | + number = {428}, | |
| 357 | + pages = {1303--1313} | |
| 358 | +} | |
| 359 | + | |
| 360 | +@article{FKO2001, | |
| 361 | + author = {Fleming, Jeff and Kirby, Chris and Ostdiek, Barbara}, | |
| 362 | + title = {The Economic Value of Volatility Timing}, | |
| 363 | + journal = {Journal of Finance}, | |
| 364 | + year = {2001}, | |
| 365 | + volume = {56}, | |
| 366 | + number = {1}, | |
| 367 | + pages = {329--352} | |
| 368 | +} | |
| 369 | + | |
| 370 | +@article{FKO2003, | |
| 371 | + author = {Fleming, Jeff and Kirby, Chris and Ostdiek, Barbara}, | |
| 372 | + title = {The Economic Value of Volatility Timing Using ``Realized'' Volatility}, | |
| 373 | + journal = {Journal of Financial Economics}, | |
| 374 | + year = {2003}, | |
| 375 | + volume = {67}, | |
| 376 | + number = {3}, | |
| 377 | + pages = {473--509} | |
| 378 | +} | |
| 379 | + | |
| 380 | +@article{Kupiec1995, | |
| 381 | + author = {Kupiec, Paul H.}, | |
| 382 | + title = {Techniques for Verifying the Accuracy of Risk Measurement Models}, | |
| 383 | + journal = {Journal of Derivatives}, | |
| 384 | + year = {1995}, | |
| 385 | + volume = {3}, | |
| 386 | + number = {2}, | |
| 387 | + pages = {73--84} | |
| 388 | +} | |
| 389 | + | |
| 390 | +@article{Christoffersen1998, | |
| 391 | + author = {Christoffersen, Peter F.}, | |
| 392 | + title = {Evaluating Interval Forecasts}, | |
| 393 | + journal = {International Economic Review}, | |
| 394 | + year = {1998}, | |
| 395 | + volume = {39}, | |
| 396 | + number = {4}, | |
| 397 | + pages = {841--862} | |
| 398 | +} | |
| 399 | + | |
| 400 | +@article{Bucci2020, | |
| 401 | + author = {Bucci, Andrea}, | |
| 402 | + title = {Realized Volatility Forecasting with Neural Networks}, | |
| 403 | + journal = {Journal of Financial Econometrics}, | |
| 404 | + year = {2020}, | |
| 405 | + volume = {18}, | |
| 406 | + number = {3}, | |
| 407 | + pages = {502--531} | |
| 408 | +} | |
| 409 | + | |
| 410 | +@article{ChristensenSiggaardVeliyev2023, | |
| 411 | + author = {Christensen, Kim and Siggaard, Mathias and Veliyev, Bezirgen}, | |
| 412 | + title = {A Machine Learning Approach to Volatility Forecasting}, | |
| 413 | + journal = {Journal of Financial Econometrics}, | |
| 414 | + year = {2023}, | |
| 415 | + volume = {21}, | |
| 416 | + number = {5}, | |
| 417 | + pages = {1680--1727} | |
| 418 | +} | |
| 419 | + | |
| 420 | +@article{GuKellyXiu2020, | |
| 421 | + author = {Gu, Shihao and Kelly, Bryan and Xiu, Dacheng}, | |
| 422 | + title = {Empirical Asset Pricing via Machine Learning}, | |
| 423 | + journal = {Review of Financial Studies}, | |
| 424 | + year = {2020}, | |
| 425 | + volume = {33}, | |
| 426 | + number = {5}, | |
| 427 | + pages = {2223--2273} | |
| 428 | +} | |
| 429 | + | |
| 430 | +@article{AudrinoKnaus2016, | |
| 431 | + author = {Audrino, Francesco and Knaus, Simon D.}, | |
| 432 | + title = {Lassoing the {HAR} Model: A Model Selection Perspective on Realized Volatility Dynamics}, | |
| 433 | + journal = {Econometric Reviews}, | |
| 434 | + year = {2016}, | |
| 435 | + volume = {35}, | |
| 436 | + number = {8--10}, | |
| 437 | + pages = {1485--1521} | |
| 438 | +} | |
| 439 | + | |
| 440 | +@article{RahimikiaPoon2020, | |
| 441 | + author = {Rahimikia, Eghbal and Poon, Ser-Huang}, | |
| 442 | + title = {Machine Learning for Realised Volatility Forecasting}, | |
| 443 | + journal = {SSRN Working Paper No. 3707796}, | |
| 444 | + year = {2020} | |
| 445 | +} | |
| 446 | + | |
| 447 | +@article{Katsiampa2017, | |
| 448 | + author = {Katsiampa, Paraskevi}, | |
| 449 | + title = {Volatility Estimation for {Bitcoin}: A Comparison of {GARCH} Models}, | |
| 450 | + journal = {Economics Letters}, | |
| 451 | + year = {2017}, | |
| 452 | + volume = {158}, | |
| 453 | + pages = {3--6} | |
| 454 | +} | |
| 455 | + | |
| 456 | +@article{Duan1983, | |
| 457 | + author = {Duan, Naihua}, | |
| 458 | + title = {Smearing Estimate: A Nonparametric Retransformation Method}, | |
| 459 | + journal = {Journal of the American Statistical Association}, | |
| 460 | + year = {1983}, | |
| 461 | + volume = {78}, | |
| 462 | + number = {383}, | |
| 463 | + pages = {605--610} | |
| 464 | +} | |
| 465 | + | |
| 466 | +@article{HochreiterSchmidhuber1997, | |
| 467 | + author = {Hochreiter, Sepp and Schmidhuber, J{\"u}rgen}, | |
| 468 | + title = {Long Short-Term Memory}, | |
| 469 | + journal = {Neural Computation}, | |
| 470 | + year = {1997}, | |
| 471 | + volume = {9}, | |
| 472 | + number = {8}, | |
| 473 | + pages = {1735--1780} | |
| 474 | +} | |
| 475 | + | |
| 476 | +@inproceedings{Vaswani2017, | |
| 477 | + author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N. and Kaiser, {\L}ukasz and Polosukhin, Illia}, | |
| 478 | + title = {Attention Is All You Need}, | |
| 479 | + booktitle = {Advances in Neural Information Processing Systems}, | |
| 480 | + year = {2017}, | |
| 481 | + volume = {30}, | |
| 482 | + pages = {5998--6008} | |
| 483 | +} | |
| 484 | + | |
| 485 | +@inproceedings{LundbergLee2017, | |
| 486 | + author = {Lundberg, Scott M. and Lee, Su-In}, | |
| 487 | + title = {A Unified Approach to Interpreting Model Predictions}, | |
| 488 | + booktitle = {Advances in Neural Information Processing Systems}, | |
| 489 | + year = {2017}, | |
| 490 | + volume = {30}, | |
| 491 | + pages = {4765--4774} | |
| 492 | +} | |
| 493 | + | |
| 494 | +@article{Breiman2001, | |
| 495 | + author = {Breiman, Leo}, | |
| 496 | + title = {Random Forests}, | |
| 497 | + journal = {Machine Learning}, | |
| 498 | + year = {2001}, | |
| 499 | + volume = {45}, | |
| 500 | + number = {1}, | |
| 501 | + pages = {5--32} | |
| 502 | +} | |
| 503 | + | |
| 504 | +@inproceedings{ChenGuestrin2016, | |
| 505 | + author = {Chen, Tianqi and Guestrin, Carlos}, | |
| 506 | + title = {{XGBoost}: A Scalable Tree Boosting System}, | |
| 507 | + booktitle = {Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining}, | |
| 508 | + year = {2016}, | |
| 509 | + pages = {785--794} | |
| 510 | +} | |
| 511 | + | |
| 512 | +@inproceedings{Ke2017, | |
| 513 | + author = {Ke, Guolin and Meng, Qi and Finley, Thomas and Wang, Taifeng and Chen, Wei and Ma, Weidong and Ye, Qiwei and Liu, Tie-Yan}, | |
| 514 | + title = {{LightGBM}: A Highly Efficient Gradient Boosting Decision Tree}, | |
| 515 | + booktitle = {Advances in Neural Information Processing Systems}, | |
| 516 | + year = {2017}, | |
| 517 | + volume = {30}, | |
| 518 | + pages = {3146--3154} | |
| 519 | +} | |
| 520 | + | |
| 521 | +@article{Tibshirani1996, | |
| 522 | + author = {Tibshirani, Robert}, | |
| 523 | + title = {Regression Shrinkage and Selection via the Lasso}, | |
| 524 | + journal = {Journal of the Royal Statistical Society: Series B}, | |
| 525 | + year = {1996}, | |
| 526 | + volume = {58}, | |
| 527 | + number = {1}, | |
| 528 | + pages = {267--288} | |
| 529 | +} | |
| 530 | + | |
| 531 | +@article{ZouHastie2005, | |
| 532 | + author = {Zou, Hui and Hastie, Trevor}, | |
| 533 | + title = {Regularization and Variable Selection via the Elastic Net}, | |
| 534 | + journal = {Journal of the Royal Statistical Society: Series B}, | |
| 535 | + year = {2005}, | |
| 536 | + volume = {67}, | |
| 537 | + number = {2}, | |
| 538 | + pages = {301--320} | |
| 539 | +} | |
| 540 | + | |
| 541 | +@article{NeweyWest1987, | |
| 542 | + author = {Newey, Whitney K. and West, Kenneth D.}, | |
| 543 | + title = {A Simple, Positive Semi-Definite, Heteroskedasticity and Autocorrelation Consistent Covariance Matrix}, | |
| 544 | + journal = {Econometrica}, | |
| 545 | + year = {1987}, | |
| 546 | + volume = {55}, | |
| 547 | + number = {3}, | |
| 548 | + pages = {703--708} | |
| 549 | +} | |
| 550 | + | |
| 551 | +@article{Mueller1997, | |
| 552 | + author = {M{\"u}ller, Ulrich A. and Dacorogna, Michel M. and Dav{\'e}, Rakhal D. and Olsen, Richard B. and Pictet, Olivier V. and von Weizs{\"a}cker, Jacob E.}, | |
| 553 | + title = {Volatilities of Different Time Resolutions --- Analyzing the Dynamics of Market Components}, | |
| 554 | + journal = {Journal of Empirical Finance}, | |
| 555 | + year = {1997}, | |
| 556 | + volume = {4}, | |
| 557 | + number = {2--3}, | |
| 558 | + pages = {213--239} | |
| 559 | +} | |
| 560 | + | |
| 561 | +@article{PoonGranger2003, | |
| 562 | + author = {Poon, Ser-Huang and Granger, Clive W. J.}, | |
| 563 | + title = {Forecasting Volatility in Financial Markets: A Review}, | |
| 564 | + journal = {Journal of Economic Literature}, | |
| 565 | + year = {2003}, | |
| 566 | + volume = {41}, | |
| 567 | + number = {2}, | |
| 568 | + pages = {478--539} | |
| 569 | +} | |
modified
paper/sections/literature.tex
+107 −1
@@ -4,5 +4,111 @@ | ||
| 4 | 4 | % Projet : Prévision de volatilité réalisée multi-actifs |
| 5 | 5 | % (HAR-RV vs GARCH vs Machine Learning) |
| 6 | 6 | % Fichier : literature.tex |
| 7 | −% Description : Section « literature » — à rédiger après exécution du pipeline. | |
| 7 | +% Description : Section 2 — Revue de littérature et gap. | |
| 8 | 8 | % ================================================================ |
| 9 | +\section{Related literature} | |
| 10 | +\label{sec:literature} | |
| 11 | + | |
| 12 | +\subsection{Realized volatility and its measurement} | |
| 13 | + | |
| 14 | +The modern volatility-forecasting literature begins with the insight | |
| 15 | +that intraday returns turn latent variance into an effectively | |
| 16 | +observable quantity. \citet{AndersenBollerslev1998} showed that | |
| 17 | +standard models forecast well once evaluated against realized rather | |
| 18 | +than squared daily returns; \citet{ABDL2001,ABDE2001} documented the | |
| 19 | +distributional properties of realized variance for exchange rates and | |
| 20 | +equities; and \citet{BNS2002} developed its asymptotic theory. | |
| 21 | +\citet{ABDL2003} demonstrated that simple reduced-form models of | |
| 22 | +realized volatility outperform latent-variable GARCH and stochastic | |
| 23 | +volatility models --- the observation that motivates the entire | |
| 24 | +measure-then-model paradigm this paper operates in. | |
| 25 | + | |
| 26 | +At very high sampling frequencies, market microstructure noise breaks | |
| 27 | +the consistency of plain realized variance | |
| 28 | +\citep{HansenLunde2006,BandiRussell2008}. The literature responded with | |
| 29 | +noise-robust estimators: subsampled and two-scale estimators | |
| 30 | +\citep{ZMA2005}, and realized kernels \citep{BNHLS2008,BNHLS2009}. | |
| 31 | +\citet{LiuPattonSheppard2015} compared some 400 realized measures | |
| 32 | +across asset classes and found that 5-minute RV is remarkably hard to | |
| 33 | +beat --- a finding that guides our choice of subsampled 5-minute RV as | |
| 34 | +the headline measure, with 1-minute RV and realized kernels as | |
| 35 | +robustness checks. A parallel branch separates the continuous and jump | |
| 36 | +components of quadratic variation: bipower variation \citep{BNS2004}, | |
| 37 | +formal jump tests \citep{BNS2006}, and threshold refinements | |
| 38 | +\citep{CorsiPirinoReno2010}. \citet{BNKS2010} introduced realized | |
| 39 | +semivariances, whose signed decomposition \citet{PattonSheppard2015} | |
| 40 | +showed to matter for persistence: negative semivariance predicts future | |
| 41 | +volatility far more strongly than positive. | |
| 42 | + | |
| 43 | +\subsection{HAR models and extensions} | |
| 44 | + | |
| 45 | +\citet{Corsi2009} proposed the heterogeneous autoregressive (HAR) | |
| 46 | +model, whose daily--weekly--monthly cascade approximates long memory | |
| 47 | +with three OLS coefficients; it has become the workhorse benchmark of | |
| 48 | +the field. Extensions incorporate the continuous/jump split | |
| 49 | +\citep{ABD2007}, signed semivariances \citep{PattonSheppard2015}, and | |
| 50 | +attenuation-bias corrections: \citet{BPQ2016} let the daily coefficient | |
| 51 | +shrink on days when realized quarticity signals a noisy RV estimate | |
| 52 | +(HARQ). Logarithmic specifications are common because log RV is close | |
| 53 | +to Gaussian \citep{ABDL2001} and the log transform tames heteroskedastic | |
| 54 | +measurement error. On the GARCH side, \citet{HansenLunde2005} showed | |
| 55 | +that little beats a GARCH(1,1) for daily returns, while | |
| 56 | +\citet{HansenHuangShek2012} tied the latent-variance recursion to a | |
| 57 | +realized measure (Realized GARCH), typically closing most of the gap to | |
| 58 | +HAR-type models. Our benchmark suite spans exactly this spectrum --- | |
| 59 | +five HAR variants, three daily GARCH variants, and Realized GARCH --- | |
| 60 | +so that the machine-learning comparison is anchored against the | |
| 61 | +strongest available econometrics rather than a single straw man. | |
| 62 | + | |
| 63 | +\subsection{Machine learning for volatility forecasting} | |
| 64 | + | |
| 65 | +Applications of statistical learning to realized volatility are more | |
| 66 | +recent. \citet{AudrinoKnaus2016} used the LASSO to select lag | |
| 67 | +structures, finding it recovers HAR-like patterns. | |
| 68 | +\citet{Bucci2020} documented gains from recurrent neural networks for | |
| 69 | +monthly equity-index volatility. \citet{RahimikiaPoon2020} trained | |
| 70 | +learners on large limit-order-book and news datasets for intraday | |
| 71 | +volatility. Closest to our design, \citet{ChristensenSiggaardVeliyev2023} | |
| 72 | +ran a systematic horse race of regularized regressions, trees and | |
| 73 | +neural networks against HAR on 29 Dow Jones stocks, finding double-digit | |
| 74 | +QLIKE improvements concentrated in periods of market stress and in | |
| 75 | +longer horizons. In the broader return-prediction literature, | |
| 76 | +\citet{GuKellyXiu2020} established the now-standard result that trees | |
| 77 | +and shallow networks dominate linear methods and that pooling across | |
| 78 | +assets is crucial for regularization --- a theme our pooled multi-asset | |
| 79 | +model imports into the volatility context. For cryptocurrencies, | |
| 80 | +GARCH-type comparisons dominate \citep{Katsiampa2017}, with | |
| 81 | +high-frequency ML applications still scarce. | |
| 82 | + | |
| 83 | +\subsection{Forecast evaluation and economic value} | |
| 84 | + | |
| 85 | +Because volatility is observed only through noisy proxies, loss | |
| 86 | +functions must be chosen carefully: \citet{Patton2011} characterizes | |
| 87 | +the class of robust losses --- QLIKE and MSE --- for which rankings | |
| 88 | +under a proxy coincide with rankings under the truth. Statistical | |
| 89 | +comparisons rely on \citet{DieboldMariano1995} and | |
| 90 | +\citet{GiacominiWhite2006} tests, the model confidence set of | |
| 91 | +\citet{HLN2011}, and, in unstable environments, the fluctuation | |
| 92 | +analysis of \citet{GiacominiRossi2010}. Forecast combination is a | |
| 93 | +persistent empirical winner \citep{BatesGranger1969,Timmermann2006}. | |
| 94 | +The economic-value strand converts statistical accuracy into portfolio | |
| 95 | +outcomes: \citet{FKO2001,FKO2003} priced volatility timing with | |
| 96 | +mean-variance utility, and risk-management evaluation backtests VaR | |
| 97 | +coverage \citep{Kupiec1995,Christoffersen1998}. | |
| 98 | + | |
| 99 | +\subsection{Contribution relative to the literature} | |
| 100 | + | |
| 101 | +Three gaps motivate this paper. First, most ML-versus-HAR evidence | |
| 102 | +comes from single asset classes --- typically US equities | |
| 103 | +\citep{ChristensenSiggaardVeliyev2023} --- leaving open whether the | |
| 104 | +documented gains generalize to FX, futures and crypto, whose | |
| 105 | +volatility dynamics, trading sessions and noise properties differ | |
| 106 | +sharply. Second, the pooling and \emph{transferability} question --- | |
| 107 | +whether one model trained on many assets can forecast an asset it has | |
| 108 | +never seen --- has been studied for returns \citep{GuKellyXiu2020} but | |
| 109 | +barely for realized volatility. Third, statistical gains are rarely | |
| 110 | +pushed through to economic value and risk management within the same | |
| 111 | +controlled design. We address all three within a single, fully | |
| 112 | +reproducible pipeline: identical rolling protocol, identical robust | |
| 113 | +losses, identical inference, from raw 1-minute bars to utility gains | |
| 114 | +and VaR coverage, across 25 assets in four classes. | |
modified
paper/sections/methodology.tex
+276 −1
@@ -4,5 +4,280 @@ | ||
| 4 | 4 | % Projet : Prévision de volatilité réalisée multi-actifs |
| 5 | 5 | % (HAR-RV vs GARCH vs Machine Learning) |
| 6 | 6 | % Fichier : methodology.tex |
| 7 | −% Description : Section « methodology » — à rédiger après exécution du pipeline. | |
| 7 | +% Description : Section 4 — Mesures de volatilité réalisée, | |
| 8 | +% modèles, protocole d'évaluation et tests. | |
| 8 | 9 | % ================================================================ |
| 10 | +\section{Methodology} | |
| 11 | +\label{sec:methodology} | |
| 12 | + | |
| 13 | +\subsection{Realized measures} | |
| 14 | +\label{sec:rv_measures} | |
| 15 | + | |
| 16 | +Let $p_{t,i}$ denote the $i$-th intraday log price of a given asset on | |
| 17 | +trading day $t$, sampled on an equidistant grid of $n$ intraday returns | |
| 18 | +$r_{t,i} = p_{t,i} - p_{t,i-1}$. Under the standard jump--diffusion | |
| 19 | +semimartingale | |
| 20 | +\begin{equation} | |
| 21 | + dp_s = \mu_s \, ds + \sigma_s \, dW_s + \kappa_s \, dq_s , | |
| 22 | + \label{eq:sde} | |
| 23 | +\end{equation} | |
| 24 | +where $W$ is a Brownian motion, $q$ a counting process, and $\kappa$ the | |
| 25 | +jump size, the realized variance | |
| 26 | +\begin{equation} | |
| 27 | + RV_t \;=\; \sum_{i=1}^{n} r_{t,i}^2 | |
| 28 | + \;\xrightarrow{\;p\;}\; | |
| 29 | + \underbrace{\int_{t-1}^{t} \sigma^2_s \, ds}_{\text{integrated variance}} | |
| 30 | + \;+\; \underbrace{\sum_{t-1 < s \le t} \kappa_s^2}_{\text{jump variation}} | |
| 31 | + \label{eq:rv} | |
| 32 | +\end{equation} | |
| 33 | +consistently estimates total quadratic variation as $n \to \infty$ | |
| 34 | +\citep{AndersenBollerslev1998,BNS2002}. In practice market microstructure | |
| 35 | +noise contaminates returns at the very highest frequencies | |
| 36 | +\citep{HansenLunde2006,BandiRussell2008}, and we therefore compute several | |
| 37 | +noise-robust variants. | |
| 38 | + | |
| 39 | +\paragraph{Subsampled 5-minute RV.} | |
| 40 | +Our headline measure, denoted $RV^{ss}_t$, averages the realized variance | |
| 41 | +computed on five staggered 5-minute grids offset by one minute each | |
| 42 | +\citep{ZMA2005}. Following \citet{LiuPattonSheppard2015}, 5-minute | |
| 43 | +sampling is a demanding benchmark that is difficult to beat across asset | |
| 44 | +classes; subsampling recovers part of the efficiency lost to sparse | |
| 45 | +sampling. | |
| 46 | + | |
| 47 | +\paragraph{Realized kernel.} | |
| 48 | +We also compute the flat-top-free Parzen realized kernel of | |
| 49 | +\citet{BNHLS2008}, | |
| 50 | +\begin{equation} | |
| 51 | + RK_t \;=\; \gamma_0 + \sum_{h=1}^{H} k\!\left(\tfrac{h}{H+1}\right) | |
| 52 | + \left(\gamma_h + \gamma_{-h}\right), \qquad | |
| 53 | + \gamma_h = \sum_i r_{t,i}\, r_{t,i-h}, | |
| 54 | + \label{eq:rk} | |
| 55 | +\end{equation} | |
| 56 | +on 1-minute returns, where $k(\cdot)$ is the Parzen kernel and the | |
| 57 | +bandwidth follows the feasible rule $H^{*} = c^{*} \xi^{4/5} n^{3/5}$ | |
| 58 | +with $c^{*} = 3.5134$, noise-to-signal ratio | |
| 59 | +$\xi^2 = \hat\omega^2 / \widehat{IV}$, noise variance estimated as | |
| 60 | +$\hat\omega^2 = RV^{(1\mathrm{min})}_t / (2n)$, and $\widehat{IV}$ proxied | |
| 61 | +by $RV^{ss}_t$ \citep{BNHLS2009}. The non-flat-top Parzen weights | |
| 62 | +guarantee non-negativity. | |
| 63 | + | |
| 64 | +\paragraph{Bipower variation and jumps.} | |
| 65 | +The realized bipower variation of \citet{BNS2004}, | |
| 66 | +\begin{equation} | |
| 67 | + BV_t \;=\; \mu_1^{-2} \, \frac{n}{n-1} \sum_{i=2}^{n} | |
| 68 | + |r_{t,i}|\,|r_{t,i-1}|, \qquad \mu_1 = \sqrt{2/\pi}, | |
| 69 | + \label{eq:bv} | |
| 70 | +\end{equation} | |
| 71 | +is consistent for integrated variance in the presence of jumps. Jump days | |
| 72 | +are identified with the ratio-form test of \citet{BNS2006}, | |
| 73 | +\begin{equation} | |
| 74 | + Z_t \;=\; \frac{1 - BV_t / RV_t} | |
| 75 | + {\sqrt{\theta \, \max\!\left(1, \, TQ_t / BV_t^2\right) / n}} | |
| 76 | + \;\xrightarrow{d}\; N(0,1), \qquad | |
| 77 | + \theta = \tfrac{\pi^2}{4} + \pi - 5, | |
| 78 | + \label{eq:bns} | |
| 79 | +\end{equation} | |
| 80 | +where $TQ_t$ is the tripower quarticity. The jump and continuous | |
| 81 | +components are then | |
| 82 | +\begin{equation} | |
| 83 | + J_t = \mathbb{1}\{Z_t > \Phi^{-1}_{1-\alpha}\} \cdot | |
| 84 | + \max(RV_t - BV_t, 0), \qquad | |
| 85 | + C_t = RV_t - J_t, | |
| 86 | + \label{eq:cj} | |
| 87 | +\end{equation} | |
| 88 | +with $\alpha = 0.001$ \citep{ABD2007}. To limit noise distortions, | |
| 89 | +$BV_t$, $TQ_t$ and the test are computed on the sparse 5-minute grid. | |
| 90 | + | |
| 91 | +\paragraph{Realized semivariances.} | |
| 92 | +Following \citet{BNKS2010}, the positive and negative semivariances | |
| 93 | +\begin{equation} | |
| 94 | + RS^{+}_t = \sum_i r_{t,i}^2 \, \mathbb{1}\{r_{t,i} > 0\}, \qquad | |
| 95 | + RS^{-}_t = \sum_i r_{t,i}^2 \, \mathbb{1}\{r_{t,i} < 0\}, | |
| 96 | + \label{eq:semivar} | |
| 97 | +\end{equation} | |
| 98 | +decompose $RV_t$ by the sign of the underlying returns; the signed jump | |
| 99 | +variation is $SJ_t = RS^{+}_t - RS^{-}_t$ \citep{PattonSheppard2015}. | |
| 100 | +Finally, the realized quarticity $RQ_t = \tfrac{n}{3}\sum_i r_{t,i}^4$ | |
| 101 | +measures the sampling variability of $RV_t$ and feeds the HARQ model | |
| 102 | +below. | |
| 103 | + | |
| 104 | +\paragraph{Trading sessions.} | |
| 105 | +Equity and ETF measures use regular trading hours only (09:30--16:00, | |
| 106 | +exchange time); FX and futures use the full trading day on a calendar-day | |
| 107 | +partition, with holidays and thin days (fewer than 700 one-minute bars, | |
| 108 | +250 for equities) discarded; crypto trades continuously and its day is | |
| 109 | +the calendar day in UTC. Daily close-to-close returns $r_t$ (which | |
| 110 | +include the overnight component for session-limited markets) feed the | |
| 111 | +GARCH-family models and the economic-value exercises. | |
| 112 | + | |
| 113 | +\subsection{Econometric benchmarks} | |
| 114 | +\label{sec:econ_models} | |
| 115 | + | |
| 116 | +\paragraph{HAR family.} | |
| 117 | +The heterogeneous autoregressive model of \citet{Corsi2009} regresses | |
| 118 | +future RV on daily, weekly and monthly averages, | |
| 119 | +\begin{equation} | |
| 120 | + \overline{RV}_{t+1:t+h} \;=\; \beta_0 + \beta_d \, RV_t | |
| 121 | + + \beta_w \, \overline{RV}_{t-4:t} | |
| 122 | + + \beta_m \, \overline{RV}_{t-21:t} + \varepsilon_{t+h}, | |
| 123 | + \label{eq:har} | |
| 124 | +\end{equation} | |
| 125 | +where $\overline{RV}_{t+1:t+h} = h^{-1}\sum_{j=1}^h RV_{t+j}$ and all | |
| 126 | +horizons are estimated by direct projection. We consider five | |
| 127 | +extensions: HAR-J adds the jump component $J_t$; HAR-CJ replaces the RV | |
| 128 | +averages with continuous-component averages and adds $J_t$ | |
| 129 | +\citep{ABD2007}; SHAR splits the daily lag into $RS^{+}_t$ and | |
| 130 | +$RS^{-}_t$ \citep{PattonSheppard2015}; HARQ interacts the daily lag with | |
| 131 | +$\sqrt{RQ_t}$ to discount noisy RV observations \citep{BPQ2016}; and | |
| 132 | +log-HAR estimates \eqref{eq:har} in logarithms, with forecasts | |
| 133 | +retransformed as $\exp(\hat y + \hat\sigma^2_{\varepsilon}/2)$. Following | |
| 134 | +\citet{BPQ2016}, all forecasts pass through an ``insanity filter'' that | |
| 135 | +clips them to the range of the target observed in the estimation window. | |
| 136 | + | |
| 137 | +\paragraph{GARCH family.} | |
| 138 | +On daily returns we estimate GARCH(1,1) \citep{Bollerslev1986}, | |
| 139 | +GJR-GARCH \citep{GJR1993} and EGARCH \citep{Nelson1991} with Gaussian | |
| 140 | +quasi-likelihood. Because these models forecast the conditional variance | |
| 141 | +of close-to-close returns --- which includes overnight variation absent | |
| 142 | +from intraday RV --- their forecasts are mapped into RV units with the | |
| 143 | +in-sample proportionality factor $\hat c = \overline{RV} / | |
| 144 | +\overline{\hat\varepsilon^2}$ re-estimated in each rolling window, in the | |
| 145 | +spirit of \citet{HansenLunde2005}. Multi-step variance paths are | |
| 146 | +analytic for GARCH and GJR and simulated (500 paths) for EGARCH. | |
| 147 | + | |
| 148 | +\paragraph{Realized GARCH.} | |
| 149 | +The log-linear Realized GARCH(1,1) of \citet{HansenHuangShek2012} couples | |
| 150 | +the latent variance $h_t$ with the realized measure $x_t$: | |
| 151 | +\begin{align} | |
| 152 | + r_t &= \sqrt{h_t}\, z_t, \qquad z_t \sim N(0,1), \notag \\ | |
| 153 | + \log h_t &= \omega + \beta \log h_{t-1} + \gamma \log x_{t-1}, | |
| 154 | + \label{eq:rgarch} \\ | |
| 155 | + \log x_t &= \xi + \varphi \log h_t + \tau_1 z_t + \tau_2(z_t^2 - 1) | |
| 156 | + + u_t, \qquad u_t \sim N(0, \sigma_u^2). \notag | |
| 157 | +\end{align} | |
| 158 | +We estimate \eqref{eq:rgarch} by quasi-maximum likelihood and forecast | |
| 159 | +the realized measure directly, using the analytic recursion for | |
| 160 | +$\mathbb{E}[\log h_{t+k}]$ with persistence $\pi = \beta + | |
| 161 | +\gamma\varphi$ and a lognormal correction for the accumulated forecast | |
| 162 | +variance. | |
| 163 | + | |
| 164 | +\subsection{Machine-learning models} | |
| 165 | +\label{sec:ml_models} | |
| 166 | + | |
| 167 | +\paragraph{Information sets.} | |
| 168 | +All learners forecast $\log \overline{RV}_{t+1:t+h}$ and are evaluated | |
| 169 | +after retransformation with the nonparametric smearing factor of | |
| 170 | +\citet{Duan1983} and the same insanity filter as the HAR models. Two | |
| 171 | +nested predictor sets are used. The \emph{HAR set} contains exactly the | |
| 172 | +HAR information: $\log RV_t$, $\log \overline{RV}_{t-4:t}$, | |
| 173 | +$\log \overline{RV}_{t-21:t}$. The \emph{extended set} adds the jump | |
| 174 | +share $J_t / RV_t$, log semivariances, normalized signed jumps, log | |
| 175 | +realized quarticity, daily/weekly/monthly returns, log volume growth, | |
| 176 | +the day of the week, the (lagged) log VIX level, and cross-asset | |
| 177 | +averages of log RV within each asset class --- fourteen additional | |
| 178 | +predictors, all known at the end of day $t$. | |
| 179 | + | |
| 180 | +\paragraph{Learners.} | |
| 181 | +We consider penalized linear models --- ridge, LASSO | |
| 182 | +\citep{Tibshirani1996} and elastic net \citep{ZouHastie2005} on | |
| 183 | +standardized predictors --- and tree ensembles: random forests | |
| 184 | +\citep{Breiman2001}, XGBoost \citep{ChenGuestrin2016} and LightGBM | |
| 185 | +\citep{Ke2017}. Hyperparameters are tuned by walk-forward validation | |
| 186 | +inside the estimation window (the last 20\% of the window, in time | |
| 187 | +order, serves as the validation split), re-tuned once a year, and the | |
| 188 | +winning configuration is refit on the full window; no future information | |
| 189 | +enters the tuning loop at any point. | |
| 190 | + | |
| 191 | +\paragraph{Deep learners.} | |
| 192 | +Two sequence models map the last 22 days of extended features to all | |
| 193 | +three horizons jointly: a single-layer LSTM | |
| 194 | +\citep{HochreiterSchmidhuber1997} with 64 hidden units, and a | |
| 195 | +two-block Transformer encoder \citep{Vaswani2017} with model dimension | |
| 196 | +32 and 4 attention heads. Both are trained with Adam, early stopping on | |
| 197 | +a temporal validation split, and quarterly re-estimation. | |
| 198 | + | |
| 199 | +\paragraph{Pooled multi-asset model.} | |
| 200 | +Beyond the asset-by-asset models, a single pooled LightGBM is trained on | |
| 201 | +the stacked panel of all 25 assets with the ticker and asset-class | |
| 202 | +identifiers as categorical features, re-estimated monthly on the | |
| 203 | +trailing window. The pooled design tests whether volatility dynamics are | |
| 204 | +sufficiently common across assets for cross-sectional information to | |
| 205 | +sharpen asset-level forecasts \citep{GuKellyXiu2020, | |
| 206 | +ChristensenSiggaardVeliyev2023}. A leave-one-out variant excludes a | |
| 207 | +target asset from the training pool entirely and predicts it | |
| 208 | +out-of-pool, measuring the transferability of pooled dynamics to unseen | |
| 209 | +assets. Model interpretation relies on SHAP values \citep{LundbergLee2017} | |
| 210 | +and permutation importance. | |
| 211 | + | |
| 212 | +\subsection{Forecast evaluation} | |
| 213 | +\label{sec:evaluation} | |
| 214 | + | |
| 215 | +\paragraph{Rolling protocol.} | |
| 216 | +All models are estimated on a rolling window of 1{,}000 trading days and | |
| 217 | +re-estimated every 22 days; forecasts are produced daily for horizons | |
| 218 | +$h \in \{1, 5, 22\}$, defined as the \emph{average} daily variance over | |
| 219 | +the next $h$ days (cumulative variance equals $h$ times the average). | |
| 220 | +The evaluation sample therefore starts after the first | |
| 221 | +1{,}000-plus-$h$ observations of each series and runs through July | |
| 222 | +2026. | |
| 223 | + | |
| 224 | +\paragraph{Loss functions.} | |
| 225 | +Our primary loss is the QLIKE, | |
| 226 | +\begin{equation} | |
| 227 | + L^{Q}(RV, F) \;=\; \frac{RV}{F} - \log\frac{RV}{F} - 1, | |
| 228 | + \label{eq:qlike} | |
| 229 | +\end{equation} | |
| 230 | +which together with the MSE is robust to noise in the volatility proxy | |
| 231 | +\citep{Patton2011}: rankings based on \eqref{eq:qlike} are consistent | |
| 232 | +for the rankings that would obtain under the true conditional variance. | |
| 233 | +QLIKE is also scale-free, which permits averaging across assets; MSE | |
| 234 | +results are reported as a complement. | |
| 235 | + | |
| 236 | +\paragraph{Statistical tests.} | |
| 237 | +Pairwise forecast comparisons use \citet{DieboldMariano1995} tests with | |
| 238 | +Newey--West long-run variances \citep{NeweyWest1987} and bandwidth | |
| 239 | +$h-1$. Joint comparisons use the 90\% model confidence set of | |
| 240 | +\citet{HLN2011} with the $T_{\max}$ statistic and 5{,}000 stationary | |
| 241 | +bootstrap replications \citep{PolitisRomano1994}. Forecast | |
| 242 | +unbiasedness is assessed with \citet{MincerZarnowitz1969} levels | |
| 243 | +regressions, $RV_{t+1:t+h} = a + b F_{t,h} + e_{t+h}$, testing | |
| 244 | +$(a,b) = (0,1)$ with HAC standard errors. Whether machine-learning | |
| 245 | +forecasts contain information absent from HAR is tested with forecast | |
| 246 | +encompassing regressions \citep{HarveyLeybourneNewbold1998}, | |
| 247 | +$RV = a + b_1 F^{HAR} + b_2 F^{ML} + e$: rejection of $b_2 = 0$ means | |
| 248 | +HAR does not encompass the rival. Finally, we report two forecast | |
| 249 | +combinations \citep{BatesGranger1969,Timmermann2006}: the simple mean | |
| 250 | +of all individual models and inverse-MSE weights computed on a trailing | |
| 251 | +60-day window. | |
| 252 | + | |
| 253 | +\paragraph{Stability.} | |
| 254 | +Because relative performance may drift across regimes, we complement | |
| 255 | +full-sample tests with the fluctuation test of \citet{GiacominiRossi2010}, | |
| 256 | +which tracks the standardized DM statistic over a centred rolling window | |
| 257 | +of 30\% of the evaluation sample and compares its path to | |
| 258 | +regime-robust critical values. | |
| 259 | + | |
| 260 | +\subsection{Economic value} | |
| 261 | +\label{sec:econ_value_method} | |
| 262 | + | |
| 263 | +\paragraph{Volatility timing.} | |
| 264 | +For each asset and model we run the volatility-timing strategy of | |
| 265 | +\citet{FKO2001,FKO2003}: the weight on the risky asset is | |
| 266 | +$w_t = \min\!\left(\bar w, \; \sigma^{\ast} / \hat\sigma_{t+1}\right)$, | |
| 267 | +with annualized target $\sigma^{\ast} = 10\%$, leverage cap | |
| 268 | +$\bar w = 3$, and $\hat\sigma_{t+1}$ the one-day-ahead forecast | |
| 269 | +volatility. Net returns subtract proportional transaction costs of 5 | |
| 270 | +basis points on turnover (0--20 bps in sensitivity checks). We report | |
| 271 | +annualized Sharpe ratios and the \citet{FKO2001} utility gain --- the | |
| 272 | +annualized fee $\Delta$ a mean-variance investor with relative risk | |
| 273 | +aversion $\gamma = 5$ would pay to switch from HAR-based to model-based | |
| 274 | +timing. | |
| 275 | + | |
| 276 | +\paragraph{Value-at-Risk.} | |
| 277 | +One-day Gaussian VaR forecasts, $\mathrm{VaR}^{q}_{t+1} = -z_q \, | |
| 278 | +\hat\sigma^{r}_{t+1}$, are backtested at the 1\% and 5\% levels with the | |
| 279 | +unconditional-coverage test of \citet{Kupiec1995} and the | |
| 280 | +conditional-coverage test of \citet{Christoffersen1998}. Because RV | |
| 281 | +omits overnight variance, forecasts are rescaled into close-to-close | |
| 282 | +return units with a trailing 250-day variance ratio, so no future | |
| 283 | +information is used. | |
added
requirements.txt
+24 −0
@@ -0,0 +1,24 @@ | ||
| 1 | +# ================================================================ | |
| 2 | +# Auteur : Simon-Pierre Boucher | |
| 3 | +# Contact : contact@spboucher.ai | |
| 4 | +# Projet : Prévision de volatilité réalisée multi-actifs | |
| 5 | +# (HAR-RV vs GARCH vs Machine Learning) | |
| 6 | +# Fichier : requirements.txt | |
| 7 | +# Description : Dépendances Python (>= 3.11), versions épinglées | |
| 8 | +# à l'environnement de l'analyse. | |
| 9 | +# ================================================================ | |
| 10 | +numpy==2.4.4 | |
| 11 | +pandas==3.0.2 | |
| 12 | +pyarrow>=16.0 | |
| 13 | +scipy==1.17.1 | |
| 14 | +scikit-learn==1.6.1 | |
| 15 | +statsmodels==0.14.6 | |
| 16 | +arch==8.0.0 | |
| 17 | +xgboost==3.2.0 | |
| 18 | +lightgbm==4.6.0 | |
| 19 | +torch==2.12.0 | |
| 20 | +shap==0.48.0 | |
| 21 | +matplotlib==3.10.9 | |
| 22 | +requests==2.34.2 | |
| 23 | +PyYAML==6.0.3 | |
| 24 | +pytest==9.1.1 | |
modified
scripts/03_run_models.py
+16 −1
@@ -100,6 +100,19 @@ def run_deep(ticker: str) -> str: | ||
| 100 | 100 | return f"{ticker}: {len(out)} deep forecasts" |
| 101 | 101 | |
| 102 | 102 | |
| 103 | +def run_interpret() -> str: | |
| 104 | + """SHAP values and permutation importance of the pooled LightGBM.""" | |
| 105 | + from wp12 import models_ml as ml | |
| 106 | + | |
| 107 | + panel = pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 108 | + frames = ml.ml_features(panel) | |
| 109 | + shap_tab, perm_tab = ml.shap_and_permutation(frames, h=1, window=WINDOW) | |
| 110 | + out = config.path("reproduced") | |
| 111 | + shap_tab.to_csv(out / "shap_h1.csv", index=False) | |
| 112 | + perm_tab.to_csv(out / "permutation_h1.csv", index=False) | |
| 113 | + return f"interpret: {len(shap_tab)} features" | |
| 114 | + | |
| 115 | + | |
| 103 | 116 | def run_pooled() -> str: |
| 104 | 117 | """Pooled multi-asset LightGBM (single job, all tickers jointly).""" |
| 105 | 118 | from wp12 import models_ml as ml |
@@ -117,7 +130,7 @@ def main() -> None: | ||
| 117 | 130 | """Dispatch model runs across tickers with a process pool.""" |
| 118 | 131 | ap = argparse.ArgumentParser() |
| 119 | 132 | ap.add_argument("--family", default="all", |
| 120 | − choices=["econ", "ml", "deep", "pooled", "all"]) | |
| 133 | + choices=["econ", "ml", "deep", "pooled", "interpret", "all"]) | |
| 121 | 134 | ap.add_argument("--tickers", default=None, help="comma-separated subset") |
| 122 | 135 | ap.add_argument("--workers", type=int, default=8) |
| 123 | 136 | args = ap.parse_args() |
@@ -147,6 +160,8 @@ def main() -> None: | ||
| 147 | 160 | logger.error("FAILED %s %s: %s", name, tk, exc) |
| 148 | 161 | if args.family in ("pooled", "all"): |
| 149 | 162 | logger.info(run_pooled()) |
| 163 | + if args.family in ("interpret", "all"): | |
| 164 | + logger.info(run_interpret()) | |
| 150 | 165 | |
| 151 | 166 | |
| 152 | 167 | if __name__ == "__main__": |
added
scripts/04_evaluate.py
+230 −0
@@ -0,0 +1,230 @@ | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +""" | |
| 3 | +================================================================ | |
| 4 | +Auteur : Simon-Pierre Boucher | |
| 5 | +Contact : contact@spboucher.ai | |
| 6 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 7 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 8 | +Fichier : 04_evaluate.py | |
| 9 | +Description : Étape 04 — Évaluation out-of-sample : assemblage | |
| 10 | + des prévisions, combinaisons, pertes QLIKE/MSE, | |
| 11 | + Diebold-Mariano vs HAR, Model Confidence Set, | |
| 12 | + Mincer-Zarnowitz et forecast encompassing. | |
| 13 | +Usage : python scripts/04_evaluate.py | |
| 14 | +================================================================ | |
| 15 | +""" | |
| 16 | + | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import logging | |
| 20 | +import sys | |
| 21 | +from pathlib import Path | |
| 22 | + | |
| 23 | +import numpy as np | |
| 24 | +import pandas as pd | |
| 25 | + | |
| 26 | +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| 27 | + | |
| 28 | +from wp12 import config, evaluation as ev # noqa: E402 | |
| 29 | +from wp12.models_econ import horizon_target # noqa: E402 | |
| 30 | + | |
| 31 | +logger = logging.getLogger("04_evaluate") | |
| 32 | + | |
| 33 | +FC_DIR = config.path("processed") / "forecasts" | |
| 34 | +OUT = config.path("reproduced") | |
| 35 | +HORIZONS = tuple(config.load_config()["evaluation"]["horizons"]) | |
| 36 | +BENCH = "HAR" | |
| 37 | + | |
| 38 | + | |
| 39 | +def load_forecasts() -> pd.DataFrame: | |
| 40 | + """Concatenate every stored forecast file into one long table.""" | |
| 41 | + frames = [] | |
| 42 | + for f in sorted(FC_DIR.glob("*.parquet")): | |
| 43 | + df = pd.read_parquet(f) | |
| 44 | + frames.append(df) | |
| 45 | + out = pd.concat(frames, ignore_index=True) | |
| 46 | + out["date"] = pd.to_datetime(out["date"]) | |
| 47 | + logger.info("loaded %d forecasts, %d models, %d tickers", | |
| 48 | + len(out), out["model"].nunique(), out["ticker"].nunique()) | |
| 49 | + return out | |
| 50 | + | |
| 51 | + | |
| 52 | +def attach_actuals(fc: pd.DataFrame, measure: str = "rv5ss") -> pd.DataFrame: | |
| 53 | + """Join the realized horizon-average RV target onto the forecast table.""" | |
| 54 | + panel = pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 55 | + targets = [] | |
| 56 | + for tk, df in panel.groupby("ticker"): | |
| 57 | + rv = df.sort_index()[measure] | |
| 58 | + for h in HORIZONS: | |
| 59 | + t = horizon_target(rv, h).rename("actual").reset_index() | |
| 60 | + t["ticker"] = tk | |
| 61 | + t["h"] = h | |
| 62 | + targets.append(t) | |
| 63 | + tgt = pd.concat(targets, ignore_index=True) | |
| 64 | + merged = fc.merge(tgt, on=["date", "ticker", "h"], how="inner") | |
| 65 | + merged = merged.dropna(subset=["actual", "forecast"]) | |
| 66 | + merged = merged[merged["actual"] > 0] | |
| 67 | + logger.info("merged: %d forecast-actual pairs", len(merged)) | |
| 68 | + return merged | |
| 69 | + | |
| 70 | + | |
| 71 | +def add_combinations(merged: pd.DataFrame) -> pd.DataFrame: | |
| 72 | + """Append simple-mean and inverse-MSE combination forecasts.""" | |
| 73 | + combos = [] | |
| 74 | + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): | |
| 75 | + wide = sub.pivot_table(index="date", columns="model", values="forecast") | |
| 76 | + actual = sub.drop_duplicates("date").set_index("date")["actual"] | |
| 77 | + c = ev.combine_forecasts(wide, actual.reindex(wide.index)) | |
| 78 | + c = c.stack().rename("forecast").reset_index() | |
| 79 | + c.columns = ["date", "model", "forecast"] | |
| 80 | + c["ticker"] = tk | |
| 81 | + c["h"] = h | |
| 82 | + c = c.merge(actual.rename("actual").reset_index(), on="date") | |
| 83 | + combos.append(c.dropna(subset=["forecast", "actual"])) | |
| 84 | + out = pd.concat([merged] + combos, ignore_index=True) | |
| 85 | + logger.info("with combinations: %d rows", len(out)) | |
| 86 | + return out | |
| 87 | + | |
| 88 | + | |
| 89 | +def loss_tables(merged: pd.DataFrame) -> None: | |
| 90 | + """Mean losses overall / by class / by ticker, with ratios to HAR.""" | |
| 91 | + panel_cls = ( | |
| 92 | + pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 93 | + .groupby("ticker")["cls"].first() | |
| 94 | + ) | |
| 95 | + merged = merged.assign(cls=merged["ticker"].map(panel_cls)) | |
| 96 | + for loss in ("qlike", "mse"): | |
| 97 | + lf = ev.LOSSES[loss] | |
| 98 | + merged[loss] = lf(merged["actual"].to_numpy(), merged["forecast"].to_numpy()) | |
| 99 | + | |
| 100 | + per_asset = merged.groupby(["ticker", "cls", "h", "model"], observed=True)[ | |
| 101 | + ["qlike", "mse"]].mean().reset_index() | |
| 102 | + per_asset.to_csv(OUT / "losses_by_ticker.csv", index=False) | |
| 103 | + | |
| 104 | + def _with_ratio(df: pd.DataFrame, keys: list[str]) -> pd.DataFrame: | |
| 105 | + bench = df[df["model"] == BENCH].set_index(keys)["qlike"] | |
| 106 | + df = df.set_index(keys) | |
| 107 | + df["qlike_ratio"] = df["qlike"] / bench | |
| 108 | + return df.reset_index() | |
| 109 | + | |
| 110 | + overall = per_asset.groupby(["h", "model"], observed=True)[ | |
| 111 | + ["qlike", "mse"]].mean().reset_index() | |
| 112 | + _with_ratio(overall, ["h"]).to_csv(OUT / "losses_overall.csv", index=False) | |
| 113 | + | |
| 114 | + by_class = per_asset.groupby(["cls", "h", "model"], observed=True)[ | |
| 115 | + ["qlike", "mse"]].mean().reset_index() | |
| 116 | + _with_ratio(by_class, ["cls", "h"]).to_csv(OUT / "losses_by_class.csv", index=False) | |
| 117 | + logger.info("loss tables written") | |
| 118 | + | |
| 119 | + | |
| 120 | +def dm_tables(merged: pd.DataFrame) -> None: | |
| 121 | + """Diebold-Mariano tests of every model against the HAR benchmark.""" | |
| 122 | + rows = [] | |
| 123 | + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): | |
| 124 | + wide_f = sub.pivot_table(index="date", columns="model", values="forecast") | |
| 125 | + actual = sub.drop_duplicates("date").set_index("date")["actual"] | |
| 126 | + if BENCH not in wide_f.columns: | |
| 127 | + continue | |
| 128 | + lb = ev.qlike(actual.to_numpy(), wide_f[BENCH].to_numpy()) | |
| 129 | + for m in wide_f.columns: | |
| 130 | + if m == BENCH: | |
| 131 | + continue | |
| 132 | + both = np.isfinite(wide_f[m].to_numpy()) & np.isfinite(wide_f[BENCH].to_numpy()) | |
| 133 | + la = ev.qlike(actual.to_numpy()[both], wide_f[m].to_numpy()[both]) | |
| 134 | + stat, p = ev.diebold_mariano(la, lb[both], h=h) | |
| 135 | + rows.append((tk, h, m, stat, p)) | |
| 136 | + dm = pd.DataFrame(rows, columns=["ticker", "h", "model", "dm_stat", "dm_p"]) | |
| 137 | + dm.to_csv(OUT / "dm_by_ticker.csv", index=False) | |
| 138 | + summary = dm.groupby(["h", "model"], observed=True).apply( | |
| 139 | + lambda d: pd.Series({ | |
| 140 | + "n_assets": len(d), | |
| 141 | + "median_dm": d["dm_stat"].median(), | |
| 142 | + "pct_better": float((d["dm_stat"] < 0).mean()), | |
| 143 | + "pct_sig_better": float(((d["dm_stat"] < 0) & (d["dm_p"] < 0.05)).mean()), | |
| 144 | + "pct_sig_worse": float(((d["dm_stat"] > 0) & (d["dm_p"] < 0.05)).mean()), | |
| 145 | + }), include_groups=False).reset_index() | |
| 146 | + summary.to_csv(OUT / "dm_summary.csv", index=False) | |
| 147 | + logger.info("DM tables written") | |
| 148 | + | |
| 149 | + | |
| 150 | +def mcs_tables(merged: pd.DataFrame) -> None: | |
| 151 | + """90% Model Confidence Set per (ticker, horizon); inclusion shares.""" | |
| 152 | + cfg = config.load_config()["evaluation"] | |
| 153 | + rows = [] | |
| 154 | + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): | |
| 155 | + wide_f = sub.pivot_table(index="date", columns="model", values="forecast") | |
| 156 | + actual = sub.drop_duplicates("date").set_index("date")["actual"] | |
| 157 | + losses = pd.DataFrame({ | |
| 158 | + m: ev.qlike(actual.to_numpy(), wide_f[m].to_numpy()) | |
| 159 | + for m in wide_f.columns | |
| 160 | + }, index=wide_f.index).dropna() | |
| 161 | + if len(losses) < 200: | |
| 162 | + continue | |
| 163 | + mcs = ev.model_confidence_set( | |
| 164 | + losses, alpha=cfg["mcs_alpha"], n_boot=int(cfg["mcs_bootstrap"])) | |
| 165 | + mcs["ticker"] = tk | |
| 166 | + mcs["h"] = h | |
| 167 | + rows.append(mcs) | |
| 168 | + allmcs = pd.concat(rows, ignore_index=True) | |
| 169 | + allmcs.to_csv(OUT / "mcs_by_ticker.csv", index=False) | |
| 170 | + inc = allmcs.groupby(["h", "model"], observed=True)["in_mcs"].mean().rename( | |
| 171 | + "mcs_inclusion").reset_index() | |
| 172 | + inc.to_csv(OUT / "mcs_inclusion.csv", index=False) | |
| 173 | + logger.info("MCS tables written") | |
| 174 | + | |
| 175 | + | |
| 176 | +def mz_and_encompassing(merged: pd.DataFrame) -> None: | |
| 177 | + """Mincer-Zarnowitz per model and HAR-vs-ML encompassing tests.""" | |
| 178 | + rows = [] | |
| 179 | + for (tk, h, m), sub in merged.groupby(["ticker", "h", "model"], observed=True): | |
| 180 | + res = ev.mincer_zarnowitz(sub["actual"].to_numpy(), sub["forecast"].to_numpy(), h=h) | |
| 181 | + rows.append((tk, h, m, res["alpha"], res["beta"], res["r2"], res["p_joint"])) | |
| 182 | + mz = pd.DataFrame(rows, columns=["ticker", "h", "model", "alpha", "beta", "r2", "p_joint"]) | |
| 183 | + mz.to_csv(OUT / "mz_by_ticker.csv", index=False) | |
| 184 | + mz_sum = mz.groupby(["h", "model"], observed=True).apply( | |
| 185 | + lambda d: pd.Series({ | |
| 186 | + "med_beta": d["beta"].median(), "med_r2": d["r2"].median(), | |
| 187 | + "pct_reject": float((d["p_joint"] < 0.05).mean()), | |
| 188 | + }), include_groups=False).reset_index() | |
| 189 | + mz_sum.to_csv(OUT / "mz_summary.csv", index=False) | |
| 190 | + | |
| 191 | + rows = [] | |
| 192 | + rivals = [m for m in merged["model"].unique() if m != BENCH] | |
| 193 | + for (tk, h), sub in merged.groupby(["ticker", "h"], observed=True): | |
| 194 | + wide_f = sub.pivot_table(index="date", columns="model", values="forecast") | |
| 195 | + actual = sub.drop_duplicates("date").set_index("date")["actual"] | |
| 196 | + if BENCH not in wide_f.columns: | |
| 197 | + continue | |
| 198 | + for m in rivals: | |
| 199 | + if m not in wide_f.columns: | |
| 200 | + continue | |
| 201 | + res = ev.forecast_encompassing( | |
| 202 | + actual.to_numpy(), wide_f[BENCH].to_numpy(), wide_f[m].to_numpy(), h=h) | |
| 203 | + rows.append((tk, h, m, res["b1"], res["b2"], res["p_b2"])) | |
| 204 | + enc = pd.DataFrame(rows, columns=["ticker", "h", "model", "b1", "b2", "p_b2"]) | |
| 205 | + enc.to_csv(OUT / "encompassing_by_ticker.csv", index=False) | |
| 206 | + enc_sum = enc.groupby(["h", "model"], observed=True).apply( | |
| 207 | + lambda d: pd.Series({ | |
| 208 | + "med_b2": d["b2"].median(), | |
| 209 | + "pct_adds_info": float(((d["b2"] > 0) & (d["p_b2"] < 0.05)).mean()), | |
| 210 | + }), include_groups=False).reset_index() | |
| 211 | + enc_sum.to_csv(OUT / "encompassing_summary.csv", index=False) | |
| 212 | + logger.info("MZ and encompassing tables written") | |
| 213 | + | |
| 214 | + | |
| 215 | +def main() -> None: | |
| 216 | + """Run the full evaluation battery and persist intermediate tables.""" | |
| 217 | + config.setup_logging() | |
| 218 | + fc = load_forecasts() | |
| 219 | + merged = attach_actuals(fc) | |
| 220 | + merged = add_combinations(merged) | |
| 221 | + merged.to_parquet(config.path("processed") / "merged_forecasts.parquet", index=False) | |
| 222 | + loss_tables(merged) | |
| 223 | + dm_tables(merged) | |
| 224 | + mcs_tables(merged) | |
| 225 | + mz_and_encompassing(merged) | |
| 226 | + logger.info("evaluation complete") | |
| 227 | + | |
| 228 | + | |
| 229 | +if __name__ == "__main__": | |
| 230 | + main() | |
added
scripts/05_robustness.py
+220 −0
@@ -0,0 +1,220 @@ | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +""" | |
| 3 | +================================================================ | |
| 4 | +Auteur : Simon-Pierre Boucher | |
| 5 | +Contact : contact@spboucher.ai | |
| 6 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 7 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 8 | +Fichier : 05_robustness.py | |
| 9 | +Description : Étape 05 — Robustesse : sous-périodes, sensibilité | |
| 10 | + à la fréquence d'échantillonnage et à la fenêtre | |
| 11 | + d'estimation, transférabilité cross-asset (pooled | |
| 12 | + leave-one-out) et fluctuation test Giacomini-Rossi. | |
| 13 | +Usage : python scripts/05_robustness.py [--part all|subperiods|freq|window|loo|gr] | |
| 14 | +================================================================ | |
| 15 | +""" | |
| 16 | + | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import argparse | |
| 20 | +import logging | |
| 21 | +import sys | |
| 22 | +from concurrent.futures import ProcessPoolExecutor, as_completed | |
| 23 | +from pathlib import Path | |
| 24 | + | |
| 25 | +import numpy as np | |
| 26 | +import pandas as pd | |
| 27 | + | |
| 28 | +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| 29 | + | |
| 30 | +from wp12 import config, evaluation as ev, robustness as rb # noqa: E402 | |
| 31 | +from wp12.models_econ import horizon_target # noqa: E402 | |
| 32 | + | |
| 33 | +logger = logging.getLogger("05_robustness") | |
| 34 | + | |
| 35 | +OUT = config.path("reproduced") | |
| 36 | +HORIZONS = tuple(config.load_config()["evaluation"]["horizons"]) | |
| 37 | +WINDOW = int(config.load_config()["evaluation"]["estimation_window_days"]) | |
| 38 | +REFIT = int(config.load_config()["evaluation"]["reestimation_freq_days"]) | |
| 39 | + | |
| 40 | +# representative asset per class for the expensive experiments | |
| 41 | +REPRESENTATIVE = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"} | |
| 42 | +# models re-run in the sensitivity experiments | |
| 43 | +SENS_MODELS = ["HAR", "LogHAR", "LightGBM"] | |
| 44 | + | |
| 45 | + | |
| 46 | +def _merged() -> pd.DataFrame: | |
| 47 | + df = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet") | |
| 48 | + df["date"] = pd.to_datetime(df["date"]) | |
| 49 | + return df | |
| 50 | + | |
| 51 | + | |
| 52 | +def part_subperiods() -> None: | |
| 53 | + """Mean QLIKE per model within each sub-period of the config.""" | |
| 54 | + sp = {k: tuple(v) for k, v in config.load_config()["robustness"]["subperiods"].items()} | |
| 55 | + merged = _merged() | |
| 56 | + tab = rb.loss_by_period(merged, sp) | |
| 57 | + tab.to_csv(OUT / "subperiod_losses.csv", index=False) | |
| 58 | + logger.info("subperiod table written (%d rows)", len(tab)) | |
| 59 | + | |
| 60 | + | |
| 61 | +def _sens_one(job: tuple[str, str, str, int]) -> pd.DataFrame: | |
| 62 | + """Worker: re-run one (ticker, model, measure, window) variant.""" | |
| 63 | + from wp12 import models_econ as me, models_ml as ml | |
| 64 | + | |
| 65 | + ticker, model, measure, window = job | |
| 66 | + df = pd.read_parquet(config.path("processed") / f"rv_{ticker}.parquet") | |
| 67 | + if model in ("HAR", "LogHAR"): | |
| 68 | + fc = me.rolling_har_forecasts( | |
| 69 | + df, model, horizons=HORIZONS, measure=measure, | |
| 70 | + window=window, refit_every=REFIT) | |
| 71 | + else: # LightGBM extended | |
| 72 | + panel = pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 73 | + frames = ml.ml_features(panel, measure=measure) | |
| 74 | + fc = ml.rolling_ml_forecasts( | |
| 75 | + frames[ticker], "LightGBM", feature_set="extended", | |
| 76 | + horizons=HORIZONS, window=window, refit_every=REFIT) | |
| 77 | + fc["model"] = "LightGBM" | |
| 78 | + fc["ticker"] = ticker | |
| 79 | + # attach actuals in the SAME measure | |
| 80 | + rv = df[measure] | |
| 81 | + parts = [] | |
| 82 | + for h in HORIZONS: | |
| 83 | + t = horizon_target(rv, h).rename("actual").reset_index() | |
| 84 | + t["h"] = h | |
| 85 | + parts.append(t) | |
| 86 | + tgt = pd.concat(parts, ignore_index=True) | |
| 87 | + fc["date"] = pd.to_datetime(fc["date"]) | |
| 88 | + tgt["date"] = pd.to_datetime(tgt["date"]) | |
| 89 | + out = fc.merge(tgt, on=["date", "h"]).dropna(subset=["actual", "forecast"]) | |
| 90 | + return out[out["actual"] > 0] | |
| 91 | + | |
| 92 | + | |
| 93 | +def _run_jobs(jobs: list[tuple], workers: int) -> list[pd.DataFrame]: | |
| 94 | + res = [] | |
| 95 | + with ProcessPoolExecutor(max_workers=workers) as pool: | |
| 96 | + futs = {pool.submit(_sens_one, j): j for j in jobs} | |
| 97 | + for fut in as_completed(futs): | |
| 98 | + j = futs[fut] | |
| 99 | + try: | |
| 100 | + res.append(fut.result()) | |
| 101 | + logger.info("done %s", j) | |
| 102 | + except Exception as exc: # noqa: BLE001 | |
| 103 | + logger.error("FAILED %s: %s", j, exc) | |
| 104 | + return res | |
| 105 | + | |
| 106 | + | |
| 107 | +def part_frequency(workers: int) -> None: | |
| 108 | + """Sensitivity to the RV sampling scheme (1-min, 5-min ss, kernel).""" | |
| 109 | + tickers = [i["ticker"] for i in config.universe()] | |
| 110 | + freqs = {"rv1min": "rv1", "rv5min_ss": "rv5ss", "rkernel": "rk"} | |
| 111 | + results = {} | |
| 112 | + for label, measure in freqs.items(): | |
| 113 | + jobs = [(tk, m, measure, WINDOW) for tk in tickers for m in SENS_MODELS] | |
| 114 | + frames = _run_jobs(jobs, workers) | |
| 115 | + results[label] = pd.concat(frames, ignore_index=True) | |
| 116 | + tab = rb.sensitivity_table(results, "frequency") | |
| 117 | + tab.to_csv(OUT / "frequency_sensitivity.csv", index=False) | |
| 118 | + logger.info("frequency sensitivity written") | |
| 119 | + | |
| 120 | + | |
| 121 | +def part_window(workers: int) -> None: | |
| 122 | + """Sensitivity to the estimation window (500 / 1000 / 2000 days).""" | |
| 123 | + tickers = [i["ticker"] for i in config.universe()] | |
| 124 | + results = {} | |
| 125 | + for w in config.load_config()["robustness"]["window_sizes"]: | |
| 126 | + jobs = [(tk, m, "rv5ss", int(w)) for tk in tickers for m in SENS_MODELS] | |
| 127 | + frames = _run_jobs(jobs, workers) | |
| 128 | + results[str(w)] = pd.concat(frames, ignore_index=True) | |
| 129 | + tab = rb.sensitivity_table(results, "window") | |
| 130 | + tab.to_csv(OUT / "window_sensitivity.csv", index=False) | |
| 131 | + logger.info("window sensitivity written") | |
| 132 | + | |
| 133 | + | |
| 134 | +def part_loo() -> None: | |
| 135 | + """Cross-asset transferability: pooled model with one class representative | |
| 136 | + excluded from training, predicted out-of-pool.""" | |
| 137 | + from wp12 import models_ml as ml | |
| 138 | + | |
| 139 | + panel = pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 140 | + frames = ml.ml_features(panel) | |
| 141 | + parts = [] | |
| 142 | + for cls, tk in REPRESENTATIVE.items(): | |
| 143 | + fc = ml.rolling_pooled_lgbm( | |
| 144 | + frames, horizons=HORIZONS, window=WINDOW, refit_every=REFIT, exclude=tk) | |
| 145 | + fc["cls"] = cls | |
| 146 | + parts.append(fc) | |
| 147 | + loo = pd.concat(parts, ignore_index=True) | |
| 148 | + loo.to_parquet(config.path("processed") / "forecasts" / "pooled_loo.parquet", | |
| 149 | + index=False) | |
| 150 | + # compare in-pool vs out-of-pool QLIKE on the same dates | |
| 151 | + merged = _merged() | |
| 152 | + rows = [] | |
| 153 | + for cls, tk in REPRESENTATIVE.items(): | |
| 154 | + sub_loo = loo[loo["ticker"] == tk].copy() | |
| 155 | + sub_loo["date"] = pd.to_datetime(sub_loo["date"]) | |
| 156 | + base = merged[(merged["ticker"] == tk) | |
| 157 | + & (merged["model"].isin(["Pooled-LGBM", "HAR", "LightGBM-X"]))] | |
| 158 | + act = base.drop_duplicates(["date", "h"])[["date", "h", "actual"]] | |
| 159 | + sub = sub_loo.merge(act, on=["date", "h"]).dropna(subset=["actual"]) | |
| 160 | + for h in HORIZONS: | |
| 161 | + d = sub[sub["h"] == h] | |
| 162 | + q_loo = float(np.mean(ev.qlike(d["actual"].to_numpy(), d["forecast"].to_numpy()))) | |
| 163 | + for m in ["Pooled-LGBM", "HAR", "LightGBM-X"]: | |
| 164 | + b = base[(base["h"] == h) & (base["model"] == m)] | |
| 165 | + b = b.merge(d[["date"]], on="date") | |
| 166 | + q = float(np.mean(ev.qlike(b["actual"].to_numpy(), b["forecast"].to_numpy()))) | |
| 167 | + rows.append((cls, tk, h, m, q)) | |
| 168 | + rows.append((cls, tk, h, "Pooled-LGBM-LOO", q_loo)) | |
| 169 | + tab = pd.DataFrame(rows, columns=["cls", "ticker", "h", "model", "qlike"]) | |
| 170 | + tab.to_csv(OUT / "transferability.csv", index=False) | |
| 171 | + logger.info("transferability written") | |
| 172 | + | |
| 173 | + | |
| 174 | +def part_gr() -> None: | |
| 175 | + """Giacomini-Rossi fluctuation paths: best ML vs HAR, h=1, per class rep.""" | |
| 176 | + merged = _merged() | |
| 177 | + paths = [] | |
| 178 | + for cls, tk in REPRESENTATIVE.items(): | |
| 179 | + sub = merged[(merged["ticker"] == tk) & (merged["h"] == 1)] | |
| 180 | + wide = sub.pivot_table(index="date", columns="model", values="forecast") | |
| 181 | + actual = sub.drop_duplicates("date").set_index("date")["actual"] | |
| 182 | + for m in ["LightGBM-X", "Pooled-LGBM"]: | |
| 183 | + if m not in wide.columns: | |
| 184 | + continue | |
| 185 | + both = wide[[m, "HAR"]].dropna() | |
| 186 | + la = pd.Series(ev.qlike(actual[both.index].to_numpy(), both[m].to_numpy()), | |
| 187 | + index=both.index) | |
| 188 | + lb = pd.Series(ev.qlike(actual[both.index].to_numpy(), both["HAR"].to_numpy()), | |
| 189 | + index=both.index) | |
| 190 | + path = rb.giacomini_rossi_fluctuation(la, lb, h=1, mu=0.3) | |
| 191 | + path["cls"] = cls | |
| 192 | + path["ticker"] = tk | |
| 193 | + path["model"] = m | |
| 194 | + paths.append(path) | |
| 195 | + pd.concat(paths, ignore_index=True).to_csv(OUT / "gr_fluctuation.csv", index=False) | |
| 196 | + logger.info("GR fluctuation paths written") | |
| 197 | + | |
| 198 | + | |
| 199 | +def main() -> None: | |
| 200 | + """Run the requested robustness parts.""" | |
| 201 | + ap = argparse.ArgumentParser() | |
| 202 | + ap.add_argument("--part", default="all", | |
| 203 | + choices=["all", "subperiods", "freq", "window", "loo", "gr"]) | |
| 204 | + ap.add_argument("--workers", type=int, default=8) | |
| 205 | + args = ap.parse_args() | |
| 206 | + config.setup_logging() | |
| 207 | + if args.part in ("all", "subperiods"): | |
| 208 | + part_subperiods() | |
| 209 | + if args.part in ("all", "freq"): | |
| 210 | + part_frequency(args.workers) | |
| 211 | + if args.part in ("all", "window"): | |
| 212 | + part_window(args.workers) | |
| 213 | + if args.part in ("all", "loo"): | |
| 214 | + part_loo() | |
| 215 | + if args.part in ("all", "gr"): | |
| 216 | + part_gr() | |
| 217 | + | |
| 218 | + | |
| 219 | +if __name__ == "__main__": | |
| 220 | + main() | |
added
scripts/06_economic_value.py
+150 −0
@@ -0,0 +1,150 @@ | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +""" | |
| 3 | +================================================================ | |
| 4 | +Auteur : Simon-Pierre Boucher | |
| 5 | +Contact : contact@spboucher.ai | |
| 6 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 7 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 8 | +Fichier : 06_economic_value.py | |
| 9 | +Description : Étape 06 — Valeur économique : volatility timing | |
| 10 | + (Sharpe net, sensibilité aux coûts), gains | |
| 11 | + d'utilité mean-variance vs HAR, backtests VaR | |
| 12 | + (Kupiec, Christoffersen). | |
| 13 | +Usage : python scripts/06_economic_value.py | |
| 14 | +================================================================ | |
| 15 | +""" | |
| 16 | + | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import logging | |
| 20 | +import sys | |
| 21 | +from pathlib import Path | |
| 22 | + | |
| 23 | +import numpy as np | |
| 24 | +import pandas as pd | |
| 25 | + | |
| 26 | +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| 27 | + | |
| 28 | +from wp12 import backtest as bt, config # noqa: E402 | |
| 29 | + | |
| 30 | +logger = logging.getLogger("06_economic_value") | |
| 31 | + | |
| 32 | +OUT = config.path("reproduced") | |
| 33 | +CFG = config.load_config()["backtest"] | |
| 34 | + | |
| 35 | +# models carried into the economic-value exercise (h=1 forecasts) | |
| 36 | +EV_MODELS = [ | |
| 37 | + "HAR", "LogHAR", "HARQ", "SHAR", "HAR-CJ", | |
| 38 | + "GARCH", "GJR", "EGARCH", "RealGARCH", | |
| 39 | + "Ridge-X", "LASSO-X", "RF-X", "XGBoost-X", "LightGBM-X", | |
| 40 | + "LSTM", "Transformer", "Pooled-LGBM", "Comb-Mean", "Comb-InvMSE", | |
| 41 | +] | |
| 42 | + | |
| 43 | + | |
| 44 | +def _load() -> tuple[pd.DataFrame, dict[str, pd.Series], dict[str, str]]: | |
| 45 | + """Merged h=1 forecasts + daily returns per ticker + class map.""" | |
| 46 | + merged = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet") | |
| 47 | + merged["date"] = pd.to_datetime(merged["date"]) | |
| 48 | + merged = merged[(merged["h"] == 1) & merged["model"].isin(EV_MODELS)] | |
| 49 | + panel = pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 50 | + rets = {tk: d.sort_index()["ret_cc"] for tk, d in panel.groupby("ticker")} | |
| 51 | + cls = panel.groupby("ticker")["cls"].first().to_dict() | |
| 52 | + return merged, rets, cls | |
| 53 | + | |
| 54 | + | |
| 55 | +def timing_tables(merged: pd.DataFrame, rets: dict, cls: dict) -> None: | |
| 56 | + """Volatility-timing Sharpe (net) per model, plus cost sensitivity.""" | |
| 57 | + rows = [] | |
| 58 | + for (tk, m), sub in merged.groupby(["ticker", "model"], observed=True): | |
| 59 | + pv = sub.set_index("date")["forecast"] | |
| 60 | + ret = rets[tk] | |
| 61 | + for tc in CFG["tc_grid_bps"]: | |
| 62 | + led = bt.volatility_timing( | |
| 63 | + ret, pv, vol_target_ann=CFG["vol_target_ann"], | |
| 64 | + leverage_cap=CFG["leverage_cap"], tc_bps=float(tc)) | |
| 65 | + st = bt.perf_stats(led["net"]) | |
| 66 | + rows.append((tk, cls[tk], m, tc, st["ann_ret"], st["ann_vol"], | |
| 67 | + st["sharpe"], st["max_dd"], | |
| 68 | + float(led["weight"].diff().abs().mean()))) | |
| 69 | + tab = pd.DataFrame(rows, columns=[ | |
| 70 | + "ticker", "cls", "model", "tc_bps", "ann_ret", "ann_vol", | |
| 71 | + "sharpe", "max_dd", "turnover"]) | |
| 72 | + tab.to_csv(OUT / "timing_by_ticker.csv", index=False) | |
| 73 | + base = tab[tab["tc_bps"] == CFG["tc_bps"]] | |
| 74 | + summary = base.groupby(["cls", "model"], observed=True)["sharpe"].mean().reset_index() | |
| 75 | + overall = base.groupby("model", observed=True)[["sharpe", "ann_ret", "ann_vol", | |
| 76 | + "turnover"]].mean().reset_index() | |
| 77 | + overall["cls"] = "all" | |
| 78 | + summary.to_csv(OUT / "timing_by_class.csv", index=False) | |
| 79 | + overall.to_csv(OUT / "timing_overall.csv", index=False) | |
| 80 | + logger.info("timing tables written") | |
| 81 | + | |
| 82 | + | |
| 83 | +def utility_tables(merged: pd.DataFrame, rets: dict, cls: dict) -> None: | |
| 84 | + """Annualized utility gain (bps) of each model over HAR.""" | |
| 85 | + gamma = CFG["risk_aversion"] | |
| 86 | + rows = [] | |
| 87 | + for tk, sub in merged.groupby("ticker", observed=True): | |
| 88 | + ret = rets[tk] | |
| 89 | + nets = {} | |
| 90 | + for m, s in sub.groupby("model", observed=True): | |
| 91 | + pv = s.set_index("date")["forecast"] | |
| 92 | + led = bt.volatility_timing( | |
| 93 | + ret, pv, vol_target_ann=CFG["vol_target_ann"], | |
| 94 | + leverage_cap=CFG["leverage_cap"], tc_bps=float(CFG["tc_bps"])) | |
| 95 | + nets[m] = led["net"] | |
| 96 | + if "HAR" not in nets: | |
| 97 | + continue | |
| 98 | + for m, net in nets.items(): | |
| 99 | + if m == "HAR": | |
| 100 | + continue | |
| 101 | + gain = bt.utility_gain(net, nets["HAR"], risk_aversion=gamma) | |
| 102 | + rows.append((tk, cls[tk], m, gain * 1e4)) | |
| 103 | + tab = pd.DataFrame(rows, columns=["ticker", "cls", "model", "utility_gain_bps"]) | |
| 104 | + tab.to_csv(OUT / "utility_by_ticker.csv", index=False) | |
| 105 | + tab.groupby("model", observed=True)["utility_gain_bps"].agg( | |
| 106 | + ["mean", "median"]).reset_index().to_csv(OUT / "utility_summary.csv", index=False) | |
| 107 | + logger.info("utility tables written") | |
| 108 | + | |
| 109 | + | |
| 110 | +def var_tables(merged: pd.DataFrame, rets: dict, cls: dict) -> None: | |
| 111 | + """VaR 1% / 5% backtests; RV forecasts scaled to return variance with a | |
| 112 | + trailing 250-day factor (no look-ahead).""" | |
| 113 | + rows = [] | |
| 114 | + for (tk, m), sub in merged.groupby(["ticker", "model"], observed=True): | |
| 115 | + pv = sub.set_index("date")["forecast"] | |
| 116 | + ret = rets[tk] | |
| 117 | + # trailing conversion factor: return variance per unit of RV | |
| 118 | + rv_al = pv.reindex(ret.index) | |
| 119 | + c = ((ret**2).rolling(250).mean() / rv_al.rolling(250).mean()).clip(0.2, 5.0) | |
| 120 | + pv_ret = (rv_al * c).dropna() | |
| 121 | + for level in CFG["var_levels"]: | |
| 122 | + res = bt.var_backtest(ret, pv_ret, float(level)) | |
| 123 | + rows.append((tk, cls[tk], m, level, res["n"], res["viol_rate"], | |
| 124 | + res["kupiec_p"], res["christoffersen_p"])) | |
| 125 | + tab = pd.DataFrame(rows, columns=[ | |
| 126 | + "ticker", "cls", "model", "level", "n", "viol_rate", | |
| 127 | + "kupiec_p", "christoffersen_p"]) | |
| 128 | + tab.to_csv(OUT / "var_by_ticker.csv", index=False) | |
| 129 | + summary = tab.groupby(["model", "level"], observed=True).apply( | |
| 130 | + lambda d: pd.Series({ | |
| 131 | + "mean_viol_rate": d["viol_rate"].mean(), | |
| 132 | + "pct_pass_kupiec": float((d["kupiec_p"] > 0.05).mean()), | |
| 133 | + "pct_pass_cc": float((d["christoffersen_p"] > 0.05).mean()), | |
| 134 | + }), include_groups=False).reset_index() | |
| 135 | + summary.to_csv(OUT / "var_summary.csv", index=False) | |
| 136 | + logger.info("VaR tables written") | |
| 137 | + | |
| 138 | + | |
| 139 | +def main() -> None: | |
| 140 | + """Run all economic-value blocks.""" | |
| 141 | + config.setup_logging() | |
| 142 | + merged, rets, cls = _load() | |
| 143 | + timing_tables(merged, rets, cls) | |
| 144 | + utility_tables(merged, rets, cls) | |
| 145 | + var_tables(merged, rets, cls) | |
| 146 | + logger.info("economic value complete") | |
| 147 | + | |
| 148 | + | |
| 149 | +if __name__ == "__main__": | |
| 150 | + main() | |
added
scripts/07_make_figures.py
+263 −0
@@ -0,0 +1,263 @@ | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +""" | |
| 3 | +================================================================ | |
| 4 | +Auteur : Simon-Pierre Boucher | |
| 5 | +Contact : contact@spboucher.ai | |
| 6 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 7 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 8 | +Fichier : 07_make_figures.py | |
| 9 | +Description : Étape 07 — Figures publication-ready du papier | |
| 10 | + (séries RV, mesures, ACF, ratios QLIKE, MCS, | |
| 11 | + sous-périodes, SHAP, Giacomini-Rossi, timing). | |
| 12 | +Usage : python scripts/07_make_figures.py | |
| 13 | +================================================================ | |
| 14 | +""" | |
| 15 | + | |
| 16 | +from __future__ import annotations | |
| 17 | + | |
| 18 | +import logging | |
| 19 | +import sys | |
| 20 | +from pathlib import Path | |
| 21 | + | |
| 22 | +import matplotlib.pyplot as plt | |
| 23 | +import numpy as np | |
| 24 | +import pandas as pd | |
| 25 | + | |
| 26 | +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| 27 | + | |
| 28 | +from wp12 import backtest as bt, config # noqa: E402 | |
| 29 | +from wp12.plots import ( # noqa: E402 | |
| 30 | + BLUE, BLUES, GREY, INK, LIGHT, RED, TEXTWIDTH, | |
| 31 | + annualized_vol, apply_style, panel_label, | |
| 32 | +) | |
| 33 | + | |
| 34 | +logger = logging.getLogger("07_figures") | |
| 35 | + | |
| 36 | +FIG = config.path("figures") | |
| 37 | +OUT = config.path("reproduced") | |
| 38 | +REPRESENTATIVE = {"equity": "SPY", "fx": "EURUSD", "crypto": "BTC", "futures": "ES"} | |
| 39 | +CLS_LABEL = {"equity": "Equities/ETFs", "fx": "FX", "crypto": "Crypto", | |
| 40 | + "futures": "Futures"} | |
| 41 | + | |
| 42 | +# headline models shown in model-comparison figures, display order | |
| 43 | +SHOW_MODELS = [ | |
| 44 | + "GARCH", "GJR", "EGARCH", "RealGARCH", | |
| 45 | + "HAR-J", "HAR-CJ", "SHAR", "HARQ", "LogHAR", | |
| 46 | + "Ridge-H", "LASSO-H", "RF-H", "XGBoost-H", "LightGBM-H", | |
| 47 | + "Ridge-X", "LASSO-X", "RF-X", "XGBoost-X", "LightGBM-X", | |
| 48 | + "LSTM", "Transformer", "Pooled-LGBM", "Comb-Mean", "Comb-InvMSE", | |
| 49 | +] | |
| 50 | + | |
| 51 | + | |
| 52 | +def _panel() -> pd.DataFrame: | |
| 53 | + return pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 54 | + | |
| 55 | + | |
| 56 | +def fig_rv_series() -> None: | |
| 57 | + """Annualized RV time series, one panel per asset class representative.""" | |
| 58 | + panel = _panel() | |
| 59 | + fig, axes = plt.subplots(4, 1, figsize=(TEXTWIDTH, 6.6), sharex=True) | |
| 60 | + for ax, (cls, tk) in zip(axes, REPRESENTATIVE.items()): | |
| 61 | + df = panel[panel["ticker"] == tk].sort_index() | |
| 62 | + ax.plot(df.index, annualized_vol(df["rv5ss"]), lw=0.4, color=BLUE) | |
| 63 | + panel_label(ax, f"Panel {'ABCD'[list(REPRESENTATIVE).index(cls)]}. " | |
| 64 | + f"{tk} ({CLS_LABEL[cls]})") | |
| 65 | + ax.set_ylabel("Ann. vol. (%)") | |
| 66 | + ax.set_ylim(bottom=0) | |
| 67 | + axes[-1].set_xlabel("") | |
| 68 | + fig.tight_layout() | |
| 69 | + fig.savefig(FIG / "fig_rv_series.png") | |
| 70 | + plt.close(fig) | |
| 71 | + | |
| 72 | + | |
| 73 | +def fig_measures() -> None: | |
| 74 | + """Yearly mean annualized vol by RV measure — microstructure noise check.""" | |
| 75 | + panel = _panel() | |
| 76 | + fig, axes = plt.subplots(1, 2, figsize=(TEXTWIDTH, 2.6)) | |
| 77 | + for ax, tk, letter in zip(axes, ["SPY", "BTC"], "AB"): | |
| 78 | + df = panel[panel["ticker"] == tk] | |
| 79 | + yearly = df.groupby(df.index.year)[["rv1", "rv5ss", "rk"]].mean() | |
| 80 | + for col, color, ls, lab in [ | |
| 81 | + ("rv1", RED, "-", "RV 1-min"), | |
| 82 | + ("rv5ss", BLUE, "-", "RV 5-min (subsampled)"), | |
| 83 | + ("rk", INK, "--", "Realized kernel"), | |
| 84 | + ]: | |
| 85 | + ax.plot(yearly.index, annualized_vol(yearly[col]), color=color, | |
| 86 | + ls=ls, lw=1.0, label=lab) | |
| 87 | + panel_label(ax, f"Panel {letter}. {tk}") | |
| 88 | + ax.set_ylabel("Mean ann. vol. (%)") | |
| 89 | + if letter == "A": | |
| 90 | + ax.legend(loc="upper right", fontsize=7) | |
| 91 | + fig.tight_layout() | |
| 92 | + fig.savefig(FIG / "fig_measures.png") | |
| 93 | + plt.close(fig) | |
| 94 | + | |
| 95 | + | |
| 96 | +def fig_acf() -> None: | |
| 97 | + """Autocorrelation of log RV by class representative (long memory).""" | |
| 98 | + panel = _panel() | |
| 99 | + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.8)) | |
| 100 | + colors = {"equity": BLUE, "fx": INK, "crypto": RED, "futures": GREY} | |
| 101 | + for cls, tk in REPRESENTATIVE.items(): | |
| 102 | + s = np.log(panel[panel["ticker"] == tk]["rv5ss"].clip(lower=1e-12)) | |
| 103 | + acf = [s.autocorr(k) for k in range(1, 101)] | |
| 104 | + ax.plot(range(1, 101), acf, lw=1.0, color=colors[cls], | |
| 105 | + ls="--" if cls in ("fx", "futures") else "-", | |
| 106 | + label=f"{tk} ({CLS_LABEL[cls]})") | |
| 107 | + ax.axhline(0, color=GREY, lw=0.5) | |
| 108 | + ax.set_xlabel("Lag (days)") | |
| 109 | + ax.set_ylabel("ACF of log RV") | |
| 110 | + ax.legend(fontsize=7) | |
| 111 | + fig.tight_layout() | |
| 112 | + fig.savefig(FIG / "fig_acf.png") | |
| 113 | + plt.close(fig) | |
| 114 | + | |
| 115 | + | |
| 116 | +def fig_qlike_ratio() -> None: | |
| 117 | + """QLIKE ratio to HAR by model and horizon (dot plot).""" | |
| 118 | + tab = pd.read_csv(OUT / "losses_overall.csv") | |
| 119 | + models = [m for m in SHOW_MODELS if m in set(tab["model"])] | |
| 120 | + fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True) | |
| 121 | + for ax, h in zip(axes, (1, 5, 22)): | |
| 122 | + sub = tab[tab["h"] == h].set_index("model").reindex(models) | |
| 123 | + y = np.arange(len(models)) | |
| 124 | + better = sub["qlike_ratio"] < 1 | |
| 125 | + ax.scatter(sub["qlike_ratio"], y, s=14, | |
| 126 | + c=[BLUE if b else RED for b in better], zorder=3) | |
| 127 | + ax.axvline(1.0, color=GREY, lw=0.7, ls="--") | |
| 128 | + ax.set_title(f"h = {h}", fontsize=9) | |
| 129 | + ax.set_xlabel("QLIKE ratio vs HAR") | |
| 130 | + ax.set_yticks(y) | |
| 131 | + ax.set_yticklabels(models, fontsize=7) | |
| 132 | + ax.invert_yaxis() | |
| 133 | + fig.tight_layout() | |
| 134 | + fig.savefig(FIG / "fig_qlike_ratio.png") | |
| 135 | + plt.close(fig) | |
| 136 | + | |
| 137 | + | |
| 138 | +def fig_mcs() -> None: | |
| 139 | + """MCS inclusion frequency across assets, per horizon.""" | |
| 140 | + tab = pd.read_csv(OUT / "mcs_inclusion.csv") | |
| 141 | + models = ["HAR"] + [m for m in SHOW_MODELS if m in set(tab["model"])] | |
| 142 | + fig, axes = plt.subplots(1, 3, figsize=(TEXTWIDTH, 4.6), sharey=True) | |
| 143 | + for ax, h in zip(axes, (1, 5, 22)): | |
| 144 | + sub = tab[tab["h"] == h].set_index("model").reindex(models) | |
| 145 | + y = np.arange(len(models)) | |
| 146 | + ax.barh(y, sub["mcs_inclusion"], color=BLUE, height=0.65) | |
| 147 | + ax.set_yticks(y) | |
| 148 | + ax.set_yticklabels(models, fontsize=7) | |
| 149 | + ax.invert_yaxis() | |
| 150 | + ax.set_xlim(0, 1) | |
| 151 | + ax.set_title(f"h = {h}", fontsize=9) | |
| 152 | + ax.set_xlabel("Share of assets in 90% MCS") | |
| 153 | + fig.tight_layout() | |
| 154 | + fig.savefig(FIG / "fig_mcs.png") | |
| 155 | + plt.close(fig) | |
| 156 | + | |
| 157 | + | |
| 158 | +def fig_subperiods() -> None: | |
| 159 | + """QLIKE ratio vs HAR across sub-periods for selected models (h=1).""" | |
| 160 | + tab = pd.read_csv(OUT / "subperiod_losses.csv") | |
| 161 | + tab = tab[tab["h"] == 1] | |
| 162 | + bench = tab[tab["model"] == "HAR"].set_index("period")["qlike"] | |
| 163 | + show = ["LogHAR", "RealGARCH", "LightGBM-X", "Pooled-LGBM", "Comb-InvMSE", "LSTM"] | |
| 164 | + periods = ["pre2020", "covid", "inflation", "recent"] | |
| 165 | + plabel = {"pre2020": "2013–19", "covid": "2020", "inflation": "2021–22", | |
| 166 | + "recent": "2024–26"} | |
| 167 | + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.75, 2.9)) | |
| 168 | + colors = [BLUE, INK, RED, BLUES[5], GREY, BLUES[2]] | |
| 169 | + markers = ["o", "s", "^", "D", "v", "P"] | |
| 170 | + for m, c, mk in zip(show, colors, markers): | |
| 171 | + sub = tab[tab["model"] == m].set_index("period") | |
| 172 | + ratio = (sub["qlike"] / bench).reindex(periods) | |
| 173 | + ax.plot(range(len(periods)), ratio, marker=mk, ms=4, lw=1.0, | |
| 174 | + color=c, label=m) | |
| 175 | + ax.axhline(1.0, color=GREY, lw=0.7, ls="--") | |
| 176 | + ax.set_xticks(range(len(periods))) | |
| 177 | + ax.set_xticklabels([plabel[p] for p in periods]) | |
| 178 | + ax.set_ylabel("QLIKE ratio vs HAR (h=1)") | |
| 179 | + ax.legend(fontsize=7, ncol=2) | |
| 180 | + fig.tight_layout() | |
| 181 | + fig.savefig(FIG / "fig_subperiods.png") | |
| 182 | + plt.close(fig) | |
| 183 | + | |
| 184 | + | |
| 185 | +def fig_shap() -> None: | |
| 186 | + """Mean |SHAP| of the pooled LightGBM (h=1).""" | |
| 187 | + tab = pd.read_csv(OUT / "shap_h1.csv").head(14) | |
| 188 | + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.7, 3.2)) | |
| 189 | + y = np.arange(len(tab)) | |
| 190 | + ax.barh(y, tab["mean_abs_shap"], color=BLUE, height=0.65) | |
| 191 | + ax.set_yticks(y) | |
| 192 | + ax.set_yticklabels(tab["feature"], fontsize=7) | |
| 193 | + ax.invert_yaxis() | |
| 194 | + ax.set_xlabel(r"Mean $|$SHAP$|$ (log-RV units)") | |
| 195 | + fig.tight_layout() | |
| 196 | + fig.savefig(FIG / "fig_shap.png") | |
| 197 | + plt.close(fig) | |
| 198 | + | |
| 199 | + | |
| 200 | +def fig_gr() -> None: | |
| 201 | + """Giacomini-Rossi fluctuation paths, LightGBM-X vs HAR.""" | |
| 202 | + tab = pd.read_csv(OUT / "gr_fluctuation.csv", parse_dates=["date"]) | |
| 203 | + tab = tab[tab["model"] == "LightGBM-X"] | |
| 204 | + fig, axes = plt.subplots(2, 2, figsize=(TEXTWIDTH, 4.4)) | |
| 205 | + reps = {"equity": "NVDA", "fx": "GBPUSD", "crypto": "ETH", "futures": "GC"} | |
| 206 | + for ax, (cls, tk), letter in zip(axes.ravel(), reps.items(), "ABCD"): | |
| 207 | + sub = tab[tab["ticker"] == tk] | |
| 208 | + if sub.empty: | |
| 209 | + continue | |
| 210 | + ax.plot(sub["date"], sub["stat"], lw=0.7, color=BLUE) | |
| 211 | + crit = sub["crit"].iloc[0] | |
| 212 | + ax.axhline(crit, color=GREY, lw=0.6, ls="--") | |
| 213 | + ax.axhline(-crit, color=GREY, lw=0.6, ls="--") | |
| 214 | + ax.axhline(0, color=GREY, lw=0.4) | |
| 215 | + panel_label(ax, f"Panel {letter}. {tk} ({CLS_LABEL[cls]})") | |
| 216 | + ax.set_ylabel("Fluctuation stat.") | |
| 217 | + fig.tight_layout() | |
| 218 | + fig.savefig(FIG / "fig_gr.png") | |
| 219 | + plt.close(fig) | |
| 220 | + | |
| 221 | + | |
| 222 | +def fig_timing() -> None: | |
| 223 | + """Cumulative net log returns of volatility timing on SPY.""" | |
| 224 | + merged = pd.read_parquet(config.path("processed") / "merged_forecasts.parquet") | |
| 225 | + merged["date"] = pd.to_datetime(merged["date"]) | |
| 226 | + panel = _panel() | |
| 227 | + ret = panel[panel["ticker"] == "SPY"].sort_index()["ret_cc"] | |
| 228 | + cfg = config.load_config()["backtest"] | |
| 229 | + fig, ax = plt.subplots(figsize=(TEXTWIDTH * 0.85, 3.0)) | |
| 230 | + series = [("HAR", INK, "-"), ("GARCH", GREY, "-"), | |
| 231 | + ("LightGBM-X", BLUE, "-"), ("Comb-InvMSE", RED, "--")] | |
| 232 | + for m, color, ls in series: | |
| 233 | + sub = merged[(merged["ticker"] == "SPY") & (merged["h"] == 1) | |
| 234 | + & (merged["model"] == m)] | |
| 235 | + if sub.empty: | |
| 236 | + continue | |
| 237 | + pv = sub.set_index("date")["forecast"] | |
| 238 | + led = bt.volatility_timing(ret, pv, cfg["vol_target_ann"], | |
| 239 | + cfg["leverage_cap"], cfg["tc_bps"]) | |
| 240 | + ax.plot(led.index, led["net"].cumsum() * 100, lw=0.9, color=color, | |
| 241 | + ls=ls, label=m) | |
| 242 | + ax.set_ylabel("Cumulative net return (%)") | |
| 243 | + ax.legend(fontsize=7) | |
| 244 | + fig.tight_layout() | |
| 245 | + fig.savefig(FIG / "fig_timing.png") | |
| 246 | + plt.close(fig) | |
| 247 | + | |
| 248 | + | |
| 249 | +def main() -> None: | |
| 250 | + """Generate every figure of the paper.""" | |
| 251 | + config.setup_logging() | |
| 252 | + apply_style() | |
| 253 | + for fn in [fig_rv_series, fig_measures, fig_acf, fig_qlike_ratio, | |
| 254 | + fig_mcs, fig_subperiods, fig_shap, fig_gr, fig_timing]: | |
| 255 | + try: | |
| 256 | + fn() | |
| 257 | + logger.info("figure %s done", fn.__name__) | |
| 258 | + except Exception as exc: # noqa: BLE001 | |
| 259 | + logger.error("figure %s FAILED: %s", fn.__name__, exc) | |
| 260 | + | |
| 261 | + | |
| 262 | +if __name__ == "__main__": | |
| 263 | + main() | |
added
scripts/08_make_tables.py
+404 −0
@@ -0,0 +1,404 @@ | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +""" | |
| 3 | +================================================================ | |
| 4 | +Auteur : Simon-Pierre Boucher | |
| 5 | +Contact : contact@spboucher.ai | |
| 6 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 7 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 8 | +Fichier : 08_make_tables.py | |
| 9 | +Description : Étape 08 — Génération des tables LaTeX du papier | |
| 10 | + (booktabs) à partir des CSV de results/reproduced. | |
| 11 | + AUCUN chiffre n'est écrit à la main. | |
| 12 | +Usage : python scripts/08_make_tables.py | |
| 13 | +================================================================ | |
| 14 | +""" | |
| 15 | + | |
| 16 | +from __future__ import annotations | |
| 17 | + | |
| 18 | +import logging | |
| 19 | +import sys | |
| 20 | +from pathlib import Path | |
| 21 | + | |
| 22 | +import numpy as np | |
| 23 | +import pandas as pd | |
| 24 | + | |
| 25 | +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| 26 | + | |
| 27 | +from wp12 import config # noqa: E402 | |
| 28 | + | |
| 29 | +logger = logging.getLogger("08_tables") | |
| 30 | + | |
| 31 | +OUT = config.path("reproduced") | |
| 32 | +TAB = config.path("tables") | |
| 33 | + | |
| 34 | +GROUPS: list[tuple[str, list[str]]] = [ | |
| 35 | + ("A. GARCH family", ["GARCH", "GJR", "EGARCH", "RealGARCH"]), | |
| 36 | + ("B. HAR family", ["HAR", "HAR-J", "HAR-CJ", "SHAR", "HARQ", "LogHAR"]), | |
| 37 | + ("C. Machine learning --- HAR information set", | |
| 38 | + ["Ridge-H", "LASSO-H", "ElasticNet-H", "RF-H", "XGBoost-H", "LightGBM-H"]), | |
| 39 | + ("D. Machine learning --- extended information set", | |
| 40 | + ["Ridge-X", "LASSO-X", "ElasticNet-X", "RF-X", "XGBoost-X", "LightGBM-X"]), | |
| 41 | + ("E. Deep learning and pooled", ["LSTM", "Transformer", "Pooled-LGBM"]), | |
| 42 | + ("F. Forecast combinations", ["Comb-Mean", "Comb-InvMSE"]), | |
| 43 | +] | |
| 44 | + | |
| 45 | +CLS_LABEL = {"equity": "Equities/ETFs", "fx": "FX", "crypto": "Crypto", | |
| 46 | + "futures": "Futures"} | |
| 47 | + | |
| 48 | + | |
| 49 | +def _fmt(x: float, nd: int = 3, bold: bool = False) -> str: | |
| 50 | + if not np.isfinite(x): | |
| 51 | + return "---" | |
| 52 | + s = f"{x:.{nd}f}" | |
| 53 | + return f"\\textbf{{{s}}}" if bold else s | |
| 54 | + | |
| 55 | + | |
| 56 | +def _write(name: str, lines: list[str]) -> None: | |
| 57 | + f = TAB / f"{name}.tex" | |
| 58 | + f.write_text("\n".join(lines) + "\n", encoding="utf-8") | |
| 59 | + logger.info("wrote %s", f.name) | |
| 60 | + | |
| 61 | + | |
| 62 | +def table_summary_stats() -> None: | |
| 63 | + """Descriptive statistics of daily annualized RV by instrument.""" | |
| 64 | + panel = pd.read_parquet(config.path("processed") / "panel.parquet") | |
| 65 | + lines = [ | |
| 66 | + "\\begin{tabular}{lrrrrrrr}", "\\toprule", | |
| 67 | + "Ticker & Days & \\makecell{Mean vol.\\\\(\\%)} & \\makecell{Median vol.\\\\(\\%)}" | |
| 68 | + " & \\makecell{P95 vol.\\\\(\\%)} & \\makecell{Skew.\\\\$\\ln$RV}" | |
| 69 | + " & \\makecell{AC(1)\\\\$\\ln$RV} & \\makecell{Jump\\\\days (\\%)} \\\\", | |
| 70 | + "\\midrule", | |
| 71 | + ] | |
| 72 | + for cls in ["equity", "fx", "crypto", "futures"]: | |
| 73 | + sub = panel[panel["cls"] == cls] | |
| 74 | + if sub.empty: | |
| 75 | + continue | |
| 76 | + lines.append(f"\\multicolumn{{8}}{{l}}{{\\itshape {CLS_LABEL[cls]}}}\\\\") | |
| 77 | + for tk, df in sub.groupby("ticker"): | |
| 78 | + vol = np.sqrt(df["rv5ss"] * 252) * 100 | |
| 79 | + lrv = np.log(df["rv5ss"].clip(lower=1e-12)) | |
| 80 | + lines.append( | |
| 81 | + f"\\quad {tk} & {len(df):,} & {vol.mean():.1f} & {vol.median():.1f}" | |
| 82 | + f" & {vol.quantile(0.95):.1f} & {lrv.skew():.2f}" | |
| 83 | + f" & {lrv.autocorr(1):.2f} & {100 * (df['jump'] > 0).mean():.1f} \\\\" | |
| 84 | + ) | |
| 85 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 86 | + _write("summary_stats", lines) | |
| 87 | + | |
| 88 | + | |
| 89 | +def _grouped_metric_table( | |
| 90 | + name: str, wide: pd.DataFrame, cols: list[tuple[str, str]], | |
| 91 | + nd: int = 3, note_bench: str | None = "HAR", bold_min_cols: bool = True, | |
| 92 | +) -> None: | |
| 93 | + """Generic grouped model table. ``wide`` is indexed by model with the | |
| 94 | + metric columns; ``cols`` maps (column, header).""" | |
| 95 | + align = "l" + "c" * len(cols) | |
| 96 | + lines = [f"\\begin{{tabular}}{{{align}}}", "\\toprule", | |
| 97 | + "Model & " + " & ".join(h for _, h in cols) + " \\\\", "\\midrule"] | |
| 98 | + mins = {c: wide[c].min() for c, _ in cols} if bold_min_cols else {} | |
| 99 | + for gname, models in GROUPS: | |
| 100 | + avail = [m for m in models if m in wide.index] | |
| 101 | + if not avail: | |
| 102 | + continue | |
| 103 | + lines.append(f"\\multicolumn{{{len(cols) + 1}}}{{l}}{{\\itshape {gname}}}\\\\") | |
| 104 | + for m in avail: | |
| 105 | + cells = [] | |
| 106 | + for c, _ in cols: | |
| 107 | + v = wide.loc[m, c] | |
| 108 | + bold = bold_min_cols and np.isfinite(v) and v == mins[c] | |
| 109 | + cells.append(_fmt(v, nd, bold)) | |
| 110 | + lines.append(f"\\quad {m} & " + " & ".join(cells) + " \\\\") | |
| 111 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 112 | + _write(name, lines) | |
| 113 | + | |
| 114 | + | |
| 115 | +def table_losses_main() -> None: | |
| 116 | + """Headline QLIKE table: level for HAR, ratio for everything else.""" | |
| 117 | + tab = pd.read_csv(OUT / "losses_overall.csv") | |
| 118 | + wide = tab.pivot_table(index="model", columns="h", values="qlike_ratio") | |
| 119 | + wide.columns = [f"r{h}" for h in wide.columns] | |
| 120 | + ql = tab.pivot_table(index="model", columns="h", values="qlike") | |
| 121 | + ql.columns = [f"q{h}" for h in ql.columns] | |
| 122 | + wide = wide.join(ql) | |
| 123 | + cols = [("q1", "QLIKE $h{=}1$"), ("r1", "Ratio $h{=}1$"), | |
| 124 | + ("q5", "QLIKE $h{=}5$"), ("r5", "Ratio $h{=}5$"), | |
| 125 | + ("q22", "QLIKE $h{=}22$"), ("r22", "Ratio $h{=}22$")] | |
| 126 | + _grouped_metric_table("losses_main", wide, cols) | |
| 127 | + | |
| 128 | + | |
| 129 | +def table_losses_class() -> None: | |
| 130 | + """QLIKE ratio vs HAR by asset class (h=1 and h=22).""" | |
| 131 | + tab = pd.read_csv(OUT / "losses_by_class.csv") | |
| 132 | + parts = {} | |
| 133 | + for h in (1, 22): | |
| 134 | + sub = tab[tab["h"] == h].pivot_table( | |
| 135 | + index="model", columns="cls", values="qlike_ratio") | |
| 136 | + for cls in ["equity", "fx", "crypto", "futures"]: | |
| 137 | + if cls in sub.columns: | |
| 138 | + parts[f"{cls}_{h}"] = sub[cls] | |
| 139 | + wide = pd.DataFrame(parts) | |
| 140 | + cols = ([(f"{c}_1", CLS_LABEL[c].split("/")[0]) for c in | |
| 141 | + ["equity", "fx", "crypto", "futures"]] | |
| 142 | + + [(f"{c}_22", CLS_LABEL[c].split("/")[0]) for c in | |
| 143 | + ["equity", "fx", "crypto", "futures"]]) | |
| 144 | + align = "l" + "cccc" + "cccc" | |
| 145 | + lines = [f"\\begin{{tabular}}{{{align}}}", "\\toprule", | |
| 146 | + " & \\multicolumn{4}{c}{$h = 1$} & \\multicolumn{4}{c}{$h = 22$} \\\\", | |
| 147 | + "\\cmidrule(lr){2-5}\\cmidrule(lr){6-9}", | |
| 148 | + "Model & " + " & ".join(h for _, h in cols) + " \\\\", "\\midrule"] | |
| 149 | + mins = {c: wide[c].min() for c, _ in cols} | |
| 150 | + for gname, models in GROUPS: | |
| 151 | + avail = [m for m in models if m in wide.index and m != "HAR"] | |
| 152 | + if not avail: | |
| 153 | + continue | |
| 154 | + lines.append(f"\\multicolumn{{9}}{{l}}{{\\itshape {gname}}}\\\\") | |
| 155 | + for m in avail: | |
| 156 | + cells = [_fmt(wide.loc[m, c], 3, np.isfinite(wide.loc[m, c]) | |
| 157 | + and wide.loc[m, c] == mins[c]) for c, _ in cols] | |
| 158 | + lines.append(f"\\quad {m} & " + " & ".join(cells) + " \\\\") | |
| 159 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 160 | + _write("losses_class", lines) | |
| 161 | + | |
| 162 | + | |
| 163 | +def table_dm_mcs() -> None: | |
| 164 | + """DM outcomes vs HAR and MCS inclusion, side by side.""" | |
| 165 | + dm = pd.read_csv(OUT / "dm_summary.csv") | |
| 166 | + mcs = pd.read_csv(OUT / "mcs_inclusion.csv") | |
| 167 | + parts = {} | |
| 168 | + for h in (1, 5, 22): | |
| 169 | + d = dm[dm["h"] == h].set_index("model") | |
| 170 | + parts[f"sb{h}"] = d["pct_sig_better"] * 100 | |
| 171 | + parts[f"sw{h}"] = d["pct_sig_worse"] * 100 | |
| 172 | + parts[f"mcs{h}"] = mcs[mcs["h"] == h].set_index("model")["mcs_inclusion"] * 100 | |
| 173 | + wide = pd.DataFrame(parts) | |
| 174 | + lines = ["\\begin{tabular}{lrrr rrr rrr}", "\\toprule", | |
| 175 | + " & \\multicolumn{3}{c}{Sig.\\ better than HAR (\\%)}" | |
| 176 | + " & \\multicolumn{3}{c}{Sig.\\ worse than HAR (\\%)}" | |
| 177 | + " & \\multicolumn{3}{c}{In 90\\% MCS (\\%)} \\\\", | |
| 178 | + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-7}\\cmidrule(lr){8-10}", | |
| 179 | + "Model & $h{=}1$ & $h{=}5$ & $h{=}22$ & $h{=}1$ & $h{=}5$ & $h{=}22$" | |
| 180 | + " & $h{=}1$ & $h{=}5$ & $h{=}22$ \\\\", "\\midrule"] | |
| 181 | + for gname, models in GROUPS: | |
| 182 | + avail = [m for m in models if m in wide.index] | |
| 183 | + if gname.startswith("B."): | |
| 184 | + avail = ["HAR"] + [m for m in avail if m != "HAR"] if "HAR" in wide.index else avail | |
| 185 | + if not avail: | |
| 186 | + continue | |
| 187 | + lines.append(f"\\multicolumn{{10}}{{l}}{{\\itshape {gname}}}\\\\") | |
| 188 | + for m in avail: | |
| 189 | + def _c(key: str) -> str: | |
| 190 | + v = wide.loc[m, key] if key in wide.columns else np.nan | |
| 191 | + return f"{v:.0f}" if np.isfinite(v) else "---" | |
| 192 | + row = [_c(f"sb{h}") for h in (1, 5, 22)] + \ | |
| 193 | + [_c(f"sw{h}") for h in (1, 5, 22)] + \ | |
| 194 | + [_c(f"mcs{h}") for h in (1, 5, 22)] | |
| 195 | + lines.append(f"\\quad {m} & " + " & ".join(row) + " \\\\") | |
| 196 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 197 | + _write("dm_mcs", lines) | |
| 198 | + | |
| 199 | + | |
| 200 | +def table_mz_encompassing() -> None: | |
| 201 | + """Mincer-Zarnowitz medians and encompassing outcomes (h=1).""" | |
| 202 | + mz = pd.read_csv(OUT / "mz_summary.csv") | |
| 203 | + enc = pd.read_csv(OUT / "encompassing_summary.csv") | |
| 204 | + m1 = mz[mz["h"] == 1].set_index("model") | |
| 205 | + e1 = enc[enc["h"] == 1].set_index("model") | |
| 206 | + wide = pd.DataFrame({ | |
| 207 | + "beta": m1["med_beta"], "r2": m1["med_r2"], | |
| 208 | + "rej": m1["pct_reject"] * 100, | |
| 209 | + "b2": e1["med_b2"], "adds": e1["pct_adds_info"] * 100, | |
| 210 | + }) | |
| 211 | + lines = ["\\begin{tabular}{lccccc}", "\\toprule", | |
| 212 | + " & \\multicolumn{3}{c}{Mincer--Zarnowitz}" | |
| 213 | + " & \\multicolumn{2}{c}{Encompassing (vs HAR)} \\\\", | |
| 214 | + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-6}", | |
| 215 | + "Model & Med.\\ $\\hat\\beta$ & Med.\\ $R^2$ & Rej.\\ (\\%)" | |
| 216 | + " & Med.\\ $\\hat b_2$ & Adds info (\\%) \\\\", "\\midrule"] | |
| 217 | + for gname, models in GROUPS: | |
| 218 | + avail = [m for m in models if m in wide.index] | |
| 219 | + if not avail: | |
| 220 | + continue | |
| 221 | + lines.append(f"\\multicolumn{{6}}{{l}}{{\\itshape {gname}}}\\\\") | |
| 222 | + for m in avail: | |
| 223 | + r = wide.loc[m] | |
| 224 | + b2 = f"{r['b2']:.2f}" if np.isfinite(r["b2"]) else "---" | |
| 225 | + adds = f"{r['adds']:.0f}" if np.isfinite(r["adds"]) else "---" | |
| 226 | + lines.append( | |
| 227 | + f"\\quad {m} & {r['beta']:.2f} & {r['r2']:.2f} & {r['rej']:.0f}" | |
| 228 | + f" & {b2} & {adds} \\\\") | |
| 229 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 230 | + _write("mz_encompassing", lines) | |
| 231 | + | |
| 232 | + | |
| 233 | +def table_subperiods() -> None: | |
| 234 | + """QLIKE ratio vs HAR per sub-period (h=1).""" | |
| 235 | + tab = pd.read_csv(OUT / "subperiod_losses.csv") | |
| 236 | + tab = tab[tab["h"] == 1] | |
| 237 | + bench = tab[tab["model"] == "HAR"].set_index("period")["qlike"] | |
| 238 | + wide = tab.pivot_table(index="model", columns="period", values="qlike") | |
| 239 | + wide = wide.div(bench, axis=1) | |
| 240 | + periods = ["pre2020", "covid", "inflation", "recent"] | |
| 241 | + heads = ["2013--19", "2020", "2021--22", "2024--26"] | |
| 242 | + cols = [(p, h) for p, h in zip(periods, heads) if p in wide.columns] | |
| 243 | + _grouped_metric_table("subperiods", wide, cols) | |
| 244 | + | |
| 245 | + | |
| 246 | +def table_sensitivity() -> None: | |
| 247 | + """Frequency and window sensitivity (QLIKE, h=1).""" | |
| 248 | + lines = ["\\begin{tabular}{lccc c ccc}", "\\toprule", | |
| 249 | + " & \\multicolumn{3}{c}{RV measure} & &" | |
| 250 | + " \\multicolumn{3}{c}{Estimation window (days)} \\\\", | |
| 251 | + "\\cmidrule(lr){2-4}\\cmidrule(lr){6-8}", | |
| 252 | + "Model & 1-min & 5-min ss & Kernel & & 500 & 1000 & 2000 \\\\", | |
| 253 | + "\\midrule"] | |
| 254 | + freq = pd.read_csv(OUT / "frequency_sensitivity.csv") | |
| 255 | + freq = freq[freq["h"] == 1].pivot_table(index="model", columns="frequency", | |
| 256 | + values="qlike") | |
| 257 | + win = pd.read_csv(OUT / "window_sensitivity.csv") | |
| 258 | + win = win[win["h"] == 1].pivot_table(index="model", columns="window", | |
| 259 | + values="qlike") | |
| 260 | + win.columns = [str(c) for c in win.columns] | |
| 261 | + for m in ["HAR", "LogHAR", "LightGBM"]: | |
| 262 | + f = freq.loc[m] if m in freq.index else pd.Series(dtype=float) | |
| 263 | + w = win.loc[m] if m in win.index else pd.Series(dtype=float) | |
| 264 | + cells = [ | |
| 265 | + _fmt(f.get("rv1min", np.nan)), _fmt(f.get("rv5min_ss", np.nan)), | |
| 266 | + _fmt(f.get("rkernel", np.nan)), "", | |
| 267 | + _fmt(w.get("500", np.nan)), _fmt(w.get("1000", np.nan)), | |
| 268 | + _fmt(w.get("2000", np.nan)), | |
| 269 | + ] | |
| 270 | + lines.append(f"{m} & " + " & ".join(cells) + " \\\\") | |
| 271 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 272 | + _write("sensitivity", lines) | |
| 273 | + | |
| 274 | + | |
| 275 | +def table_transferability() -> None: | |
| 276 | + """Cross-asset transferability: QLIKE h=1 for the class representatives.""" | |
| 277 | + tab = pd.read_csv(OUT / "transferability.csv") | |
| 278 | + tab = tab[tab["h"] == 1] | |
| 279 | + wide = tab.pivot_table(index="model", columns="ticker", values="qlike") | |
| 280 | + order = ["HAR", "LightGBM-X", "Pooled-LGBM", "Pooled-LGBM-LOO"] | |
| 281 | + label = {"HAR": "HAR (own asset)", "LightGBM-X": "LightGBM (own asset)", | |
| 282 | + "Pooled-LGBM": "Pooled LightGBM (asset in pool)", | |
| 283 | + "Pooled-LGBM-LOO": "Pooled LightGBM (asset excluded)"} | |
| 284 | + tickers = [t for t in ["NVDA", "GBPUSD", "ETH", "GC"] if t in wide.columns] | |
| 285 | + lines = ["\\begin{tabular}{l" + "c" * len(tickers) + "}", "\\toprule", | |
| 286 | + "Model & " + " & ".join(tickers) + " \\\\", "\\midrule"] | |
| 287 | + for m in order: | |
| 288 | + if m not in wide.index: | |
| 289 | + continue | |
| 290 | + cells = [_fmt(wide.loc[m, t]) for t in tickers] | |
| 291 | + lines.append(f"{label[m]} & " + " & ".join(cells) + " \\\\") | |
| 292 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 293 | + _write("transferability", lines) | |
| 294 | + | |
| 295 | + | |
| 296 | +def table_timing() -> None: | |
| 297 | + """Volatility timing: net Sharpe, cost sensitivity, utility gains.""" | |
| 298 | + overall = pd.read_csv(OUT / "timing_overall.csv").set_index("model") | |
| 299 | + util = pd.read_csv(OUT / "utility_summary.csv").set_index("model") | |
| 300 | + byt = pd.read_csv(OUT / "timing_by_ticker.csv") | |
| 301 | + sens = byt.groupby(["model", "tc_bps"], observed=True)["sharpe"].mean().unstack() | |
| 302 | + wide = pd.DataFrame({ | |
| 303 | + "sharpe": overall["sharpe"], "ann_ret": overall["ann_ret"] * 100, | |
| 304 | + "turn": overall["turnover"], | |
| 305 | + "s0": sens.get(0), "s10": sens.get(10), "s20": sens.get(20), | |
| 306 | + "ug": util["mean"], | |
| 307 | + }) | |
| 308 | + lines = ["\\begin{tabular}{lccc ccc c}", "\\toprule", | |
| 309 | + " & \\multicolumn{3}{c}{Net of 5 bps} &" | |
| 310 | + " \\multicolumn{3}{c}{Sharpe at cost (bps)} & \\\\", | |
| 311 | + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-7}", | |
| 312 | + "Model & Sharpe & \\makecell{Ann.\\ ret.\\\\(\\%)} & Turnover" | |
| 313 | + " & 0 & 10 & 20 & \\makecell{Utility gain\\\\vs HAR (bps)} \\\\", | |
| 314 | + "\\midrule"] | |
| 315 | + smax = wide["sharpe"].max() | |
| 316 | + for gname, models in GROUPS: | |
| 317 | + avail = [m for m in models if m in wide.index] | |
| 318 | + if not avail: | |
| 319 | + continue | |
| 320 | + lines.append(f"\\multicolumn{{8}}{{l}}{{\\itshape {gname}}}\\\\") | |
| 321 | + for m in avail: | |
| 322 | + r = wide.loc[m] | |
| 323 | + ug = f"{r['ug']:.0f}" if np.isfinite(r["ug"]) else "---" | |
| 324 | + sh = _fmt(r["sharpe"], 2, r["sharpe"] == smax) | |
| 325 | + lines.append( | |
| 326 | + f"\\quad {m} & {sh} & {r['ann_ret']:.1f} & {r['turn']:.3f}" | |
| 327 | + f" & {_fmt(r['s0'], 2)} & {_fmt(r['s10'], 2)} & {_fmt(r['s20'], 2)}" | |
| 328 | + f" & {ug} \\\\") | |
| 329 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 330 | + _write("timing", lines) | |
| 331 | + | |
| 332 | + | |
| 333 | +def table_var() -> None: | |
| 334 | + """VaR backtest outcomes at 1% and 5%.""" | |
| 335 | + tab = pd.read_csv(OUT / "var_summary.csv") | |
| 336 | + parts = {} | |
| 337 | + for lv in (0.01, 0.05): | |
| 338 | + s = tab[tab["level"] == lv].set_index("model") | |
| 339 | + parts[f"vr{lv}"] = s["mean_viol_rate"] * 100 | |
| 340 | + parts[f"ku{lv}"] = s["pct_pass_kupiec"] * 100 | |
| 341 | + parts[f"cc{lv}"] = s["pct_pass_cc"] * 100 | |
| 342 | + wide = pd.DataFrame(parts) | |
| 343 | + lines = ["\\begin{tabular}{lccc ccc}", "\\toprule", | |
| 344 | + " & \\multicolumn{3}{c}{VaR 1\\%} & \\multicolumn{3}{c}{VaR 5\\%} \\\\", | |
| 345 | + "\\cmidrule(lr){2-4}\\cmidrule(lr){5-7}", | |
| 346 | + "Model & \\makecell{Viol.\\\\(\\%)} & \\makecell{Kupiec\\\\pass (\\%)}" | |
| 347 | + " & \\makecell{CC\\\\pass (\\%)}" | |
| 348 | + " & \\makecell{Viol.\\\\(\\%)} & \\makecell{Kupiec\\\\pass (\\%)}" | |
| 349 | + " & \\makecell{CC\\\\pass (\\%)} \\\\", "\\midrule"] | |
| 350 | + for gname, models in GROUPS: | |
| 351 | + avail = [m for m in models if m in wide.index] | |
| 352 | + if not avail: | |
| 353 | + continue | |
| 354 | + lines.append(f"\\multicolumn{{7}}{{l}}{{\\itshape {gname}}}\\\\") | |
| 355 | + for m in avail: | |
| 356 | + r = wide.loc[m] | |
| 357 | + lines.append( | |
| 358 | + f"\\quad {m} & {r['vr0.01']:.2f} & {r['ku0.01']:.0f} & {r['cc0.01']:.0f}" | |
| 359 | + f" & {r['vr0.05']:.2f} & {r['ku0.05']:.0f} & {r['cc0.05']:.0f} \\\\") | |
| 360 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 361 | + _write("var", lines) | |
| 362 | + | |
| 363 | + | |
| 364 | +def table_by_ticker_appendix() -> None: | |
| 365 | + """Appendix: QLIKE h=1 per ticker for headline models.""" | |
| 366 | + tab = pd.read_csv(OUT / "losses_by_ticker.csv") | |
| 367 | + tab = tab[tab["h"] == 1] | |
| 368 | + models = ["HAR", "LogHAR", "RealGARCH", "LightGBM-X", "Pooled-LGBM", | |
| 369 | + "LSTM", "Comb-InvMSE"] | |
| 370 | + wide = tab.pivot_table(index=["cls", "ticker"], columns="model", | |
| 371 | + values="qlike") | |
| 372 | + avail = [m for m in models if m in wide.columns] | |
| 373 | + lines = ["\\begin{tabular}{l" + "c" * len(avail) + "}", "\\toprule", | |
| 374 | + "Ticker & " + " & ".join(avail) + " \\\\", "\\midrule"] | |
| 375 | + for cls in ["equity", "fx", "crypto", "futures"]: | |
| 376 | + if cls not in wide.index.get_level_values(0): | |
| 377 | + continue | |
| 378 | + lines.append(f"\\multicolumn{{{len(avail) + 1}}}{{l}}" | |
| 379 | + f"{{\\itshape {CLS_LABEL[cls]}}}\\\\") | |
| 380 | + sub = wide.loc[cls] | |
| 381 | + for tk, r in sub.iterrows(): | |
| 382 | + best = r[avail].min() | |
| 383 | + cells = [_fmt(r[m], 3, np.isfinite(r[m]) and r[m] == best) | |
| 384 | + for m in avail] | |
| 385 | + lines.append(f"\\quad {tk} & " + " & ".join(cells) + " \\\\") | |
| 386 | + lines += ["\\bottomrule", "\\end{tabular}"] | |
| 387 | + _write("by_ticker_appendix", lines) | |
| 388 | + | |
| 389 | + | |
| 390 | +def main() -> None: | |
| 391 | + """Generate every LaTeX table from the reproduced CSVs.""" | |
| 392 | + config.setup_logging() | |
| 393 | + for fn in [table_summary_stats, table_losses_main, table_losses_class, | |
| 394 | + table_dm_mcs, table_mz_encompassing, table_subperiods, | |
| 395 | + table_sensitivity, table_transferability, table_timing, | |
| 396 | + table_var, table_by_ticker_appendix]: | |
| 397 | + try: | |
| 398 | + fn() | |
| 399 | + except Exception as exc: # noqa: BLE001 | |
| 400 | + logger.error("table %s FAILED: %s", fn.__name__, exc) | |
| 401 | + | |
| 402 | + | |
| 403 | +if __name__ == "__main__": | |
| 404 | + main() | |
added
scripts/09_build_paper.py
+106 −0
@@ -0,0 +1,106 @@ | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +""" | |
| 3 | +================================================================ | |
| 4 | +Auteur : Simon-Pierre Boucher | |
| 5 | +Contact : contact@spboucher.ai | |
| 6 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 7 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 8 | +Fichier : 09_build_paper.py | |
| 9 | +Description : Étape 09 — Vérification des headers de tous les | |
| 10 | + fichiers de code, présence des tables/figures | |
| 11 | + référencées, et compilation du PDF (latexmk). | |
| 12 | +Usage : python scripts/09_build_paper.py [--no-compile] | |
| 13 | +================================================================ | |
| 14 | +""" | |
| 15 | + | |
| 16 | +from __future__ import annotations | |
| 17 | + | |
| 18 | +import argparse | |
| 19 | +import logging | |
| 20 | +import re | |
| 21 | +import subprocess | |
| 22 | +import sys | |
| 23 | +from pathlib import Path | |
| 24 | + | |
| 25 | +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| 26 | + | |
| 27 | +from wp12 import config # noqa: E402 | |
| 28 | + | |
| 29 | +logger = logging.getLogger("09_build_paper") | |
| 30 | + | |
| 31 | +ROOT = config.ROOT | |
| 32 | +HEADER_MARK = "Auteur : Simon-Pierre Boucher" | |
| 33 | + | |
| 34 | + | |
| 35 | +def check_headers() -> bool: | |
| 36 | + """Every .py / config / paper source file must carry the project header.""" | |
| 37 | + ok = True | |
| 38 | + files = ( | |
| 39 | + list((ROOT / "src").rglob("*.py")) | |
| 40 | + + list((ROOT / "scripts").glob("*.py")) | |
| 41 | + + list((ROOT / "tests").glob("*.py")) | |
| 42 | + + [ROOT / "config.yaml", ROOT / "requirements.txt"] | |
| 43 | + + list((ROOT / "paper").glob("*.tex")) | |
| 44 | + + list((ROOT / "paper" / "sections").glob("*.tex")) | |
| 45 | + ) | |
| 46 | + for f in files: | |
| 47 | + if "__pycache__" in str(f): | |
| 48 | + continue | |
| 49 | + head = f.read_text(encoding="utf-8", errors="ignore")[:600] | |
| 50 | + if HEADER_MARK not in head: | |
| 51 | + logger.error("MISSING HEADER: %s", f.relative_to(ROOT)) | |
| 52 | + ok = False | |
| 53 | + logger.info("header check: %d files, %s", len(files), "OK" if ok else "FAILURES") | |
| 54 | + return ok | |
| 55 | + | |
| 56 | + | |
| 57 | +def check_inputs() -> bool: | |
| 58 | + """All tables/figures referenced by the paper must exist on disk.""" | |
| 59 | + ok = True | |
| 60 | + tex = "" | |
| 61 | + for f in [ROOT / "paper" / "main.tex", *(ROOT / "paper" / "sections").glob("*.tex")]: | |
| 62 | + tex += f.read_text(encoding="utf-8") | |
| 63 | + for m in re.finditer(r"\\input\{(\.\./results/tables/[^}]+)\}", tex): | |
| 64 | + p = (ROOT / "paper" / m.group(1)).resolve() | |
| 65 | + p = p if p.suffix else p.with_suffix(".tex") | |
| 66 | + if not p.exists(): | |
| 67 | + logger.error("missing table: %s", m.group(1)) | |
| 68 | + ok = False | |
| 69 | + for m in re.finditer(r"\\includegraphics\[[^\]]*\]\{([^}]+)\}", tex): | |
| 70 | + name = m.group(1) | |
| 71 | + if name.startswith("uq_logo"): | |
| 72 | + continue | |
| 73 | + if not (ROOT / "figures" / name).exists(): | |
| 74 | + logger.error("missing figure: %s", name) | |
| 75 | + ok = False | |
| 76 | + logger.info("inputs check: %s", "OK" if ok else "FAILURES") | |
| 77 | + return ok | |
| 78 | + | |
| 79 | + | |
| 80 | +def compile_pdf() -> bool: | |
| 81 | + """Compile the paper with latexmk (pdflatex + bibtex).""" | |
| 82 | + res = subprocess.run( | |
| 83 | + ["latexmk", "-pdf", "-interaction=nonstopmode", "main.tex"], | |
| 84 | + cwd=ROOT / "paper", capture_output=True, text=True, timeout=600, | |
| 85 | + ) | |
| 86 | + if res.returncode != 0: | |
| 87 | + logger.error("latexmk failed:\n%s", res.stdout[-3000:]) | |
| 88 | + return False | |
| 89 | + logger.info("PDF compiled: paper/main.pdf") | |
| 90 | + return True | |
| 91 | + | |
| 92 | + | |
| 93 | +def main() -> None: | |
| 94 | + """Run all pre-flight checks, then compile.""" | |
| 95 | + ap = argparse.ArgumentParser() | |
| 96 | + ap.add_argument("--no-compile", action="store_true") | |
| 97 | + args = ap.parse_args() | |
| 98 | + config.setup_logging() | |
| 99 | + ok = check_headers() & check_inputs() | |
| 100 | + if not args.no_compile: | |
| 101 | + ok &= compile_pdf() | |
| 102 | + sys.exit(0 if ok else 1) | |
| 103 | + | |
| 104 | + | |
| 105 | +if __name__ == "__main__": | |
| 106 | + main() | |
added
src/wp12/plots.py
+77 −0
@@ -0,0 +1,77 @@ | ||
| 1 | +""" | |
| 2 | +================================================================ | |
| 3 | +Auteur : Simon-Pierre Boucher | |
| 4 | +Contact : contact@spboucher.ai | |
| 5 | +Projet : Prévision de volatilité réalisée multi-actifs | |
| 6 | + (HAR-RV vs GARCH vs Machine Learning) | |
| 7 | +Fichier : plots.py | |
| 8 | +Description : Style matplotlib commun et figures publication- | |
| 9 | + ready du papier (calibre revue) — mêmes conventions | |
| 10 | + que la série de working papers UQO. | |
| 11 | +================================================================ | |
| 12 | +""" | |
| 13 | + | |
| 14 | +from __future__ import annotations | |
| 15 | + | |
| 16 | +import logging | |
| 17 | + | |
| 18 | +import matplotlib | |
| 19 | + | |
| 20 | +matplotlib.use("Agg") | |
| 21 | +import matplotlib.pyplot as plt # noqa: E402 | |
| 22 | +import numpy as np # noqa: E402 | |
| 23 | +import pandas as pd # noqa: E402 | |
| 24 | + | |
| 25 | +logger = logging.getLogger(__name__) | |
| 26 | + | |
| 27 | +INK = "#1a1a1a" # primary marks | |
| 28 | +BLUE = "#2e6da4" # accent series | |
| 29 | +RED = "#c04848" # contrast series / polarity | |
| 30 | +GREY = "#8a8a8a" # reference lines | |
| 31 | +LIGHT = "#d9d9d9" # fills, bands | |
| 32 | +GRID = "#e3e3e3" | |
| 33 | + | |
| 34 | +BLUES = ["#c6d7e8", "#9dbcd8", "#74a1c8", "#4b86b8", "#2e6da4", "#1d4a75"] | |
| 35 | + | |
| 36 | +TEXTWIDTH = 6.3 # \textwidth in inches | |
| 37 | + | |
| 38 | + | |
| 39 | +def apply_style() -> None: | |
| 40 | + """Apply the shared print-journal matplotlib style.""" | |
| 41 | + plt.rcParams.update({ | |
| 42 | + "font.family": "serif", | |
| 43 | + "font.serif": ["Times New Roman", "Times", "STIXGeneral", "DejaVu Serif"], | |
| 44 | + "mathtext.fontset": "stix", | |
| 45 | + "font.size": 9, | |
| 46 | + "axes.labelsize": 9, | |
| 47 | + "xtick.labelsize": 8, | |
| 48 | + "ytick.labelsize": 8, | |
| 49 | + "legend.fontsize": 8, | |
| 50 | + "figure.dpi": 150, | |
| 51 | + "savefig.dpi": 300, | |
| 52 | + "axes.spines.top": False, | |
| 53 | + "axes.spines.right": False, | |
| 54 | + "axes.linewidth": 0.6, | |
| 55 | + "xtick.major.width": 0.6, | |
| 56 | + "ytick.major.width": 0.6, | |
| 57 | + "xtick.major.size": 3, | |
| 58 | + "ytick.major.size": 3, | |
| 59 | + "xtick.direction": "out", | |
| 60 | + "ytick.direction": "out", | |
| 61 | + "axes.grid": False, | |
| 62 | + "grid.color": GRID, | |
| 63 | + "grid.linewidth": 0.5, | |
| 64 | + "legend.frameon": False, | |
| 65 | + "savefig.bbox": "tight", | |
| 66 | + "savefig.pad_inches": 0.02, | |
| 67 | + }) | |
| 68 | + | |
| 69 | + | |
| 70 | +def panel_label(ax, text: str) -> None: | |
| 71 | + """Bold flush-left panel header above the axes (JF house style).""" | |
| 72 | + ax.set_title(text, loc="left", fontweight="bold", fontsize=9, pad=6) | |
| 73 | + | |
| 74 | + | |
| 75 | +def annualized_vol(rv: pd.Series) -> pd.Series: | |
| 76 | + """Convert daily realized variance to annualized percent volatility.""" | |
| 77 | + return np.sqrt(rv * 252) * 100 | |
| 78 | ||