@article{bradley1952rank, author = {Bradley, Ralph Allan and Terry, Milton E.}, title = {Rank Analysis of Incomplete Block Designs: {I}. The Method of Paired Comparisons}, journal = {Biometrika}, volume = {39}, number = {3/4}, pages = {324--345}, year = {1952} } @book{embretson2000item, author = {Embretson, Susan E. and Reise, Steven P.}, title = {Item Response Theory for Psychologists}, publisher = {Lawrence Erlbaum Associates}, year = {2000} } @incollection{birnbaum1968some, author = {Birnbaum, Allan}, title = {Some Latent Trait Models and Their Use in Inferring an Examinee's Ability}, booktitle = {Statistical Theories of Mental Test Scores}, editor = {Lord, Frederic M. and Novick, Melvin R.}, publisher = {Addison-Wesley}, year = {1968} } @article{brier1950verification, author = {Brier, Glenn W.}, title = {Verification of Forecasts Expressed in Terms of Probability}, journal = {Monthly Weather Review}, volume = {78}, number = {1}, pages = {1--3}, year = {1950} } @inproceedings{guo2017calibration, author = {Guo, Chuan and Pleiss, Geoff and Sun, Yu and Weinberger, Kilian Q.}, title = {On Calibration of Modern Neural Networks}, booktitle = {Proceedings of the 34th International Conference on Machine Learning (ICML)}, year = {2017} } @inproceedings{hendrycks2021mmlu, author = {Hendrycks, Dan and Burns, Collin and Basart, Steven and Zou, Andy and Mazeika, Mantas and Song, Dawn and Steinhardt, Jacob}, title = {Measuring Massive Multitask Language Understanding}, booktitle = {International Conference on Learning Representations (ICLR)}, year = {2021} } @inproceedings{wang2024mmlupro, author = {Wang, Yubo and Ma, Xueguang and Zhang, Ge and others}, title = {{MMLU-Pro}: A More Robust and Challenging Multi-Task Language Understanding Benchmark}, booktitle = {Advances in Neural Information Processing Systems (NeurIPS)}, year = {2024} } @article{liang2023helm, author = {Liang, Percy and Bommasani, Rishi and Lee, Tony and others}, title = {Holistic Evaluation of Language Models}, journal = {Transactions on Machine Learning Research}, year = {2023} } @inproceedings{chiang2024chatbot, author = {Chiang, Wei-Lin and Zheng, Lianmin and Sheng, Ying and others}, title = {Chatbot Arena: An Open Platform for Evaluating {LLMs} by Human Preference}, booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)}, year = {2024} } @article{white2024livebench, author = {White, Colin and Dooley, Samuel and Roberts, Manley and others}, title = {{LiveBench}: A Challenging, Contamination-Free {LLM} Benchmark}, journal = {arXiv preprint arXiv:2406.19314}, year = {2024} } @article{mirzadeh2024gsmsymbolic, author = {Mirzadeh, Iman and Alizadeh, Keivan and Shahrokhi, Hooman and Tuzel, Oncel and Bengio, Samy and Farajtabar, Mehrdad}, title = {{GSM-Symbolic}: Understanding the Limitations of Mathematical Reasoning in Large Language Models}, journal = {arXiv preprint arXiv:2410.05229}, year = {2024} } @inproceedings{shi2023distracted, author = {Shi, Freda and Chen, Xinyun and Misra, Kanishka and Scales, Nathan and Dohan, David and Chi, Ed and Sch{\"a}rli, Nathanael and Zhou, Denny}, title = {Large Language Models Can Be Easily Distracted by Irrelevant Context}, booktitle = {Proceedings of the 40th International Conference on Machine Learning (ICML)}, year = {2023} } @inproceedings{wu2024reasoning, author = {Wu, Zhaofeng and Qiu, Linlu and Ross, Alexis and others}, title = {Reasoning or Reciting? Exploring the Capabilities and Limitations of Language Models Through Counterfactual Tasks}, booktitle = {Proceedings of NAACL}, year = {2024} } @inproceedings{dziri2023faith, author = {Dziri, Nouha and Lu, Ximing and Sclar, Melanie and others}, title = {Faith and Fate: Limits of Transformers on Compositionality}, booktitle = {Advances in Neural Information Processing Systems (NeurIPS)}, year = {2023} } @inproceedings{lin2025zebralogic, author = {Lin, Bill Yuchen and Bras, Ronan Le and Richardson, Kyle and others}, title = {{ZebraLogic}: On the Scaling Limits of {LLMs} for Logical Reasoning}, booktitle = {Proceedings of the 42nd International Conference on Machine Learning (ICML)}, year = {2025} } @article{zhou2023ifeval, author = {Zhou, Jeffrey and Lu, Tianjian and Mishra, Swaroop and others}, title = {Instruction-Following Evaluation for Large Language Models}, journal = {arXiv preprint arXiv:2311.07911}, year = {2023} } @inproceedings{polo2024tinybenchmarks, author = {Polo, Felipe Maia and Weber, Lucas and Choshen, Leshem and Sun, Yuekai and Xu, Gongjun and Yurochkin, Mikhail}, title = {{tinyBenchmarks}: Evaluating {LLMs} with Fewer Examples}, booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)}, year = {2024} } @inproceedings{kipnis2025metabench, author = {Kipnis, Alex and Voudouris, Konstantinos and Buschoff, Luca M. Schulze and Schulz, Eric}, title = {metabench: A Sparse Benchmark of Reasoning and Knowledge in Large Language Models}, booktitle = {International Conference on Learning Representations (ICLR)}, year = {2025} } @article{yao2024taubench, author = {Yao, Shunyu and Shinn, Noah and Razavi, Pedram and Narasimhan, Karthik}, title = {$\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains}, journal = {arXiv preprint arXiv:2406.12045}, year = {2024} } @inproceedings{patil2025bfcl, author = {Patil, Shishir G. and Mao, Huanzhi and Cheng-Jie Ji, Charlie and others}, title = {The {Berkeley} Function Calling Leaderboard ({BFCL}): From Tool Use to Agentic Evaluation of Large Language Models}, booktitle = {Proceedings of the 42nd International Conference on Machine Learning (ICML)}, year = {2025} } @article{mialon2023gaia, author = {Mialon, Gr{\'e}goire and Fourrier, Cl{\'e}mentine and Swift, Craig and Wolf, Thomas and LeCun, Yann and Scialom, Thomas}, title = {{GAIA}: A Benchmark for General {AI} Assistants}, journal = {arXiv preprint arXiv:2311.12983}, year = {2023} } @inproceedings{jimenez2024swebench, author = {Jimenez, Carlos E. and Yang, John and Wettig, Alexander and Yao, Shunyu and Pei, Kexin and Press, Ofir and Narasimhan, Karthik}, title = {{SWE-bench}: Can Language Models Resolve Real-World {GitHub} Issues?}, booktitle = {International Conference on Learning Representations (ICLR)}, year = {2024} } @article{gu2024cruxeval, author = {Gu, Alex and Roziere, Baptiste and Leather, Hugh and Solar-Lezama, Armando and Synnaeve, Gabriel and Wang, Sida I.}, title = {{CRUXEval}: A Benchmark for Code Reasoning, Understanding and Execution}, journal = {arXiv preprint arXiv:2401.03065}, year = {2024} } @article{tbench2025, author = {The Terminal-Bench Team}, title = {Terminal-Bench: A Benchmark for {AI} Agents in Terminal Environments}, journal = {https://www.tbench.ai}, year = {2025} } @inproceedings{tian2023just, author = {Tian, Katherine and Mitchell, Eric and Zhou, Allan and Sharma, Archit and Rafailov, Rafael and Yao, Huaxiu and Finn, Chelsea and Manning, Christopher D.}, title = {Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback}, booktitle = {Proceedings of EMNLP}, year = {2023} } @article{yu2024xfinder, author = {Yu, Qingchen and Zheng, Zifan and Song, Shichao and others}, title = {{xFinder}: Robust and Pinpoint Answer Extraction for Large Language Models}, journal = {arXiv preprint arXiv:2405.11874}, year = {2024} } @inproceedings{sclar2024formatspread, author = {Sclar, Melanie and Choi, Yejin and Tsvetkov, Yulia and Suhr, Alane}, title = {Quantifying Language Models' Sensitivity to Spurious Features in Prompt Design}, booktitle = {International Conference on Learning Representations (ICLR)}, year = {2024} } @article{phan2025hle, author = {Phan, Long and Gatti, Alice and Han, Ziwen and others}, title = {Humanity's Last Exam}, journal = {Nature}, year = {2025} } @article{fan2024nphardeval, author = {Fan, Lizhou and Hua, Wenyue and Li, Lingyao and Ling, Haoyang and Zhang, Yongfeng}, title = {{NPHardEval}: Dynamic Benchmark on Reasoning Ability of Large Language Models via Complexity Classes}, journal = {arXiv preprint arXiv:2312.14890}, year = {2024} } @inproceedings{greenberg2020smoosh, author = {Greenberg, Michael and Blatt, Austin J.}, title = {Executable Formal Semantics for the {POSIX} Shell}, booktitle = {Proceedings of the ACM on Programming Languages (POPL)}, year = {2020} } @article{elo1978rating, author = {Elo, Arpad E.}, title = {The Rating of Chessplayers, Past and Present}, journal = {Arco Publishing}, year = {1978} } @article{hunter2004mm, author = {Hunter, David R.}, title = {{MM} Algorithms for Generalized {Bradley-Terry} Models}, journal = {The Annals of Statistics}, volume = {32}, number = {1}, pages = {384--406}, year = {2004} } @article{oren2024proving, author = {Oren, Yonatan and Meister, Nicole and Chatterji, Niladri and Ladhak, Faisal and Hashimoto, Tatsunori B.}, title = {Proving Test Set Contamination in Black Box Language Models}, journal = {International Conference on Learning Representations (ICLR)}, year = {2024} } @article{yu2025rhorizon, author = {Yu, Sitong and others}, title = {{R-Horizon}: How Far Can Your Large Reasoning Model Really Go in Breadth and Depth?}, journal = {arXiv preprint arXiv:2510.08189}, year = {2025} }