SPB Git

spb/llmindex Public

The discriminative, contamination-resistant, fully transparent LLM ranking — updated live.

TypeScript 77.9% TeX 15.2% Python 3.7% SQL 1.4% JavaScript 1.1% Shell 0.5%
9.7 KB · 249 lines bibtex
Raw Blame History
1@article{bradley1952rank,2  author  = {Bradley, Ralph Allan and Terry, Milton E.},3  title   = {Rank Analysis of Incomplete Block Designs: {I}. The Method of Paired Comparisons},4  journal = {Biometrika},5  volume  = {39},6  number  = {3/4},7  pages   = {324--345},8  year    = {1952}9}1011@book{embretson2000item,12  author    = {Embretson, Susan E. and Reise, Steven P.},13  title     = {Item Response Theory for Psychologists},14  publisher = {Lawrence Erlbaum Associates},15  year      = {2000}16}1718@incollection{birnbaum1968some,19  author    = {Birnbaum, Allan},20  title     = {Some Latent Trait Models and Their Use in Inferring an Examinee's Ability},21  booktitle = {Statistical Theories of Mental Test Scores},22  editor    = {Lord, Frederic M. and Novick, Melvin R.},23  publisher = {Addison-Wesley},24  year      = {1968}25}2627@article{brier1950verification,28  author  = {Brier, Glenn W.},29  title   = {Verification of Forecasts Expressed in Terms of Probability},30  journal = {Monthly Weather Review},31  volume  = {78},32  number  = {1},33  pages   = {1--3},34  year    = {1950}35}3637@inproceedings{guo2017calibration,38  author    = {Guo, Chuan and Pleiss, Geoff and Sun, Yu and Weinberger, Kilian Q.},39  title     = {On Calibration of Modern Neural Networks},40  booktitle = {Proceedings of the 34th International Conference on Machine Learning (ICML)},41  year      = {2017}42}4344@inproceedings{hendrycks2021mmlu,45  author    = {Hendrycks, Dan and Burns, Collin and Basart, Steven and Zou, Andy and Mazeika, Mantas and Song, Dawn and Steinhardt, Jacob},46  title     = {Measuring Massive Multitask Language Understanding},47  booktitle = {International Conference on Learning Representations (ICLR)},48  year      = {2021}49}5051@inproceedings{wang2024mmlupro,52  author    = {Wang, Yubo and Ma, Xueguang and Zhang, Ge and others},53  title     = {{MMLU-Pro}: A More Robust and Challenging Multi-Task Language Understanding Benchmark},54  booktitle = {Advances in Neural Information Processing Systems (NeurIPS)},55  year      = {2024}56}5758@article{liang2023helm,59  author  = {Liang, Percy and Bommasani, Rishi and Lee, Tony and others},60  title   = {Holistic Evaluation of Language Models},61  journal = {Transactions on Machine Learning Research},62  year    = {2023}63}6465@inproceedings{chiang2024chatbot,66  author    = {Chiang, Wei-Lin and Zheng, Lianmin and Sheng, Ying and others},67  title     = {Chatbot Arena: An Open Platform for Evaluating {LLMs} by Human Preference},68  booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)},69  year      = {2024}70}7172@article{white2024livebench,73  author  = {White, Colin and Dooley, Samuel and Roberts, Manley and others},74  title   = {{LiveBench}: A Challenging, Contamination-Free {LLM} Benchmark},75  journal = {arXiv preprint arXiv:2406.19314},76  year    = {2024}77}7879@article{mirzadeh2024gsmsymbolic,80  author  = {Mirzadeh, Iman and Alizadeh, Keivan and Shahrokhi, Hooman and Tuzel, Oncel and Bengio, Samy and Farajtabar, Mehrdad},81  title   = {{GSM-Symbolic}: Understanding the Limitations of Mathematical Reasoning in Large Language Models},82  journal = {arXiv preprint arXiv:2410.05229},83  year    = {2024}84}8586@inproceedings{shi2023distracted,87  author    = {Shi, Freda and Chen, Xinyun and Misra, Kanishka and Scales, Nathan and Dohan, David and Chi, Ed and Sch{\"a}rli, Nathanael and Zhou, Denny},88  title     = {Large Language Models Can Be Easily Distracted by Irrelevant Context},89  booktitle = {Proceedings of the 40th International Conference on Machine Learning (ICML)},90  year      = {2023}91}9293@inproceedings{wu2024reasoning,94  author    = {Wu, Zhaofeng and Qiu, Linlu and Ross, Alexis and others},95  title     = {Reasoning or Reciting? Exploring the Capabilities and Limitations of Language Models Through Counterfactual Tasks},96  booktitle = {Proceedings of NAACL},97  year      = {2024}98}99100@inproceedings{dziri2023faith,101  author    = {Dziri, Nouha and Lu, Ximing and Sclar, Melanie and others},102  title     = {Faith and Fate: Limits of Transformers on Compositionality},103  booktitle = {Advances in Neural Information Processing Systems (NeurIPS)},104  year      = {2023}105}106107@inproceedings{lin2025zebralogic,108  author    = {Lin, Bill Yuchen and Bras, Ronan Le and Richardson, Kyle and others},109  title     = {{ZebraLogic}: On the Scaling Limits of {LLMs} for Logical Reasoning},110  booktitle = {Proceedings of the 42nd International Conference on Machine Learning (ICML)},111  year      = {2025}112}113114@article{zhou2023ifeval,115  author  = {Zhou, Jeffrey and Lu, Tianjian and Mishra, Swaroop and others},116  title   = {Instruction-Following Evaluation for Large Language Models},117  journal = {arXiv preprint arXiv:2311.07911},118  year    = {2023}119}120121@inproceedings{polo2024tinybenchmarks,122  author    = {Polo, Felipe Maia and Weber, Lucas and Choshen, Leshem and Sun, Yuekai and Xu, Gongjun and Yurochkin, Mikhail},123  title     = {{tinyBenchmarks}: Evaluating {LLMs} with Fewer Examples},124  booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)},125  year      = {2024}126}127128@inproceedings{kipnis2025metabench,129  author    = {Kipnis, Alex and Voudouris, Konstantinos and Buschoff, Luca M. Schulze and Schulz, Eric},130  title     = {metabench: A Sparse Benchmark of Reasoning and Knowledge in Large Language Models},131  booktitle = {International Conference on Learning Representations (ICLR)},132  year      = {2025}133}134135@article{yao2024taubench,136  author  = {Yao, Shunyu and Shinn, Noah and Razavi, Pedram and Narasimhan, Karthik},137  title   = {$\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains},138  journal = {arXiv preprint arXiv:2406.12045},139  year    = {2024}140}141142@inproceedings{patil2025bfcl,143  author    = {Patil, Shishir G. and Mao, Huanzhi and Cheng-Jie Ji, Charlie and others},144  title     = {The {Berkeley} Function Calling Leaderboard ({BFCL}): From Tool Use to Agentic Evaluation of Large Language Models},145  booktitle = {Proceedings of the 42nd International Conference on Machine Learning (ICML)},146  year      = {2025}147}148149@article{mialon2023gaia,150  author  = {Mialon, Gr{\'e}goire and Fourrier, Cl{\'e}mentine and Swift, Craig and Wolf, Thomas and LeCun, Yann and Scialom, Thomas},151  title   = {{GAIA}: A Benchmark for General {AI} Assistants},152  journal = {arXiv preprint arXiv:2311.12983},153  year    = {2023}154}155156@inproceedings{jimenez2024swebench,157  author    = {Jimenez, Carlos E. and Yang, John and Wettig, Alexander and Yao, Shunyu and Pei, Kexin and Press, Ofir and Narasimhan, Karthik},158  title     = {{SWE-bench}: Can Language Models Resolve Real-World {GitHub} Issues?},159  booktitle = {International Conference on Learning Representations (ICLR)},160  year      = {2024}161}162163@article{gu2024cruxeval,164  author  = {Gu, Alex and Roziere, Baptiste and Leather, Hugh and Solar-Lezama, Armando and Synnaeve, Gabriel and Wang, Sida I.},165  title   = {{CRUXEval}: A Benchmark for Code Reasoning, Understanding and Execution},166  journal = {arXiv preprint arXiv:2401.03065},167  year    = {2024}168}169170@article{tbench2025,171  author  = {The Terminal-Bench Team},172  title   = {Terminal-Bench: A Benchmark for {AI} Agents in Terminal Environments},173  journal = {https://www.tbench.ai},174  year    = {2025}175}176177@inproceedings{tian2023just,178  author    = {Tian, Katherine and Mitchell, Eric and Zhou, Allan and Sharma, Archit and Rafailov, Rafael and Yao, Huaxiu and Finn, Chelsea and Manning, Christopher D.},179  title     = {Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback},180  booktitle = {Proceedings of EMNLP},181  year      = {2023}182}183184@article{yu2024xfinder,185  author  = {Yu, Qingchen and Zheng, Zifan and Song, Shichao and others},186  title   = {{xFinder}: Robust and Pinpoint Answer Extraction for Large Language Models},187  journal = {arXiv preprint arXiv:2405.11874},188  year    = {2024}189}190191@inproceedings{sclar2024formatspread,192  author    = {Sclar, Melanie and Choi, Yejin and Tsvetkov, Yulia and Suhr, Alane},193  title     = {Quantifying Language Models' Sensitivity to Spurious Features in Prompt Design},194  booktitle = {International Conference on Learning Representations (ICLR)},195  year      = {2024}196}197198@article{phan2025hle,199  author  = {Phan, Long and Gatti, Alice and Han, Ziwen and others},200  title   = {Humanity's Last Exam},201  journal = {Nature},202  year    = {2025}203}204205@article{fan2024nphardeval,206  author  = {Fan, Lizhou and Hua, Wenyue and Li, Lingyao and Ling, Haoyang and Zhang, Yongfeng},207  title   = {{NPHardEval}: Dynamic Benchmark on Reasoning Ability of Large Language Models via Complexity Classes},208  journal = {arXiv preprint arXiv:2312.14890},209  year    = {2024}210}211212@inproceedings{greenberg2020smoosh,213  author    = {Greenberg, Michael and Blatt, Austin J.},214  title     = {Executable Formal Semantics for the {POSIX} Shell},215  booktitle = {Proceedings of the ACM on Programming Languages (POPL)},216  year      = {2020}217}218219@article{elo1978rating,220  author    = {Elo, Arpad E.},221  title     = {The Rating of Chessplayers, Past and Present},222  journal   = {Arco Publishing},223  year      = {1978}224}225226@article{hunter2004mm,227  author  = {Hunter, David R.},228  title   = {{MM} Algorithms for Generalized {Bradley-Terry} Models},229  journal = {The Annals of Statistics},230  volume  = {32},231  number  = {1},232  pages   = {384--406},233  year    = {2004}234}235236@article{oren2024proving,237  author  = {Oren, Yonatan and Meister, Nicole and Chatterji, Niladri and Ladhak, Faisal and Hashimoto, Tatsunori B.},238  title   = {Proving Test Set Contamination in Black Box Language Models},239  journal = {International Conference on Learning Representations (ICLR)},240  year    = {2024}241}242243@article{yu2025rhorizon,244  author  = {Yu, Sitong and others},245  title   = {{R-Horizon}: How Far Can Your Large Reasoning Model Really Go in Breadth and Depth?},246  journal = {arXiv preprint arXiv:2510.08189},247  year    = {2025}248}249