spb/llmindex Public
The discriminative, contamination-resistant, fully transparent LLM ranking — updated live.
TypeScript 77.9%
TeX 15.2%
Python 3.7%
SQL 1.4%
JavaScript 1.1%
Shell 0.5%
1@article{bradley1952rank,2 author = {Bradley, Ralph Allan and Terry, Milton E.},3 title = {Rank Analysis of Incomplete Block Designs: {I}. The Method of Paired Comparisons},4 journal = {Biometrika},5 volume = {39},6 number = {3/4},7 pages = {324--345},8 year = {1952}9}1011@book{embretson2000item,12 author = {Embretson, Susan E. and Reise, Steven P.},13 title = {Item Response Theory for Psychologists},14 publisher = {Lawrence Erlbaum Associates},15 year = {2000}16}1718@incollection{birnbaum1968some,19 author = {Birnbaum, Allan},20 title = {Some Latent Trait Models and Their Use in Inferring an Examinee's Ability},21 booktitle = {Statistical Theories of Mental Test Scores},22 editor = {Lord, Frederic M. and Novick, Melvin R.},23 publisher = {Addison-Wesley},24 year = {1968}25}2627@article{brier1950verification,28 author = {Brier, Glenn W.},29 title = {Verification of Forecasts Expressed in Terms of Probability},30 journal = {Monthly Weather Review},31 volume = {78},32 number = {1},33 pages = {1--3},34 year = {1950}35}3637@inproceedings{guo2017calibration,38 author = {Guo, Chuan and Pleiss, Geoff and Sun, Yu and Weinberger, Kilian Q.},39 title = {On Calibration of Modern Neural Networks},40 booktitle = {Proceedings of the 34th International Conference on Machine Learning (ICML)},41 year = {2017}42}4344@inproceedings{hendrycks2021mmlu,45 author = {Hendrycks, Dan and Burns, Collin and Basart, Steven and Zou, Andy and Mazeika, Mantas and Song, Dawn and Steinhardt, Jacob},46 title = {Measuring Massive Multitask Language Understanding},47 booktitle = {International Conference on Learning Representations (ICLR)},48 year = {2021}49}5051@inproceedings{wang2024mmlupro,52 author = {Wang, Yubo and Ma, Xueguang and Zhang, Ge and others},53 title = {{MMLU-Pro}: A More Robust and Challenging Multi-Task Language Understanding Benchmark},54 booktitle = {Advances in Neural Information Processing Systems (NeurIPS)},55 year = {2024}56}5758@article{liang2023helm,59 author = {Liang, Percy and Bommasani, Rishi and Lee, Tony and others},60 title = {Holistic Evaluation of Language Models},61 journal = {Transactions on Machine Learning Research},62 year = {2023}63}6465@inproceedings{chiang2024chatbot,66 author = {Chiang, Wei-Lin and Zheng, Lianmin and Sheng, Ying and others},67 title = {Chatbot Arena: An Open Platform for Evaluating {LLMs} by Human Preference},68 booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)},69 year = {2024}70}7172@article{white2024livebench,73 author = {White, Colin and Dooley, Samuel and Roberts, Manley and others},74 title = {{LiveBench}: A Challenging, Contamination-Free {LLM} Benchmark},75 journal = {arXiv preprint arXiv:2406.19314},76 year = {2024}77}7879@article{mirzadeh2024gsmsymbolic,80 author = {Mirzadeh, Iman and Alizadeh, Keivan and Shahrokhi, Hooman and Tuzel, Oncel and Bengio, Samy and Farajtabar, Mehrdad},81 title = {{GSM-Symbolic}: Understanding the Limitations of Mathematical Reasoning in Large Language Models},82 journal = {arXiv preprint arXiv:2410.05229},83 year = {2024}84}8586@inproceedings{shi2023distracted,87 author = {Shi, Freda and Chen, Xinyun and Misra, Kanishka and Scales, Nathan and Dohan, David and Chi, Ed and Sch{\"a}rli, Nathanael and Zhou, Denny},88 title = {Large Language Models Can Be Easily Distracted by Irrelevant Context},89 booktitle = {Proceedings of the 40th International Conference on Machine Learning (ICML)},90 year = {2023}91}9293@inproceedings{wu2024reasoning,94 author = {Wu, Zhaofeng and Qiu, Linlu and Ross, Alexis and others},95 title = {Reasoning or Reciting? Exploring the Capabilities and Limitations of Language Models Through Counterfactual Tasks},96 booktitle = {Proceedings of NAACL},97 year = {2024}98}99100@inproceedings{dziri2023faith,101 author = {Dziri, Nouha and Lu, Ximing and Sclar, Melanie and others},102 title = {Faith and Fate: Limits of Transformers on Compositionality},103 booktitle = {Advances in Neural Information Processing Systems (NeurIPS)},104 year = {2023}105}106107@inproceedings{lin2025zebralogic,108 author = {Lin, Bill Yuchen and Bras, Ronan Le and Richardson, Kyle and others},109 title = {{ZebraLogic}: On the Scaling Limits of {LLMs} for Logical Reasoning},110 booktitle = {Proceedings of the 42nd International Conference on Machine Learning (ICML)},111 year = {2025}112}113114@article{zhou2023ifeval,115 author = {Zhou, Jeffrey and Lu, Tianjian and Mishra, Swaroop and others},116 title = {Instruction-Following Evaluation for Large Language Models},117 journal = {arXiv preprint arXiv:2311.07911},118 year = {2023}119}120121@inproceedings{polo2024tinybenchmarks,122 author = {Polo, Felipe Maia and Weber, Lucas and Choshen, Leshem and Sun, Yuekai and Xu, Gongjun and Yurochkin, Mikhail},123 title = {{tinyBenchmarks}: Evaluating {LLMs} with Fewer Examples},124 booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)},125 year = {2024}126}127128@inproceedings{kipnis2025metabench,129 author = {Kipnis, Alex and Voudouris, Konstantinos and Buschoff, Luca M. Schulze and Schulz, Eric},130 title = {metabench: A Sparse Benchmark of Reasoning and Knowledge in Large Language Models},131 booktitle = {International Conference on Learning Representations (ICLR)},132 year = {2025}133}134135@article{yao2024taubench,136 author = {Yao, Shunyu and Shinn, Noah and Razavi, Pedram and Narasimhan, Karthik},137 title = {$\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains},138 journal = {arXiv preprint arXiv:2406.12045},139 year = {2024}140}141142@inproceedings{patil2025bfcl,143 author = {Patil, Shishir G. and Mao, Huanzhi and Cheng-Jie Ji, Charlie and others},144 title = {The {Berkeley} Function Calling Leaderboard ({BFCL}): From Tool Use to Agentic Evaluation of Large Language Models},145 booktitle = {Proceedings of the 42nd International Conference on Machine Learning (ICML)},146 year = {2025}147}148149@article{mialon2023gaia,150 author = {Mialon, Gr{\'e}goire and Fourrier, Cl{\'e}mentine and Swift, Craig and Wolf, Thomas and LeCun, Yann and Scialom, Thomas},151 title = {{GAIA}: A Benchmark for General {AI} Assistants},152 journal = {arXiv preprint arXiv:2311.12983},153 year = {2023}154}155156@inproceedings{jimenez2024swebench,157 author = {Jimenez, Carlos E. and Yang, John and Wettig, Alexander and Yao, Shunyu and Pei, Kexin and Press, Ofir and Narasimhan, Karthik},158 title = {{SWE-bench}: Can Language Models Resolve Real-World {GitHub} Issues?},159 booktitle = {International Conference on Learning Representations (ICLR)},160 year = {2024}161}162163@article{gu2024cruxeval,164 author = {Gu, Alex and Roziere, Baptiste and Leather, Hugh and Solar-Lezama, Armando and Synnaeve, Gabriel and Wang, Sida I.},165 title = {{CRUXEval}: A Benchmark for Code Reasoning, Understanding and Execution},166 journal = {arXiv preprint arXiv:2401.03065},167 year = {2024}168}169170@article{tbench2025,171 author = {The Terminal-Bench Team},172 title = {Terminal-Bench: A Benchmark for {AI} Agents in Terminal Environments},173 journal = {https://www.tbench.ai},174 year = {2025}175}176177@inproceedings{tian2023just,178 author = {Tian, Katherine and Mitchell, Eric and Zhou, Allan and Sharma, Archit and Rafailov, Rafael and Yao, Huaxiu and Finn, Chelsea and Manning, Christopher D.},179 title = {Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback},180 booktitle = {Proceedings of EMNLP},181 year = {2023}182}183184@article{yu2024xfinder,185 author = {Yu, Qingchen and Zheng, Zifan and Song, Shichao and others},186 title = {{xFinder}: Robust and Pinpoint Answer Extraction for Large Language Models},187 journal = {arXiv preprint arXiv:2405.11874},188 year = {2024}189}190191@inproceedings{sclar2024formatspread,192 author = {Sclar, Melanie and Choi, Yejin and Tsvetkov, Yulia and Suhr, Alane},193 title = {Quantifying Language Models' Sensitivity to Spurious Features in Prompt Design},194 booktitle = {International Conference on Learning Representations (ICLR)},195 year = {2024}196}197198@article{phan2025hle,199 author = {Phan, Long and Gatti, Alice and Han, Ziwen and others},200 title = {Humanity's Last Exam},201 journal = {Nature},202 year = {2025}203}204205@article{fan2024nphardeval,206 author = {Fan, Lizhou and Hua, Wenyue and Li, Lingyao and Ling, Haoyang and Zhang, Yongfeng},207 title = {{NPHardEval}: Dynamic Benchmark on Reasoning Ability of Large Language Models via Complexity Classes},208 journal = {arXiv preprint arXiv:2312.14890},209 year = {2024}210}211212@inproceedings{greenberg2020smoosh,213 author = {Greenberg, Michael and Blatt, Austin J.},214 title = {Executable Formal Semantics for the {POSIX} Shell},215 booktitle = {Proceedings of the ACM on Programming Languages (POPL)},216 year = {2020}217}218219@article{elo1978rating,220 author = {Elo, Arpad E.},221 title = {The Rating of Chessplayers, Past and Present},222 journal = {Arco Publishing},223 year = {1978}224}225226@article{hunter2004mm,227 author = {Hunter, David R.},228 title = {{MM} Algorithms for Generalized {Bradley-Terry} Models},229 journal = {The Annals of Statistics},230 volume = {32},231 number = {1},232 pages = {384--406},233 year = {2004}234}235236@article{oren2024proving,237 author = {Oren, Yonatan and Meister, Nicole and Chatterji, Niladri and Ladhak, Faisal and Hashimoto, Tatsunori B.},238 title = {Proving Test Set Contamination in Black Box Language Models},239 journal = {International Conference on Learning Representations (ICLR)},240 year = {2024}241}242243@article{yu2025rhorizon,244 author = {Yu, Sitong and others},245 title = {{R-Horizon}: How Far Can Your Large Reasoning Model Really Go in Breadth and Depth?},246 journal = {arXiv preprint arXiv:2510.08189},247 year = {2025}248}249