@article{atreya2025roboarena,
  title = {{RoboArena}: {Distributed} {Real}-{World} {Evaluation} of {Generalist} {Robot} {Policies}},
  shorttitle = {{RoboArena}},
  url = {http://arxiv.org/abs/2506.18123},
  doi = {10.48550/arXiv.2506.18123},
  abstract = {Comprehensive, unbiased, and comparable evaluation of modern generalist policies is uniquely challenging: existing approaches for robot benchmarking typically rely on heavy standardization, either by specifying fixed evaluation tasks and environments, or by hosting centralized ''robot challenges'', and do not readily scale to evaluating generalist policies across a broad range of tasks and environments. In this work, we propose RoboArena, a new approach for scalable evaluation of generalist robot policies in the real world. Instead of standardizing evaluations around fixed tasks, environments, or locations, we propose to crowd-source evaluations across a distributed network of evaluators. Importantly, evaluators can freely choose the tasks and environments they evaluate on, enabling easy scaling of diversity, but they are required to perform double-blind evaluations over pairs of policies. Then, by aggregating preference feedback from pairwise comparisons across diverse tasks and environments, we can derive a ranking of policies. We instantiate our approach across a network of evaluators at seven academic institutions using the DROID robot platform. Through more than 600 pairwise real-robot evaluation episodes across seven generalist policies, we demonstrate that our crowd-sourced approach can more accurately rank the performance of existing generalist policies than conventional, centralized evaluation approaches, while being more scalable, resilient, and trustworthy. We open our evaluation network to the community and hope that it can enable more accessible comparisons of generalist robot policies.},
  urldate = {2025-09-23},
  journal = {CORL},
  author = {Atreya, Pranav and Pertsch, Karl and Lee, Tony and Kim, Moo Jin and Jain, Arhan and Kuramshin, Artur and Eppner, Clemens and Neary, Cyrus and Hu, Edward and Ramos, Fabio and Tremblay, Jonathan and Arora, Kanav and Ellis, Kirsty and Macesanu, Luca and Leonard, Matthew and Cho, Meedeum and Aslan, Ozgur and Dass, Shivin and Wang, Jie and Yuan, Xingfang and Yang, Xuning and Gupta, Abhishek and Jayaraman, Dinesh and Berseth, Glen and Daniilidis, Kostas and Martin-Martin, Roberto and Lee, Youngwoon and Liang, Percy and Finn, Chelsea and Levine, Sergey},
  month = {Nov},
  year = {2025},
  pub_type = {conference},
  award = {Oral Presentation},
  note = {arXiv:2506.18123 [cs]},
  keywords = {Computer Science - Machine Learning, Computer Science - Robotics},
  annote = {Comment: Website: https://robo-arena.github.io/},
  file = {Preprint PDF:/Users/dineshj/Zotero/storage/8AGWFMBI/Atreya et al. - 2025 - RoboArena Distributed Real-World Evaluation of Generalist Robot Policies.pdf:application/pdf;Snapshot:/Users/dineshj/Zotero/storage/LQJH37S2/2506.html:text/html},
  url_pdf = {/publication/atreya-2025-roboarena/atreya-2025-roboarena.pdf},
  url_project = {https://robo-arena.github.io/},
  url_code = {https://github.com/robo-arena/roboarena},
  url_dataset = {https://huggingface.co/datasets/RoboArena/DataDump_02-03-2026},
  url_arxiv = {http://arxiv.org/abs/2506.18123},
}
