@article{anupam2026roboreason,
  title = {{RoboReason}: Boosting Robotics-Relevant {VLM} Reasoning Through Scalable Data Generation},
  author = {Anupam, Sagnik and Lal, Utkarsh and Edara, Aneesh and Zhang, Botong and Tang, Yao and Gu, Jiatao and Jayaraman*, Dinesh and Bastani*, Osbert},
  journal = {EMNLP},
  year = {2026},
  month = {Oct},
  pub_type = {conference},
  abstract = {Vision-Language Models (VLMs) have demonstrated impressive performance when integrated into robot manipulation pipelines, but real-world evaluation of VLM capabilities on downstream robotics tasks is both difficult and expensive to scale. We propose that spatial and temporal ordering tasks using images from stereo and wrist cameras attached to real-world robots are particularly well-suited for evaluating visual reasoning capabilities, as they assess skills directly applicable to robots. We evaluate VLM reasoning abilities on three questions: ordering shuffled frames conditioned on a language description of the task, ordering objects depthwise based on distance from the camera, and ordering objects from left-to-right in pixel coordinates. The complexity of these tasks can also be modulated by changing the size of the set to sort. Our experiments show that several state-of-the-art vision-language models struggle on ordering tasks. To fix this, we propose a scalable autonomous training data generation pipeline, and show that both supervised finetuning and reinforcement-learning on the generated data can significantly improve VLM reasoning performance both on spatiotemporal ordering tasks as well as downstream robotics and visual reasoning tasks.},
}
