@inproceedings{grover2026cosplan,title={CoSPlan: Corrective Sequential Planning via Scene Graph Incremental Updates},author={Grover, Shresth and Pathak, Priyank and Kumar, Akash and Rawat, Yogesh S.},booktitle={European Conference on Computer Vision (ECCV)},year={2026},}
VISTA: A Benchmark for Spatio-Temporal Interaction Understanding in Videos
Alejandro Aparcedo, Akash Kumar, Aaryan Garg, Dalton Pham, Wen-Kai Chen, Anirudh Bharadwaj, Aman Chadha, and Yogesh Rawat
In IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), 2026
@inproceedings{aparcedo2026vista,title={VISTA: A Benchmark for Spatio-Temporal Interaction Understanding in Videos},author={Aparcedo, Alejandro and Kumar, Akash and Garg, Aaryan and Pham, Dalton and Chen, Wen-Kai and Bharadwaj, Anirudh and Chadha, Aman and Rawat, Yogesh},booktitle={IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)},pages={8411--8422},year={2026},}
VLA-Thinker: Boosting Vision-Language-Action Models through Thinking-with-Image Reasoning
Chaoyang Wang, Wenrui Bao, Sicheng Gao, Bingxin Xu, Yu Tian, Yogesh S Rawat, Yunhao Ge, and Yuzhang Shang
@article{wang2026vla,title={VLA-Thinker: Boosting Vision-Language-Action Models through Thinking-with-Image Reasoning},author={Wang, Chaoyang and Bao, Wenrui and Gao, Sicheng and Xu, Bingxin and Tian, Yu and Rawat, Yogesh S and Ge, Yunhao and Shang, Yuzhang},journal={arXiv preprint arXiv:2603.14523},year={2026},}
2025
LR0.FM: Low-Resolution Zero-Shot Classification Benchmark for Foundation Models
Priyank Pathak, Shyam Marjit, Shruti Vyas, and Yogesh S. Rawat
In International Conference on Learning Representations (ICLR), 2025
@inproceedings{pathak2025lr0fm,title={LR0.FM: Low-Resolution Zero-Shot Classification Benchmark for Foundation Models},author={Pathak, Priyank and Marjit, Shyam and Vyas, Shruti and Rawat, Yogesh S.},booktitle={International Conference on Learning Representations (ICLR)},year={2025},}
Contextual Self-paced Learning for Weakly Supervised Spatio-Temporal Video Grounding
Akash Kumar, Zsolt Kira, and Yogesh Singh Rawat
In International Conference on Learning Representations (ICLR), 2025
@inproceedings{kumar2025cospal,title={Contextual Self-paced Learning for Weakly Supervised Spatio-Temporal Video Grounding},author={Kumar, Akash and Kira, Zsolt and Rawat, Yogesh Singh},booktitle={International Conference on Learning Representations (ICLR)},year={2025},}
STPro: Spatial and Temporal Progressive Learning for Weakly Supervised Spatio-Temporal Grounding
Aaryan Garg, Akash Kumar, and Yogesh Rawat
In IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2025
@inproceedings{garg2025stpro,title={STPro: Spatial and Temporal Progressive Learning for Weakly Supervised Spatio-Temporal Grounding},author={Garg, Aaryan and Kumar, Akash and Rawat, Yogesh},booktitle={IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},year={2025},}
Understanding Depth and Height Perception in Large Visual-Language Models
Shehreen Azad, Yash Jain, Rishit Garg, Yogesh S Rawat, and Vibhav Vineet
In IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), 2025
@inproceedings{azad2024geometer,title={Understanding Depth and Height Perception in Large Visual-Language Models},author={Azad, Shehreen and Jain, Yash and Garg, Rishit and Rawat, Yogesh S and Vineet, Vibhav},booktitle={IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)},year={2025},}
iSafetyBench: A video-language benchmark for safety in industrial environment
Raiyaan Abdullah, Yogesh Singh Rawat, and Shruti Vyas
In IEEE/CVF International Conference on Computer Vision Workshops (ICCVW), 2025
@inproceedings{abdullah2025isafetybench,title={iSafetyBench: A video-language benchmark for safety in industrial environment},author={Abdullah, Raiyaan and Rawat, Yogesh Singh and Vyas, Shruti},booktitle={IEEE/CVF International Conference on Computer Vision Workshops (ICCVW)},pages={1433--1442},year={2025},}
Re: Verse-Can Your VLM Read a Manga?
Aaditya Baranwal, Madhav Kataria, Naitik Agrawal, Yogesh S Rawat, and Shruti Vyas
In IEEE/CVF International Conference on Computer Vision Workshops (ICCVW), 2025
@inproceedings{baranwal2025re,title={Re: Verse-Can Your VLM Read a Manga?},author={Baranwal, Aaditya and Kataria, Madhav and Agrawal, Naitik and Rawat, Yogesh S and Vyas, Shruti},booktitle={IEEE/CVF International Conference on Computer Vision Workshops (ICCVW)},pages={3761--3771},year={2025},}
2024
Probing conceptual understanding of large visual-language models
Madeline Schiappa, Raiyaan Abdullah, Shehreen Azad, Jared Claypoole, Michael Cogswell, Ajay Divakaran, and Yogesh Rawat
In IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), 2024
@inproceedings{schiappa2024probing,title={Probing conceptual understanding of large visual-language models},author={Schiappa, Madeline and Abdullah, Raiyaan and Azad, Shehreen and Claypoole, Jared and Cogswell, Michael and Divakaran, Ajay and Rawat, Yogesh},booktitle={IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)},pages={1797--1807},year={2024},}
Navigating Hallucinations for Reasoning of Unintentional Activities
Shresth Grover, Vibhav Vineet, and Yogesh S. Rawat
In Findings of the Association for Computational Linguistics: EMNLP, 2024
@inproceedings{grover2024navigating,title={Navigating Hallucinations for Reasoning of Unintentional Activities},author={Grover, Shresth and Vineet, Vibhav and Rawat, Yogesh S.},booktitle={Findings of the Association for Computational Linguistics: EMNLP},year={2024},}