@misc{indiciaec8e1398718e2, title = {VARP: Reinforcement Learning from Vision-Language Model Feedback with Agent Regularized Preferences}, author = {Anukriti Singh and Amisha Bhaskar and Peihong Yu and Souradip Chakraborty and Ruthwik Dasyam and Amrit Bedi and Pratap Tokekar}, year = {2025}, url = {https://arxiv.org/abs/2503.13817}, note = {Source identifier: 2503.13817} }