@misc{indiciae987cc8d097f5, title = {Policy Learning from Large Vision-Language Model Feedback without Reward Modeling}, author = {Tung M. Luu and Donghoon Lee and Younghwan Lee and Chang D. Yoo}, year = {2025}, url = {https://arxiv.org/abs/2507.23391}, note = {Source identifier: 2507.23391} }