@misc{indiciae66990faaf958, title = {VLRM: Vision-Language Models act as Reward Models for Image Captioning}, author = {Maksim Dzabraev and Alexander Kunitsyn and Andrei Ivaniuta}, year = {2024}, url = {https://arxiv.org/abs/2404.01911}, note = {Source identifier: 2404.01911} }