@misc{indiciae8b76ca2c333a, title = {Visualising Policy-Reward Interplay to Inform Zeroth-Order Preference Optimisation of Large Language Models}, author = {Alessio Galatolo and Zhenbang Dai and Katie Winkle and Meriem Beloucif}, year = {2025}, url = {https://arxiv.org/abs/2503.03460}, note = {Source identifier: 2503.03460} }