@misc{indiciae270e68ad9aab, title = {Reward Evolution with Graph-of-Thoughts: A Bi-Level Language Model Framework for Reinforcement Learning}, author = {Changwei Yao and Xinzi Liu and Chen Li and Marios Savvides}, year = {2026}, url = {https://arxiv.org/abs/2509.16136}, note = {Source identifier: 2509.16136} }