@misc{indiciae8db8e9e195f5, title = {Multi-objective Reinforcement Learning with Nonlinear Preferences: Provable Approximation for Maximizing Expected Scalarized Return}, author = {Nianli Peng and Muhang Tian and Brandon Fain}, year = {2025}, url = {https://arxiv.org/abs/2311.02544}, note = {Source identifier: 2311.02544} }