@misc{indiciae53dfb58d5efe, title = {Uni-O4: Unifying Online and Offline Deep Reinforcement Learning with Multi-Step On-Policy Optimization}, author = {Kun Lei and Zhengmao He and Chenhao Lu and Kaizhe Hu and Yang Gao and Huazhe Xu}, year = {2024}, url = {https://arxiv.org/abs/2311.03351}, note = {Source identifier: 2311.03351} }