@misc{indiciae9a333a8ff357, title = {Off-policy Maximum Entropy Reinforcement Learning : Soft Actor-Critic with Advantage Weighted Mixture Policy(SAC-AWMP)}, author = {Zhimin Hou and Kuangen Zhang and Yi Wan and Dongyu Li and Chenglong Fu and Haoyong Yu}, year = {2020}, url = {https://arxiv.org/abs/2002.02829}, note = {Source identifier: 2002.02829} }