@misc{indiciae21e70876ae96, title = {Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor}, author = {Tuomas Haarnoja and Aurick Zhou and Pieter Abbeel and Sergey Levine}, year = {2018}, url = {https://arxiv.org/abs/1801.01290}, note = {Source identifier: 1801.01290} }