@misc{indiciae493764b5bde7, title = {On the model-based stochastic value gradient for continuous reinforcement learning}, author = {Brandon Amos and Samuel Stanton and Denis Yarats and Andrew Gordon Wilson}, year = {2021}, url = {https://arxiv.org/abs/2008.12775}, note = {Source identifier: 2008.12775} }