@misc{indiciae04a892cddb24, title = {Conservative Q-Improvement: Reinforcement Learning for an Interpretable Decision-Tree Policy}, author = {Aaron M. Roth and Nicholay Topin and Pooyan Jamshidi and Manuela Veloso}, year = {2019}, url = {https://arxiv.org/abs/1907.01180}, note = {Source identifier: 1907.01180} }