@misc{indiciaeaf08782a066e, title = {How Ensembles of Distilled Policies Improve Generalisation in Reinforcement Learning}, author = {Max Weltevrede and Moritz A. Zanger and Matthijs T. J. Spaan and Wendelin Böhmer}, year = {2025}, url = {https://arxiv.org/abs/2505.16581}, note = {Source identifier: 2505.16581} }