@misc{indiciaec3fa11a267eb, title = {Q-Learning Lagrange Policies for Multi-Action Restless Bandits}, author = {Jackson A. Killian and Arpita Biswas and Sanket Shah and Milind Tambe}, year = {2021}, doi = {10.1145/3447548.3467370}, url = {https://arxiv.org/abs/2106.12024}, note = {Source identifier: 2106.12024} }