@misc{indiciae5c41a51efe32, title = {Unified theory of upper confidence bound policies for bandit problems targeting total reward, maximal reward, and more}, author = {Nobuaki Kikkawa and Hiroshi Ohno}, year = {2024}, url = {https://arxiv.org/abs/2411.00339}, note = {Source identifier: 2411.00339} }