@misc{indiciaee2a19ba9305c, title = {From Bandits Model to Deep Deterministic Policy Gradient, Reinforcement Learning with Contextual Information}, author = {Zhendong Shi and Xiaoli Wei and Ercan E. Kuruoglu}, year = {2023}, url = {https://arxiv.org/abs/2310.00642}, note = {Source identifier: 2310.00642} }