@misc{indiciaea010564e83c6, title = {Statistically Efficient Variance Reduction with Double Policy Estimation for Off-Policy Evaluation in Sequence-Modeled Reinforcement Learning}, author = {Hanhan Zhou and Tian Lan and Vaneet Aggarwal}, year = {2023}, url = {https://arxiv.org/abs/2308.14897}, note = {Source identifier: 2308.14897} }