@misc{indiciae1715c7376a1a, title = {Hindsight-Anchored Policy Optimization: Learning Through Hindsight with Thompson Sampling-Inspired Adaptive Gating}, author = {Yuning Wu and Ke Wang and Haoran Liu and Chaoqun Jia and Devin Chen and Kai Wei}, year = {2026}, url = {https://arxiv.org/abs/2603.11321}, note = {Source identifier: 2603.11321} }