@misc{indiciaee1a67d3b9ce8, title = {Satori: Reinforcement Learning with Chain-of-Action-Thought Enhances LLM Reasoning via Autoregressive Search}, author = {Maohao Shen and Guangtao Zeng and Zhenting Qi and Zhang-Wei Hong and Zhenfang Chen and Wei Lu and Gregory Wornell and Subhro Das and David Cox and Chuang Gan}, year = {2025}, url = {https://arxiv.org/abs/2502.02508}, note = {Source identifier: 2502.02508} }