@misc{indiciae40764b6bcc20, title = {TD-Grokking: Learning from Zero-Reward Problems by Training-Time Decomposition}, author = {Ningyuan Xi and Hao Xu and Hongsheng Xin and Ning Miao}, year = {2026}, url = {https://arxiv.org/abs/2606.09883}, note = {Source identifier: 2606.09883} }