@misc{indiciae13c614aafd89, title = {Test-time reward-guided alignment of language models by importance sampling on pre-logit space}, author = {Sekitoshi Kanai and Tsukasa Yoshida and Hiroshi Takahashi and Haru Kuroki and Kazumune Hashimoto}, year = {2026}, url = {https://arxiv.org/abs/2510.26219}, note = {Source identifier: 2510.26219} }