@misc{indiciae6aaa6775cc48, title = {How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework}, author = {Zi Liang and Liantong Yu and Shiyu Zhang and Qingqing Ye and Haibo Hu}, year = {2026}, url = {https://arxiv.org/abs/2507.19219}, note = {Source identifier: 2507.19219} }