@misc{indiciae5bcc3816ceed, title = {SciCode-Verified: How Benchmark Defects Underestimated the Scientific-Coding Ability of Language Models}, author = {Sihan Hu and Lyuhan Huang and Youjin Deng and Kun Chen}, year = {2026}, url = {https://arxiv.org/abs/2608.04975}, note = {Source identifier: 2608.04975} }