@misc{indiciae8b23b230d874, title = {MMLU-SR: A Benchmark for Stress-Testing Reasoning Capability of Large Language Models}, author = {Wentian Wang and Sarthak Jain and Paul Kantor and Jacob Feldman and Lazaros Gallos and Hao Wang}, year = {2024}, url = {https://arxiv.org/abs/2406.15468}, note = {Source identifier: 2406.15468} }