@misc{indiciae7eb2841ed59b, title = {Examining the robustness of LLM evaluation to the distributional assumptions of benchmarks}, author = {Melissa Ailem and Katerina Marazopoulou and Charlotte Siska and James Bono}, year = {2024}, url = {https://arxiv.org/abs/2404.16966}, note = {Source identifier: 2404.16966} }