@misc{indiciaeafe3d8401823, title = {SciEx: Benchmarking Large Language Models on Scientific Exams with Human Expert Grading and Automatic Grading}, author = {Tu Anh Dinh and Carlos Mullov and Leonard Bärmann and Zhaolin Li and Danni Liu and Simon Reiß and Jueun Lee and Nathan Lerzer and Fabian Ternava and Jianfeng Gao and Tobias Röddiger and Alexander Waibel and Tamim Asfour and Michael Beigl and Rainer Stiefelhagen and Carsten Dachsbacher and Klemens Böhm and Jan Niehues}, year = {2024}, url = {https://arxiv.org/abs/2406.10421}, note = {Source identifier: 2406.10421} }