@misc{indiciae1597fbe35ce9, title = {DeepSeek-R1 vs. o3-mini: How Well can Reasoning LLMs Evaluate MT and Summarization?}, author = {Daniil Larionov and Sotaro Takeshita and Ran Zhang and Yanran Chen and Christoph Leiter and Zhipin Wang and Christian Greisinger and Steffen Eger}, year = {2025}, url = {https://arxiv.org/abs/2504.08120}, note = {Source identifier: 2504.08120} }