@misc{indiciaef4542bc089ab, title = {Can Large Language Models be Trusted for Evaluation? Scalable Meta-Evaluation of LLMs as Evaluators via Agent Debate}, author = {Steffi Chern and Ethan Chern and Graham Neubig and Pengfei Liu}, year = {2024}, url = {https://arxiv.org/abs/2401.16788}, note = {Source identifier: 2401.16788} }