@misc{indiciaec39369fcc52e, title = {Evaluation Metrics in the Era of GPT-4: Reliably Evaluating Large Language Models on Sequence to Sequence Tasks}, author = {Andrea Sottana and Bin Liang and Kai Zou and Zheng Yuan}, year = {2023}, url = {https://arxiv.org/abs/2310.13800}, note = {Source identifier: 2310.13800} }