@misc{indiciae025a7648c521, title = {Benchmarking Failures in Tool-Augmented Language Models}, author = {Eduardo TreviƱo and Hugo Contant and James Ngai and Graham Neubig and Zora Zhiruo Wang}, year = {2025}, url = {https://arxiv.org/abs/2503.14227}, note = {Source identifier: 2503.14227} }