@misc{indiciae25f3534ddc57, title = {Benchmarking LLM Judges for Voice-Agent Evaluation: Reliability, Calibration, and Human Oversight}, author = {Anupam Purwar and Shashank Singh and Kritika Srivastava}, year = {2026}, url = {https://arxiv.org/abs/2608.24314}, note = {Source identifier: 2608.24314} }