@misc{indiciaeba0a7fa1debd, title = {Do Benchmarks Underestimate LLM Performance? Evaluating Hallucination Detection With LLM-First Human-Adjudicated Assessment}, author = {I. F. Atasoy and B. Mutlu and E. A. Sezer and A. Wahdan}, year = {2026}, url = {https://arxiv.org/abs/2605.08462}, note = {Source identifier: 2605.08462} }