@misc{indiciae0922442f2457, title = {A Tool for Benchmarking Large Language Models' Robustness in Assessing the Realism of Driving Scenarios}, author = {Jiahui Wu and Chengjie Lu and Aitor Arrieta and Shaukat Ali}, year = {2025}, url = {https://arxiv.org/abs/2511.04267}, note = {Source identifier: 2511.04267} }