@misc{indiciae2339ff64f72d, title = {Enhancing Video Representations with Spatiotemporal-Semantic Residual to Mitigate Hallucinations in Video Large Multimodal Models}, author = {Yuansheng Gao and Jinman Zhao and Tong Zhang and Xingguo Xu and Wenbin Xing and Han Bao and Zonghui Wang and Wenzhi Chen}, year = {2026}, url = {https://arxiv.org/abs/2601.22574}, note = {Source identifier: 2601.22574} }