@misc{indiciae985b6eb943b4, title = {Language-driven Description Generation and Common Sense Reasoning for Video Action Recognition}, author = {Xiaodan Hu and Chuhang Zou and Suchen Wang and Jaechul Kim and Narendra Ahuja}, year = {2025}, url = {https://arxiv.org/abs/2506.16701}, note = {Source identifier: 2506.16701} }