@misc{indiciae2d6b02b8bd92, title = {Vision-Language Meets the Skeleton: Progressively Distillation with Cross-Modal Knowledge for 3D Action Representation Learning}, author = {Yang Chen and Tian He and Junfeng Fu and Ling Wang and Jingcai Guo and Ting Hu and Hong Cheng}, year = {2024}, url = {https://arxiv.org/abs/2405.20606}, note = {Source identifier: 2405.20606} }