@misc{indiciae247c34dc5769, title = {CLAP: Contrastive Latent Action Pretraining for Learning Vision-Language-Action Models from Human Videos}, author = {Chubin Zhang and Jianan Wang and Zifeng Gao and Yue Su and Tianru Dai and Cai Zhou and Jiwen Lu and Yansong Tang}, year = {2026}, url = {https://arxiv.org/abs/2601.04061}, note = {Source identifier: 2601.04061} }