@misc{indiciae96f48ce969e0, title = {Compositional Context Fine-Tuning Vision-Language Model for Complex Assembly Action Understanding from Videos}, author = {Hao Zheng and Jinyi Huang and Tiantian Zheng and Xun Xu and Tuka Alhanai}, year = {2026}, url = {https://arxiv.org/abs/2607.10797}, note = {Source identifier: 2607.10797} }