@misc{indiciaeb38e56fe59f6, title = {Benchmarking and Mechanistic Analysis of Vision-Language Models for Cross-Depiction Assembly Instruction Alignment}, author = {Zhuchenyang Liu and Yao Zhang and Yu Xiao}, year = {2026}, url = {https://arxiv.org/abs/2604.00913}, note = {Source identifier: 2604.00913} }