@misc{indiciae1b990eeb8641, title = {LEMON: How Well Do MLLMs Perform Temporal Multimodal Understanding on Instructional Videos?}, author = {Zhuang Yu and Lei Shen and Jing Zhao and Shiliang Sun}, year = {2026}, url = {https://arxiv.org/abs/2601.20705}, note = {Source identifier: 2601.20705} }