@misc{indiciae1ed456a16a66, title = {VinQA: Visual Elements Interleaved Long-form Answer Generation for Real-World Multimodal Document QA}, author = {Young Rok Jang and Hyesoo Kong and Kyunghwan An and Jae Sub Huh and Gyeonghun Kim and Stanley Jungkyu Choi}, year = {2026}, url = {https://arxiv.org/abs/2606.16092}, note = {Source identifier: 2606.16092} }