@misc{indiciae86b5abb998ac, title = {Image2Struct: Benchmarking Structure Extraction for Vision-Language Models}, author = {Josselin Somerville Roberts and Tony Lee and Chi Heem Wong and Michihiro Yasunaga and Yifan Mai and Percy Liang}, year = {2024}, url = {https://arxiv.org/abs/2410.22456}, note = {Source identifier: 2410.22456} }