@misc{indiciaef8a9f0ca8671, title = {Scene-Graph ViT: End-to-End Open-Vocabulary Visual Relationship Detection}, author = {Tim Salzmann and Markus Ryll and Alex Bewley and Matthias Minderer}, year = {2024}, url = {https://arxiv.org/abs/2403.14270}, note = {Source identifier: 2403.14270} }