@misc{indiciae789248b4edca, title = {T-REN: Learning Text-Aligned Region Tokens Improves Dense Vision-Language Alignment and Scalability}, author = {Savya Khosla and Sethuraman T V and Aryan Chadha and Alex Schwing and Derek Hoiem}, year = {2026}, url = {https://arxiv.org/abs/2604.18573}, note = {Source identifier: 2604.18573} }