@misc{indiciae9e60f0051a26, title = {Scaling Text-to-Image Diffusion Transformers with Representation Autoencoders}, author = {Shengbang Tong and Boyang Zheng and Ziteng Wang and Bingda Tang and Nanye Ma and Ellis Brown and Jihan Yang and Rob Fergus and Yann LeCun and Saining Xie}, year = {2026}, url = {https://arxiv.org/abs/2601.16208}, note = {Source identifier: 2601.16208} }