@misc{indiciaeee41e7c99fee, title = {Is Random Attention Sufficient for Sequence Modeling? Disentangling Trainable Components in the Transformer}, author = {Yihe Dong and Lorenzo Noci and Mikhail Khodak and Mufan Li}, year = {2025}, url = {https://arxiv.org/abs/2506.01115}, note = {Source identifier: 2506.01115} }