@misc{indiciaedd3d19eef422, title = {Linear attention is (maybe) all you need (to understand transformer optimization)}, author = {Kwangjun Ahn and Xiang Cheng and Minhak Song and Chulhee Yun and Ali Jadbabaie and Suvrit Sra}, year = {2024}, url = {https://arxiv.org/abs/2310.01082}, note = {Source identifier: 2310.01082} }