@misc{indiciae89e9cc0c5396, title = {One Step of Gradient Descent is Provably the Optimal In-Context Learner with One Layer of Linear Self-Attention}, author = {Arvind Mahankali and Tatsunori B. Hashimoto and Tengyu Ma}, year = {2023}, url = {https://arxiv.org/abs/2307.03576}, note = {Source identifier: 2307.03576} }