@misc{indiciae5ae2d192b0fe, title = {How Transformers Utilize Multi-Head Attention in In-Context Learning? A Case Study on Sparse Linear Regression}, author = {Xingwu Chen and Lei Zhao and Difan Zou}, year = {2024}, url = {https://arxiv.org/abs/2408.04532}, note = {Source identifier: 2408.04532} }