@misc{indiciae3f1f22c99036, title = {Softmax \$\textbackslash{}geq\$ Linear: Transformers may learn to classify in-context by kernel gradient descent}, author = {Sara Dragutinović and Andrew M. Saxe and Aaditya K. Singh}, year = {2025}, url = {https://arxiv.org/abs/2510.10425}, note = {Source identifier: 2510.10425} }