@article{softmaxgeqlineartransformersmaylearntocl, title = {Softmax $\geq$ Linear: Transformers may learn to classify in-context by kernel gradient descent}, author = {Sara Dragutinović and Andrew M. Saxe and Aaditya K. Singh}, year = {2025}, eprint = {2510.10425}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2510.10425}, }