@misc{indiciae2cf2378bc760, title = {Why Larger Models Learn More: Effects of Capacity, Interference, and Rare-Task Retention}, author = {Jing Huang and Daniel Wurgaft and Rachit Bansal and Laura Ruis and Naomi Saphra and David Alvarez-Melis and Andrew Kyle Lampinen and Christopher Potts and Ekdeep Singh Lubana}, year = {2026}, url = {https://arxiv.org/abs/2605.29548}, note = {Source identifier: 2605.29548} }