@misc{indiciae1b2c8aba8ddc, title = {Small Batch Size Training for Language Models: When Vanilla SGD Works, and Why Gradient Accumulation Is Wasteful}, author = {Martin Marek and Sanae Lotfi and Aditya Somasundaram and Andrew Gordon Wilson and Micah Goldblum}, year = {2025}, url = {https://arxiv.org/abs/2507.07101}, note = {Source identifier: 2507.07101} }