@misc{indiciae4f1ef3353590, title = {Efficient Training of Language Models with Compact and Consistent Next Token Distributions}, author = {Ashutosh Sathe and Sunita Sarawagi}, year = {2024}, url = {https://arxiv.org/abs/2407.02819}, note = {Source identifier: 2407.02819} }