@misc{indiciae46896b8d0bb5, title = {Efficient INT8 Inference of Small NLP Models on Server CPUs with PyTorch Native Stack}, author = {Weiwen Xia and Yuxin Cui and E Cao}, year = {2026}, url = {https://arxiv.org/abs/2608.18182}, note = {Source identifier: 2608.18182} }