@misc{indiciae36da1043fe22, title = {LATTE: Low-Precision Approximate Attention with Head-wise Trainable Threshold for Efficient Transformer}, author = {Jiing-Ping Wang and Ming-Guang Lin and An-Yeu and Wu}, year = {2024}, url = {https://arxiv.org/abs/2404.07519}, note = {Source identifier: 2404.07519} }