@misc{indiciae49e3a11a2227, title = {Discovering the Gems in Early Layers: Accelerating Long-Context LLMs with 1000x Input Token Reduction}, author = {Zhenmei Shi and Yifei Ming and Xuan-Phi Nguyen and Yingyu Liang and Shafiq Joty}, year = {2024}, url = {https://arxiv.org/abs/2409.17422}, note = {Source identifier: 2409.17422} }