@misc{indiciae9c3f6bb78828, title = {Exploiting Inter-Layer Expert Affinity for Accelerating Mixture-of-Experts Model Inference}, author = {Jinghan Yao and Quentin Anthony and Aamir Shafi and Hari Subramoni and Dhabaleswar K. and Panda}, year = {2024}, url = {https://arxiv.org/abs/2401.08383}, note = {Source identifier: 2401.08383} }