@misc{indiciae5caa673a9f81, title = {Duplex: A Device for Large Language Models with Mixture of Experts, Grouped Query Attention, and Continuous Batching}, author = {Sungmin Yun and Kwanhee Kyung and Juhwan Cho and Jaewan Choi and Jongmin Kim and Byeongho Kim and Sukhan Lee and Kyomin Sohn and Jung Ho Ahn}, year = {2024}, doi = {10.1109/micro61859.2024.00105}, url = {https://arxiv.org/abs/2409.01141}, note = {Source identifier: 2409.01141} }