@misc{indiciae9c749a9385e4, title = {Architecting Long-Context LLM Acceleration with Packing-Prefetch Scheduler and Ultra-Large Capacity On-Chip Memories}, author = {Ming-Yen Lee and Faaiq Waqar and Hanchen Yang and Muhammed Ahosan Ul Karim and Harsono Simka and Shimeng Yu}, year = {2025}, url = {https://arxiv.org/abs/2508.08457}, note = {Source identifier: 2508.08457} }