@misc{indiciae48c0241c27b2, title = {GPT Semantic Cache: Reducing LLM Costs and Latency via Semantic Embedding Caching}, author = {Sajal Regmi and Chetan Phakami Pun}, year = {2024}, url = {https://arxiv.org/abs/2411.05276}, note = {Source identifier: 2411.05276} }