Publication / poster / sc24 poster / 2024
Uncover the Overhead and Resource Usage for Handling KV Cache Overflow in LLM Inference
The International Conference for High Performance Computing, Networking, Storage, and Analysis (SC’24)
Citation
@misc{ye2024uncoveroverhead,
grc_key = {ye2024uncoveroverhead},
grc_slug = {ye-2024-uncover-overhead-3814},
author = {Ye, J. and Nicolae, B. and Kougkas, A. and Sun, X.-H.},
title = {{Uncover the Overhead and Resource Usage for Handling KV Cache Overflow in LLM Inference}},
howpublished = {The International Conference for High Performance Computing, Networking, Storage, and Analysis (SC'24)},
year = {2024},
month = nov,
url = {http://cs.iit.edu/%7Escs/assets/files/ye2024kvcache_poster.pdf},
source_url = {http://cs.iit.edu/%7Escs/assets/files/ye2024kvcache_poster.pdf},
keywords = {KV Cache, LLM Inference},
}