@article{batchmaxhigherllmthroughputusinglarger, title = {Batch-Max: Higher LLM Throughput using Larger Batch Sizes and KV Cache Compression}, author = {Michael R. Metel and Boxing Chen and Mehdi Rezagholizadeh}, year = {2024}, eprint = {2412.05693}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2412.05693v1}, }