@article{batchllmoptimizinglargebatchedllm, title = {BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching}, author = {Zhen Zheng and Xin Ji and Taosong Fang and Fanghao Zhou and Chuanjie Liu and Gang Peng}, year = {2024}, eprint = {2412.03594}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2412.03594v2}, }