@article{infmllmefficientstreaminginferenceof, title = {Inf-MLLM: Efficient Streaming Inference of Multimodal Large Language Models on a Single GPU}, author = {Zhenyu Ning and Jieru Zhao and Qihao Jin and Wenchao Ding and Minyi Guo}, year = {2024}, eprint = {2409.09086}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2409.09086v1}, }