@article{flashcommunicationreducingtensor, title = {Flash Communication: Reducing Tensor Parallelization Bottleneck for Fast Large Language Model Inference}, author = {Qingyuan Li and Bo Zhang and Liang Ye and Yifan Zhang and Wei Wu and Yerui Sun and Lin Ma and Yuchen Xie}, year = {2024}, eprint = {2412.04964}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2412.04964v2}, }