@article{asynctlsefficientgenerativellminferencew, title = {AsyncTLS: Efficient Generative LLM Inference with Asynchronous Two-level Sparse Attention}, author = {Yuxuan Hu and Jianchao Tan and Jiaqi Zhang and Wen Zan and Pingwei Sun and Yifan Lu and Yerui Sun and Yuchen Xie and Xunliang Cai and Jing Zhang}, year = {2026}, eprint = {2604.07815}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2604.07815}, }