@article{vqlogitscompressingtheoutputbottleneck, title = {VQ-Logits: Compressing the Output Bottleneck of Large Language Models via Vector Quantized Logits}, author = {Jintian Shao and Hongyi Huang and Jiayi Wu and Yiming Cheng and Zhiyu Wu and You Shan and Mingkai Zheng}, year = {2025}, eprint = {2505.10202}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2505.10202v1}, }