@article{prefixingattentionsinkscanmitigate, title = {Prefixing Attention Sinks can Mitigate Activation Outliers for Large Language Model Quantization}, author = {Seungwoo Son and Wonpyo Park and Woohyun Han and Kyuyeun Kim and Jaeho Lee}, year = {2024}, eprint = {2406.12016}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2406.12016v2}, }