@article{enhancingandacceleratinglargelanguage, title = {Enhancing and Accelerating Large Language Models via Instruction-Aware Contextual Compression}, author = {Haowen Hou and Fei Ma and Binwen Bai and Xinxin Zhu and Fei Yu}, year = {2024}, eprint = {2408.15491}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2408.15491v1}, }