@article{cacheawarepromptcompressionatwotiercostm, title = {Cache-Aware Prompt Compression:A Two-Tier Cost Model for LLM API Caching}, author = {Yan Song}, year = {2026}, eprint = {2607.15516}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2607.15516}, }