@article{clapsepleveragingcontrastivepretrained, title = {CLAPSep: Leveraging Contrastive Pre-trained Model for Multi-Modal Query-Conditioned Target Sound Extraction}, author = {Hao Ma and Zhiyuan Peng and Xu Li and Mingjie Shao and Xixin Wu and Ju Liu}, year = {2024}, eprint = {2402.17455}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2402.17455v5}, }