@article{speechclipselfsupervisedmultitask, title = {SpeechCLIP+: Self-supervised multi-task representation learning for speech via CLIP and speech-image data}, author = {Hsuan-Fu Wang and Yi-Jen Shih and Heng-Jui Chang and Layne Berry and Puyuan Peng and Hung-Yi Lee and Hsin-Min Wang and David Harwath}, year = {2024}, eprint = {2402.06959}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2402.06959v1}, }