@article{bestrq2contextualizethenpredictatwostepa, title = {BEST-RQ-2: Contextualize-Then-Predict, a Two-Step Approach for Self-Supervised Audio Representations}, author = {Ludovic K. Tuncay and Etienne Labbé and Thomas Pellegrini}, year = {2026}, eprint = {2606.30700}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2606.30700}, }