{"task":"Text-To-Speech Synthesis","dataset":"LJSpeech","metric_names":["Audio Quality MOS","Pleasantness MOS","Word Error Rate (WER)","MOS","WER (%)"],"rows":[{"id":8212,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"NaturalSpeech","metrics":{"Audio Quality MOS":"4.56"},"paper_url":"https://arxiv.org/abs/2205.04421v2","paper_title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","paper_date":"2022-05-09","code_links":[{"title":"microsoft/NeuralSpeech","url":"https://github.com/microsoft/NeuralSpeech"},{"title":"daniilrobnikov/vits2","url":"https://github.com/daniilrobnikov/vits2"},{"title":"heatz123/naturalspeech","url":"https://github.com/heatz123/naturalspeech"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8213,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"VITS","metrics":{"Audio Quality MOS":"4.43"},"paper_url":"https://arxiv.org/abs/2205.04421v2","paper_title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","paper_date":"2022-05-09","code_links":[{"title":"microsoft/NeuralSpeech","url":"https://github.com/microsoft/NeuralSpeech"},{"title":"daniilrobnikov/vits2","url":"https://github.com/daniilrobnikov/vits2"},{"title":"heatz123/naturalspeech","url":"https://github.com/heatz123/naturalspeech"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8214,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Grad-TTS + HiFiGAN (1000 steps)","metrics":{"Audio Quality MOS":"4.37"},"paper_url":"https://arxiv.org/abs/2105.06337v2","paper_title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","paper_date":"2021-05-13","code_links":[{"title":"huawei-noah/Speech-Backbones","url":"https://github.com/huawei-noah/Speech-Backbones"},{"title":"keonlee9420/DiffGAN-TTS","url":"https://github.com/keonlee9420/DiffGAN-TTS"},{"title":"keonlee9420/DiffSinger","url":"https://github.com/keonlee9420/DiffSinger"},{"title":"WelkinYang/GradTTS","url":"https://github.com/WelkinYang/GradTTS"},{"title":"playvoice/grad-svc","url":"https://github.com/playvoice/grad-svc"},{"title":"majidAdibian77/ResGrad","url":"https://github.com/majidAdibian77/ResGrad"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8215,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Glow-TTS + HiFiGAN","metrics":{"Audio Quality MOS":"4.34"},"paper_url":"https://arxiv.org/abs/2005.11129v1","paper_title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","paper_date":"2020-05-22","code_links":[{"title":"coqui-ai/TTS","url":"https://github.com/coqui-ai/TTS"},{"title":"jaywalnut310/glow-tts","url":"https://github.com/jaywalnut310/glow-tts"},{"title":"supertone-inc/super-monotonic-align","url":"https://github.com/supertone-inc/super-monotonic-align"},{"title":"keonlee9420/VAENAR-TTS","url":"https://github.com/keonlee9420/VAENAR-TTS"},{"title":"ankurdhuriya/multispeaker-glow-tts","url":"https://github.com/ankurdhuriya/multispeaker-glow-tts"},{"title":"revsic/tf-glow-tts","url":"https://github.com/revsic/tf-glow-tts"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8216,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"FastSpeech 2 + HiFiGAN","metrics":{"Audio Quality MOS":"4.34"},"paper_url":"https://arxiv.org/abs/2205.04421v2","paper_title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","paper_date":"2022-05-09","code_links":[{"title":"microsoft/NeuralSpeech","url":"https://github.com/microsoft/NeuralSpeech"},{"title":"daniilrobnikov/vits2","url":"https://github.com/daniilrobnikov/vits2"},{"title":"heatz123/naturalspeech","url":"https://github.com/heatz123/naturalspeech"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8217,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"FastSpeech 2 + HiFiGAN","metrics":{"Audio Quality MOS":"4.32"},"paper_url":"https://arxiv.org/abs/2006.04558v8","paper_title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","paper_date":"2020-06-08","code_links":[{"title":"coqui-ai/TTS","url":"https://github.com/coqui-ai/TTS"},{"title":"PaddlePaddle/PaddleSpeech","url":"https://github.com/PaddlePaddle/PaddleSpeech"},{"title":"TensorSpeech/TensorflowTTS","url":"https://github.com/TensorSpeech/TensorflowTTS"},{"title":"ming024/FastSpeech2","url":"https://github.com/ming024/FastSpeech2"},{"title":"as-ideas/TransformerTTS","url":"https://github.com/as-ideas/TransformerTTS"},{"title":"xcmyz/FastSpeech","url":"https://github.com/xcmyz/FastSpeech"},{"title":"keonlee9420/PortaSpeech","url":"https://github.com/keonlee9420/PortaSpeech"},{"title":"keonlee9420/Comprehensive-Transformer-TTS","url":"https://github.com/keonlee9420/Comprehensive-Transformer-TTS"},{"title":"keonlee9420/Expressive-FastSpeech2","url":"https://github.com/keonlee9420/Expressive-FastSpeech2"},{"title":"KevinMIN95/StyleSpeech","url":"https://github.com/KevinMIN95/StyleSpeech"},{"title":"keonlee9420/DiffSinger","url":"https://github.com/keonlee9420/DiffSinger"},{"title":"rishikksh20/FastSpeech2","url":"https://github.com/rishikksh20/FastSpeech2"},{"title":"keonlee9420/StyleSpeech","url":"https://github.com/keonlee9420/StyleSpeech"},{"title":"keonlee9420/STYLER","url":"https://github.com/keonlee9420/STYLER"},{"title":"keonlee9420/Comprehensive-E2E-TTS","url":"https://github.com/keonlee9420/Comprehensive-E2E-TTS"},{"title":"galaxycong/hpmdubbing","url":"https://github.com/galaxycong/hpmdubbing"},{"title":"ga642381/FastSpeech2","url":"https://github.com/ga642381/FastSpeech2"},{"title":"rishikksh20/LightSpeech","url":"https://github.com/rishikksh20/LightSpeech"},{"title":"ai-unicamp/tts-objective-metrics","url":"https://github.com/ai-unicamp/tts-objective-metrics"},{"title":"komyeongjin/specdiff-gan","url":"https://github.com/komyeongjin/specdiff-gan"},{"title":"wataru-nakata/fastspeech2-jsut","url":"https://github.com/wataru-nakata/fastspeech2-jsut"},{"title":"shivammehta25/BetterFastSpeech2","url":"https://github.com/shivammehta25/BetterFastSpeech2"},{"title":"roedoejet/fastspeech2","url":"https://github.com/roedoejet/fastspeech2"},{"title":"roedoejet/fastspeech2_acl2022_reproducibility","url":"https://github.com/roedoejet/fastspeech2_acl2022_reproducibility"},{"title":"majidAdibian77/ResGrad","url":"https://github.com/majidAdibian77/ResGrad"},{"title":"dathudeptrai/TensorflowTTS","url":"https://github.com/dathudeptrai/TensorflowTTS"},{"title":"yoshifumi-nakano/visual-text-to-speech","url":"https://github.com/yoshifumi-nakano/visual-text-to-speech"},{"title":"RayeRen/RayeRen","url":"https://github.com/RayeRen/RayeRen"},{"title":"ndkgit339/fastspeech2-filled_pause_speech_synthesis","url":"https://github.com/ndkgit339/fastspeech2-filled_pause_speech_synthesis"},{"title":"tartunlp/transformertts","url":"https://github.com/tartunlp/transformertts"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/fastspeech2_conformer"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-2/fastspeech2_conformer"},{"title":"zhangbo2008/fastSpeeck2_chinese_train","url":"https://github.com/zhangbo2008/fastSpeeck2_chinese_train"},{"title":"Munna-Manoj/Team6_FastSpeech2_TTS","url":"https://github.com/Munna-Manoj/Team6_FastSpeech2_TTS"},{"title":"mtresearcher/FastSpeech2","url":"https://github.com/mtresearcher/FastSpeech2"},{"title":"cadia-lvl/fastspeech2","url":"https://github.com/cadia-lvl/fastspeech2"},{"title":"OlaWod/my-fastspeech2","url":"https://github.com/OlaWod/my-fastspeech2"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8218,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"FastDiff (4 steps)","metrics":{"Audio Quality MOS":"4.28"},"paper_url":"https://arxiv.org/abs/2204.09934v1","paper_title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","paper_date":"2022-04-21","code_links":[{"title":"Rongjiehuang/ProDiff","url":"https://github.com/Rongjiehuang/ProDiff"},{"title":"Rongjiehuang/FastDiff","url":"https://github.com/Rongjiehuang/FastDiff"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8219,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"FastDiff-TTS","metrics":{"Audio Quality MOS":"4.03"},"paper_url":"https://arxiv.org/abs/2204.09934v1","paper_title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","paper_date":"2022-04-21","code_links":[{"title":"Rongjiehuang/ProDiff","url":"https://github.com/Rongjiehuang/ProDiff"},{"title":"Rongjiehuang/FastDiff","url":"https://github.com/Rongjiehuang/FastDiff"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8220,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Transformer TTS (Mel + WaveGlow)","metrics":{"Audio Quality MOS":"3.88"},"paper_url":"http://arxiv.org/abs/1809.08895v3","paper_title":"Neural Speech Synthesis with Transformer Network","paper_date":"2018-09-19","code_links":[{"title":"PaddlePaddle/PaddleSpeech","url":"https://github.com/PaddlePaddle/PaddleSpeech"},{"title":"as-ideas/TransformerTTS","url":"https://github.com/as-ideas/TransformerTTS"},{"title":"soobinseo/transformer-tts","url":"https://github.com/soobinseo/transformer-tts"},{"title":"choiHkk/Transformer-TTS","url":"https://github.com/choiHkk/Transformer-TTS"},{"title":"tartunlp/transformertts","url":"https://github.com/tartunlp/transformertts"},{"title":"Munna-Manoj/Team6_FastSpeech2_TTS","url":"https://github.com/Munna-Manoj/Team6_FastSpeech2_TTS"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8221,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"FastSpeech (Mel + WaveGlow)","metrics":{"Audio Quality MOS":"3.84"},"paper_url":"https://arxiv.org/abs/1905.09263v5","paper_title":"FastSpeech: Fast, Robust and Controllable Text to Speech","paper_date":"2019-05-22","code_links":[{"title":"coqui-ai/TTS","url":"https://github.com/coqui-ai/TTS"},{"title":"PaddlePaddle/PaddleSpeech","url":"https://github.com/PaddlePaddle/PaddleSpeech"},{"title":"ming024/FastSpeech2","url":"https://github.com/ming024/FastSpeech2"},{"title":"as-ideas/TransformerTTS","url":"https://github.com/as-ideas/TransformerTTS"},{"title":"xcmyz/FastSpeech","url":"https://github.com/xcmyz/FastSpeech"},{"title":"kdaip/stabletts","url":"https://github.com/kdaip/stabletts"},{"title":"keonlee9420/PortaSpeech","url":"https://github.com/keonlee9420/PortaSpeech"},{"title":"keonlee9420/Comprehensive-Transformer-TTS","url":"https://github.com/keonlee9420/Comprehensive-Transformer-TTS"},{"title":"keonlee9420/Expressive-FastSpeech2","url":"https://github.com/keonlee9420/Expressive-FastSpeech2"},{"title":"rishikksh20/FastSpeech2","url":"https://github.com/rishikksh20/FastSpeech2"},{"title":"keonlee9420/StyleSpeech","url":"https://github.com/keonlee9420/StyleSpeech"},{"title":"keonlee9420/STYLER","url":"https://github.com/keonlee9420/STYLER"},{"title":"ga642381/FastSpeech2","url":"https://github.com/ga642381/FastSpeech2"},{"title":"rishikksh20/LightSpeech","url":"https://github.com/rishikksh20/LightSpeech"},{"title":"as-ideas/deepforcedaligner","url":"https://github.com/as-ideas/deepforcedaligner"},{"title":"dathudeptrai/TensorflowTTS","url":"https://github.com/dathudeptrai/TensorflowTTS"},{"title":"jkyunnng/happyquokka_system_for_eeg_challenge","url":"https://github.com/jkyunnng/happyquokka_system_for_eeg_challenge"},{"title":"tartunlp/transformertts","url":"https://github.com/tartunlp/transformertts"},{"title":"bloodraven66/deepforcedaligner","url":"https://github.com/bloodraven66/deepforcedaligner"},{"title":"erasedwalt/FastSpeech","url":"https://github.com/erasedwalt/FastSpeech"},{"title":"2023-MindSpore-1/ms-code-15","url":"https://github.com/2023-MindSpore-1/ms-code-15/tree/main/FastSpeech"},{"title":"cadia-lvl/fastspeech2","url":"https://github.com/cadia-lvl/fastspeech2"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8222,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"OverFlow","metrics":{"Audio Quality MOS":"3.37","Word Error Rate (WER)":"2.30"},"paper_url":"https://arxiv.org/abs/2211.06892v2","paper_title":"OverFlow: Putting flows on top of neural transducers for better TTS","paper_date":"2022-11-13","code_links":[{"title":"coqui-ai/TTS","url":"https://github.com/coqui-ai/TTS"},{"title":"shivammehta25/OverFlow","url":"https://github.com/shivammehta25/OverFlow"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8223,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Merlin","metrics":{"Audio Quality MOS":"2.4"},"paper_url":"https://arxiv.org/abs/1905.09263v5","paper_title":"FastSpeech: Fast, Robust and Controllable Text to Speech","paper_date":"2019-05-22","code_links":[{"title":"coqui-ai/TTS","url":"https://github.com/coqui-ai/TTS"},{"title":"PaddlePaddle/PaddleSpeech","url":"https://github.com/PaddlePaddle/PaddleSpeech"},{"title":"ming024/FastSpeech2","url":"https://github.com/ming024/FastSpeech2"},{"title":"as-ideas/TransformerTTS","url":"https://github.com/as-ideas/TransformerTTS"},{"title":"xcmyz/FastSpeech","url":"https://github.com/xcmyz/FastSpeech"},{"title":"kdaip/stabletts","url":"https://github.com/kdaip/stabletts"},{"title":"keonlee9420/PortaSpeech","url":"https://github.com/keonlee9420/PortaSpeech"},{"title":"keonlee9420/Comprehensive-Transformer-TTS","url":"https://github.com/keonlee9420/Comprehensive-Transformer-TTS"},{"title":"keonlee9420/Expressive-FastSpeech2","url":"https://github.com/keonlee9420/Expressive-FastSpeech2"},{"title":"rishikksh20/FastSpeech2","url":"https://github.com/rishikksh20/FastSpeech2"},{"title":"keonlee9420/StyleSpeech","url":"https://github.com/keonlee9420/StyleSpeech"},{"title":"keonlee9420/STYLER","url":"https://github.com/keonlee9420/STYLER"},{"title":"ga642381/FastSpeech2","url":"https://github.com/ga642381/FastSpeech2"},{"title":"rishikksh20/LightSpeech","url":"https://github.com/rishikksh20/LightSpeech"},{"title":"as-ideas/deepforcedaligner","url":"https://github.com/as-ideas/deepforcedaligner"},{"title":"dathudeptrai/TensorflowTTS","url":"https://github.com/dathudeptrai/TensorflowTTS"},{"title":"jkyunnng/happyquokka_system_for_eeg_challenge","url":"https://github.com/jkyunnng/happyquokka_system_for_eeg_challenge"},{"title":"tartunlp/transformertts","url":"https://github.com/tartunlp/transformertts"},{"title":"bloodraven66/deepforcedaligner","url":"https://github.com/bloodraven66/deepforcedaligner"},{"title":"erasedwalt/FastSpeech","url":"https://github.com/erasedwalt/FastSpeech"},{"title":"2023-MindSpore-1/ms-code-15","url":"https://github.com/2023-MindSpore-1/ms-code-15/tree/main/FastSpeech"},{"title":"cadia-lvl/fastspeech2","url":"https://github.com/cadia-lvl/fastspeech2"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8224,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"temp","metrics":{"Audio Quality MOS":"1.25"},"paper_url":null,"paper_title":null,"paper_date":null,"code_links":[],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8225,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Flowtron","metrics":{"Pleasantness MOS":"3.665"},"paper_url":"https://arxiv.org/abs/2005.05957v3","paper_title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","paper_date":"2020-05-12","code_links":[{"title":"NVIDIA/flowtron","url":"https://github.com/NVIDIA/flowtron"},{"title":"NVIDIA/radtts","url":"https://github.com/NVIDIA/radtts"},{"title":"KathyReid/opensource-voice-tools","url":"https://github.com/KathyReid/opensource-voice-tools"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8226,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Tacotron 2","metrics":{"Pleasantness MOS":"3.521"},"paper_url":"https://arxiv.org/abs/2005.05957v3","paper_title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","paper_date":"2020-05-12","code_links":[{"title":"NVIDIA/flowtron","url":"https://github.com/NVIDIA/flowtron"},{"title":"NVIDIA/radtts","url":"https://github.com/NVIDIA/radtts"},{"title":"KathyReid/opensource-voice-tools","url":"https://github.com/KathyReid/opensource-voice-tools"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":1,"source":"archive","tags":[]},{"id":8227,"task":"Text-To-Speech Synthesis","parent_task":null,"dataset":"LJSpeech","model_name":"Matcha-TTS","metrics":{"MOS":"3.84","WER (%)":"2.09"},"paper_url":"https://arxiv.org/abs/2309.03199v2","paper_title":"Matcha-TTS: A fast TTS architecture with conditional flow matching","paper_date":"2023-09-06","code_links":[{"title":"shivammehta25/Matcha-TTS","url":"https://github.com/shivammehta25/Matcha-TTS"}],"metrics_order":"[\"Audio Quality MOS\", \"Pleasantness MOS\", \"Word Error Rate (WER)\", \"MOS\", \"WER (%)\"]","area":"Audio","uses_additional_data":0,"source":"archive","tags":[]}]}