{"task":"Language Modelling","dataset":"enwik8","metric_names":["Bit per Character (BPC)","Number of params"],"rows":[{"id":33699,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"GPT-2 (48 layers, h=1600)","metrics":{"Bit per Character (BPC)":"0.93","Number of params":"1542M"},"paper_url":"https://d4mucfpksywv.cloudfront.net/better-language-models/language-models.pdf","paper_title":"Language Models are Unsupervised Multitask Learners","paper_date":"2019-02-14","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"openai/gpt-2","url":"https://github.com/openai/gpt-2"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/gpt"},{"title":"minimaxir/gpt-2-simple","url":"https://github.com/minimaxir/gpt-2-simple"},{"title":"imcaspar/gpt2-ml","url":"https://github.com/imcaspar/gpt2-ml"},{"title":"huggingface/swift-coreml-transformers","url":"https://github.com/huggingface/swift-coreml-transformers"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/blob/master/research/nlp/gpt2"},{"title":"jankrepl/mildlyoverfitted","url":"https://github.com/jankrepl/mildlyoverfitted"},{"title":"affjljoo3581/GPT2","url":"https://github.com/affjljoo3581/GPT2"},{"title":"akanyaani/gpt-2-tensorflow2.0","url":"https://github.com/akanyaani/gpt-2-tensorflow2.0"},{"title":"lvyufeng/bert4ms","url":"https://github.com/lvyufeng/bert4ms"},{"title":"milmor/GPT","url":"https://github.com/milmor/GPT"},{"title":"abhaskumarsinha/MinimalGPT","url":"https://github.com/abhaskumarsinha/MinimalGPT"},{"title":"aananda-giri/gpt2-nepali","url":"https://github.com/aananda-giri/gpt2-nepali"},{"title":"akanyaani/minGPTF","url":"https://github.com/akanyaani/minGPTF"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"VachanVY/gpt.jax","url":"https://github.com/VachanVY/gpt.jax"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/gpt2"},{"title":"ramanakshay/nanogpt","url":"https://github.com/ramanakshay/nanogpt"},{"title":"2023-MindSpore-1/ms-code-154","url":"https://github.com/2023-MindSpore-1/ms-code-154"},{"title":"varun-suresh/experiments-with-gpt2","url":"https://github.com/varun-suresh/experiments-with-gpt2"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":1,"source":"archive","tags":[]},{"id":33700,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer-XL (24 layers, RMS dynamic eval, decay)","metrics":{"Bit per Character (BPC)":"0.940","Number of params":"277M"},"paper_url":"http://arxiv.org/abs/1904.08378v1","paper_title":"Dynamic Evaluation of Transformer Language Models","paper_date":"2019-04-17","code_links":[{"title":"benkrause/dynamiceval-transformer","url":"https://github.com/benkrause/dynamiceval-transformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":1,"source":"archive","tags":[]},{"id":33701,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Focus","metrics":{"Bit per Character (BPC)":"0.940","Number of params":"22M"},"paper_url":"https://arxiv.org/abs/2305.14952v2","paper_title":"Focus Your Attention (with Adaptive IIR Filters)","paper_date":"2023-05-24","code_links":[],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33702,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Expire-Span (24 layers)","metrics":{"Bit per Character (BPC)":"0.95","Number of params":"208M"},"paper_url":"https://arxiv.org/abs/2105.06548v2","paper_title":"Not All Memories are Created Equal: Learning to Forget by Expiring","paper_date":"2021-05-13","code_links":[{"title":"facebookresearch/transformer-sequential","url":"https://github.com/facebookresearch/transformer-sequential"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33703,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"SRU++ Large","metrics":{"Bit per Character (BPC)":"0.95","Number of params":"195M"},"paper_url":"https://arxiv.org/abs/2102.12459v3","paper_title":"When Attention Meets Fast Recurrence: Training Language Models with Reduced Compute","paper_date":"2021-02-24","code_links":[{"title":"asappresearch/sru","url":"https://github.com/asappresearch/sru"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33704,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Feedback Transformer","metrics":{"Bit per Character (BPC)":"0.96","Number of params":"77M"},"paper_url":"https://arxiv.org/abs/2002.09402v3","paper_title":"Addressing Some Limitations of Transformers with Feedback Memory","paper_date":"2020-02-21","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"facebookresearch/transformer-sequential","url":"https://github.com/facebookresearch/transformer-sequential"},{"title":"lucidrains/feedback-transformer-pytorch","url":"https://github.com/lucidrains/feedback-transformer-pytorch"},{"title":"rajaswa/feedback-and-memory-in-transformers","url":"https://github.com/rajaswa/feedback-and-memory-in-transformers"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33705,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Sandwich Transformer (adaptive span)","metrics":{"Bit per Character (BPC)":"0.968","Number of params":"209M"},"paper_url":"https://arxiv.org/abs/1911.03864v2","paper_title":"Improving Transformer Models by Reordering their Sublayers","paper_date":"2019-11-10","code_links":[{"title":"ofirpress/sandwich_transformer","url":"https://github.com/ofirpress/sandwich_transformer"},{"title":"JunnYu/x-transformers-paddle","url":"https://github.com/JunnYu/x-transformers-paddle"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33706,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Compressive Transformer (24 layers)","metrics":{"Bit per Character (BPC)":"0.97","Number of params":"277M"},"paper_url":"https://arxiv.org/abs/1911.05507v1","paper_title":"Compressive Transformers for Long-Range Sequence Modelling","paper_date":"2019-11-13","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"google-deepmind/pg19","url":"https://github.com/google-deepmind/pg19"},{"title":"deepmind/pg19","url":"https://github.com/deepmind/pg19"},{"title":"lucidrains/block-recurrent-transformer-pytorch","url":"https://github.com/lucidrains/block-recurrent-transformer-pytorch"},{"title":"lucidrains/compressive-transformer-pytorch","url":"https://github.com/lucidrains/compressive-transformer-pytorch"},{"title":"ViktorStagge/CompressiveTransformer","url":"https://github.com/ViktorStagge/CompressiveTransformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33707,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer-LS (large)","metrics":{"Bit per Character (BPC)":"0.97","Number of params":"110M"},"paper_url":"https://arxiv.org/abs/2107.02192v3","paper_title":"Long-Short Transformer: Efficient Transformers for Language and Vision","paper_date":"2021-07-05","code_links":[{"title":"keonlee9420/Comprehensive-Transformer-TTS","url":"https://github.com/keonlee9420/Comprehensive-Transformer-TTS"},{"title":"NVIDIA/transformer-ls","url":"https://github.com/NVIDIA/transformer-ls"},{"title":"lucidrains/long-short-transformer","url":"https://github.com/lucidrains/long-short-transformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33708,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"SRU++ Base","metrics":{"Bit per Character (BPC)":"0.97","Number of params":"108M"},"paper_url":"https://arxiv.org/abs/2102.12459v3","paper_title":"When Attention Meets Fast Recurrence: Training Language Models with Reduced Compute","paper_date":"2021-02-24","code_links":[{"title":"asappresearch/sru","url":"https://github.com/asappresearch/sru"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33709,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer (24 layers, 8k adaptive span)","metrics":{"Bit per Character (BPC)":"0.98","Number of params":"209M"},"paper_url":"https://arxiv.org/abs/1905.07799v2","paper_title":"Adaptive Attention Span in Transformers","paper_date":"2019-05-19","code_links":[{"title":"facebookresearch/adaptive-span","url":"https://github.com/facebookresearch/adaptive-span"},{"title":"jerrodparker20/adaptive-transformers-in-rl","url":"https://github.com/jerrodparker20/adaptive-transformers-in-rl"},{"title":"prajjwal1/fluence","url":"https://github.com/prajjwal1/fluence"},{"title":"lancopku/Explicit-Sparse-Transformer","url":"https://github.com/lancopku/Explicit-Sparse-Transformer"},{"title":"ofirpress/sandwich_transformer","url":"https://github.com/ofirpress/sandwich_transformer"},{"title":"prajjwal1/adaptive_transformer","url":"https://github.com/prajjwal1/adaptive_transformer"},{"title":"JoeRoussy/adaptive-attention-in-cv","url":"https://github.com/JoeRoussy/adaptive-attention-in-cv"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/7/Knowing-When-to-Look-Adaptive-Attention"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33710,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer-XL (24 layers)","metrics":{"Bit per Character (BPC)":"0.99","Number of params":"277M"},"paper_url":"https://arxiv.org/abs/1901.02860v3","paper_title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","paper_date":"2019-01-09","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"NVIDIA/DeepLearningExamples","url":"https://github.com/NVIDIA/DeepLearningExamples/tree/master/TensorFlow/LanguageModeling/Transformer-XL"},{"title":"NVIDIA/DeepLearningExamples","url":"https://github.com/NVIDIA/DeepLearningExamples/tree/master/PyTorch/LanguageModeling/Transformer-XL"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/transformer-xl"},{"title":"kimiyoung/transformer-xl","url":"https://github.com/kimiyoung/transformer-xl"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine"},{"title":"sooftware/conformer","url":"https://github.com/sooftware/conformer"},{"title":"mustafaaljadery/gemma-2b-10m","url":"https://github.com/mustafaaljadery/gemma-2b-10m"},{"title":"sooftware/nlp-attentions","url":"https://github.com/sooftware/nlp-attentions"},{"title":"sooftware/Attention-Implementation","url":"https://github.com/sooftware/Attention-Implementation"},{"title":"sooftware/attentions","url":"https://github.com/sooftware/attentions"},{"title":"sh951011/Attention-Implementation","url":"https://github.com/sh951011/Attention-Implementation"},{"title":"google-research/meliad","url":"https://github.com/google-research/meliad"},{"title":"shanghai-digital-brain-laboratory/bdm-db1","url":"https://github.com/shanghai-digital-brain-laboratory/bdm-db1"},{"title":"facebookresearch/code-prediction-transformer","url":"https://github.com/facebookresearch/code-prediction-transformer"},{"title":"okkteam/Transformer-Transducer","url":"https://github.com/okkteam/Transformer-Transducer"},{"title":"wxt1997/Transformer-Transducer","url":"https://github.com/wxt1997/Transformer-Transducer"},{"title":"lvyufeng/bert4ms","url":"https://github.com/lvyufeng/bert4ms"},{"title":"TimDettmers/transformer-xl","url":"https://github.com/TimDettmers/transformer-xl"},{"title":"inzva/fake-academic-paper-generation","url":"https://github.com/inzva/fake-academic-paper-generation"},{"title":"Machine-Learning-Tokyo/Poetry-GAN","url":"https://github.com/Machine-Learning-Tokyo/Poetry-GAN"},{"title":"benkrause/dynamiceval-transformer","url":"https://github.com/benkrause/dynamiceval-transformer"},{"title":"jincan333/lot","url":"https://github.com/jincan333/lot"},{"title":"huggingface/xlnet","url":"https://github.com/huggingface/xlnet"},{"title":"aiha-lab/Attention-Head-Pruning","url":"https://github.com/aiha-lab/Attention-Head-Pruning"},{"title":"cedrickchee/pytorch-pretrained-BERT","url":"https://github.com/cedrickchee/pytorch-pretrained-BERT"},{"title":"park-cheol/ASR-Conformer","url":"https://github.com/park-cheol/ASR-Conformer"},{"title":"Jmkernes/PAR-Transformer-XL","url":"https://github.com/Jmkernes/PAR-Transformer-XL"},{"title":"zhdbwe/Paper-DailyReading","url":"https://github.com/zhdbwe/Paper-DailyReading"},{"title":"AIResearchHub/transformergallery","url":"https://github.com/AIResearchHub/transformergallery"},{"title":"cmunnis/BERT_vs_Transformer-XL","url":"https://github.com/cmunnis/BERT_vs_Transformer-XL"},{"title":"jiean001/models_m","url":"https://github.com/jiean001/models_m/tree/main/transformer_xl"},{"title":"samwisegamjeee/pytorch-transformers","url":"https://github.com/samwisegamjeee/pytorch-transformers"},{"title":"SambhawDrag/XLNet.jl","url":"https://github.com/SambhawDrag/XLNet.jl"},{"title":"listenviolet/XLNet","url":"https://github.com/listenviolet/XLNet"},{"title":"2023-MindSpore-1/ms-code-220","url":"https://github.com/2023-MindSpore-1/ms-code-220/tree/main/transformer_xl"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33711,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Longformer (30 layers, h=512)","metrics":{"Bit per Character (BPC)":"0.99","Number of params":"102M"},"paper_url":"https://arxiv.org/abs/2004.05150v2","paper_title":"Longformer: The Long-Document Transformer","paper_date":"2020-04-10","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"mistralai/mistral-src","url":"https://github.com/mistralai/mistral-src"},{"title":"facebookresearch/xformers","url":"https://github.com/facebookresearch/xformers"},{"title":"allenai/longformer","url":"https://github.com/allenai/longformer"},{"title":"mim-solutions/roberta_for_longer_texts","url":"https://github.com/mim-solutions/roberta_for_longer_texts"},{"title":"mim-solutions/bert_for_longer_texts","url":"https://github.com/mim-solutions/bert_for_longer_texts"},{"title":"microsoft/dialoglm","url":"https://github.com/microsoft/dialoglm"},{"title":"schenliu/longformer-chinese","url":"https://github.com/schenliu/longformer-chinese"},{"title":"jaketae/pytorch-malware-detection","url":"https://github.com/jaketae/pytorch-malware-detection"},{"title":"amoramine/Pegasus_Longformer_summarization","url":"https://github.com/amoramine/Pegasus_Longformer_summarization"},{"title":"amoramine/Pegasus_with_Longformer_summarization","url":"https://github.com/amoramine/Pegasus_with_Longformer_summarization"},{"title":"amazon-science/efficient-longdoc-classification","url":"https://github.com/amazon-science/efficient-longdoc-classification"},{"title":"han-shi/SparseBERT","url":"https://github.com/han-shi/SparseBERT"},{"title":"naver-ai/simseek","url":"https://github.com/naver-ai/simseek"},{"title":"kit-mrt/road-barlow-twins","url":"https://github.com/kit-mrt/road-barlow-twins"},{"title":"kit-mrt/red-motion","url":"https://github.com/kit-mrt/red-motion"},{"title":"Phrase-in-Context/eval","url":"https://github.com/Phrase-in-Context/eval"},{"title":"lucashueda/long_sentence_transformer","url":"https://github.com/lucashueda/long_sentence_transformer"},{"title":"AIResearchHub/transformergallery","url":"https://github.com/AIResearchHub/transformergallery"},{"title":"a-rios/ats-models","url":"https://github.com/a-rios/ats-models"},{"title":"2023-MindSpore-1/ms-code-155","url":"https://github.com/2023-MindSpore-1/ms-code-155"},{"title":"2023-MindSpore-1/ms-code-161","url":"https://github.com/2023-MindSpore-1/ms-code-161"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33712,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Sparse Transformer (30 layers, fixed attn)","metrics":{"Bit per Character (BPC)":"0.99","Number of params":"95M"},"paper_url":"http://arxiv.org/abs/1904.10509v1","paper_title":"Generating Long Sequences with Sparse Transformers","paper_date":"2019-04-23","code_links":[{"title":"mistralai/mistral-src","url":"https://github.com/mistralai/mistral-src"},{"title":"openai/sparse_attention","url":"https://github.com/openai/sparse_attention"},{"title":"wilson1yan/VideoGPT","url":"https://github.com/wilson1yan/VideoGPT"},{"title":"ptillet/torch-blocksparse","url":"https://github.com/ptillet/torch-blocksparse"},{"title":"han-shi/SparseBERT","url":"https://github.com/han-shi/SparseBERT"},{"title":"jonahwinninghoff/Text-Summarization","url":"https://github.com/jonahwinninghoff/Text-Summarization"},{"title":"MindCode-4/code-11","url":"https://github.com/MindCode-4/code-11/tree/main/factorized-attention"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33713,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Routing Transformer (12 layers)","metrics":{"Bit per Character (BPC)":"0.99"},"paper_url":"https://arxiv.org/abs/2003.05997v5","paper_title":"Efficient Content-Based Sparse Attention with Routing Transformers","paper_date":"2020-03-12","code_links":[{"title":"lucidrains/local-attention","url":"https://github.com/lucidrains/local-attention"},{"title":"lucidrains/routing-transformer","url":"https://github.com/lucidrains/routing-transformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33714,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer-LS (small)","metrics":{"Bit per Character (BPC)":"0.99"},"paper_url":"https://arxiv.org/abs/2107.02192v3","paper_title":"Long-Short Transformer: Efficient Transformers for Language and Vision","paper_date":"2021-07-05","code_links":[{"title":"keonlee9420/Comprehensive-Transformer-TTS","url":"https://github.com/keonlee9420/Comprehensive-Transformer-TTS"},{"title":"NVIDIA/transformer-ls","url":"https://github.com/NVIDIA/transformer-ls"},{"title":"lucidrains/long-short-transformer","url":"https://github.com/lucidrains/long-short-transformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33715,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Hourglass","metrics":{"Bit per Character (BPC)":"0.997"},"paper_url":"https://arxiv.org/abs/2110.13711v2","paper_title":"Hierarchical Transformers Are More Efficient Language Models","paper_date":"2021-10-26","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"google/trax","url":"https://github.com/google/trax/blob/master/trax/models/research/hourglass.py"},{"title":"lucidrains/hourglass-transformer-pytorch","url":"https://github.com/lucidrains/hourglass-transformer-pytorch"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33716,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Longformer (12 layers, h=512)","metrics":{"Bit per Character (BPC)":"1.00","Number of params":"41M"},"paper_url":"https://arxiv.org/abs/2004.05150v2","paper_title":"Longformer: The Long-Document Transformer","paper_date":"2020-04-10","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"mistralai/mistral-src","url":"https://github.com/mistralai/mistral-src"},{"title":"facebookresearch/xformers","url":"https://github.com/facebookresearch/xformers"},{"title":"allenai/longformer","url":"https://github.com/allenai/longformer"},{"title":"mim-solutions/roberta_for_longer_texts","url":"https://github.com/mim-solutions/roberta_for_longer_texts"},{"title":"mim-solutions/bert_for_longer_texts","url":"https://github.com/mim-solutions/bert_for_longer_texts"},{"title":"microsoft/dialoglm","url":"https://github.com/microsoft/dialoglm"},{"title":"schenliu/longformer-chinese","url":"https://github.com/schenliu/longformer-chinese"},{"title":"jaketae/pytorch-malware-detection","url":"https://github.com/jaketae/pytorch-malware-detection"},{"title":"amoramine/Pegasus_Longformer_summarization","url":"https://github.com/amoramine/Pegasus_Longformer_summarization"},{"title":"amoramine/Pegasus_with_Longformer_summarization","url":"https://github.com/amoramine/Pegasus_with_Longformer_summarization"},{"title":"amazon-science/efficient-longdoc-classification","url":"https://github.com/amazon-science/efficient-longdoc-classification"},{"title":"han-shi/SparseBERT","url":"https://github.com/han-shi/SparseBERT"},{"title":"naver-ai/simseek","url":"https://github.com/naver-ai/simseek"},{"title":"kit-mrt/road-barlow-twins","url":"https://github.com/kit-mrt/road-barlow-twins"},{"title":"kit-mrt/red-motion","url":"https://github.com/kit-mrt/red-motion"},{"title":"Phrase-in-Context/eval","url":"https://github.com/Phrase-in-Context/eval"},{"title":"lucashueda/long_sentence_transformer","url":"https://github.com/lucashueda/long_sentence_transformer"},{"title":"AIResearchHub/transformergallery","url":"https://github.com/AIResearchHub/transformergallery"},{"title":"a-rios/ats-models","url":"https://github.com/a-rios/ats-models"},{"title":"2023-MindSpore-1/ms-code-155","url":"https://github.com/2023-MindSpore-1/ms-code-155"},{"title":"2023-MindSpore-1/ms-code-161","url":"https://github.com/2023-MindSpore-1/ms-code-161"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33717,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"All-attention network (18 layers)","metrics":{"Bit per Character (BPC)":"1.01","Number of params":"39M"},"paper_url":"https://arxiv.org/abs/1907.01470v1","paper_title":"Augmenting Self-attention with Persistent Memory","paper_date":"2019-07-02","code_links":[{"title":"lucidrains/x-transformers","url":"https://github.com/lucidrains/x-transformers"},{"title":"facebookresearch/adaptive-span","url":"https://github.com/facebookresearch/adaptive-span"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33718,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer (12 layers, 8k adaptive span)","metrics":{"Bit per Character (BPC)":"1.02","Number of params":"39M"},"paper_url":"https://arxiv.org/abs/1905.07799v2","paper_title":"Adaptive Attention Span in Transformers","paper_date":"2019-05-19","code_links":[{"title":"facebookresearch/adaptive-span","url":"https://github.com/facebookresearch/adaptive-span"},{"title":"jerrodparker20/adaptive-transformers-in-rl","url":"https://github.com/jerrodparker20/adaptive-transformers-in-rl"},{"title":"prajjwal1/fluence","url":"https://github.com/prajjwal1/fluence"},{"title":"lancopku/Explicit-Sparse-Transformer","url":"https://github.com/lancopku/Explicit-Sparse-Transformer"},{"title":"ofirpress/sandwich_transformer","url":"https://github.com/ofirpress/sandwich_transformer"},{"title":"prajjwal1/adaptive_transformer","url":"https://github.com/prajjwal1/adaptive_transformer"},{"title":"JoeRoussy/adaptive-attention-in-cv","url":"https://github.com/JoeRoussy/adaptive-attention-in-cv"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/7/Knowing-When-to-Look-Adaptive-Attention"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33719,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"BP-Transformer (12 layers)","metrics":{"Bit per Character (BPC)":"1.02","Number of params":"38M"},"paper_url":"https://arxiv.org/abs/1911.04070v1","paper_title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","paper_date":"2019-11-11","code_links":[{"title":"dmlc/dgl","url":"https://github.com/dmlc/dgl/tree/master/examples/pytorch/transformer"},{"title":"yzh119/BPT","url":"https://github.com/yzh119/BPT"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33720,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer+SSA","metrics":{"Bit per Character (BPC)":"1.024"},"paper_url":"https://arxiv.org/abs/2306.01705v1","paper_title":"The Information Pathways Hypothesis: Transformers are Dynamic Self-Ensembles","paper_date":"2023-06-02","code_links":[{"title":"shamim-hussain/ssa","url":"https://github.com/shamim-hussain/ssa"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33721,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer-XL (18 layers)","metrics":{"Bit per Character (BPC)":"1.03","Number of params":"88M"},"paper_url":"https://arxiv.org/abs/1901.02860v3","paper_title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","paper_date":"2019-01-09","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"NVIDIA/DeepLearningExamples","url":"https://github.com/NVIDIA/DeepLearningExamples/tree/master/TensorFlow/LanguageModeling/Transformer-XL"},{"title":"NVIDIA/DeepLearningExamples","url":"https://github.com/NVIDIA/DeepLearningExamples/tree/master/PyTorch/LanguageModeling/Transformer-XL"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/transformer-xl"},{"title":"kimiyoung/transformer-xl","url":"https://github.com/kimiyoung/transformer-xl"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine"},{"title":"sooftware/conformer","url":"https://github.com/sooftware/conformer"},{"title":"mustafaaljadery/gemma-2b-10m","url":"https://github.com/mustafaaljadery/gemma-2b-10m"},{"title":"sooftware/nlp-attentions","url":"https://github.com/sooftware/nlp-attentions"},{"title":"sooftware/Attention-Implementation","url":"https://github.com/sooftware/Attention-Implementation"},{"title":"sooftware/attentions","url":"https://github.com/sooftware/attentions"},{"title":"sh951011/Attention-Implementation","url":"https://github.com/sh951011/Attention-Implementation"},{"title":"google-research/meliad","url":"https://github.com/google-research/meliad"},{"title":"shanghai-digital-brain-laboratory/bdm-db1","url":"https://github.com/shanghai-digital-brain-laboratory/bdm-db1"},{"title":"facebookresearch/code-prediction-transformer","url":"https://github.com/facebookresearch/code-prediction-transformer"},{"title":"okkteam/Transformer-Transducer","url":"https://github.com/okkteam/Transformer-Transducer"},{"title":"wxt1997/Transformer-Transducer","url":"https://github.com/wxt1997/Transformer-Transducer"},{"title":"lvyufeng/bert4ms","url":"https://github.com/lvyufeng/bert4ms"},{"title":"TimDettmers/transformer-xl","url":"https://github.com/TimDettmers/transformer-xl"},{"title":"inzva/fake-academic-paper-generation","url":"https://github.com/inzva/fake-academic-paper-generation"},{"title":"Machine-Learning-Tokyo/Poetry-GAN","url":"https://github.com/Machine-Learning-Tokyo/Poetry-GAN"},{"title":"benkrause/dynamiceval-transformer","url":"https://github.com/benkrause/dynamiceval-transformer"},{"title":"jincan333/lot","url":"https://github.com/jincan333/lot"},{"title":"huggingface/xlnet","url":"https://github.com/huggingface/xlnet"},{"title":"aiha-lab/Attention-Head-Pruning","url":"https://github.com/aiha-lab/Attention-Head-Pruning"},{"title":"cedrickchee/pytorch-pretrained-BERT","url":"https://github.com/cedrickchee/pytorch-pretrained-BERT"},{"title":"park-cheol/ASR-Conformer","url":"https://github.com/park-cheol/ASR-Conformer"},{"title":"Jmkernes/PAR-Transformer-XL","url":"https://github.com/Jmkernes/PAR-Transformer-XL"},{"title":"zhdbwe/Paper-DailyReading","url":"https://github.com/zhdbwe/Paper-DailyReading"},{"title":"AIResearchHub/transformergallery","url":"https://github.com/AIResearchHub/transformergallery"},{"title":"cmunnis/BERT_vs_Transformer-XL","url":"https://github.com/cmunnis/BERT_vs_Transformer-XL"},{"title":"jiean001/models_m","url":"https://github.com/jiean001/models_m/tree/main/transformer_xl"},{"title":"samwisegamjeee/pytorch-transformers","url":"https://github.com/samwisegamjeee/pytorch-transformers"},{"title":"SambhawDrag/XLNet.jl","url":"https://github.com/SambhawDrag/XLNet.jl"},{"title":"listenviolet/XLNet","url":"https://github.com/listenviolet/XLNet"},{"title":"2023-MindSpore-1/ms-code-220","url":"https://github.com/2023-MindSpore-1/ms-code-220/tree/main/transformer_xl"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33722,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Skip Cross-Head Transformer-XL","metrics":{"Bit per Character (BPC)":"1.033","Number of params":"41M"},"paper_url":"https://arxiv.org/abs/2311.08123v1","paper_title":"Memory-efficient Stochastic methods for Memory-based Transformers","paper_date":"2023-11-14","code_links":[{"title":"vishwajit-vishnu/memory-efficient-stochastic-methods-for-memory-based-transformers","url":"https://github.com/vishwajit-vishnu/memory-efficient-stochastic-methods-for-memory-based-transformers"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33723,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer (64 layers)","metrics":{"Bit per Character (BPC)":"1.06","Number of params":"235M"},"paper_url":"http://arxiv.org/abs/1808.04444v2","paper_title":"Character-Level Language Modeling with Deeper Self-Attention","paper_date":"2018-08-09","code_links":[{"title":"facebookresearch/code-prediction-transformer","url":"https://github.com/facebookresearch/code-prediction-transformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33724,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Transformer-XL (12 layers)","metrics":{"Bit per Character (BPC)":"1.06","Number of params":"41M"},"paper_url":"https://arxiv.org/abs/1901.02860v3","paper_title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","paper_date":"2019-01-09","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"NVIDIA/DeepLearningExamples","url":"https://github.com/NVIDIA/DeepLearningExamples/tree/master/TensorFlow/LanguageModeling/Transformer-XL"},{"title":"NVIDIA/DeepLearningExamples","url":"https://github.com/NVIDIA/DeepLearningExamples/tree/master/PyTorch/LanguageModeling/Transformer-XL"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/transformer-xl"},{"title":"kimiyoung/transformer-xl","url":"https://github.com/kimiyoung/transformer-xl"},{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine"},{"title":"sooftware/conformer","url":"https://github.com/sooftware/conformer"},{"title":"mustafaaljadery/gemma-2b-10m","url":"https://github.com/mustafaaljadery/gemma-2b-10m"},{"title":"sooftware/nlp-attentions","url":"https://github.com/sooftware/nlp-attentions"},{"title":"sooftware/Attention-Implementation","url":"https://github.com/sooftware/Attention-Implementation"},{"title":"sooftware/attentions","url":"https://github.com/sooftware/attentions"},{"title":"sh951011/Attention-Implementation","url":"https://github.com/sh951011/Attention-Implementation"},{"title":"google-research/meliad","url":"https://github.com/google-research/meliad"},{"title":"shanghai-digital-brain-laboratory/bdm-db1","url":"https://github.com/shanghai-digital-brain-laboratory/bdm-db1"},{"title":"facebookresearch/code-prediction-transformer","url":"https://github.com/facebookresearch/code-prediction-transformer"},{"title":"okkteam/Transformer-Transducer","url":"https://github.com/okkteam/Transformer-Transducer"},{"title":"wxt1997/Transformer-Transducer","url":"https://github.com/wxt1997/Transformer-Transducer"},{"title":"lvyufeng/bert4ms","url":"https://github.com/lvyufeng/bert4ms"},{"title":"TimDettmers/transformer-xl","url":"https://github.com/TimDettmers/transformer-xl"},{"title":"inzva/fake-academic-paper-generation","url":"https://github.com/inzva/fake-academic-paper-generation"},{"title":"Machine-Learning-Tokyo/Poetry-GAN","url":"https://github.com/Machine-Learning-Tokyo/Poetry-GAN"},{"title":"benkrause/dynamiceval-transformer","url":"https://github.com/benkrause/dynamiceval-transformer"},{"title":"jincan333/lot","url":"https://github.com/jincan333/lot"},{"title":"huggingface/xlnet","url":"https://github.com/huggingface/xlnet"},{"title":"aiha-lab/Attention-Head-Pruning","url":"https://github.com/aiha-lab/Attention-Head-Pruning"},{"title":"cedrickchee/pytorch-pretrained-BERT","url":"https://github.com/cedrickchee/pytorch-pretrained-BERT"},{"title":"park-cheol/ASR-Conformer","url":"https://github.com/park-cheol/ASR-Conformer"},{"title":"Jmkernes/PAR-Transformer-XL","url":"https://github.com/Jmkernes/PAR-Transformer-XL"},{"title":"zhdbwe/Paper-DailyReading","url":"https://github.com/zhdbwe/Paper-DailyReading"},{"title":"AIResearchHub/transformergallery","url":"https://github.com/AIResearchHub/transformergallery"},{"title":"cmunnis/BERT_vs_Transformer-XL","url":"https://github.com/cmunnis/BERT_vs_Transformer-XL"},{"title":"jiean001/models_m","url":"https://github.com/jiean001/models_m/tree/main/transformer_xl"},{"title":"samwisegamjeee/pytorch-transformers","url":"https://github.com/samwisegamjeee/pytorch-transformers"},{"title":"SambhawDrag/XLNet.jl","url":"https://github.com/SambhawDrag/XLNet.jl"},{"title":"listenviolet/XLNet","url":"https://github.com/listenviolet/XLNet"},{"title":"2023-MindSpore-1/ms-code-220","url":"https://github.com/2023-MindSpore-1/ms-code-220/tree/main/transformer_xl"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33725,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"SHA-RNN (4 layers, h=1024, attention head per layer)","metrics":{"Bit per Character (BPC)":"1.068","Number of params":"54M"},"paper_url":"https://arxiv.org/abs/1911.11423v2","paper_title":"Single Headed Attention RNN: Stop Thinking With Your Head","paper_date":"2019-11-26","code_links":[{"title":"Smerity/sha-rnn","url":"https://github.com/Smerity/sha-rnn"},{"title":"floleuerer/fastai_ulmfit","url":"https://github.com/floleuerer/fastai_ulmfit"},{"title":"saattrupdan/scholarly","url":"https://github.com/saattrupdan/scholarly"},{"title":"alisafaya/SHA-RNN.jl","url":"https://github.com/alisafaya/SHA-RNN.jl"},{"title":"Tobias-K93/media-bias-prediction","url":"https://github.com/Tobias-K93/media-bias-prediction"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33726,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"SHA-RNN (4 layers, h=1024, single attention head)","metrics":{"Bit per Character (BPC)":"1.076","Number of params":"52M"},"paper_url":"https://arxiv.org/abs/1911.11423v2","paper_title":"Single Headed Attention RNN: Stop Thinking With Your Head","paper_date":"2019-11-26","code_links":[{"title":"Smerity/sha-rnn","url":"https://github.com/Smerity/sha-rnn"},{"title":"floleuerer/fastai_ulmfit","url":"https://github.com/floleuerer/fastai_ulmfit"},{"title":"saattrupdan/scholarly","url":"https://github.com/saattrupdan/scholarly"},{"title":"alisafaya/SHA-RNN.jl","url":"https://github.com/alisafaya/SHA-RNN.jl"},{"title":"Tobias-K93/media-bias-prediction","url":"https://github.com/Tobias-K93/media-bias-prediction"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33727,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"64-layer Character Transformer Model","metrics":{"Bit per Character (BPC)":"1.11","Number of params":"44M"},"paper_url":"http://arxiv.org/abs/1808.04444v2","paper_title":"Character-Level Language Modeling with Deeper Self-Attention","paper_date":"2018-08-09","code_links":[{"title":"facebookresearch/code-prediction-transformer","url":"https://github.com/facebookresearch/code-prediction-transformer"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33728,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Mogrifier LSTM","metrics":{"Bit per Character (BPC)":"1.146","Number of params":"48M"},"paper_url":"https://arxiv.org/abs/1909.01792v2","paper_title":"Mogrifier LSTM","paper_date":"2019-09-04","code_links":[{"title":"deepmind/lamb","url":"https://github.com/deepmind/lamb"},{"title":"RMichaelSwan/MogrifierLSTM","url":"https://github.com/RMichaelSwan/MogrifierLSTM"},{"title":"microcoder-py/mogrifier-lstm","url":"https://github.com/microcoder-py/mogrifier-lstm"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33729,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"LSTM","metrics":{"Bit per Character (BPC)":"1.195","Number of params":"48M"},"paper_url":"https://arxiv.org/abs/1909.01792v2","paper_title":"Mogrifier LSTM","paper_date":"2019-09-04","code_links":[{"title":"deepmind/lamb","url":"https://github.com/deepmind/lamb"},{"title":"RMichaelSwan/MogrifierLSTM","url":"https://github.com/RMichaelSwan/MogrifierLSTM"},{"title":"microcoder-py/mogrifier-lstm","url":"https://github.com/microcoder-py/mogrifier-lstm"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33730,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Cluster-Former (#C=512)","metrics":{"Bit per Character (BPC)":"1.22"},"paper_url":"https://arxiv.org/abs/2009.06097v2","paper_title":"Cluster-Former: Clustering-based Sparse Transformer for Long-Range Dependency Encoding","paper_date":"2020-09-13","code_links":[],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33731,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"AWD-LSTM (3 layers)","metrics":{"Bit per Character (BPC)":"1.232","Number of params":"47M"},"paper_url":"http://arxiv.org/abs/1803.08240v1","paper_title":"An Analysis of Neural Language Modeling at Multiple Scales","paper_date":"2018-03-22","code_links":[{"title":"salesforce/awd-lstm-lm","url":"https://github.com/salesforce/awd-lstm-lm"},{"title":"Han-JD/GRU-D","url":"https://github.com/Han-JD/GRU-D"},{"title":"jb33k/awd-lstm-lm-ThinkNet","url":"https://github.com/jb33k/awd-lstm-lm-ThinkNet"},{"title":"mnhng/hier-char-emb","url":"https://github.com/mnhng/hier-char-emb"},{"title":"SachinIchake/KALM","url":"https://github.com/SachinIchake/KALM"},{"title":"AtheMathmo/lookahead-lstm","url":"https://github.com/AtheMathmo/lookahead-lstm"},{"title":"llppff/ptb-lstmorqrnn-pytorch","url":"https://github.com/llppff/ptb-lstmorqrnn-pytorch"},{"title":"philippwirth/awd-lstm-test","url":"https://github.com/philippwirth/awd-lstm-test"},{"title":"ari-holtzman/genlm","url":"https://github.com/ari-holtzman/genlm"},{"title":"arvieFrydenlund/awd-lstm-lm","url":"https://github.com/arvieFrydenlund/awd-lstm-lm"},{"title":"soyoung97/awd-lstm-gru","url":"https://github.com/soyoung97/awd-lstm-gru"},{"title":"philippwirth/treelangrnn","url":"https://github.com/philippwirth/treelangrnn"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33732,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Large mLSTM","metrics":{"Bit per Character (BPC)":"1.24","Number of params":"46M"},"paper_url":"http://arxiv.org/abs/1609.07959v3","paper_title":"Multiplicative LSTM for sequence modelling","paper_date":"2016-09-26","code_links":[{"title":"astakara48/python_project","url":"https://github.com/astakara48/python_project"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33733,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Large FS-LSTM-4","metrics":{"Bit per Character (BPC)":" 1.25","Number of params":"47M"},"paper_url":"http://arxiv.org/abs/1705.08639v2","paper_title":"Fast-Slow Recurrent Neural Networks","paper_date":"2017-05-24","code_links":[{"title":"amujika/Fast-Slow-LSTM","url":"https://github.com/amujika/Fast-Slow-LSTM"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33734,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Recurrent Highway Networks","metrics":{"Bit per Character (BPC)":"1.27","Number of params":"46M"},"paper_url":"http://arxiv.org/abs/1607.03474v5","paper_title":"Recurrent Highway Networks","paper_date":"2016-07-12","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"julian121266/RecurrentHighwayNetworks","url":"https://github.com/julian121266/RecurrentHighwayNetworks"},{"title":"jzilly/RecurrentHighwayNetworks","url":"https://github.com/jzilly/RecurrentHighwayNetworks"},{"title":"vermaMachineLearning/Pytorch-JIT-Recurrent-Highway-Network","url":"https://github.com/vermaMachineLearning/Pytorch-JIT-Recurrent-Highway-Network"},{"title":"davidsvaughn/dts-tf","url":"https://github.com/davidsvaughn/dts-tf"},{"title":"nanzhaogang/contrib","url":"https://github.com/nanzhaogang/contrib/tree/master/application/recurrent-highway-network"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33735,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"ByteNet","metrics":{"Bit per Character (BPC)":"1.31"},"paper_url":"http://arxiv.org/abs/1610.10099v2","paper_title":"Neural Machine Translation in Linear Time","paper_date":"2016-10-31","code_links":[{"title":"paarthneekhara/byteNet-tensorflow","url":"https://github.com/paarthneekhara/byteNet-tensorflow"},{"title":"microsoft/protein-sequence-models","url":"https://github.com/microsoft/protein-sequence-models"},{"title":"randomrandom/deep-atrous-cnn-sentiment","url":"https://github.com/randomrandom/deep-atrous-cnn-sentiment"},{"title":"kingstarcraft/speech-to-text-wavenet2","url":"https://github.com/kingstarcraft/speech-to-text-wavenet2"},{"title":"sriharireddypusapati/speech-to-text-wavenet2","url":"https://github.com/sriharireddypusapati/speech-to-text-wavenet2"},{"title":"kinimod23/ATS_Project","url":"https://github.com/kinimod23/ATS_Project"},{"title":"Vikas-Sony/speech-to-text","url":"https://github.com/Vikas-Sony/speech-to-text"},{"title":"freedombenLiu/speech-to-text-wavenet","url":"https://github.com/freedombenLiu/speech-to-text-wavenet"},{"title":"adityaagrawal7/speech-to-text-wavenet","url":"https://github.com/adityaagrawal7/speech-to-text-wavenet"},{"title":"liguigui/speech-to-text-wavenet","url":"https://github.com/liguigui/speech-to-text-wavenet"},{"title":"Shivendra-psc/speechbot","url":"https://github.com/Shivendra-psc/speechbot"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33736,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"LN HM-LSTM","metrics":{"Bit per Character (BPC)":"1.32","Number of params":"35M"},"paper_url":"http://arxiv.org/abs/1609.01704v7","paper_title":"Hierarchical Multiscale Recurrent Neural Networks","paper_date":"2016-09-06","code_links":[{"title":"bolducp/hierarchical-rnn","url":"https://github.com/bolducp/hierarchical-rnn"},{"title":"kaiu85/hm-rnn","url":"https://github.com/kaiu85/hm-rnn"},{"title":"nikolasthuesen/HMLSTM","url":"https://github.com/nikolasthuesen/HMLSTM"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33737,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"SHA-LSTM (4 layers, h=1024, no attention head)","metrics":{"Bit per Character (BPC)":"1.33","Number of params":"51M"},"paper_url":"https://arxiv.org/abs/1911.11423v2","paper_title":"Single Headed Attention RNN: Stop Thinking With Your Head","paper_date":"2019-11-26","code_links":[{"title":"Smerity/sha-rnn","url":"https://github.com/Smerity/sha-rnn"},{"title":"floleuerer/fastai_ulmfit","url":"https://github.com/floleuerer/fastai_ulmfit"},{"title":"saattrupdan/scholarly","url":"https://github.com/saattrupdan/scholarly"},{"title":"alisafaya/SHA-RNN.jl","url":"https://github.com/alisafaya/SHA-RNN.jl"},{"title":"Tobias-K93/media-bias-prediction","url":"https://github.com/Tobias-K93/media-bias-prediction"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33738,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"Hypernetworks","metrics":{"Bit per Character (BPC)":"1.34","Number of params":"27M"},"paper_url":"http://arxiv.org/abs/1609.09106v4","paper_title":"HyperNetworks","paper_date":"2016-09-27","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"g1910/HyperNetworks","url":"https://github.com/g1910/HyperNetworks"},{"title":"tjuhaoxiaotian/pymarl3","url":"https://github.com/tjuhaoxiaotian/pymarl3"},{"title":"chrhenning/hypnettorch","url":"https://github.com/chrhenning/hypnettorch"},{"title":"shyamsn97/hyper-nn","url":"https://github.com/shyamsn97/hyper-nn"},{"title":"gahaalt/continual-learning-overview","url":"https://github.com/gahaalt/continual-learning-overview"},{"title":"gahaalt/continual-learning-with-hypernets","url":"https://github.com/gahaalt/continual-learning-with-hypernets"},{"title":"gtegner/hyper-gan","url":"https://github.com/gtegner/hyper-gan"},{"title":"pennfranc/hypnettorch","url":"https://github.com/pennfranc/hypnettorch"},{"title":"cellistigs/ensemble_attention","url":"https://github.com/cellistigs/ensemble_attention"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33739,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"LSTM (7 layers)","metrics":{"Bit per Character (BPC)":"1.67"},"paper_url":"http://arxiv.org/abs/1308.0850v5","paper_title":"Generating Sequences With Recurrent Neural Networks","paper_date":"2013-08-04","code_links":[{"title":"karpathy/char-rnn","url":"https://github.com/karpathy/char-rnn"},{"title":"sjvasquez/handwriting-synthesis","url":"https://github.com/sjvasquez/handwriting-synthesis"},{"title":"karpathy/makemore","url":"https://github.com/karpathy/makemore"},{"title":"sherjilozair/char-rnn-tensorflow","url":"https://github.com/sherjilozair/char-rnn-tensorflow"},{"title":"szcom/rnnlib","url":"https://github.com/szcom/rnnlib"},{"title":"hardmaru/write-rnn-tensorflow","url":"https://github.com/hardmaru/write-rnn-tensorflow"},{"title":"grzego/handwriting-generation","url":"https://github.com/grzego/handwriting-generation"},{"title":"larspars/word-rnn","url":"https://github.com/larspars/word-rnn"},{"title":"greydanus/scribe","url":"https://github.com/greydanus/scribe"},{"title":"jparkhill/TensorMol","url":"https://github.com/jparkhill/TensorMol"},{"title":"swechhachoudhary/Handwriting-synthesis","url":"https://github.com/swechhachoudhary/Handwriting-synthesis"},{"title":"hardmaru/sketch-rnn-datasets","url":"https://github.com/hardmaru/sketch-rnn-datasets"},{"title":"kedartatwawadi/NN_compression","url":"https://github.com/kedartatwawadi/NN_compression"},{"title":"cpmpercussion/keras-mdn-layer","url":"https://github.com/cpmpercussion/keras-mdn-layer"},{"title":"snowkylin/rnn-handwriting-generation","url":"https://github.com/snowkylin/rnn-handwriting-generation"},{"title":"pnshiralkar/text-to-handwriting","url":"https://github.com/pnshiralkar/text-to-handwriting"},{"title":"edwin-de-jong/incremental-sequence-learning","url":"https://github.com/edwin-de-jong/incremental-sequence-learning"},{"title":"huyhoang17/Vietnamese_Handwriting_Recognition","url":"https://github.com/huyhoang17/Vietnamese_Handwriting_Recognition"},{"title":"cody2007/arcane_fortune","url":"https://github.com/cody2007/arcane_fortune"},{"title":"adbrebs/handwriting","url":"https://github.com/adbrebs/handwriting"},{"title":"altsoph/paranoid_transformer","url":"https://github.com/altsoph/paranoid_transformer"},{"title":"jolinxql/LSTM","url":"https://github.com/jolinxql/LSTM"},{"title":"AnesBenmerzoug/Handwriting-Model","url":"https://github.com/AnesBenmerzoug/Handwriting-Model"},{"title":"TengHu/Generating-Sequences-With-Recurrent-Neural-Networks","url":"https://github.com/TengHu/Generating-Sequences-With-Recurrent-Neural-Networks"},{"title":"AniketBajpai/deep-handwriting-generation","url":"https://github.com/AniketBajpai/deep-handwriting-generation"},{"title":"ikonthomas/Digitize-Handwritten-Texts","url":"https://github.com/ikonthomas/Digitize-Handwritten-Texts"},{"title":"raymondhs/char-rnn-truecase","url":"https://github.com/raymondhs/char-rnn-truecase"},{"title":"rahulbhalley/rnns-in-pytorch","url":"https://github.com/rahulbhalley/rnns-in-pytorch"},{"title":"sotelo/scribe","url":"https://github.com/sotelo/scribe"},{"title":"rahul96rajan/text-2-strokes","url":"https://github.com/rahul96rajan/text-2-strokes"},{"title":"Vrushank264/Handwriting-Generation-using-LSTM","url":"https://github.com/Vrushank264/Handwriting-Generation-using-LSTM"},{"title":"SamuelNguyen1998/Vietnamese_Handwriting_Recognition","url":"https://github.com/SamuelNguyen1998/Vietnamese_Handwriting_Recognition"},{"title":"YukiDaSlayer316/R-Tutorial","url":"https://github.com/YukiDaSlayer316/R-Tutorial"},{"title":"yukikongju/R-Tutorial","url":"https://github.com/yukikongju/R-Tutorial"},{"title":"rahulbhalley/Recurrent-Nets-in-Pytorch","url":"https://github.com/rahulbhalley/Recurrent-Nets-in-Pytorch"},{"title":"edwinzhng/handwriting-synthesis","url":"https://github.com/edwinzhng/handwriting-synthesis"},{"title":"CongBao/ChatBot","url":"https://github.com/CongBao/ChatBot"},{"title":"raymondhs/char-rnn-truecase","url":"https://gitlab.com/raymondhs/char-rnn-truecase"},{"title":"robertknight/textgen","url":"https://github.com/robertknight/textgen"},{"title":"vladbataev/handwriting_synthesis","url":"https://github.com/vladbataev/handwriting_synthesis"},{"title":"feiwu77777/Handwriting_gen","url":"https://github.com/feiwu77777/Handwriting_gen"},{"title":"feiwu77777/Handwriting_generation","url":"https://github.com/feiwu77777/Handwriting_generation"},{"title":"ciaua/HandwritingPyTorch","url":"https://github.com/ciaua/HandwritingPyTorch"},{"title":"Muiiya/research","url":"https://github.com/Muiiya/research"},{"title":"mohitRohatgi/handwritingGenerator","url":"https://github.com/mohitRohatgi/handwritingGenerator"},{"title":"ZhouYC627/NN_Compression","url":"https://github.com/ZhouYC627/NN_Compression"},{"title":"Julie260/DeepLearningProject","url":"https://github.com/Julie260/DeepLearningProject"},{"title":"CambridgeIIS/Gesture-Keyboard-Traj-Gen","url":"https://github.com/CambridgeIIS/Gesture-Keyboard-Traj-Gen"},{"title":"AsianZeus/Ooze-Handwritten-Text-Generator","url":"https://github.com/AsianZeus/Ooze-Handwritten-Text-Generator"},{"title":"HEIMaxER/descript-research-test","url":"https://github.com/HEIMaxER/descript-research-test"},{"title":"srikanth-sfu/lyrebird","url":"https://github.com/srikanth-sfu/lyrebird"},{"title":"badhandas/hw_synthesis","url":"https://github.com/badhandas/hw_synthesis"},{"title":"azfarkhoja305/Handwritten-RNNs","url":"https://github.com/azfarkhoja305/Handwritten-RNNs"},{"title":"KurtAhn/lyrebird","url":"https://github.com/KurtAhn/lyrebird"},{"title":"akhilkumarco007/lyrebird-egg-master","url":"https://github.com/akhilkumarco007/lyrebird-egg-master"},{"title":"Akella17/Handwriting_Synthesis","url":"https://github.com/Akella17/Handwriting_Synthesis"},{"title":"adrienphilardeau/Descript-Research-Test","url":"https://github.com/adrienphilardeau/Descript-Research-Test"},{"title":"rockyyliang/handwrite","url":"https://github.com/rockyyliang/handwrite"},{"title":"canneltigrou/testHandwritting","url":"https://github.com/canneltigrou/testHandwritting"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33740,"task":"Language Modelling","parent_task":null,"dataset":"enwik8","model_name":"All-attention network (36 layers)","metrics":{"Number of params":"114M"},"paper_url":"https://arxiv.org/abs/1907.01470v1","paper_title":"Augmenting Self-attention with Persistent Memory","paper_date":"2019-07-02","code_links":[{"title":"lucidrains/x-transformers","url":"https://github.com/lucidrains/x-transformers"},{"title":"facebookresearch/adaptive-span","url":"https://github.com/facebookresearch/adaptive-span"}],"metrics_order":"[\"Bit per Character (BPC)\", \"Number of params\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]}]}