{"task":"Language Modelling","dataset":"C4","metric_names":["Perplexity","TPUv3 Hours","Steps"],"rows":[{"id":33377,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"Primer","metrics":{"Perplexity":"12.35","Steps":"1M","TPUv3 Hours":"17.3K"},"paper_url":"https://arxiv.org/abs/2109.08668v2","paper_title":"Primer: Searching for Efficient Transformers for Language Modeling","paper_date":"2021-09-17","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/primer"},{"title":"lucidrains/FLASH-pytorch","url":"https://github.com/lucidrains/FLASH-pytorch"},{"title":"JunnYu/x-transformers-paddle","url":"https://github.com/JunnYu/x-transformers-paddle"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33378,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"Zeropoint LLM.int8 13B (vector-wise + decomp)","metrics":{"Perplexity":"12.45"},"paper_url":"https://arxiv.org/abs/2208.07339v2","paper_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","paper_date":"2022-08-15","code_links":[{"title":"timdettmers/bitsandbytes","url":"https://github.com/timdettmers/bitsandbytes"},{"title":"huggingface/transformers-bloom-inference","url":"https://github.com/huggingface/transformers-bloom-inference"},{"title":"kohjingyu/fromage","url":"https://github.com/kohjingyu/fromage"},{"title":"alextmallen/adaptive-retrieval","url":"https://github.com/alextmallen/adaptive-retrieval"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33379,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"T5++","metrics":{"Perplexity":"12.69","Steps":"1M","TPUv3 Hours":"16.5K"},"paper_url":"https://arxiv.org/abs/2109.08668v2","paper_title":"Primer: Searching for Efficient Transformers for Language Modeling","paper_date":"2021-09-17","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/primer"},{"title":"lucidrains/FLASH-pytorch","url":"https://github.com/lucidrains/FLASH-pytorch"},{"title":"JunnYu/x-transformers-paddle","url":"https://github.com/JunnYu/x-transformers-paddle"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33380,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"Original T5","metrics":{"Perplexity":"13.25","Steps":"1M","TPUv3 Hours":"15.7K"},"paper_url":"https://arxiv.org/abs/2109.08668v2","paper_title":"Primer: Searching for Efficient Transformers for Language Modeling","paper_date":"2021-09-17","code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/primer"},{"title":"lucidrains/FLASH-pytorch","url":"https://github.com/lucidrains/FLASH-pytorch"},{"title":"JunnYu/x-transformers-paddle","url":"https://github.com/JunnYu/x-transformers-paddle"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33381,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"LLM.float32 6.7B","metrics":{"Perplexity":"13.3"},"paper_url":"https://arxiv.org/abs/2208.07339v2","paper_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","paper_date":"2022-08-15","code_links":[{"title":"timdettmers/bitsandbytes","url":"https://github.com/timdettmers/bitsandbytes"},{"title":"huggingface/transformers-bloom-inference","url":"https://github.com/huggingface/transformers-bloom-inference"},{"title":"kohjingyu/fromage","url":"https://github.com/kohjingyu/fromage"},{"title":"alextmallen/adaptive-retrieval","url":"https://github.com/alextmallen/adaptive-retrieval"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33382,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"LLM.float32 2.7B","metrics":{"Perplexity":"14.43"},"paper_url":"https://arxiv.org/abs/2208.07339v2","paper_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","paper_date":"2022-08-15","code_links":[{"title":"timdettmers/bitsandbytes","url":"https://github.com/timdettmers/bitsandbytes"},{"title":"huggingface/transformers-bloom-inference","url":"https://github.com/huggingface/transformers-bloom-inference"},{"title":"kohjingyu/fromage","url":"https://github.com/kohjingyu/fromage"},{"title":"alextmallen/adaptive-retrieval","url":"https://github.com/alextmallen/adaptive-retrieval"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33383,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"N-Grammer 343M","metrics":{"Perplexity":"14.79"},"paper_url":"https://arxiv.org/abs/2207.06366v1","paper_title":"N-Grammer: Augmenting Transformers with latent n-grams","paper_date":"2022-07-13","code_links":[{"title":"tensorflow/lingvo","url":"https://github.com/tensorflow/lingvo"},{"title":"yiyixuxu/n-grammer-flax","url":"https://github.com/yiyixuxu/n-grammer-flax"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33384,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"N-Grammer 288M","metrics":{"Perplexity":"15.01"},"paper_url":"https://arxiv.org/abs/2207.06366v1","paper_title":"N-Grammer: Augmenting Transformers with latent n-grams","paper_date":"2022-07-13","code_links":[{"title":"tensorflow/lingvo","url":"https://github.com/tensorflow/lingvo"},{"title":"yiyixuxu/n-grammer-flax","url":"https://github.com/yiyixuxu/n-grammer-flax"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]},{"id":33385,"task":"Language Modelling","parent_task":null,"dataset":"C4","model_name":"LLM.float32 1.3B","metrics":{"Perplexity":"15.91"},"paper_url":"https://arxiv.org/abs/2208.07339v2","paper_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","paper_date":"2022-08-15","code_links":[{"title":"timdettmers/bitsandbytes","url":"https://github.com/timdettmers/bitsandbytes"},{"title":"huggingface/transformers-bloom-inference","url":"https://github.com/huggingface/transformers-bloom-inference"},{"title":"kohjingyu/fromage","url":"https://github.com/kohjingyu/fromage"},{"title":"alextmallen/adaptive-retrieval","url":"https://github.com/alextmallen/adaptive-retrieval"}],"metrics_order":"[\"Perplexity\", \"TPUv3 Hours\", \"Steps\"]","area":"Medical","uses_additional_data":0,"source":"archive","tags":[]}]}