{"task":"Code Generation","dataset":"MBPP","metric_names":["Accuracy"],"rows":[{"id":29425,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"EG-CFG (DeepSeek-V3-0324)","metrics":{"Accuracy":"96.6"},"paper_url":"https://arxiv.org/abs/2506.10948v1","paper_title":"Execution Guided Line-by-Line Code Generation","paper_date":"2025-06-12","code_links":[{"title":"boazlavon/eg_cfg","url":"https://github.com/boazlavon/eg_cfg"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29426,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"QualityFlow (Sonnet-3.5)","metrics":{"Accuracy":"94.2"},"paper_url":"https://arxiv.org/abs/2501.17167v2","paper_title":"QualityFlow: An Agentic Workflow for Program Synthesis Controlled by LLM Quality Checks","paper_date":"2025-01-20","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29427,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"o1-mini + MapCoder (Hamming.ai)","metrics":{"Accuracy":"93.2"},"paper_url":"https://arxiv.org/abs/2405.11403v1","paper_title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","paper_date":"2024-05-18","code_links":[{"title":"md-ashraful-pramanik/mapcoder","url":"https://github.com/md-ashraful-pramanik/mapcoder"},{"title":"Luoji-zju/Agents4PLC_release","url":"https://github.com/Luoji-zju/Agents4PLC_release"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29428,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"MGDebugger (DeepSeek-V3-0324)","metrics":{"Accuracy":"92.4"},"paper_url":"https://arxiv.org/abs/2410.01215v2","paper_title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","paper_date":"2024-10-02","code_links":[{"title":"YerbaPage/MGDebugger","url":"https://github.com/YerbaPage/MGDebugger"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29429,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-4 + AgentCoder","metrics":{"Accuracy":"91.8"},"paper_url":"https://arxiv.org/abs/2312.13010v3","paper_title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","paper_date":"2023-12-20","code_links":[{"title":"huangd1999/AgentCoder","url":"https://github.com/huangd1999/AgentCoder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29430,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"CodeSim (GPT4o)","metrics":{"Accuracy":"90.7"},"paper_url":"https://arxiv.org/abs/2502.05664v1","paper_title":"CODESIM: Multi-Agent Code Generation and Problem Solving through Simulation-Driven Planning and Debugging","paper_date":"2025-02-08","code_links":[{"title":"kagnlp/CodeGenerator","url":"https://github.com/kagnlp/CodeGenerator"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29431,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Jiutian-大模型","metrics":{"Accuracy":"90.0"},"paper_url":null,"paper_title":null,"paper_date":null,"code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29432,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (ChatGPT) + AgentCoder","metrics":{"Accuracy":"89.9"},"paper_url":"https://arxiv.org/abs/2312.13010v3","paper_title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","paper_date":"2023-12-20","code_links":[{"title":"huangd1999/AgentCoder","url":"https://github.com/huangd1999/AgentCoder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29433,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"MapCoder (GPT-4o)","metrics":{"Accuracy":"89.7"},"paper_url":"https://arxiv.org/abs/2405.11403v1","paper_title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","paper_date":"2024-05-18","code_links":[{"title":"md-ashraful-pramanik/mapcoder","url":"https://github.com/md-ashraful-pramanik/mapcoder"},{"title":"Luoji-zju/Agents4PLC_release","url":"https://github.com/Luoji-zju/Agents4PLC_release"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29434,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-4 (ChatGPT Plus)","metrics":{"Accuracy":"87.5"},"paper_url":"https://arxiv.org/abs/2307.12488v5","paper_title":"How Does Naming Affect LLMs on Code Analysis Tasks?","paper_date":"2023-07-24","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29435,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Claude 3 Opus","metrics":{"Accuracy":"86.4"},"paper_url":"https://www.anthropic.com/news/claude-3-family","paper_title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","paper_date":"2024-03-04","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29436,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"LPW (GPT-4o)","metrics":{"Accuracy":"84.8"},"paper_url":"https://arxiv.org/abs/2411.14503v3","paper_title":"Planning-Driven Programming: A Large Language Model Programming Workflow","paper_date":"2024-11-21","code_links":[{"title":"you68681/lpw","url":"https://github.com/you68681/lpw"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29437,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo + FlowGenScrum + Test","metrics":{"Accuracy":"83.8±0.6"},"paper_url":"https://arxiv.org/abs/2403.15852v2","paper_title":"SOEN-101: Code Generation by Emulating Software Process Models Using Large Language Model Agents","paper_date":"2024-03-23","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29438,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"AFlow(GPT-4o-mini)","metrics":{"Accuracy":"83.4"},"paper_url":"https://arxiv.org/abs/2410.10762v4","paper_title":"AFlow: Automating Agentic Workflow Generation","paper_date":"2024-10-14","code_links":[{"title":"geekan/metagpt","url":"https://github.com/geekan/metagpt"},{"title":"evoagentx/evoagentx","url":"https://github.com/evoagentx/evoagentx"},{"title":"qixucen/atom","url":"https://github.com/qixucen/atom"},{"title":"foundationagents/aflow","url":"https://github.com/foundationagents/aflow"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29439,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (ChatGPT)","metrics":{"Accuracy":"83.2"},"paper_url":"https://arxiv.org/abs/2307.12488v5","paper_title":"How Does Naming Affect LLMs on Code Analysis Tasks?","paper_date":"2023-07-24","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29440,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"EG-CFG (DeepSeek Coder 1.3b Instruct)","metrics":{"Accuracy":"83.2"},"paper_url":"https://arxiv.org/abs/2506.10948v1","paper_title":"Execution Guided Line-by-Line Code Generation","paper_date":"2025-06-12","code_links":[{"title":"boazlavon/eg_cfg","url":"https://github.com/boazlavon/eg_cfg"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29441,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"MapCoder (GPT-4)","metrics":{"Accuracy":"83.1"},"paper_url":"https://arxiv.org/abs/2405.11403v1","paper_title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","paper_date":"2024-05-18","code_links":[{"title":"md-ashraful-pramanik/mapcoder","url":"https://github.com/md-ashraful-pramanik/mapcoder"},{"title":"Luoji-zju/Agents4PLC_release","url":"https://github.com/Luoji-zju/Agents4PLC_release"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29442,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"o1-mini + Language Agent Tree Search (Hamming.ai)","metrics":{"Accuracy":"82.3"},"paper_url":"https://arxiv.org/abs/2310.04406v3","paper_title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","paper_date":"2023-10-06","code_links":[{"title":"lapisrocks/languageagenttreesearch","url":"https://github.com/lapisrocks/languageagenttreesearch"},{"title":"andyz245/LanguageAgentTreeSearch","url":"https://github.com/andyz245/LanguageAgentTreeSearch"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29443,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-4 (Bing Chat)","metrics":{"Accuracy":"82"},"paper_url":"https://arxiv.org/abs/2307.12488v5","paper_title":"How Does Naming Affect LLMs on Code Analysis Tasks?","paper_date":"2023-07-24","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29444,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo + Language Agent Tree Search","metrics":{"Accuracy":"81.1"},"paper_url":"https://arxiv.org/abs/2310.04406v3","paper_title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","paper_date":"2023-10-06","code_links":[{"title":"lapisrocks/languageagenttreesearch","url":"https://github.com/lapisrocks/languageagenttreesearch"},{"title":"andyz245/LanguageAgentTreeSearch","url":"https://github.com/andyz245/LanguageAgentTreeSearch"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29445,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"MGDebugger (CodeQwen1.5)","metrics":{"Accuracy":"80.8"},"paper_url":"https://arxiv.org/abs/2410.01215v2","paper_title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","paper_date":"2024-10-02","code_links":[{"title":"YerbaPage/MGDebugger","url":"https://github.com/YerbaPage/MGDebugger"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29446,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Claude 3 Haiku","metrics":{"Accuracy":"80.4"},"paper_url":"https://www.anthropic.com/news/claude-3-family","paper_title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","paper_date":"2024-03-04","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29447,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-4 (Self-Debugging with unit tests + trace)","metrics":{"Accuracy":"80.2"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29448,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-4 (few-shot)","metrics":{"Accuracy":"80"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29449,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Claude 3 Sonnet","metrics":{"Accuracy":"79.4"},"paper_url":"https://www.anthropic.com/news/claude-3-family","paper_title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","paper_date":"2024-03-04","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29450,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Bard (PaLM 2/chat-bison-001)","metrics":{"Accuracy":"76.2"},"paper_url":"https://arxiv.org/abs/2307.12488v5","paper_title":"How Does Naming Affect LLMs on Code Analysis Tasks?","paper_date":"2023-07-24","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29451,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (Self-Debugging with unit tests + trace)","metrics":{"Accuracy":"72.8"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29452,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Claude","metrics":{"Accuracy":"71.4"},"paper_url":"https://arxiv.org/abs/2307.12488v5","paper_title":"How Does Naming Affect LLMs on Code Analysis Tasks?","paper_date":"2023-07-24","code_links":[],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29453,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B (Self-Debugging with unit tests + trace)","metrics":{"Accuracy":"70.8"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29454,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (few-shot)","metrics":{"Accuracy":"70.8"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29455,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"DeepSeek-Coder-Instruct 33B (few-shot)","metrics":{"Accuracy":"70"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29456,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo + INTERVENOR","metrics":{"Accuracy":"69.8"},"paper_url":"https://arxiv.org/abs/2311.09868v5","paper_title":"INTERVENOR: Prompting the Coding Ability of Large Language Models with the Interactive Chain of Repair","paper_date":"2023-11-16","code_links":[{"title":"neuir/intervenor","url":"https://github.com/neuir/intervenor"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29457,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B + LEVER","metrics":{"Accuracy":"68.9"},"paper_url":"https://arxiv.org/abs/2302.08468v3","paper_title":"LEVER: Learning to Verify Language-to-Code Generation with Execution","paper_date":"2023-02-16","code_links":[{"title":"niansong1996/lever","url":"https://github.com/niansong1996/lever"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29458,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B + CodeT","metrics":{"Accuracy":"67.7"},"paper_url":"https://arxiv.org/abs/2207.10397v2","paper_title":"CodeT: Code Generation with Generated Tests","paper_date":"2022-07-21","code_links":[{"title":"microsoft/codet","url":"https://github.com/microsoft/codet"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29459,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (3-shot)","metrics":{"Accuracy":"67.6"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29460,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B + Reviewer","metrics":{"Accuracy":"66.9"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29461,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B + Coder-Reviewer","metrics":{"Accuracy":"66.4"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29462,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"StarCoder2-15B","metrics":{"Accuracy":"66.2"},"paper_url":"https://arxiv.org/abs/2402.19173v1","paper_title":"StarCoder 2 and The Stack v2: The Next Generation","paper_date":"2024-02-29","code_links":[{"title":"bigcode-project/starcoder2","url":"https://github.com/bigcode-project/starcoder2"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/starcoder"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/starcoder2"},{"title":"ana-oprescu/greenllms","url":"https://github.com/ana-oprescu/greenllms"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29463,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"DeepSeek-Coder-Base 33B (few-shot)","metrics":{"Accuracy":"66"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29464,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Python 70B (3-shot)","metrics":{"Accuracy":"65.5"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29465,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"DeepSeek-Coder-Instruct 6.7B (few-shot)","metrics":{"Accuracy":"65.4"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29466,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B + MBR-Exec","metrics":{"Accuracy":"63"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29467,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama 70B (3-shot)","metrics":{"Accuracy":"62.4"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29468,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Instruct 70B (3-shot)","metrics":{"Accuracy":"62.2"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29469,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-001 175B + CodeT","metrics":{"Accuracy":"61.9"},"paper_url":"https://arxiv.org/abs/2207.10397v2","paper_title":"CodeT: Code Generation with Generated Tests","paper_date":"2022-07-21","code_links":[{"title":"microsoft/codet","url":"https://github.com/microsoft/codet"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29470,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-002 175B (3-shot)","metrics":{"Accuracy":"61.4"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29471,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Unnatural Code Llama 34B (3-shot)","metrics":{"Accuracy":"61.2"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29472,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Mixtral 8x7B (3-shot)","metrics":{"Accuracy":"60.7"},"paper_url":"https://arxiv.org/abs/2401.04088v1","paper_title":"Mixtral of Experts","paper_date":"2024-01-08","code_links":[{"title":"jingyaogong/minimind","url":"https://github.com/jingyaogong/minimind"},{"title":"hit-scir/chinese-mixtral-8x7b","url":"https://github.com/hit-scir/chinese-mixtral-8x7b"},{"title":"ymcui/chinese-mixtral","url":"https://github.com/ymcui/chinese-mixtral"},{"title":"consequentai/fneval","url":"https://github.com/consequentai/fneval"},{"title":"kamanphoebe/look-into-moes","url":"https://github.com/kamanphoebe/look-into-moes"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/2/mixtral"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29473,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"DeepSeek-Coder-Base 6.7B (few-shot)","metrics":{"Accuracy":"60.6"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29474,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-davinci-001 175B + MBR-Exec","metrics":{"Accuracy":"58.2"},"paper_url":"https://arxiv.org/abs/2204.11454v2","paper_title":"Natural Language to Code Translation with Execution","paper_date":"2022-04-25","code_links":[{"title":"facebookresearch/mbr-exec","url":"https://github.com/facebookresearch/mbr-exec"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29475,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Instruct 34B (3-shot)","metrics":{"Accuracy":"57"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29476,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Python 34B (3-shot)","metrics":{"Accuracy":"56.2"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29477,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-cushman-001 12B (CodeT)","metrics":{"Accuracy":"55.4"},"paper_url":"https://arxiv.org/abs/2207.10397v2","paper_title":"CodeT: Code Generation with Generated Tests","paper_date":"2022-07-21","code_links":[{"title":"microsoft/codet","url":"https://github.com/microsoft/codet"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29478,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama 34B (3-shot)","metrics":{"Accuracy":"55"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29479,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"StarCoder 15.5B (Self-Debugging with unit tests + trace)","metrics":{"Accuracy":"53.2"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29480,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"StarCoder 15.5B","metrics":{"Accuracy":"52.7"},"paper_url":"https://arxiv.org/abs/2305.06161v2","paper_title":"StarCoder: may the source be with you!","paper_date":"2023-05-09","code_links":[{"title":"bigcode-project/starcoder","url":"https://github.com/bigcode-project/starcoder"},{"title":"ise-uiuc/magicoder","url":"https://github.com/ise-uiuc/magicoder"},{"title":"nuprl/multipl-e","url":"https://github.com/nuprl/multipl-e"},{"title":"mcgill-nlp/length-generalization","url":"https://github.com/mcgill-nlp/length-generalization"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29481,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo","metrics":{"Accuracy":"52.2"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29482,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"WizardCoder 15B","metrics":{"Accuracy":"51.8"},"paper_url":"https://arxiv.org/abs/2306.08568v1","paper_title":"WizardCoder: Empowering Code Large Language Models with Evol-Instruct","paper_date":"2023-06-14","code_links":[{"title":"nlpxucan/wizardlm","url":"https://github.com/nlpxucan/wizardlm"},{"title":"nickrosh/evol-teacher","url":"https://github.com/nickrosh/evol-teacher"},{"title":"kyle-lyu/data-efficient-finetuning","url":"https://github.com/kyle-lyu/data-efficient-finetuning"},{"title":"kyle-lyu/codeact","url":"https://github.com/kyle-lyu/codeact"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":1,"source":"archive","tags":[]},{"id":29483,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"PaLM 2-S* (few-shot)","metrics":{"Accuracy":"50"},"paper_url":"https://arxiv.org/abs/2305.10403v3","paper_title":"PaLM 2 Technical Report","paper_date":"2023-05-17","code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29484,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"CodeGen-Mono 16B + CodeT","metrics":{"Accuracy":"49.5"},"paper_url":"https://arxiv.org/abs/2207.10397v2","paper_title":"CodeT: Code Generation with Generated Tests","paper_date":"2022-07-21","code_links":[{"title":"microsoft/codet","url":"https://github.com/microsoft/codet"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29485,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Instruct 13B (3-shot)","metrics":{"Accuracy":"49.4"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29486,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"DeepSeek-Coder-Instruct 1.3B (few-shot)","metrics":{"Accuracy":"49.4"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29487,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"StarCoderBase 15.5B","metrics":{"Accuracy":"49"},"paper_url":"https://arxiv.org/abs/2305.06161v2","paper_title":"StarCoder: may the source be with you!","paper_date":"2023-05-09","code_links":[{"title":"bigcode-project/starcoder","url":"https://github.com/bigcode-project/starcoder"},{"title":"ise-uiuc/magicoder","url":"https://github.com/ise-uiuc/magicoder"},{"title":"nuprl/multipl-e","url":"https://github.com/nuprl/multipl-e"},{"title":"mcgill-nlp/length-generalization","url":"https://github.com/mcgill-nlp/length-generalization"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29488,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Python 13B (3-shot)","metrics":{"Accuracy":"49"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29489,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Qwen2idae-16x14B (4-shot)","metrics":{"Accuracy":"48.6"},"paper_url":"https://arxiv.org/abs/2401.02731v4","paper_title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","paper_date":"2024-01-05","code_links":[{"title":"wuhy68/parameter-efficient-moe","url":"https://github.com/wuhy68/parameter-efficient-moe"},{"title":"ShayekhBinIslam/openrag","url":"https://github.com/ShayekhBinIslam/openrag"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29490,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"code-cushman-001 12B + MBR-Exec","metrics":{"Accuracy":"48.3"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29491,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Python 7B (3-shot)","metrics":{"Accuracy":"47.6"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29492,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Mistral 7B (3-shot)","metrics":{"Accuracy":"47.5"},"paper_url":"https://arxiv.org/abs/2310.06825v1","paper_title":"Mistral 7B","paper_date":"2023-10-10","code_links":[{"title":"mistralai/mistral-src","url":"https://github.com/mistralai/mistral-src"},{"title":"facebookresearch/fairseq2","url":"https://github.com/facebookresearch/fairseq2"},{"title":"mgmalek/efficient_cross_entropy","url":"https://github.com/mgmalek/efficient_cross_entropy"},{"title":"ninglab/ecellm","url":"https://github.com/ninglab/ecellm"},{"title":"knowlab/bi-weekly-paper-presentation","url":"https://github.com/knowlab/bi-weekly-paper-presentation"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/2/mistral"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29493,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"CodeGen 16B + MBR-Exec","metrics":{"Accuracy":"47.3"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29494,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"StarCoder 15.5B (3-shot)","metrics":{"Accuracy":"47.2"},"paper_url":"https://arxiv.org/abs/2304.05128v2","paper_title":"Teaching Large Language Models to Self-Debug","paper_date":"2023-04-11","code_links":[{"title":"amazon-science/SDFeedback","url":"https://github.com/amazon-science/SDFeedback"},{"title":"amazon-science/self_debug","url":"https://github.com/amazon-science/self_debug"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29495,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"PaLM Coder 540B","metrics":{"Accuracy":"47"},"paper_url":"https://arxiv.org/abs/2204.02311v5","paper_title":"PaLM: Scaling Language Modeling with Pathways","paper_date":"2022-04-05","code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29496,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama 13B (3-shot)","metrics":{"Accuracy":"47"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29497,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"CodeGen 16B + Coder-Reviewer","metrics":{"Accuracy":"46.2"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29498,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"DeepSeek-Coder-Base 1.3B (few-shot)","metrics":{"Accuracy":"46.2"},"paper_url":"https://arxiv.org/abs/2401.14196v2","paper_title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","paper_date":"2024-01-25","code_links":[{"title":"deepseek-ai/DeepSeek-Coder","url":"https://github.com/deepseek-ai/DeepSeek-Coder"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29499,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (few-shot)","metrics":{"Accuracy":"45.4"},"paper_url":"https://arxiv.org/abs/2311.09868v5","paper_title":"INTERVENOR: Prompting the Coding Ability of Large Language Models with the Interactive Chain of Repair","paper_date":"2023-11-16","code_links":[{"title":"neuir/intervenor","url":"https://github.com/neuir/intervenor"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29500,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Llama 2 70B (zero-shot)","metrics":{"Accuracy":"45"},"paper_url":"https://arxiv.org/abs/2307.09288v2","paper_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","paper_date":"2023-07-18","code_links":[{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"Lightning-AI/lit-gpt","url":"https://github.com/Lightning-AI/lit-gpt"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"IBM/Dromedary","url":"https://github.com/IBM/Dromedary"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"xverse-ai/xverse-13b","url":"https://github.com/xverse-ai/xverse-13b"},{"title":"usyd-fsalab/fp6_llm","url":"https://github.com/usyd-fsalab/fp6_llm"},{"title":"rijgersberg/geitje","url":"https://github.com/rijgersberg/geitje"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"glb400/Toy-RecLM","url":"https://github.com/glb400/Toy-RecLM"},{"title":"ninglab/ecellm","url":"https://github.com/ninglab/ecellm"},{"title":"zurichnlp/contradecode","url":"https://github.com/zurichnlp/contradecode"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"idiap/abroad-re","url":"https://github.com/idiap/abroad-re"},{"title":"meetyou-ai-lab/can-mc-evaluate-llms","url":"https://github.com/meetyou-ai-lab/can-mc-evaluate-llms"},{"title":"xuetianci/pacit","url":"https://github.com/xuetianci/pacit"},{"title":"coastalcph/eu-politics-llms","url":"https://github.com/coastalcph/eu-politics-llms"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29501,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama - Instruct 7B (3-shot)","metrics":{"Accuracy":"44.4"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29502,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"CodeGen 16B + Reviewer","metrics":{"Accuracy":"44.1"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29503,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"phi-1.5-web 1.3B","metrics":{"Accuracy":"43.5"},"paper_url":"https://arxiv.org/abs/2309.05463v1","paper_title":"Textbooks Are All You Need II: phi-1.5 technical report","paper_date":"2023-09-11","code_links":[{"title":"knowlab/bi-weekly-paper-presentation","url":"https://github.com/knowlab/bi-weekly-paper-presentation"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29504,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Branch-Train-Merge 4x7B (top-2)","metrics":{"Accuracy":"42.6"},"paper_url":"https://arxiv.org/abs/2403.07816v1","paper_title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","paper_date":"2024-03-12","code_links":[{"title":"Leeroo-AI/mergoo","url":"https://github.com/Leeroo-AI/mergoo"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29505,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Code Llama 7B (3-shot)","metrics":{"Accuracy":"41.4"},"paper_url":"https://arxiv.org/abs/2308.12950v3","paper_title":"Code Llama: Open Foundation Models for Code","paper_date":"2023-08-24","code_links":[{"title":"facebookresearch/codellama","url":"https://github.com/facebookresearch/codellama"},{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29506,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Camelidae-8×34B (4-shot)","metrics":{"Accuracy":"41.4"},"paper_url":"https://arxiv.org/abs/2401.02731v4","paper_title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","paper_date":"2024-01-05","code_links":[{"title":"wuhy68/parameter-efficient-moe","url":"https://github.com/wuhy68/parameter-efficient-moe"},{"title":"ShayekhBinIslam/openrag","url":"https://github.com/ShayekhBinIslam/openrag"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29507,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"GPT-3.5 Turbo (0-shot)","metrics":{"Accuracy":"39.8"},"paper_url":"https://arxiv.org/abs/2311.09868v5","paper_title":"INTERVENOR: Prompting the Coding Ability of Large Language Models with the Interactive Chain of Repair","paper_date":"2023-11-16","code_links":[{"title":"neuir/intervenor","url":"https://github.com/neuir/intervenor"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29508,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Branch-Train-MiX 4x7B (sampling top-2 experts)","metrics":{"Accuracy":"39.4"},"paper_url":"https://arxiv.org/abs/2403.07816v1","paper_title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","paper_date":"2024-03-12","code_links":[{"title":"Leeroo-AI/mergoo","url":"https://github.com/Leeroo-AI/mergoo"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29509,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"LLaMA 65B (0-shot)","metrics":{"Accuracy":"37.7"},"paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","paper_date":"2023-02-27","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"ggml-org/llama.cpp","url":"https://github.com/ggml-org/llama.cpp"},{"title":"ggerganov/llama.cpp","url":"https://github.com/ggerganov/llama.cpp"},{"title":"meta-llama/llama","url":"https://github.com/meta-llama/llama"},{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"tatsu-lab/stanford_alpaca","url":"https://github.com/tatsu-lab/stanford_alpaca"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"nebuly-ai/nebullvm","url":"https://github.com/nebuly-ai/nebullvm/tree/main/apps/accelerate/chatllama"},{"title":"Lightning-AI/lit-llama","url":"https://github.com/Lightning-AI/lit-llama"},{"title":"facico/chinese-vicuna","url":"https://github.com/facico/chinese-vicuna"},{"title":"qwopqwop200/GPTQ-for-LLaMa","url":"https://github.com/qwopqwop200/GPTQ-for-LLaMa"},{"title":"phoebussi/alpaca-cot","url":"https://github.com/phoebussi/alpaca-cot"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"xusenlinzy/api-for-open-llm","url":"https://github.com/xusenlinzy/api-for-open-llm"},{"title":"guinmoon/llmfarm","url":"https://github.com/guinmoon/llmfarm"},{"title":"aethercortex/llama-x","url":"https://github.com/aethercortex/llama-x"},{"title":"beomi/koalpaca","url":"https://github.com/beomi/koalpaca"},{"title":"freedomintelligence/huatuogpt","url":"https://github.com/freedomintelligence/huatuogpt"},{"title":"icalk-nlp/educhat","url":"https://github.com/icalk-nlp/educhat"},{"title":"ecnu-icalk/educhat","url":"https://github.com/ecnu-icalk/educhat"},{"title":"ganjinzero/rrhf","url":"https://github.com/ganjinzero/rrhf"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"chaoyi-wu/pmc-llama","url":"https://github.com/chaoyi-wu/pmc-llama"},{"title":"kbressem/medalpaca","url":"https://github.com/kbressem/medalpaca"},{"title":"chaoyi-wu/finetune_llama","url":"https://github.com/chaoyi-wu/finetune_llama"},{"title":"ofa-sys/expertllama","url":"https://github.com/ofa-sys/expertllama"},{"title":"xiaoman-zhang/PMC-VQA","url":"https://github.com/xiaoman-zhang/PMC-VQA"},{"title":"fsoft-ai4code/codecapybara","url":"https://github.com/fsoft-ai4code/codecapybara"},{"title":"kayvr/token-hawk","url":"https://github.com/kayvr/token-hawk"},{"title":"ntunlplab/traditional-chinese-alpaca","url":"https://github.com/ntunlplab/traditional-chinese-alpaca"},{"title":"teelinsan/camoscio","url":"https://github.com/teelinsan/camoscio"},{"title":"replicate/cog_stanford_alpaca","url":"https://github.com/replicate/cog_stanford_alpaca"},{"title":"greenbitai/low_bit_llama","url":"https://github.com/greenbitai/low_bit_llama"},{"title":"krafton-ai/korani","url":"https://github.com/krafton-ai/korani"},{"title":"yuanmu97/secure-transformer-inference","url":"https://github.com/yuanmu97/secure-transformer-inference"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"hamishivi/easylm","url":"https://github.com/hamishivi/easylm"},{"title":"xvyaward/owq","url":"https://github.com/xvyaward/owq"},{"title":"vcskaushik/LLMzip","url":"https://github.com/vcskaushik/LLMzip"},{"title":"batsresearch/alfred","url":"https://github.com/batsresearch/alfred"},{"title":"grantslatton/llama.cpp","url":"https://github.com/grantslatton/llama.cpp"},{"title":"zihanzhaosjtu/librisqa","url":"https://github.com/zihanzhaosjtu/librisqa"},{"title":"fajri91/indommlu","url":"https://github.com/fajri91/indommlu"},{"title":"akanyaani/miniLLAMA","url":"https://github.com/akanyaani/miniLLAMA"},{"title":"stanfordbdhg/llama.cpp","url":"https://github.com/stanfordbdhg/llama.cpp"},{"title":"facebookresearch/chai","url":"https://github.com/facebookresearch/chai"},{"title":"aozhongzhang/magr","url":"https://github.com/aozhongzhang/magr"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"ohadrubin/rpt","url":"https://github.com/ohadrubin/rpt"},{"title":"ecolab-postech/owq","url":"https://github.com/ecolab-postech/owq"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama2"},{"title":"Mind23-2/MindCode-140","url":"https://github.com/Mind23-2/MindCode-140"},{"title":"longhao-chen/aicas2024","url":"https://github.com/longhao-chen/aicas2024"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/llama"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/llama"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29510,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"PaLM 540B","metrics":{"Accuracy":"36.8"},"paper_url":"https://arxiv.org/abs/2204.02311v5","paper_title":"PaLM: Scaling Language Modeling with Pathways","paper_date":"2022-04-05","code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29511,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"SantaCoder 1.1B","metrics":{"Accuracy":"35"},"paper_url":"https://arxiv.org/abs/2305.06161v2","paper_title":"StarCoder: may the source be with you!","paper_date":"2023-05-09","code_links":[{"title":"bigcode-project/starcoder","url":"https://github.com/bigcode-project/starcoder"},{"title":"ise-uiuc/magicoder","url":"https://github.com/ise-uiuc/magicoder"},{"title":"nuprl/multipl-e","url":"https://github.com/nuprl/multipl-e"},{"title":"mcgill-nlp/length-generalization","url":"https://github.com/mcgill-nlp/length-generalization"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29512,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"InCoder 6.7B + CodeT","metrics":{"Accuracy":"34.4"},"paper_url":"https://arxiv.org/abs/2207.10397v2","paper_title":"CodeT: Code Generation with Generated Tests","paper_date":"2022-07-21","code_links":[{"title":"microsoft/codet","url":"https://github.com/microsoft/codet"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":597600,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Programming","metrics":{"Accuracy":"34"},"paper_url":"https://paperswithcode.com/paper/context-augmented-code-generation-using-programming-knowledge-graphs","paper_title":"Context-Augmented Code Generation Using Programming Knowledge Graphs","paper_date":"2026-01-28","code_links":[],"metrics_order":null,"area":null,"uses_additional_data":null,"source":"auto","tags":[]},{"id":29513,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Llama 2 34B (0-shot)","metrics":{"Accuracy":"33"},"paper_url":"https://arxiv.org/abs/2307.09288v2","paper_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","paper_date":"2023-07-18","code_links":[{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"Lightning-AI/lit-gpt","url":"https://github.com/Lightning-AI/lit-gpt"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"IBM/Dromedary","url":"https://github.com/IBM/Dromedary"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"xverse-ai/xverse-13b","url":"https://github.com/xverse-ai/xverse-13b"},{"title":"usyd-fsalab/fp6_llm","url":"https://github.com/usyd-fsalab/fp6_llm"},{"title":"rijgersberg/geitje","url":"https://github.com/rijgersberg/geitje"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"glb400/Toy-RecLM","url":"https://github.com/glb400/Toy-RecLM"},{"title":"ninglab/ecellm","url":"https://github.com/ninglab/ecellm"},{"title":"zurichnlp/contradecode","url":"https://github.com/zurichnlp/contradecode"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"idiap/abroad-re","url":"https://github.com/idiap/abroad-re"},{"title":"meetyou-ai-lab/can-mc-evaluate-llms","url":"https://github.com/meetyou-ai-lab/can-mc-evaluate-llms"},{"title":"xuetianci/pacit","url":"https://github.com/xuetianci/pacit"},{"title":"coastalcph/eu-politics-llms","url":"https://github.com/coastalcph/eu-politics-llms"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29514,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Llama 2 13B (0-shot)","metrics":{"Accuracy":"30.6"},"paper_url":"https://arxiv.org/abs/2307.09288v2","paper_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","paper_date":"2023-07-18","code_links":[{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"Lightning-AI/lit-gpt","url":"https://github.com/Lightning-AI/lit-gpt"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"IBM/Dromedary","url":"https://github.com/IBM/Dromedary"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"xverse-ai/xverse-13b","url":"https://github.com/xverse-ai/xverse-13b"},{"title":"usyd-fsalab/fp6_llm","url":"https://github.com/usyd-fsalab/fp6_llm"},{"title":"rijgersberg/geitje","url":"https://github.com/rijgersberg/geitje"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"glb400/Toy-RecLM","url":"https://github.com/glb400/Toy-RecLM"},{"title":"ninglab/ecellm","url":"https://github.com/ninglab/ecellm"},{"title":"zurichnlp/contradecode","url":"https://github.com/zurichnlp/contradecode"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"idiap/abroad-re","url":"https://github.com/idiap/abroad-re"},{"title":"meetyou-ai-lab/can-mc-evaluate-llms","url":"https://github.com/meetyou-ai-lab/can-mc-evaluate-llms"},{"title":"xuetianci/pacit","url":"https://github.com/xuetianci/pacit"},{"title":"coastalcph/eu-politics-llms","url":"https://github.com/coastalcph/eu-politics-llms"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29515,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"LLaMA 33B (0-shot)","metrics":{"Accuracy":"30.2"},"paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","paper_date":"2023-02-27","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"ggml-org/llama.cpp","url":"https://github.com/ggml-org/llama.cpp"},{"title":"ggerganov/llama.cpp","url":"https://github.com/ggerganov/llama.cpp"},{"title":"meta-llama/llama","url":"https://github.com/meta-llama/llama"},{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"tatsu-lab/stanford_alpaca","url":"https://github.com/tatsu-lab/stanford_alpaca"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"nebuly-ai/nebullvm","url":"https://github.com/nebuly-ai/nebullvm/tree/main/apps/accelerate/chatllama"},{"title":"Lightning-AI/lit-llama","url":"https://github.com/Lightning-AI/lit-llama"},{"title":"facico/chinese-vicuna","url":"https://github.com/facico/chinese-vicuna"},{"title":"qwopqwop200/GPTQ-for-LLaMa","url":"https://github.com/qwopqwop200/GPTQ-for-LLaMa"},{"title":"phoebussi/alpaca-cot","url":"https://github.com/phoebussi/alpaca-cot"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"xusenlinzy/api-for-open-llm","url":"https://github.com/xusenlinzy/api-for-open-llm"},{"title":"guinmoon/llmfarm","url":"https://github.com/guinmoon/llmfarm"},{"title":"aethercortex/llama-x","url":"https://github.com/aethercortex/llama-x"},{"title":"beomi/koalpaca","url":"https://github.com/beomi/koalpaca"},{"title":"freedomintelligence/huatuogpt","url":"https://github.com/freedomintelligence/huatuogpt"},{"title":"icalk-nlp/educhat","url":"https://github.com/icalk-nlp/educhat"},{"title":"ecnu-icalk/educhat","url":"https://github.com/ecnu-icalk/educhat"},{"title":"ganjinzero/rrhf","url":"https://github.com/ganjinzero/rrhf"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"chaoyi-wu/pmc-llama","url":"https://github.com/chaoyi-wu/pmc-llama"},{"title":"kbressem/medalpaca","url":"https://github.com/kbressem/medalpaca"},{"title":"chaoyi-wu/finetune_llama","url":"https://github.com/chaoyi-wu/finetune_llama"},{"title":"ofa-sys/expertllama","url":"https://github.com/ofa-sys/expertllama"},{"title":"xiaoman-zhang/PMC-VQA","url":"https://github.com/xiaoman-zhang/PMC-VQA"},{"title":"fsoft-ai4code/codecapybara","url":"https://github.com/fsoft-ai4code/codecapybara"},{"title":"kayvr/token-hawk","url":"https://github.com/kayvr/token-hawk"},{"title":"ntunlplab/traditional-chinese-alpaca","url":"https://github.com/ntunlplab/traditional-chinese-alpaca"},{"title":"teelinsan/camoscio","url":"https://github.com/teelinsan/camoscio"},{"title":"replicate/cog_stanford_alpaca","url":"https://github.com/replicate/cog_stanford_alpaca"},{"title":"greenbitai/low_bit_llama","url":"https://github.com/greenbitai/low_bit_llama"},{"title":"krafton-ai/korani","url":"https://github.com/krafton-ai/korani"},{"title":"yuanmu97/secure-transformer-inference","url":"https://github.com/yuanmu97/secure-transformer-inference"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"hamishivi/easylm","url":"https://github.com/hamishivi/easylm"},{"title":"xvyaward/owq","url":"https://github.com/xvyaward/owq"},{"title":"vcskaushik/LLMzip","url":"https://github.com/vcskaushik/LLMzip"},{"title":"batsresearch/alfred","url":"https://github.com/batsresearch/alfred"},{"title":"grantslatton/llama.cpp","url":"https://github.com/grantslatton/llama.cpp"},{"title":"zihanzhaosjtu/librisqa","url":"https://github.com/zihanzhaosjtu/librisqa"},{"title":"fajri91/indommlu","url":"https://github.com/fajri91/indommlu"},{"title":"akanyaani/miniLLAMA","url":"https://github.com/akanyaani/miniLLAMA"},{"title":"stanfordbdhg/llama.cpp","url":"https://github.com/stanfordbdhg/llama.cpp"},{"title":"facebookresearch/chai","url":"https://github.com/facebookresearch/chai"},{"title":"aozhongzhang/magr","url":"https://github.com/aozhongzhang/magr"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"ohadrubin/rpt","url":"https://github.com/ohadrubin/rpt"},{"title":"ecolab-postech/owq","url":"https://github.com/ecolab-postech/owq"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama2"},{"title":"Mind23-2/MindCode-140","url":"https://github.com/Mind23-2/MindCode-140"},{"title":"longhao-chen/aicas2024","url":"https://github.com/longhao-chen/aicas2024"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/llama"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/llama"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29516,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"InCoder 6.7B + MBR-Exec","metrics":{"Accuracy":"26.7"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29517,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"InCoder 6.7B + Coder-Reviewer","metrics":{"Accuracy":"26.1"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29518,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"InCoder 6.7B + Reviewer","metrics":{"Accuracy":"24.4"},"paper_url":"https://arxiv.org/abs/2211.16490v1","paper_title":"Coder Reviewer Reranking for Code Generation","paper_date":"2022-11-29","code_links":[{"title":"facebookresearch/coder_reviewer_reranking","url":"https://github.com/facebookresearch/coder_reviewer_reranking"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29519,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"CodeGeeX-13B","metrics":{"Accuracy":"24.4"},"paper_url":"https://arxiv.org/abs/2303.17568v2","paper_title":"CodeGeeX: A Pre-Trained Model for Code Generation with Multilingual Benchmarking on HumanEval-X","paper_date":"2023-03-30","code_links":[{"title":"THUDM/CodeGeeX","url":"https://github.com/THUDM/CodeGeeX"},{"title":"thudm/codegeex2","url":"https://github.com/thudm/codegeex2"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29520,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"LLaMA 13B (0-shot)","metrics":{"Accuracy":"22"},"paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","paper_date":"2023-02-27","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"ggml-org/llama.cpp","url":"https://github.com/ggml-org/llama.cpp"},{"title":"ggerganov/llama.cpp","url":"https://github.com/ggerganov/llama.cpp"},{"title":"meta-llama/llama","url":"https://github.com/meta-llama/llama"},{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"tatsu-lab/stanford_alpaca","url":"https://github.com/tatsu-lab/stanford_alpaca"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"nebuly-ai/nebullvm","url":"https://github.com/nebuly-ai/nebullvm/tree/main/apps/accelerate/chatllama"},{"title":"Lightning-AI/lit-llama","url":"https://github.com/Lightning-AI/lit-llama"},{"title":"facico/chinese-vicuna","url":"https://github.com/facico/chinese-vicuna"},{"title":"qwopqwop200/GPTQ-for-LLaMa","url":"https://github.com/qwopqwop200/GPTQ-for-LLaMa"},{"title":"phoebussi/alpaca-cot","url":"https://github.com/phoebussi/alpaca-cot"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"xusenlinzy/api-for-open-llm","url":"https://github.com/xusenlinzy/api-for-open-llm"},{"title":"guinmoon/llmfarm","url":"https://github.com/guinmoon/llmfarm"},{"title":"aethercortex/llama-x","url":"https://github.com/aethercortex/llama-x"},{"title":"beomi/koalpaca","url":"https://github.com/beomi/koalpaca"},{"title":"freedomintelligence/huatuogpt","url":"https://github.com/freedomintelligence/huatuogpt"},{"title":"icalk-nlp/educhat","url":"https://github.com/icalk-nlp/educhat"},{"title":"ecnu-icalk/educhat","url":"https://github.com/ecnu-icalk/educhat"},{"title":"ganjinzero/rrhf","url":"https://github.com/ganjinzero/rrhf"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"chaoyi-wu/pmc-llama","url":"https://github.com/chaoyi-wu/pmc-llama"},{"title":"kbressem/medalpaca","url":"https://github.com/kbressem/medalpaca"},{"title":"chaoyi-wu/finetune_llama","url":"https://github.com/chaoyi-wu/finetune_llama"},{"title":"ofa-sys/expertllama","url":"https://github.com/ofa-sys/expertllama"},{"title":"xiaoman-zhang/PMC-VQA","url":"https://github.com/xiaoman-zhang/PMC-VQA"},{"title":"fsoft-ai4code/codecapybara","url":"https://github.com/fsoft-ai4code/codecapybara"},{"title":"kayvr/token-hawk","url":"https://github.com/kayvr/token-hawk"},{"title":"ntunlplab/traditional-chinese-alpaca","url":"https://github.com/ntunlplab/traditional-chinese-alpaca"},{"title":"teelinsan/camoscio","url":"https://github.com/teelinsan/camoscio"},{"title":"replicate/cog_stanford_alpaca","url":"https://github.com/replicate/cog_stanford_alpaca"},{"title":"greenbitai/low_bit_llama","url":"https://github.com/greenbitai/low_bit_llama"},{"title":"krafton-ai/korani","url":"https://github.com/krafton-ai/korani"},{"title":"yuanmu97/secure-transformer-inference","url":"https://github.com/yuanmu97/secure-transformer-inference"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"hamishivi/easylm","url":"https://github.com/hamishivi/easylm"},{"title":"xvyaward/owq","url":"https://github.com/xvyaward/owq"},{"title":"vcskaushik/LLMzip","url":"https://github.com/vcskaushik/LLMzip"},{"title":"batsresearch/alfred","url":"https://github.com/batsresearch/alfred"},{"title":"grantslatton/llama.cpp","url":"https://github.com/grantslatton/llama.cpp"},{"title":"zihanzhaosjtu/librisqa","url":"https://github.com/zihanzhaosjtu/librisqa"},{"title":"fajri91/indommlu","url":"https://github.com/fajri91/indommlu"},{"title":"akanyaani/miniLLAMA","url":"https://github.com/akanyaani/miniLLAMA"},{"title":"stanfordbdhg/llama.cpp","url":"https://github.com/stanfordbdhg/llama.cpp"},{"title":"facebookresearch/chai","url":"https://github.com/facebookresearch/chai"},{"title":"aozhongzhang/magr","url":"https://github.com/aozhongzhang/magr"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"ohadrubin/rpt","url":"https://github.com/ohadrubin/rpt"},{"title":"ecolab-postech/owq","url":"https://github.com/ecolab-postech/owq"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama2"},{"title":"Mind23-2/MindCode-140","url":"https://github.com/Mind23-2/MindCode-140"},{"title":"longhao-chen/aicas2024","url":"https://github.com/longhao-chen/aicas2024"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/llama"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/llama"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29521,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Llama 2 7B (0-shot)","metrics":{"Accuracy":"20.8"},"paper_url":"https://arxiv.org/abs/2307.09288v2","paper_title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","paper_date":"2023-07-18","code_links":[{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"Lightning-AI/lit-gpt","url":"https://github.com/Lightning-AI/lit-gpt"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"IBM/Dromedary","url":"https://github.com/IBM/Dromedary"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"xverse-ai/xverse-13b","url":"https://github.com/xverse-ai/xverse-13b"},{"title":"usyd-fsalab/fp6_llm","url":"https://github.com/usyd-fsalab/fp6_llm"},{"title":"rijgersberg/geitje","url":"https://github.com/rijgersberg/geitje"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"glb400/Toy-RecLM","url":"https://github.com/glb400/Toy-RecLM"},{"title":"ninglab/ecellm","url":"https://github.com/ninglab/ecellm"},{"title":"zurichnlp/contradecode","url":"https://github.com/zurichnlp/contradecode"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"idiap/abroad-re","url":"https://github.com/idiap/abroad-re"},{"title":"meetyou-ai-lab/can-mc-evaluate-llms","url":"https://github.com/meetyou-ai-lab/can-mc-evaluate-llms"},{"title":"xuetianci/pacit","url":"https://github.com/xuetianci/pacit"},{"title":"coastalcph/eu-politics-llms","url":"https://github.com/coastalcph/eu-politics-llms"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29522,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"InCoder 6.7B (0-shot)","metrics":{"Accuracy":"19.4"},"paper_url":"https://arxiv.org/abs/2204.05999v3","paper_title":"InCoder: A Generative Model for Code Infilling and Synthesis","paper_date":"2022-04-12","code_links":[{"title":"dpfried/incoder","url":"https://github.com/dpfried/incoder"},{"title":"openai/human-eval-infilling","url":"https://github.com/openai/human-eval-infilling"},{"title":"eth-sri/sven","url":"https://github.com/eth-sri/sven"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29523,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"LLaMA 7B (0-shot)","metrics":{"Accuracy":"17.7"},"paper_url":"https://arxiv.org/abs/2302.13971v1","paper_title":"LLaMA: Open and Efficient Foundation Language Models","paper_date":"2023-02-27","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"ggml-org/llama.cpp","url":"https://github.com/ggml-org/llama.cpp"},{"title":"ggerganov/llama.cpp","url":"https://github.com/ggerganov/llama.cpp"},{"title":"meta-llama/llama","url":"https://github.com/meta-llama/llama"},{"title":"facebookresearch/llama","url":"https://github.com/facebookresearch/llama"},{"title":"tatsu-lab/stanford_alpaca","url":"https://github.com/tatsu-lab/stanford_alpaca"},{"title":"llamafamily/llama-chinese","url":"https://github.com/llamafamily/llama-chinese"},{"title":"flagalpha/llama2-chinese","url":"https://github.com/flagalpha/llama2-chinese"},{"title":"nebuly-ai/nebullvm","url":"https://github.com/nebuly-ai/nebullvm/tree/main/apps/accelerate/chatllama"},{"title":"Lightning-AI/lit-llama","url":"https://github.com/Lightning-AI/lit-llama"},{"title":"facico/chinese-vicuna","url":"https://github.com/facico/chinese-vicuna"},{"title":"qwopqwop200/GPTQ-for-LLaMa","url":"https://github.com/qwopqwop200/GPTQ-for-LLaMa"},{"title":"phoebussi/alpaca-cot","url":"https://github.com/phoebussi/alpaca-cot"},{"title":"young-geng/easylm","url":"https://github.com/young-geng/easylm"},{"title":"xusenlinzy/api-for-open-llm","url":"https://github.com/xusenlinzy/api-for-open-llm"},{"title":"guinmoon/llmfarm","url":"https://github.com/guinmoon/llmfarm"},{"title":"aethercortex/llama-x","url":"https://github.com/aethercortex/llama-x"},{"title":"beomi/koalpaca","url":"https://github.com/beomi/koalpaca"},{"title":"freedomintelligence/huatuogpt","url":"https://github.com/freedomintelligence/huatuogpt"},{"title":"icalk-nlp/educhat","url":"https://github.com/icalk-nlp/educhat"},{"title":"ecnu-icalk/educhat","url":"https://github.com/ecnu-icalk/educhat"},{"title":"ganjinzero/rrhf","url":"https://github.com/ganjinzero/rrhf"},{"title":"squeezeailab/squeezellm","url":"https://github.com/squeezeailab/squeezellm"},{"title":"chaoyi-wu/pmc-llama","url":"https://github.com/chaoyi-wu/pmc-llama"},{"title":"kbressem/medalpaca","url":"https://github.com/kbressem/medalpaca"},{"title":"chaoyi-wu/finetune_llama","url":"https://github.com/chaoyi-wu/finetune_llama"},{"title":"ofa-sys/expertllama","url":"https://github.com/ofa-sys/expertllama"},{"title":"xiaoman-zhang/PMC-VQA","url":"https://github.com/xiaoman-zhang/PMC-VQA"},{"title":"fsoft-ai4code/codecapybara","url":"https://github.com/fsoft-ai4code/codecapybara"},{"title":"kayvr/token-hawk","url":"https://github.com/kayvr/token-hawk"},{"title":"ntunlplab/traditional-chinese-alpaca","url":"https://github.com/ntunlplab/traditional-chinese-alpaca"},{"title":"teelinsan/camoscio","url":"https://github.com/teelinsan/camoscio"},{"title":"replicate/cog_stanford_alpaca","url":"https://github.com/replicate/cog_stanford_alpaca"},{"title":"greenbitai/low_bit_llama","url":"https://github.com/greenbitai/low_bit_llama"},{"title":"krafton-ai/korani","url":"https://github.com/krafton-ai/korani"},{"title":"yuanmu97/secure-transformer-inference","url":"https://github.com/yuanmu97/secure-transformer-inference"},{"title":"xzhang97666/alpacare","url":"https://github.com/xzhang97666/alpacare"},{"title":"hamishivi/easylm","url":"https://github.com/hamishivi/easylm"},{"title":"xvyaward/owq","url":"https://github.com/xvyaward/owq"},{"title":"vcskaushik/LLMzip","url":"https://github.com/vcskaushik/LLMzip"},{"title":"batsresearch/alfred","url":"https://github.com/batsresearch/alfred"},{"title":"grantslatton/llama.cpp","url":"https://github.com/grantslatton/llama.cpp"},{"title":"zihanzhaosjtu/librisqa","url":"https://github.com/zihanzhaosjtu/librisqa"},{"title":"fajri91/indommlu","url":"https://github.com/fajri91/indommlu"},{"title":"akanyaani/miniLLAMA","url":"https://github.com/akanyaani/miniLLAMA"},{"title":"stanfordbdhg/llama.cpp","url":"https://github.com/stanfordbdhg/llama.cpp"},{"title":"facebookresearch/chai","url":"https://github.com/facebookresearch/chai"},{"title":"aozhongzhang/magr","url":"https://github.com/aozhongzhang/magr"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"ohadrubin/rpt","url":"https://github.com/ohadrubin/rpt"},{"title":"ecolab-postech/owq","url":"https://github.com/ecolab-postech/owq"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/llama2"},{"title":"Mind23-2/MindCode-140","url":"https://github.com/Mind23-2/MindCode-140"},{"title":"longhao-chen/aicas2024","url":"https://github.com/longhao-chen/aicas2024"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/llama"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/llama"}],"metrics_order":"[\"Accuracy\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":597586,"task":"Code Generation","parent_task":null,"dataset":"MBPP","model_name":"Group-Based","metrics":{"Accuracy":"5.5"},"paper_url":"https://paperswithcode.com/paper/beyond-kl-divergence-policy-optimization-with-flexible-bregman-divergences-for-llm-reasoning","paper_title":"Beyond KL Divergence: Policy Optimization with Flexible Bregman Divergences for LLM Reasoning","paper_date":"2026-02-04","code_links":[],"metrics_order":null,"area":null,"uses_additional_data":null,"source":"auto","tags":[]}]}