{"task":"Code Generation","dataset":"BigCodeBench-Complete","metric_names":["Pass@1"],"rows":[{"id":29542,"task":"Code Generation","parent_task":null,"dataset":"BigCodeBench-Complete","model_name":"GPT-4o-2024-05-13","metrics":{"Pass@1":"61.1"},"paper_url":"https://arxiv.org/abs/2406.15877v4","paper_title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","paper_date":"2024-06-22","code_links":[{"title":"mlfoundations/Evalchemy","url":"https://github.com/mlfoundations/Evalchemy"},{"title":"bigcode-project/bigcodebench","url":"https://github.com/bigcode-project/bigcodebench"},{"title":"bigcode-project/bigcodebench-annotation","url":"https://github.com/bigcode-project/bigcodebench-annotation"},{"title":"stovecat/convcodeworld","url":"https://github.com/stovecat/convcodeworld"}],"metrics_order":"[\"Pass@1\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]},{"id":29543,"task":"Code Generation","parent_task":null,"dataset":"BigCodeBench-Complete","model_name":"DeepSeek-Coder-V2-Instruct","metrics":{"Pass@1":"59.7"},"paper_url":"https://arxiv.org/abs/2406.15877v4","paper_title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","paper_date":"2024-06-22","code_links":[{"title":"mlfoundations/Evalchemy","url":"https://github.com/mlfoundations/Evalchemy"},{"title":"bigcode-project/bigcodebench","url":"https://github.com/bigcode-project/bigcodebench"},{"title":"bigcode-project/bigcodebench-annotation","url":"https://github.com/bigcode-project/bigcodebench-annotation"},{"title":"stovecat/convcodeworld","url":"https://github.com/stovecat/convcodeworld"}],"metrics_order":"[\"Pass@1\"]","area":"Natural Language Processing","uses_additional_data":0,"source":"archive","tags":[]}]}