{"url":"/dataset/humaneval","name":"HumanEval","full_name":null,"description_markdown":"This is an evaluation harness for the HumanEval problem solving dataset described in the paper \"Evaluating Large Language Models Trained on Code\". It used to measure functional correctness for synthesizing programs from docstrings. It consists of 164 original programming problems, assessing language comprehension, algorithms, and simple mathematics, with some\r\ncomparable to simple software interview questions.\r\n\r\nSource: [Evaluating Large Language Models Trained on Code](/paper/evaluating-large-language-models-trained-on)\r\nImage Source: [Evaluating Large Language Models Trained on Code](/paper/evaluating-large-language-models-trained-on)","description_withheld":null,"homepage":"https://github.com/openai/human-eval","introduced_date":"2021-07-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","first_author":"Mark Chen","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"},{"name":"Semantic Textual Similarity","url":"/task/semantic-textual-similarity","datasets_with_task":"/datasets/task/semantic-textual-similarity"},{"name":"HumanEval","url":"/task/humaneval","datasets_with_task":"/datasets/task/humaneval"},{"name":"MTEB Benchmark","url":"/task/mteb-benchmark","datasets_with_task":"/datasets/task/mteb-benchmark"},{"name":"STS Benchmark","url":"/task/sts-benchmark","datasets_with_task":"/datasets/task/sts-benchmark"},{"name":"AllNLI Triplet","url":"/task/allnli-triplet","datasets_with_task":"/datasets/task/allnli-triplet"}],"languages":[{"name":"Arabic","url":"/datasets/language/arabic"}],"variants":["STS Benchmark","HumanEval","HumanEval!","humaneval (0-shots)","MTEB Benchmark","AllNLI Triplet"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/openai/openai_humaneval","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/openai_humaneval","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/openai/human-eval","url":"https://github.com/openai/human-eval","frameworks":[]}],"num_papers_in_archive":1201,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/code-generation-on-humaneval","task":"Code Generation","dataset_variant":"HumanEval","rows":8,"metrics":["Pass@1"],"first_row_in_archive_order":{"model":"DeepSeek-R1 (MGDebugger)","paper":"/paper/from-code-to-correctness-closing-the-last","metrics":{"Pass@1":"100"},"code_links":[{"title":"YerbaPage/MGDebugger","url":"https://github.com/YerbaPage/MGDebugger"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/execution-guided-line-by-line-code-generation","title":"Execution Guided Line-by-Line Code Generation","date":"2025-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/qualityflow-an-agentic-workflow-for-program","title":"QualityFlow: An Agentic Workflow for Program Synthesis Controlled by LLM Quality Checks","date":"2025-01-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/planning-driven-programming-a-large-language","title":"Planning-Driven Programming: A Large Language Model Programming Workflow","date":"2024-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/from-code-to-correctness-closing-the-last","title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","date":"2024-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":8,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mapcoder-multi-agent-code-generation-for","title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","date":"2024-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ldb-a-large-language-model-debugger-via","title":"Debug like a Human: A Large Language Model Debugger via Verifying Runtime Execution Step-by-step","date":"2024-02-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":8,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/l2mac-large-language-model-automatic-computer","title":"L2MAC: Large Language Model Automatic Computer for Extensive Code Generation","date":"2023-10-02","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":54,"samples_ran":26,"samples_unverified":28,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}