{"url":"/dataset/humanevalpack","name":"HumanEvalPack","full_name":"HumanEvalPack","description_markdown":"HumanEvalPack is an extension of OpenAI's HumanEval to cover 6 total languages across 3 tasks. The evaluation suite is fully created by humans.","description_withheld":null,"homepage":"https://huggingface.co/datasets/bigcode/humanevalpack","introduced_date":"2023-08-14","introduced_date_note":null,"introduced_by":{"paper":"/paper/octopack-instruction-tuning-code-large","title":"OctoPack: Instruction Tuning Code Large Language Models","first_author":"Niklas Muennighoff","url":null},"license":null,"modalities":[],"tasks":[{"name":"Program Repair","url":"/task/program-repair","datasets_with_task":"/datasets/task/program-repair"}],"languages":[],"variants":["HumanEvalPack"],"data_loaders":[],"num_papers_in_archive":26,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/program-repair-on-humanevalpack","task":"Program Repair","dataset_variant":"HumanEvalPack","rows":1,"metrics":["Pass@1"],"first_row_in_archive_order":{"model":"MGDebugger (DeepSeek-Coder-V2-Lite)","paper":"/paper/from-code-to-correctness-closing-the-last","metrics":{"Pass@1":"97.6"},"code_links":[{"title":"YerbaPage/MGDebugger","url":"https://github.com/YerbaPage/MGDebugger"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/from-code-to-correctness-closing-the-last","title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","date":"2024-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":8,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":9,"samples_ran":8,"samples_unverified":1,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}