{"url":"/dataset/apps","name":"APPS","full_name":"Automated Programming Progress Standard","description_markdown":"The APPS dataset consists of problems collected from different open-access coding websites such as Codeforces, Kattis, and more. The APPS benchmark attempts to mirror how humans programmers are evaluated by posing coding problems in unrestricted natural language and evaluating the correctness of solutions. The problems range in difficulty from introductory to collegiate competition level and measure coding ability as well as problem-solving. \r\n\r\nThe Automated Programming Progress Standard, abbreviated APPS, consists of 10,000 coding problems in total, with 131,836 test cases for checking solutions and 232,444 ground-truth solutions written by humans. Problems can be complicated, as the average length of a problem is 293.2 words. The data are split evenly into training and test sets, with 5,000 problems each. In the test set, every problem has multiple test cases, and the average number of test cases is 21.2. Each test case is specifically designed for the corresponding problem, enabling us to rigorously evaluate program functionality.\r\n\r\nSource: [Measuring Coding Challenge Competence With APPS](https://arxiv.org/pdf/2105.09938v1.pdf)\r\n\r\nImage source: [Measuring Coding Challenge Competence With APPS](https://arxiv.org/pdf/2105.09938v1.pdf)","description_withheld":null,"homepage":"https://github.com/hendrycks/apps","introduced_date":"2021-05-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/measuring-coding-challenge-competence-with","title":"Measuring Coding Challenge Competence With APPS","first_author":"Dan Hendrycks","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["APPS"],"data_loaders":[],"num_papers_in_archive":180,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/code-generation-on-apps","task":"Code Generation","dataset_variant":"APPS","rows":18,"metrics":["Introductory Pass@1","Interview Pass@1","Competition Pass@1","Competition Pass@any","Interview Pass@any","Introductory Pass@any","Competition Pass@5","Interview Pass@5","Introductory Pass@5","Competition Pass@1000","Interview Pass@1000","Introductory Pass@1000","Pass@1"],"first_row_in_archive_order":{"model":"LPW (GPT-4o)","paper":"/paper/planning-driven-programming-a-large-language","metrics":{"Competition Pass@1":"34.8","Interview Pass@1":"65.2","Introductory Pass@1":"87.2","Pass@1":"62.6"},"code_links":[{"title":"you68681/lpw","url":"https://github.com/you68681/lpw"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/codesim-multi-agent-code-generation-and-1","title":"CODESIM: Multi-Agent Code Generation and Problem Solving through Simulation-Driven Planning and Debugging","date":"2025-02-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/planning-driven-programming-a-large-language","title":"Planning-Driven Programming: A Large Language Model Programming Workflow","date":"2024-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mapcoder-multi-agent-code-generation-for","title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","date":"2024-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deepseek-coder-when-the-large-language-model","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","date":"2024-01-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":9,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/motcoder-elevating-large-language-models-with","title":"MoTCoder: Elevating Large Language Models with Modular of Thought for Challenging Programming Tasks","date":"2023-12-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/codechain-towards-modular-code-generation","title":"CodeChain: Towards Modular Code Generation Through Chain of Self-revisions with Representative Sub-modules","date":"2023-10-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/codet-code-generation-with-generated-tests","title":"CodeT: Code Generation with Generated Tests","date":"2022-07-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/competition-level-code-generation-with-1","title":"Competition-Level Code Generation with AlphaCode","date":"2022-02-08","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":39,"samples_ran":6,"samples_unverified":33,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/measuring-coding-challenge-competence-with","title":"Measuring Coding Challenge Competence With APPS","date":"2021-05-20","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":96,"samples_ran":27,"samples_unverified":69,"pointer_only_for_licence":10,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}