{"url":"/dataset/minif2f","name":"MiniF2F","full_name":null,"description_markdown":"**MiniF2F** is a dataset of formal Olympiad-level mathematics problems statements intended to provide a unified cross-system benchmark for neural theorem proving. The miniF2F benchmark currently targets Metamath, Lean, and Isabelle and consists of 488 problem statements drawn from the AIME, AMC, and the International Mathematical Olympiad (IMO), as well as material from high-school and undergraduate mathematics courses.","description_withheld":null,"homepage":"https://github.com/openai/miniF2F","introduced_date":"2021-08-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/minif2f-a-cross-system-benchmark-for-formal","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","first_author":"Kunhao Zheng","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Automated Theorem Proving","url":"/task/automated-theorem-proving","datasets_with_task":"/datasets/task/automated-theorem-proving"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MiniF2F","miniF2F-valid","miniF2F-test"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Tonic/MiniF2F","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":84,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/automated-theorem-proving-on-minif2f-test","task":"Automated Theorem Proving","dataset_variant":"miniF2F-test","rows":29,"metrics":["cumulative","Pass@1","Pass@32","Pass@64","Pass@100","ITP","pass@1024","pass@8192"],"first_row_in_archive_order":{"model":"Kimina-Prover-Preview","paper":"/paper/kimina-prover-preview-towards-large-formal","metrics":{"ITP":"Lean","Pass@1":"52.94","Pass@32":"68.85","cumulative":"80.74","pass@1024":"77.87","pass@8192":"80.74"},"code_links":[{"title":"moonshotai/kimina-prover-preview","url":"https://github.com/moonshotai/kimina-prover-preview"},{"title":"cmu-l3/llmlean","url":"https://github.com/cmu-l3/llmlean"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/automated-theorem-proving-on-minif2f-valid","task":"Automated Theorem Proving","dataset_variant":"miniF2F-valid","rows":10,"metrics":["Pass@8","Pass@64","Pass@1","Pass@100"],"first_row_in_archive_order":{"model":"Lean GPT-f","paper":"/paper/minif2f-a-cross-system-benchmark-for-formal","metrics":{"Pass@1":"23.9","Pass@8":"29.3"},"code_links":[{"title":"openai/minif2f","url":"https://github.com/openai/minif2f"},{"title":"facebookresearch/minif2f","url":"https://github.com/facebookresearch/minif2f"},{"title":"yangky11/minif2f-lean4","url":"https://github.com/yangky11/minif2f-lean4"},{"title":"rah4927/lean-dojo-mew","url":"https://github.com/rah4927/lean-dojo-mew"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/kimina-prover-preview-towards-large-formal","title":"Kimina-Prover Preview: Towards Large Formal Reasoning Models with Reinforcement Learning","date":"2025-04-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-neural-theorem-proving-via-fine","title":"Efficient Neural Theorem Proving via Fine-grained Proof Structure Analysis","date":"2025-01-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/subgoalxl-subgoal-based-expert-learning-for","title":"SubgoalXL: Subgoal-based Expert Learning for Theorem Proving","date":"2024-08-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deepseek-prover-v1-5-harnessing-proof","title":"DeepSeek-Prover-V1.5: Harnessing Proof Assistant Feedback for Reinforcement Learning and Monte-Carlo Tree Search","date":"2024-08-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":4,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deepseek-prover-advancing-theorem-proving-in","title":"DeepSeek-Prover: Advancing Theorem Proving in LLMs through Large-Scale Synthetic Data","date":"2024-05-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-data-ability-boundary","title":"An Empirical Study of Data Ability Boundary in LLMs' Math Reasoning","date":"2024-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llemma-an-open-language-model-for-mathematics","title":"Llemma: An Open Language Model For Mathematics","date":"2023-10-16","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-language-agent-approach-to-formal-theorem","title":"An In-Context Learning Agent for Formal Theorem-Proving","date":"2023-10-06","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/lego-prover-neural-theorem-proving-with","title":"LEGO-Prover: Neural Theorem Proving with Growing Libraries","date":"2023-10-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/lyra-orchestrating-dual-correction-in","title":"Lyra: Orchestrating Dual Correction in Automated Theorem Proving","date":"2023-09-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/decomposing-the-enigma-subgoal-based","title":"Decomposing the Enigma: Subgoal-based Demonstration Learning for Formal Theorem Proving","date":"2023-05-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/draft-sketch-and-prove-guiding-formal-theorem","title":"Draft, Sketch, and Prove: Guiding Formal Theorem Provers with Informal Proofs","date":"2022-10-21","rows_on_this_dataset":3,"code_links":3,"syntology":null},{"paper":"/paper/hypertree-proof-search-for-neural-theorem","title":"HyperTree Proof Search for Neural Theorem Proving","date":"2022-05-23","rows_on_this_dataset":8,"code_links":0,"syntology":null},{"paper":"/paper/thor-wielding-hammers-to-integrate-language","title":"Thor: Wielding Hammers to Integrate Language Models and Automated Theorem Provers","date":"2022-05-22","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/formal-mathematics-statement-curriculum","title":"Formal Mathematics Statement Curriculum Learning","date":"2022-02-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/minif2f-a-cross-system-benchmark-for-formal","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","date":"2021-08-31","rows_on_this_dataset":6,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/proof-artifact-co-training-for-theorem","title":"Proof Artifact Co-training for Theorem Proving with Language Models","date":"2021-02-11","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":47,"samples_ran":35,"samples_unverified":12,"pointer_only_for_licence":14,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}