{"url":"/task/automated-theorem-proving","name":"Automated Theorem Proving","slug":"automated-theorem-proving","description_markdown":"The goal of **Automated Theorem Proving** is to automatically generate a proof, given a conjecture (the target theorem) and a knowledge base of known facts, all expressed in a formal language. Automated Theorem Proving is useful in a wide range of applications, including the verification and synthesis of software and hardware systems.\r\n\r\n\r\n<span class=\"description-source\">Source: [Learning to Prove Theorems by Learning to Generate Theorems ](https://arxiv.org/abs/2002.07019)</span>","categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":288,"papers_with_code":110,"benchmarks":9,"benchmark_tables_in_archive":9,"benchmark_tables_shown":9,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/automated-theorem-proving-on-minif2f-test","slug":"automated-theorem-proving-on-minif2f-test","dataset":"miniF2F-test","dataset_url":"/dataset/minif2f","rows_in_archive":29,"metrics":["cumulative","Pass@1","Pass@32","Pass@64","Pass@100","ITP","pass@1024","pass@8192"],"first_row_in_archive_order":{"model":"Kimina-Prover-Preview","paper_title":"Kimina-Prover Preview: Towards Large Formal Reasoning Models with Reinforcement Learning","paper_url":"/paper/kimina-prover-preview-towards-large-formal","paper_date":"2025-04-15","arxiv_id":"2504.11354","code_links":[{"title":"moonshotai/kimina-prover-preview","url":"https://github.com/moonshotai/kimina-prover-preview"},{"title":"cmu-l3/llmlean","url":"https://github.com/cmu-l3/llmlean"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/automated-theorem-proving-on-minif2f-valid","slug":"automated-theorem-proving-on-minif2f-valid","dataset":"miniF2F-valid","dataset_url":"/dataset/minif2f","rows_in_archive":10,"metrics":["Pass@8","Pass@64","Pass@1","Pass@100"],"first_row_in_archive_order":{"model":"Lean GPT-f","paper_title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","paper_url":"/paper/minif2f-a-cross-system-benchmark-for-formal","paper_date":"2021-08-31","arxiv_id":"2109.00110","code_links":[{"title":"openai/minif2f","url":"https://github.com/openai/minif2f"},{"title":"facebookresearch/minif2f","url":"https://github.com/facebookresearch/minif2f"},{"title":"yangky11/minif2f-lean4","url":"https://github.com/yangky11/minif2f-lean4"},{"title":"rah4927/lean-dojo-mew","url":"https://github.com/rah4927/lean-dojo-mew"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/automated-theorem-proving-on-holstep","slug":"automated-theorem-proving-on-holstep","dataset":"HolStep (Conditional)","dataset_url":"/dataset/holstep","rows_in_archive":5,"metrics":["Classification Accuracy"],"first_row_in_archive_order":{"model":"MPNN-DagLSTM","paper_title":"Improving Graph Neural Network Representations of Logical Formulae with Subgraph Pooling","paper_url":"/paper/improving-graph-neural-network","paper_date":"2019-11-15","arxiv_id":"1911.06904","code_links":[{"title":"IBM/LogicalFormulaEmbedder","url":"https://github.com/IBM/LogicalFormulaEmbedder"}],"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}}},{"leaderboard":"/sota/automated-theorem-proving-on-holist-benchmark","slug":"automated-theorem-proving-on-holist-benchmark","dataset":"HOList benchmark","dataset_url":"/dataset/holist","rows_in_archive":4,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"4-hop GNN, sub-expression sharing","paper_title":"Graph Representations for Higher-Order Logic and Theorem Proving","paper_url":"/paper/graph-representations-for-higher-order-logic","paper_date":"2019-05-24","arxiv_id":"1905.10006","code_links":[],"syntology":null}},{"leaderboard":"/sota/automated-theorem-proving-on-holstep-1","slug":"automated-theorem-proving-on-holstep-1","dataset":"HolStep (Unconditional)","dataset_url":"/dataset/holstep","rows_in_archive":4,"metrics":["Classification Accuracy"],"first_row_in_archive_order":{"model":"FormulaNet","paper_title":"Premise Selection for Theorem Proving by Deep Graph Embedding","paper_url":"/paper/premise-selection-for-theorem-proving-by-deep","paper_date":"2017-09-28","arxiv_id":"1709.09994","code_links":[{"title":"princeton-vl/FormulaNet","url":"https://github.com/princeton-vl/FormulaNet"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/automated-theorem-proving-on-metamath-setmm","slug":"automated-theorem-proving-on-metamath-setmm","dataset":"Metamath set.mm","dataset_url":null,"rows_in_archive":4,"metrics":["Percentage correct","Pass@32"],"first_row_in_archive_order":{"model":"GPT-f","paper_title":"Generative Language Modeling for Automated Theorem Proving","paper_url":"/paper/generative-language-modeling-for-automated","paper_date":"2020-09-07","arxiv_id":"2009.03393","code_links":[],"syntology":null}},{"leaderboard":"/sota/automated-theorem-proving-on-minif2f-1","slug":"automated-theorem-proving-on-minif2f-1","dataset":"miniF2F-curriculum","dataset_url":null,"rows_in_archive":4,"metrics":["Pass@64"],"first_row_in_archive_order":{"model":"Evariste-7d","paper_title":"HyperTree Proof Search for Neural Theorem Proving","paper_url":"/paper/hypertree-proof-search-for-neural-theorem","paper_date":"2022-05-23","arxiv_id":"2205.11491","code_links":[],"syntology":null}},{"leaderboard":"/sota/automated-theorem-proving-on-compcert","slug":"automated-theorem-proving-on-compcert","dataset":"CompCert","dataset_url":null,"rows_in_archive":2,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"Proverbot9001","paper_title":"Generating Correctness Proofs with Neural Networks","paper_url":"/paper/generating-correctness-proofs-with-neural-1","paper_date":"2020-05-28","arxiv_id":"1907.07794","code_links":[{"title":"UCSD-PL/proverbot9001","url":"https://github.com/UCSD-PL/proverbot9001"}],"syntology":null}},{"leaderboard":"/sota/automated-theorem-proving-on-coqgym","slug":"automated-theorem-proving-on-coqgym","dataset":"CoqGym","dataset_url":null,"rows_in_archive":1,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"ASTactic","paper_title":"Learning to Prove Theorems via Interacting with Proof Assistants","paper_url":"/paper/learning-to-prove-theorems-via-interacting","paper_date":"2019-05-21","arxiv_id":"1905.09381","code_links":[{"title":"princeton-vl/CoqGym","url":"https://github.com/princeton-vl/CoqGym"}],"syntology":null}}],"datasets":[{"url":"/dataset/minif2f","name":"MiniF2F","full_name":"","num_papers_in_archive":84},{"url":"/dataset/geometry3k","name":"Geometry3K","full_name":"","num_papers_in_archive":45},{"url":"/dataset/proofnet","name":"ProofNet","full_name":"","num_papers_in_archive":27},{"url":"/dataset/med","name":"MED","full_name":"Monotonicity Entailment Dataset","num_papers_in_archive":18},{"url":"/dataset/holstep","name":"HolStep","full_name":"","num_papers_in_archive":10},{"url":"/dataset/kinship","name":"Kinship","full_name":null,"num_papers_in_archive":10},{"url":"/dataset/holist","name":"HOList","full_name":"HOList","num_papers_in_archive":8},{"url":"/dataset/gamepad-environment","name":"GamePad Environment","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/mathematical-proofs","name":"Mathematical Proofs"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":110,"tagged_in_all":288,"items":[{"url":"/paper/minictx-neural-theorem-proving-with-long","title":"miniCTX: Neural Theorem Proving with (Long-)Contexts","date":"2024-08-05","arxiv_id":"2408.03350","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/llemma-an-open-language-model-for-mathematics","title":"Llemma: An Open Language Model For Mathematics","date":"2023-10-16","arxiv_id":"2310.10631","repositories_listed":4,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/minif2f-a-cross-system-benchmark-for-formal","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","date":"2021-08-31","arxiv_id":"2109.00110","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/proof-artifact-co-training-for-theorem","title":"Proof Artifact Co-training for Theorem Proving with Language Models","date":"2021-02-11","arxiv_id":"2102.06203","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/holophrasm-a-neural-automated-theorem-prover","title":"Holophrasm: a neural Automated Theorem Prover for higher-order logic","date":"2016-08-08","arxiv_id":"1608.02644","repositories_listed":4,"syntology":{"n":14,"n_ran":0,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/leandojo-theorem-proving-with-retrieval-1","title":"LeanDojo: Theorem Proving with Retrieval-Augmented Language Models","date":"2023-06-27","arxiv_id":"2306.15626","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/draft-sketch-and-prove-guiding-formal-theorem","title":"Draft, Sketch, and Prove: Guiding Formal Theorem Provers with Informal Proofs","date":"2022-10-21","arxiv_id":"2210.12283","repositories_listed":3,"syntology":null},{"url":"/paper/holist-an-environment-for-machine-learning-of","title":"HOList: An Environment for Machine Learning of Higher-Order Theorem Proving","date":"2019-04-05","arxiv_id":"1904.03241","repositories_listed":3,"syntology":null},{"url":"/paper/deeptheorem-advancing-llm-reasoning-for","title":"DeepTheorem: Advancing LLM Reasoning for Theorem Proving Through Natural Language and Reinforcement Learning","date":"2025-05-29","arxiv_id":"2505.23754","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/kimina-prover-preview-towards-large-formal","title":"Kimina-Prover Preview: Towards Large Formal Reasoning Models with Reinforcement Learning","date":"2025-04-15","arxiv_id":"2504.11354","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/rm-p-small-roof-w-small-ala-multilingual","title":"ProofWala: Multilingual Proof Data Synthesis and Theorem-Proving","date":"2025-02-07","arxiv_id":"2502.04671","repositories_listed":2,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/pantograph-a-machine-to-machine-interaction","title":"Pantograph: A Machine-to-Machine Interaction Interface for Advanced Theorem Proving, High Level Reasoning, and Data Extraction in Lean 4","date":"2024-10-21","arxiv_id":"2410.16429","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/deepseek-prover-v1-5-harnessing-proof","title":"DeepSeek-Prover-V1.5: Harnessing Proof Assistant Feedback for Reinforcement Learning and Monte-Carlo Tree Search","date":"2024-08-15","arxiv_id":"2408.08152","repositories_listed":2,"syntology":{"n":10,"n_ran":4,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/putnambench-evaluating-neural-theorem-provers","title":"PutnamBench: Evaluating Neural Theorem-Provers on the Putnam Mathematical Competition","date":"2024-07-15","arxiv_id":"2407.11214","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/learning-formal-mathematics-from-intrinsic","title":"Learning Formal Mathematics From Intrinsic Motivation","date":"2024-06-30","arxiv_id":"2407.00695","repositories_listed":2,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/towards-large-language-models-as-copilots-for","title":"Lean Copilot: Large Language Models as Copilots for Theorem Proving in Lean","date":"2024-04-18","arxiv_id":"2404.12534","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/proofnet-autoformalizing-and-formally-proving","title":"ProofNet: Autoformalizing and Formally Proving Undergraduate-Level Mathematics","date":"2023-02-24","arxiv_id":"2302.12433","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/linear-algebra-with-transformers-1","title":"Linear algebra with transformers","date":"2021-12-03","arxiv_id":"2112.01898","repositories_listed":2,"syntology":null},{"url":"/paper/learning-symbolic-rules-for-reasoning-in-1","title":"Learning Symbolic Rules for Reasoning in Quasi-Natural Language","date":"2021-11-23","arxiv_id":"2111.12038","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-match-mathematical-statements","title":"Learning to Match Mathematical Statements with Proofs","date":"2021-02-03","arxiv_id":"2102.02110","repositories_listed":2,"syntology":null},{"url":"/paper/measuring-systematic-generalization-in-neural","title":"Measuring Systematic Generalization in Neural Proof Generation with Transformers","date":"2020-09-30","arxiv_id":"2009.14786","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/logical-neural-networks","title":"Logical Neural Networks","date":"2020-06-23","arxiv_id":"2006.13155","repositories_listed":2,"syntology":{"n":15,"n_ran":2,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/learning-to-prove-theorems-by-learning-to-1","title":"Learning to Prove Theorems by Learning to Generate Theorems","date":"2020-02-17","arxiv_id":"2002.07019","repositories_listed":2,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/deepmath-deep-sequence-models-for-premise","title":"DeepMath - Deep Sequence Models for Premise Selection","date":"2016-06-14","arxiv_id":"1606.04442","repositories_listed":2,"syntology":null},{"url":"/paper/criticlean-critic-guided-reinforcement","title":"CriticLean: Critic-Guided Reinforcement Learning for Mathematical Formalization","date":"2025-07-08","arxiv_id":"2507.06181","repositories_listed":1,"syntology":null},{"url":"/paper/safe-enhancing-mathematical-reasoning-in","title":"Safe: Enhancing Mathematical Reasoning in Large Language Models via Retrospective Step-aware Formal Verification","date":"2025-06-05","arxiv_id":"2506.04592","repositories_listed":1,"syntology":null},{"url":"/paper/leanexplore-a-search-engine-for-lean-4","title":"LeanExplore: A search engine for Lean 4 declarations","date":"2025-06-04","arxiv_id":"2506.11085","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/autoformalization-in-the-era-of-large","title":"Autoformalization in the Era of Large Language Models: A Survey","date":"2025-05-29","arxiv_id":"2505.23486","repositories_listed":1,"syntology":null},{"url":"/paper/enumerate-conjecture-prove-formally-solving","title":"Enumerate-Conjecture-Prove: Formally Solving Answer-Construction Problems in Math Competitions","date":"2025-05-24","arxiv_id":"2505.18492","repositories_listed":1,"syntology":null},{"url":"/paper/mirb-mathematical-information-retrieval","title":"MIRB: Mathematical Information Retrieval Benchmark","date":"2025-05-21","arxiv_id":"2505.15585","repositories_listed":1,"syntology":null}],"syntology_records":19,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}