{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/4","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":9,"rows_per_page":100,"rows":[301,400],"of":805,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning","prev":"/task/mathematical-reasoning/papers/3","next":"/task/mathematical-reasoning/papers/5","papers":[{"url":"/paper/planning-and-editing-what-you-retrieve-for","slug":"planning-and-editing-what-you-retrieve-for","title":"Planning and Editing What You Retrieve for Enhanced Tool Learning","date":"2024-03-30","arxiv_id":"2404.00450","repositories_listed":1,"syntology":null},{"url":"/paper/instructing-large-language-models-to-identify","slug":"instructing-large-language-models-to-identify","title":"Instructing Large Language Models to Identify and Ignore Irrelevant Conditions","date":"2024-03-19","arxiv_id":"2403.12744","repositories_listed":1,"syntology":null},{"url":"/paper/rat-retrieval-augmented-thoughts-elicit","slug":"rat-retrieval-augmented-thoughts-elicit","title":"RAT: Retrieval Augmented Thoughts Elicit Context-Aware Reasoning in Long-Horizon Generation","date":"2024-03-08","arxiv_id":"2403.05313","repositories_listed":1,"syntology":null},{"url":"/paper/mathscale-scaling-instruction-tuning-for","slug":"mathscale-scaling-instruction-tuning-for","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","date":"2024-03-05","arxiv_id":"2403.02884","repositories_listed":1,"syntology":null},{"url":"/paper/masked-thought-simply-masking-partial","slug":"masked-thought-simply-masking-partial","title":"Masked Thought: Simply Masking Partial Reasoning Steps Can Improve Mathematical Reasoning Learning of Language Models","date":"2024-03-04","arxiv_id":"2403.02178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/masked-thought-simply-masking-partial#ran","syntology_url":"https://syntology.ai/paper/2403.02178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02178"}},"official":{"repos":["changyuchen347/maskedthought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gsm-plus-a-comprehensive-benchmark-for","slug":"gsm-plus-a-comprehensive-benchmark-for","title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","date":"2024-02-29","arxiv_id":"2402.19255","repositories_listed":1,"syntology":null},{"url":"/paper/mathsensei-a-tool-augmented-large-language","slug":"mathsensei-a-tool-augmented-large-language","title":"MATHSENSEI: A Tool-Augmented Large Language Model for Mathematical Reasoning","date":"2024-02-27","arxiv_id":"2402.17231","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-humans-write-code-large-models-do-it","slug":"how-do-humans-write-code-large-models-do-it","title":"How Do Humans Write Code? Large Models Do It the Same Way Too","date":"2024-02-24","arxiv_id":"2402.15729","repositories_listed":1,"syntology":null},{"url":"/paper/stepwise-self-consistent-mathematical","slug":"stepwise-self-consistent-mathematical","title":"Stepwise Self-Consistent Mathematical Reasoning with Large Language Models","date":"2024-02-24","arxiv_id":"2402.17786","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stepwise-self-consistent-mathematical#ran","syntology_url":"https://syntology.ai/paper/2402.17786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17786"}},"official":{"repos":["zhao-zilong/ssc-cot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conceptmath-a-bilingual-concept-wise","slug":"conceptmath-a-bilingual-concept-wise","title":"ConceptMath: A Bilingual Concept-wise Benchmark for Measuring Mathematical Reasoning of Large Language Models","date":"2024-02-22","arxiv_id":"2402.14660","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conceptmath-a-bilingual-concept-wise#ran","syntology_url":"https://syntology.ai/paper/2402.14660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14660"}},"official":{"repos":["conceptmath/conceptmath"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-multimodal-mathematical-reasoning","slug":"measuring-multimodal-mathematical-reasoning","title":"Measuring Multimodal Mathematical Reasoning with MATH-Vision Dataset","date":"2024-02-22","arxiv_id":"2402.14804","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-check-unleashing-potentials-for","slug":"learning-to-check-unleashing-potentials-for","title":"Learning to Check: Unleashing Potentials for Self-Correction in Large Language Models","date":"2024-02-20","arxiv_id":"2402.13035","repositories_listed":1,"syntology":null},{"url":"/paper/reformatted-alignment","slug":"reformatted-alignment","title":"Reformatted Alignment","date":"2024-02-19","arxiv_id":"2402.12219","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reformatted-alignment#ran","syntology_url":"https://syntology.ai/paper/2402.12219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12219"}},"official":{"repos":["gair-nlp/realign"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-failure-integrating-negative","slug":"learning-from-failure-integrating-negative","title":"Learning From Failure: Integrating Negative Examples when Fine-tuning Large Language Models as Agents","date":"2024-02-18","arxiv_id":"2402.11651","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-failure-integrating-negative#ran","syntology_url":"https://syntology.ai/paper/2402.11651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11651"}},"official":{"repos":["reason-wang/nat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/perils-of-self-feedback-self-bias-amplifies","slug":"perils-of-self-feedback-self-bias-amplifies","title":"Pride and Prejudice: LLM Amplifies Self-Bias in Self-Refinement","date":"2024-02-18","arxiv_id":"2402.11436","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perils-of-self-feedback-self-bias-amplifies#ran","syntology_url":"https://syntology.ai/paper/2402.11436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11436"}},"official":{"repos":["xu1998hz/llm_self_bias"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-is-tree-search-useful-for-llm-planning","slug":"when-is-tree-search-useful-for-llm-planning","title":"When is Tree Search Useful for LLM Planning? It Depends on the Discriminator","date":"2024-02-16","arxiv_id":"2402.10890","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/when-is-tree-search-useful-for-llm-planning#ran","syntology_url":"https://syntology.ai/paper/2402.10890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10890"}},"official":{"repos":["osu-nlp-group/llm-planning-eval"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/mustard-mastering-uniform-synthesis-of","slug":"mustard-mastering-uniform-synthesis-of","title":"MUSTARD: Mastering Uniform Synthesis of Theorem and Proof Data","date":"2024-02-14","arxiv_id":"2402.08957","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mustard-mastering-uniform-synthesis-of#ran","syntology_url":"https://syntology.ai/paper/2402.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08957"}},"official":{"repos":["eleanor-h/mustard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-reasoning-in-generative-large","slug":"probabilistic-reasoning-in-generative-large","title":"Reasoning over Uncertain Text by Generative Large Language Models","date":"2024-02-14","arxiv_id":"2402.09614","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-speculative-sampling-requires","slug":"eagle-speculative-sampling-requires","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","date":"2024-01-26","arxiv_id":"2401.15077","repositories_listed":1,"syntology":null},{"url":"/paper/superclue-math6-graded-multi-step-math","slug":"superclue-math6-graded-multi-step-math","title":"SuperCLUE-Math6: Graded Multi-Step Math Reasoning Benchmark for LLMs in Chinese","date":"2024-01-22","arxiv_id":"2401.11819","repositories_listed":1,"syntology":null},{"url":"/paper/langbridge-multilingual-reasoning-without","slug":"langbridge-multilingual-reasoning-without","title":"LangBridge: Multilingual Reasoning Without Multilingual Supervision","date":"2024-01-19","arxiv_id":"2401.10695","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/langbridge-multilingual-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2401.10695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10695"}},"official":{"repos":["kaistAI/LangBridge"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/augmenting-math-word-problems-via-iterative","slug":"augmenting-math-word-problems-via-iterative","title":"Augmenting Math Word Problems via Iterative Question Composing","date":"2024-01-17","arxiv_id":"2401.09003","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-math-word-problems-via-iterative#ran","syntology_url":"https://syntology.ai/paper/2401.09003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09003"}},"official":{"repos":["iiis-ai/iterativequestioncomposing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/question-translation-training-for-better","slug":"question-translation-training-for-better","title":"Question Translation Training for Better Multilingual Reasoning","date":"2024-01-15","arxiv_id":"2401.07817","repositories_listed":1,"syntology":null},{"url":"/paper/sciglm-training-scientific-language-models","slug":"sciglm-training-scientific-language-models","title":"SciInstruct: a Self-Reflective Instruction Annotated Dataset for Training Scientific Language Models","date":"2024-01-15","arxiv_id":"2401.07950","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sciglm-training-scientific-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.07950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07950"}},"official":{"repos":["thudm/sciglm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mapo-advancing-multilingual-reasoning-through","slug":"mapo-advancing-multilingual-reasoning-through","title":"MAPO: Advancing Multilingual Reasoning through Multilingual Alignment-as-Preference Optimization","date":"2024-01-12","arxiv_id":"2401.06838","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mapo-advancing-multilingual-reasoning-through#ran","syntology_url":"https://syntology.ai/paper/2401.06838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06838"}},"official":{"repos":["njunlp/mapo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-ai-for-math-part-i-mathpile-a","slug":"generative-ai-for-math-part-i-mathpile-a","title":"MathPile: A Billion-Token-Scale Pretraining Corpus for Math","date":"2023-12-28","arxiv_id":"2312.17120","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generative-ai-for-math-part-i-mathpile-a#ran","syntology_url":"https://syntology.ai/paper/2312.17120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17120"}},"official":{"repos":["gair-nlp/mathpile"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-in-depth-look-at-gemini-s-language#ran","syntology_url":"https://syntology.ai/paper/2312.11444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11444"}},"official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-complex-mathematical-reasoning-via","slug":"modeling-complex-mathematical-reasoning-via","title":"Modeling Complex Mathematical Reasoning via Large Language Model based MathAgent","date":"2023-12-14","arxiv_id":"2312.08926","repositories_listed":1,"syntology":null},{"url":"/paper/frugal-lms-trained-to-invoke-symbolic-solvers","slug":"frugal-lms-trained-to-invoke-symbolic-solvers","title":"Frugal LMs Trained to Invoke Symbolic Solvers Achieve Parameter-Efficient Arithmetic Reasoning","date":"2023-12-09","arxiv_id":"2312.05571","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/frugal-lms-trained-to-invoke-symbolic-solvers#ran","syntology_url":"https://syntology.ai/paper/2312.05571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.05571"}},"official":{"repos":["joykirat18/syrelm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/speak-like-a-native-prompting-large-language","slug":"speak-like-a-native-prompting-large-language","title":"AlignedCoT: Prompting Large Language Models via Native-Speaking Demonstrations","date":"2023-11-22","arxiv_id":"2311.13538","repositories_listed":1,"syntology":null},{"url":"/paper/outcome-supervised-verifiers-for-planning-in","slug":"outcome-supervised-verifiers-for-planning-in","title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","date":"2023-11-16","arxiv_id":"2311.09724","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/outcome-supervised-verifiers-for-planning-in#ran","syntology_url":"https://syntology.ai/paper/2311.09724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09724"}},"official":{"repos":["freedomintelligence/ovm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/veritymath-advancing-mathematical-reasoning","slug":"veritymath-advancing-mathematical-reasoning","title":"VerityMath: Advancing Mathematical Reasoning by Self-Verification Through Unit Consistency","date":"2023-11-13","arxiv_id":"2311.07172","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/veritymath-advancing-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2311.07172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07172"}},"official":{"repos":["vernontoh/veritymath"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/athena-mathematical-reasoning-with-thought","slug":"athena-mathematical-reasoning-with-thought","title":"ATHENA: Mathematical Reasoning with Thought Expansion","date":"2023-11-02","arxiv_id":"2311.01036","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-mistakes-makes-llm-better","slug":"learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","arxiv_id":"2310.20689","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-mistakes-makes-llm-better#ran","syntology_url":"https://syntology.ai/paper/2310.20689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20689"}},"official":{"repos":["microsoft/lema"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skymath-technical-report","slug":"skymath-technical-report","title":"SkyMath: Technical Report","date":"2023-10-25","arxiv_id":"2310.16713","repositories_listed":1,"syntology":null},{"url":"/paper/mcc-kd-multi-cot-consistent-knowledge","slug":"mcc-kd-multi-cot-consistent-knowledge","title":"MCC-KD: Multi-CoT Consistent Knowledge Distillation","date":"2023-10-23","arxiv_id":"2310.14747","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mcc-kd-multi-cot-consistent-knowledge#ran","syntology_url":"https://syntology.ai/paper/2310.14747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14747"}},"official":{"repos":["homzer/MCC-KD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/maf-multi-aspect-feedback-for-improving","slug":"maf-multi-aspect-feedback-for-improving","title":"MAF: Multi-Aspect Feedback for Improving Reasoning in Large Language Models","date":"2023-10-19","arxiv_id":"2310.12426","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/maf-multi-aspect-feedback-for-improving#ran","syntology_url":"https://syntology.ai/paper/2310.12426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12426"}},"official":{"repos":["deepakn97/maf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trigo-benchmarking-formal-mathematical-proof","slug":"trigo-benchmarking-formal-mathematical-proof","title":"TRIGO: Benchmarking Formal Mathematical Proof Reduction for Generative Language Models","date":"2023-10-16","arxiv_id":"2310.10180","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/trigo-benchmarking-formal-mathematical-proof#ran","syntology_url":"https://syntology.ai/paper/2310.10180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10180"}},"official":{"repos":["menik1126/TRIGO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/an-expression-tree-decoding-strategy-for","slug":"an-expression-tree-decoding-strategy-for","title":"An Expression Tree Decoding Strategy for Mathematical Equation Generation","date":"2023-10-14","arxiv_id":"2310.09619","repositories_listed":1,"syntology":null},{"url":"/paper/trace-a-comprehensive-benchmark-for-continual","slug":"trace-a-comprehensive-benchmark-for-continual","title":"TRACE: A Comprehensive Benchmark for Continual Learning in Large Language Models","date":"2023-10-10","arxiv_id":"2310.06762","repositories_listed":1,"syntology":null},{"url":"/paper/query-and-response-augmentation-cannot-help","slug":"query-and-response-augmentation-cannot-help","title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","date":"2023-10-09","arxiv_id":"2310.05506","repositories_listed":1,"syntology":null},{"url":"/paper/ada-instruct-adapting-instruction-generators","slug":"ada-instruct-adapting-instruction-generators","title":"Ada-Instruct: Adapting Instruction Generators for Complex Reasoning","date":"2023-10-06","arxiv_id":"2310.04484","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder-seamless-code-integration-in-llms","slug":"mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","arxiv_id":"2310.03731","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-seamless-code-integration-in-llms#ran","syntology_url":"https://syntology.ai/paper/2310.03731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03731"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/tora-a-tool-integrated-reasoning-agent-for","slug":"tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","arxiv_id":"2309.17452","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tora-a-tool-integrated-reasoning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2309.17452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17452"}},"official":{"repos":["microsoft/tora"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/code-soliloquies-for-accurate-calculations-in","slug":"code-soliloquies-for-accurate-calculations-in","title":"Code Soliloquies for Accurate Calculations in Large Language Models","date":"2023-09-21","arxiv_id":"2309.12161","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/code-soliloquies-for-accurate-calculations-in#ran","syntology_url":"https://syntology.ai/paper/2309.12161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12161"}},"official":{"repos":["luffycodes/tutorbot-spock-phys"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/metamath-bootstrap-your-own-mathematical","slug":"metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","arxiv_id":"2309.12284","repositories_listed":1,"syntology":{"n":22,"n_ran":15,"n_constructed":0,"n_ran_checked":1,"n_instrument":14,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/metamath-bootstrap-your-own-mathematical#ran","syntology_url":"https://syntology.ai/paper/2309.12284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12284"}},"official":{"repos":["meta-math/MetaMath"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/auto-regressive-next-token-predictors-are","slug":"auto-regressive-next-token-predictors-are","title":"Auto-Regressive Next-Token Predictors are Universal Learners","date":"2023-09-13","arxiv_id":"2309.06979","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/auto-regressive-next-token-predictors-are#ran","syntology_url":"https://syntology.ai/paper/2309.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06979"}},"official":{"repos":["emalach/linearlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mammoth-building-math-generalist-models","slug":"mammoth-building-math-generalist-models","title":"MAmmoTH: Building Math Generalist Models through Hybrid Instruction Tuning","date":"2023-09-11","arxiv_id":"2309.05653","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammoth-building-math-generalist-models#ran","syntology_url":"https://syntology.ai/paper/2309.05653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05653"}},"official":null}},{"url":"/paper/when-do-program-of-thoughts-work-for","slug":"when-do-program-of-thoughts-work-for","title":"When Do Program-of-Thoughts Work for Reasoning?","date":"2023-08-29","arxiv_id":"2308.15452","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-learning-via-weight-aware","slug":"semi-supervised-learning-via-weight-aware","title":"Semi-Supervised Learning via Weight-aware Distillation under Class Distribution Mismatch","date":"2023-08-23","arxiv_id":"2308.11874","repositories_listed":1,"syntology":null},{"url":"/paper/wizardmath-empowering-mathematical-reasoning","slug":"wizardmath-empowering-mathematical-reasoning","title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","date":"2023-08-18","arxiv_id":"2308.09583","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":16,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/wizardmath-empowering-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2308.09583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09583"}},"official":null}},{"url":"/paper/separate-the-wheat-from-the-chaff-model","slug":"separate-the-wheat-from-the-chaff-model","title":"Separate the Wheat from the Chaff: Model Deficiency Unlearning via Parameter-Efficient Module Operation","date":"2023-08-16","arxiv_id":"2308.08090","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/separate-the-wheat-from-the-chaff-model#ran","syntology_url":"https://syntology.ai/paper/2308.08090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08090"}},"official":{"repos":["hitsz-tmg/ext-sub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-challenging-math-word-problems-using","slug":"solving-challenging-math-word-problems-using","title":"Solving Challenging Math Word Problems Using GPT-4 Code Interpreter with Code-based Self-Verification","date":"2023-08-15","arxiv_id":"2308.07921","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-relationship-on-learning-mathematical","slug":"scaling-relationship-on-learning-mathematical","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","date":"2023-08-03","arxiv_id":"2308.01825","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-relationship-on-learning-mathematical#ran","syntology_url":"https://syntology.ai/paper/2308.01825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01825"}},"official":{"repos":["ofa-sys/gsm8k-screl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/math-word-problem-solving-by-generating","slug":"math-word-problem-solving-by-generating","title":"Math Word Problem Solving by Generating Linguistic Variants of Problem Statements","date":"2023-06-24","arxiv_id":"2306.13899","repositories_listed":1,"syntology":null},{"url":"/paper/efficiently-measuring-the-cognitive-ability","slug":"efficiently-measuring-the-cognitive-ability","title":"Position: AI Evaluation Should Learn from How We Test Humans","date":"2023-06-18","arxiv_id":"2306.10512","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-really-good-logical","slug":"are-large-language-models-really-good-logical","title":"Are Large Language Models Really Good Logical Reasoners? A Comprehensive Evaluation and Beyond","date":"2023-06-16","arxiv_id":"2306.09841","repositories_listed":1,"syntology":null},{"url":"/paper/turning-large-language-models-into-cognitive","slug":"turning-large-language-models-into-cognitive","title":"Turning large language models into cognitive models","date":"2023-06-06","arxiv_id":"2306.03917","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-language-models-for-mathematics","slug":"evaluating-language-models-for-mathematics","title":"Evaluating Language Models for Mathematics through Interactions","date":"2023-06-02","arxiv_id":"2306.01694","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/evaluating-language-models-for-mathematics#ran","syntology_url":"https://syntology.ai/paper/2306.01694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01694"}},"official":{"repos":["collinskatie/checkmate"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-multi-step-reasoning-from-arithmetic","slug":"learning-multi-step-reasoning-from-arithmetic","title":"Learning Multi-Step Reasoning by Solving Arithmetic Tasks","date":"2023-06-02","arxiv_id":"2306.01707","repositories_listed":1,"syntology":null},{"url":"/paper/gorilla-large-language-model-connected-with","slug":"gorilla-large-language-model-connected-with","title":"Gorilla: Large Language Model Connected with Massive APIs","date":"2023-05-24","arxiv_id":"2305.15334","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-arithmetic-reasoning-in","slug":"understanding-arithmetic-reasoning-in","title":"A Mechanistic Interpretation of Arithmetic Reasoning in Language Models using Causal Mediation Analysis","date":"2023-05-24","arxiv_id":"2305.15054","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":3,"n_ran_checked":3,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/understanding-arithmetic-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2305.15054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15054"}},"official":{"repos":["alestolfo/lm-arithmetic"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fedcbo-reaching-group-consensus-in-clustered","slug":"fedcbo-reaching-group-consensus-in-clustered","title":"FedCBO: Reaching Group Consensus in Clustered Federated Learning through Consensus-based Optimization","date":"2023-05-04","arxiv_id":"2305.02894","repositories_listed":1,"syntology":null},{"url":"/paper/nature-language-reasoning-a-survey","slug":"nature-language-reasoning-a-survey","title":"Natural Language Reasoning, A Survey","date":"2023-03-26","arxiv_id":"2303.14725","repositories_listed":1,"syntology":null},{"url":"/paper/mathprompter-mathematical-reasoning-using","slug":"mathprompter-mathematical-reasoning-using","title":"MathPrompter: Mathematical Reasoning using Large Language Models","date":"2023-03-04","arxiv_id":"2303.05398","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mathprompter-mathematical-reasoning-using#ran","syntology_url":"https://syntology.ai/paper/2303.05398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05398"}},"official":null}},{"url":"/paper/a-multi-modal-neural-geometric-solver-with","slug":"a-multi-modal-neural-geometric-solver-with","title":"A Multi-Modal Neural Geometric Solver with Textual Clauses Parsed from Diagram","date":"2023-02-22","arxiv_id":"2302.11097","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-multi-modal-neural-geometric-solver-with#ran","syntology_url":"https://syntology.ai/paper/2302.11097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11097"}},"official":{"repos":["mingliangzhang2018/pgps"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-for-robotics-design-principles-and","slug":"chatgpt-for-robotics-design-principles-and","title":"ChatGPT for Robotics: Design Principles and Model Abilities","date":"2023-02-20","arxiv_id":"2306.17582","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/chatgpt-for-robotics-design-principles-and#ran","syntology_url":"https://syntology.ai/paper/2306.17582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17582"}},"official":{"repos":["microsoft/promptcraft-robotics"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/tree-based-representation-and-generation-of","slug":"tree-based-representation-and-generation-of","title":"Tree-Based Representation and Generation of Natural and Mathematical Language","date":"2023-02-15","arxiv_id":"2302.07974","repositories_listed":1,"syntology":null},{"url":"/paper/explanation-selection-using-unlabeled-data","slug":"explanation-selection-using-unlabeled-data","title":"Explanation Selection Using Unlabeled Data for Chain-of-Thought Prompting","date":"2023-02-09","arxiv_id":"2302.04813","repositories_listed":1,"syntology":null},{"url":"/paper/techniques-to-improve-neural-math-word","slug":"techniques-to-improve-neural-math-word","title":"Techniques to Improve Neural Math Word Problem Solvers","date":"2023-02-06","arxiv_id":"2302.03145","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-learning-for-mathematical","slug":"a-survey-of-deep-learning-for-mathematical","title":"A Survey of Deep Learning for Mathematical Reasoning","date":"2022-12-20","arxiv_id":"2212.10535","repositories_listed":1,"syntology":null},{"url":"/paper/peano-learning-formal-mathematical-reasoning","slug":"peano-learning-formal-mathematical-reasoning","title":"Peano: Learning Formal Mathematical Reasoning","date":"2022-11-29","arxiv_id":"2211.15864","repositories_listed":1,"syntology":null},{"url":"/paper/galactica-a-large-language-model-for-science-1","slug":"galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/galactica-a-large-language-model-for-science-1#ran","syntology_url":"https://syntology.ai/paper/2211.09085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09085"}},"official":{"repos":["paperswithcode/galai"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lemma-bootstrapping-high-level-mathematical","slug":"lemma-bootstrapping-high-level-mathematical","title":"LEMMA: Bootstrapping High-Level Mathematical Reasoning with Learned Symbolic Abstractions","date":"2022-11-16","arxiv_id":"2211.08671","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lemma-bootstrapping-high-level-mathematical#ran","syntology_url":"https://syntology.ai/paper/2211.08671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.08671"}},"official":{"repos":["uranium11010/lemma"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-barriers-to-skill-injection-in","slug":"overcoming-barriers-to-skill-injection-in","title":"Overcoming Barriers to Skill Injection in Language Modeling: Case Study in Arithmetic","date":"2022-11-03","arxiv_id":"2211.02098","repositories_listed":1,"syntology":null},{"url":"/paper/blank-collapse-compressing-ctc-emission-for","slug":"blank-collapse-compressing-ctc-emission-for","title":"Blank Collapse: Compressing CTC emission for the faster decoding","date":"2022-10-31","arxiv_id":"2210.17017","repositories_listed":1,"syntology":null},{"url":"/paper/lila-a-unified-benchmark-for-mathematical","slug":"lila-a-unified-benchmark-for-mathematical","title":"Lila: A Unified Benchmark for Mathematical Reasoning","date":"2022-10-31","arxiv_id":"2210.17517","repositories_listed":1,"syntology":null},{"url":"/paper/a-causal-framework-to-quantify-the-robustness","slug":"a-causal-framework-to-quantify-the-robustness","title":"A Causal Framework to Quantify the Robustness of Mathematical Reasoning with Language Models","date":"2022-10-21","arxiv_id":"2210.12023","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-causal-framework-to-quantify-the-robustness#ran","syntology_url":"https://syntology.ai/paper/2210.12023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12023"}},"official":{"repos":["alestolfo/causal-math"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-view-reasoning-consistent-contrastive","slug":"multi-view-reasoning-consistent-contrastive","title":"Multi-View Reasoning: Consistent Contrastive Learning for Math Word Problem","date":"2022-10-21","arxiv_id":"2210.11694","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-view-reasoning-consistent-contrastive#ran","syntology_url":"https://syntology.ai/paper/2210.11694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11694"}},"official":{"repos":["zwq2018/multi-view-consistency-for-mwp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-formula-learner-for-solving","slug":"weakly-supervised-formula-learner-for-solving","title":"Weakly Supervised Formula Learner for Solving Mathematical Problems","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/clevr-math-a-dataset-for-compositional","slug":"clevr-math-a-dataset-for-compositional","title":"CLEVR-Math: A Dataset for Compositional Language, Visual and Mathematical Reasoning","date":"2022-08-10","arxiv_id":"2208.05358","repositories_listed":1,"syntology":null},{"url":"/paper/transformers-discover-an-elementary","slug":"transformers-discover-an-elementary","title":"Transformers discover an elementary calculation system exploiting local attention and grid-like problem representation","date":"2022-07-06","arxiv_id":"2207.02536","repositories_listed":1,"syntology":null},{"url":"/paper/a-neural-network-solves-and-generates","slug":"a-neural-network-solves-and-generates","title":"A Neural Network Solves, Explains, and Generates University Math Problems by Program Synthesis and Few-Shot Learning at Human Level","date":"2021-12-31","arxiv_id":"2112.15594","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-neural-network-solves-and-generates#ran","syntology_url":"https://syntology.ai/paper/2112.15594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.15594"}},"official":{"repos":["idrori/mathq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/iconqa-a-new-benchmark-for-abstract-diagram#ran","syntology_url":"https://syntology.ai/paper/2110.13214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13214"}},"official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-with-transformer-based-models-deep","slug":"reasoning-with-transformer-based-models-deep","title":"Reasoning with Transformer-based Models: Deep Learning, but Shallow Reasoning","date":"2021-06-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/geoqa-a-geometric-question-answering","slug":"geoqa-a-geometric-question-answering","title":"GeoQA: A Geometric Question Answering Benchmark Towards Multimodal Numerical Reasoning","date":"2021-05-30","arxiv_id":"2105.14517","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"0 ran · 5 unverified","sample_list":"/paper/geoqa-a-geometric-question-answering#ran","syntology_url":"https://syntology.ai/paper/2105.14517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14517"}},"official":{"repos":["chen-judge/GeoQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/compositional-processing-emerges-in-neural","slug":"compositional-processing-emerges-in-neural","title":"Compositional Processing Emerges in Neural Networks Solving Math Problems","date":"2021-05-19","arxiv_id":"2105.08961","repositories_listed":1,"syntology":null},{"url":"/paper/inter-gps-interpretable-geometry-problem","slug":"inter-gps-interpretable-geometry-problem","title":"Inter-GPS: Interpretable Geometry Problem Solving with Formal Language and Symbolic Reasoning","date":"2021-05-10","arxiv_id":"2105.04165","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inter-gps-interpretable-geometry-problem#ran","syntology_url":"https://syntology.ai/paper/2105.04165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04165"}},"official":{"repos":["lupantech/InterGPS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lime-learning-inductive-bias-for-primitives-1","slug":"lime-learning-inductive-bias-for-primitives-1","title":"LIME: Learning Inductive Bias for Primitives of Mathematical Reasoning","date":"2021-01-15","arxiv_id":"2101.06223","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lime-learning-inductive-bias-for-primitives-1#ran","syntology_url":"https://syntology.ai/paper/2101.06223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.06223"}},"official":{"repos":["tonywu95/lime"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reverse-operation-based-data-augmentation-for","slug":"reverse-operation-based-data-augmentation-for","title":"Reverse Operation based Data Augmentation for Solving Math Word Problems","date":"2020-10-04","arxiv_id":"2010.01556","repositories_listed":1,"syntology":null},{"url":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-prove-theorems-via-interacting","slug":"learning-to-prove-theorems-via-interacting","title":"Learning to Prove Theorems via Interacting with Proof Assistants","date":"2019-05-21","arxiv_id":"1905.09381","repositories_listed":1,"syntology":null},{"url":null,"slug":"var-math-probing-true-mathematical-reasoning","title":"VAR-MATH: Probing True Mathematical Reasoning in Large Language Models via Symbolic Multi-Instance Benchmarks","date":"2025-07-17","arxiv_id":"2507.12885","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-deep-learning-for-geometry","title":"A Survey of Deep Learning for Geometry Problem Solving","date":"2025-07-16","arxiv_id":"2507.11936","repositories_listed":0,"syntology":null},{"url":null,"slug":"kismath-do-llms-have-knowledge-of-implicit","title":"KisMATH: Do LLMs Have Knowledge of Implicit Structures in Mathematical Reasoning?","date":"2025-07-15","arxiv_id":"2507.11408","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-external-tools-with-large","title":"Integrating External Tools with Large Language Models to Improve Accuracy","date":"2025-07-09","arxiv_id":"2507.08034","repositories_listed":0,"syntology":null},{"url":"/paper/agentic-r1-distilled-dual-strategy-reasoning","slug":"agentic-r1-distilled-dual-strategy-reasoning","title":"Agentic-R1: Distilled Dual-Strategy Reasoning","date":"2025-07-08","arxiv_id":"2507.05707","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agentic-r1-distilled-dual-strategy-reasoning#ran","syntology_url":"https://syntology.ai/paper/2507.05707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05707"}},"official":null}}],"record_sha256":"6b0534e92ebacafdc63e5831f8c7b177733bd145a3a29399cf1cded78aeddb43","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}