{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/3","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":9,"rows_per_page":100,"rows":[201,300],"of":805,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning","prev":"/task/mathematical-reasoning/papers/2","next":"/task/mathematical-reasoning/papers/4","papers":[{"url":"/paper/template-driven-llm-paraphrased-framework-for","slug":"template-driven-llm-paraphrased-framework-for","title":"Template-Driven LLM-Paraphrased Framework for Tabular Math Word Problem Generation","date":"2024-12-20","arxiv_id":"2412.15594","repositories_listed":1,"syntology":null},{"url":"/paper/critical-questions-of-thought-steering-llm","slug":"critical-questions-of-thought-steering-llm","title":"Critical-Questions-of-Thought: Steering LLM reasoning with Argumentative Querying","date":"2024-12-19","arxiv_id":"2412.15177","repositories_listed":1,"syntology":null},{"url":"/paper/multilingpot-enhancing-mathematical-reasoning","slug":"multilingpot-enhancing-mathematical-reasoning","title":"MultiLingPoT: Enhancing Mathematical Reasoning with Multilingual Program Fine-tuning","date":"2024-12-17","arxiv_id":"2412.12609","repositories_listed":1,"syntology":null},{"url":"/paper/coinmath-harnessing-the-power-of-coding","slug":"coinmath-harnessing-the-power-of-coding","title":"CoinMath: Harnessing the Power of Coding Instruction for Math LLMs","date":"2024-12-16","arxiv_id":"2412.11699","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-regularized-process-reward-model","slug":"entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","arxiv_id":"2412.11006","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-process-reward-model#ran","syntology_url":"https://syntology.ai/paper/2412.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11006"}},"official":{"repos":["hanningzhang/er-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/processbench-identifying-process-errors-in","slug":"processbench-identifying-process-errors-in","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","date":"2024-12-09","arxiv_id":"2412.06559","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/processbench-identifying-process-errors-in#ran","syntology_url":"https://syntology.ai/paper/2412.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06559"}},"official":{"repos":["qwenlm/processbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taco-learning-multi-modal-action-models-with","slug":"taco-learning-multi-modal-action-models-with","title":"TACO: Learning Multi-modal Action Models with Synthetic Chains-of-Thought-and-Action","date":"2024-12-07","arxiv_id":"2412.05479","repositories_listed":1,"syntology":null},{"url":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critical-tokens-matter-token-level#ran","syntology_url":"https://syntology.ai/paper/2411.19943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19943"}},"official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/initialization-using-update-approximation-is","slug":"initialization-using-update-approximation-is","title":"Initialization using Update Approximation is a Silver Bullet for Extremely Efficient Low-Rank Fine-Tuning","date":"2024-11-29","arxiv_id":"2411.19557","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-examples-high-level-automated","slug":"beyond-examples-high-level-automated","title":"Beyond Examples: High-level Automated Reasoning Paradigm in In-Context Learning via MCTS","date":"2024-11-27","arxiv_id":"2411.18478","repositories_listed":1,"syntology":null},{"url":"/paper/training-and-evaluating-language-models-with","slug":"training-and-evaluating-language-models-with","title":"Training and Evaluating Language Models with Template-based Data Generation","date":"2024-11-27","arxiv_id":"2411.18104","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-and-evaluating-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2411.18104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18104"}},"official":{"repos":["iiis-ai/templatemath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/o1-replication-journey-part-2-surpassing-o1","slug":"o1-replication-journey-part-2-surpassing-o1","title":"O1 Replication Journey -- Part 2: Surpassing O1-preview through Simple Distillation, Big Progress or Bitter Lesson?","date":"2024-11-25","arxiv_id":"2411.16489","repositories_listed":1,"syntology":null},{"url":"/paper/preference-optimization-for-reasoning-with","slug":"preference-optimization-for-reasoning-with","title":"Preference Optimization for Reasoning with Pseudo Feedback","date":"2024-11-25","arxiv_id":"2411.16345","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preference-optimization-for-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16345"}},"official":null}},{"url":"/paper/atomthink-a-slow-thinking-framework-for","slug":"atomthink-a-slow-thinking-framework-for","title":"AtomThink: A Slow Thinking Framework for Multimodal Mathematical Reasoning","date":"2024-11-18","arxiv_id":"2411.11930","repositories_listed":1,"syntology":null},{"url":"/paper/pspo-an-effective-process-supervised-policy","slug":"pspo-an-effective-process-supervised-policy","title":"PSPO*: An Effective Process-supervised Policy Optimization for Reasoning Alignment","date":"2024-11-18","arxiv_id":"2411.11681","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pspo-an-effective-process-supervised-policy#ran","syntology_url":"https://syntology.ai/paper/2411.11681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11681"}},"official":{"repos":["direct-bit/pspo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utmath-math-evaluation-with-unit-test-via","slug":"utmath-math-evaluation-with-unit-test-via","title":"UTMath: Math Evaluation with Unit Test via Reasoning-to-Coding Thoughts","date":"2024-11-11","arxiv_id":"2411.07240","repositories_listed":1,"syntology":null},{"url":"/paper/gap-filling-prompting-enhances-code-assisted","slug":"gap-filling-prompting-enhances-code-assisted","title":"Gap-Filling Prompting Enhances Code-Assisted Mathematical Reasoning","date":"2024-11-08","arxiv_id":"2411.05407","repositories_listed":1,"syntology":null},{"url":"/paper/library-learning-doesn-t-the-curious-case-of","slug":"library-learning-doesn-t-the-curious-case-of","title":"Library Learning Doesn't: The Curious Case of the Single-Use \"Library\"","date":"2024-10-26","arxiv_id":"2410.20274","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/library-learning-doesn-t-the-curious-case-of#ran","syntology_url":"https://syntology.ai/paper/2410.20274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20274"}},"official":{"repos":["ikb-a/curious-case"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/assessing-the-creativity-of-llms-in-proposing","slug":"assessing-the-creativity-of-llms-in-proposing","title":"Assessing the Creativity of LLMs in Proposing Novel Solutions to Mathematical Problems","date":"2024-10-24","arxiv_id":"2410.18336","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/assessing-the-creativity-of-llms-in-proposing#ran","syntology_url":"https://syntology.ai/paper/2410.18336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18336"}},"official":{"repos":["junyiye/creativemath"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/siked-self-guided-iterative-knowledge","slug":"siked-self-guided-iterative-knowledge","title":"SIKeD: Self-guided Iterative Knowledge Distillation for mathematical reasoning","date":"2024-10-24","arxiv_id":"2410.18574","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-reasoning-capability-of-llms-via","slug":"unleashing-reasoning-capability-of-llms-via","title":"Unleashing Reasoning Capability of LLMs via Scalable Question Synthesis from Scratch","date":"2024-10-24","arxiv_id":"2410.18693","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unleashing-reasoning-capability-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.18693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18693"}},"official":{"repos":["yyding1/scalequest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/not-all-votes-count-programs-as-verifiers","slug":"not-all-votes-count-programs-as-verifiers","title":"Not All Votes Count! Programs as Verifiers Improve Self-Consistency of Language Models for Math Reasoning","date":"2024-10-16","arxiv_id":"2410.12608","repositories_listed":1,"syntology":null},{"url":"/paper/comat-chain-of-mathematically-annotated","slug":"comat-chain-of-mathematically-annotated","title":"CoMAT: Chain of Mathematically Annotated Thought Improves Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10336","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comat-chain-of-mathematically-annotated#ran","syntology_url":"https://syntology.ai/paper/2410.10336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10336"}},"official":{"repos":["joshuaongg21/comat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hardmath-a-benchmark-dataset-for-challenging","slug":"hardmath-a-benchmark-dataset-for-challenging","title":"HARDMath: A Benchmark Dataset for Challenging Problems in Applied Mathematics","date":"2024-10-13","arxiv_id":"2410.09988","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder2-better-math-reasoning-from","slug":"mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","arxiv_id":"2410.08196","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mathcoder2-better-math-reasoning-from#ran","syntology_url":"https://syntology.ai/paper/2410.08196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08196"}},"official":{"repos":["mathllm/mathcoder2"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-inspired-integrated-prompting","slug":"teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","arxiv_id":"2410.08068","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-inspired-integrated-prompting#ran","syntology_url":"https://syntology.ai/paper/2410.08068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08068"}},"official":{"repos":["sallytan13/teaching-inspired-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/give-me-a-hint-can-llms-take-a-hint-to-solve","slug":"give-me-a-hint-can-llms-take-a-hint-to-solve","title":"Give me a hint: Can LLMs take a hint to solve math problems?","date":"2024-10-08","arxiv_id":"2410.05915","repositories_listed":1,"syntology":null},{"url":"/paper/leanagent-lifelong-learning-for-formal","slug":"leanagent-lifelong-learning-for-formal","title":"LeanAgent: Lifelong Learning for Formal Theorem Proving","date":"2024-10-08","arxiv_id":"2410.06209","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/leanagent-lifelong-learning-for-formal#ran","syntology_url":"https://syntology.ai/paper/2410.06209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06209"}},"official":null}},{"url":"/paper/polymath-a-challenging-multi-modal","slug":"polymath-a-challenging-multi-modal","title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","date":"2024-10-06","arxiv_id":"2410.14702","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polymath-a-challenging-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2410.14702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14702"}},"official":{"repos":["polymathbenchmark/PolyMATH"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tubench-benchmarking-large-vision-language","slug":"tubench-benchmarking-large-vision-language","title":"TUBench: Benchmarking Large Vision-Language Models on Trustworthiness with Unanswerable Questions","date":"2024-10-05","arxiv_id":"2410.04107","repositories_listed":1,"syntology":null},{"url":"/paper/table-question-answering-for-low-resourced","slug":"table-question-answering-for-low-resourced","title":"Table Question Answering for Low-resourced Indic Languages","date":"2024-10-04","arxiv_id":"2410.03576","repositories_listed":1,"syntology":null},{"url":"/paper/guided-stream-of-search-learning-to-better","slug":"guided-stream-of-search-learning-to-better","title":"Guided Stream of Search: Learning to Better Search with Language Models via Optimal Path Guidance","date":"2024-10-03","arxiv_id":"2410.02992","repositories_listed":1,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":10,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-stream-of-search-learning-to-better#ran","syntology_url":"https://syntology.ai/paper/2410.02992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02992"}},"official":{"repos":["symoon11/guided-stream-of-search"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/llama-berry-pairwise-optimization-for-o1-like","slug":"llama-berry-pairwise-optimization-for-o1-like","title":"LLaMA-Berry: Pairwise Optimization for O1-like Olympiad-Level Mathematical Reasoning","date":"2024-10-03","arxiv_id":"2410.02884","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llama-berry-pairwise-optimization-for-o1-like#ran","syntology_url":"https://syntology.ai/paper/2410.02884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02884"}},"official":null}},{"url":"/paper/openmathinstruct-2-accelerating-ai-for-math","slug":"openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","arxiv_id":"2410.01560","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/openmathinstruct-2-accelerating-ai-for-math#ran","syntology_url":"https://syntology.ai/paper/2410.01560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01560"}},"official":null}},{"url":"/paper/scheherazade-evaluating-chain-of-thought-math","slug":"scheherazade-evaluating-chain-of-thought-math","title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","date":"2024-09-30","arxiv_id":"2410.00151","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scheherazade-evaluating-chain-of-thought-math#ran","syntology_url":"https://syntology.ai/paper/2410.00151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00151"}},"official":{"repos":["yoshikitakashima/scheherazade-code-data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pace-marrying-generalization-in-parameter","slug":"pace-marrying-generalization-in-parameter","title":"PACE: Marrying generalization in PArameter-efficient fine-tuning with Consistency rEgularization","date":"2024-09-25","arxiv_id":"2409.17137","repositories_listed":1,"syntology":{"n":18,"n_ran":8,"n_constructed":1,"n_ran_checked":7,"n_instrument":1,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/pace-marrying-generalization-in-parameter#ran","syntology_url":"https://syntology.ai/paper/2409.17137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17137"}},"official":{"repos":["maxwellyaoni/pace"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":7,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/romath-a-mathematical-reasoning-benchmark-in","slug":"romath-a-mathematical-reasoning-benchmark-in","title":"RoMath: A Mathematical Reasoning Benchmark in Romanian","date":"2024-09-17","arxiv_id":"2409.11074","repositories_listed":1,"syntology":null},{"url":"/paper/ai-for-mathematics-mathematical-formalized","slug":"ai-for-mathematics-mathematical-formalized","title":"Mathematical Formalized Problem Solving and Theorem Proving in Different Fields in Lean 4","date":"2024-09-09","arxiv_id":"2409.05977","repositories_listed":1,"syntology":null},{"url":"/paper/diagram-formalization-enhanced-multi-modal","slug":"diagram-formalization-enhanced-multi-modal","title":"Diagram Formalization Enhanced Multi-Modal Geometry Problem Solver","date":"2024-09-06","arxiv_id":"2409.04214","repositories_listed":1,"syntology":null},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multimath-bridging-visual-and-mathematical","slug":"multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","arxiv_id":"2409.00147","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":6,"n_pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 3 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimath-bridging-visual-and-mathematical#ran","syntology_url":"https://syntology.ai/paper/2409.00147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00147"}},"official":{"repos":["pengshuai-rin/multimath"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-for-math","slug":"benchmarking-large-language-models-for-math","title":"Benchmarking Large Language Models for Math Reasoning Tasks","date":"2024-08-20","arxiv_id":"2408.10839","repositories_listed":1,"syntology":null},{"url":"/paper/math-puma-progressive-upward-multimodal","slug":"math-puma-progressive-upward-multimodal","title":"Math-PUMA: Progressive Upward Multimodal Alignment to Enhance Mathematical Reasoning","date":"2024-08-16","arxiv_id":"2408.08640","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-puma-progressive-upward-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.08640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08640"}},"official":{"repos":["wwzhuang01/math-puma"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathscape-evaluating-mllms-in-multimodal-math","slug":"mathscape-evaluating-mllms-in-multimodal-math","title":"MathScape: Evaluating MLLMs in multimodal Math Scenarios through a Hierarchical Benchmark","date":"2024-08-14","arxiv_id":"2408.07543","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathscape-evaluating-mllms-in-multimodal-math#ran","syntology_url":"https://syntology.ai/paper/2408.07543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07543"}},"official":{"repos":["PKU-Baichuan-MLSystemLab/MathScape"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/maqa-evaluating-uncertainty-quantification-in","slug":"maqa-evaluating-uncertainty-quantification-in","title":"MAQA: Evaluating Uncertainty Quantification in LLMs Regarding Data Uncertainty","date":"2024-08-13","arxiv_id":"2408.06816","repositories_listed":1,"syntology":null},{"url":"/paper/extend-model-merging-from-fine-tuned-to-pre","slug":"extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","arxiv_id":"2408.03092","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extend-model-merging-from-fine-tuned-to-pre#ran","syntology_url":"https://syntology.ai/paper/2408.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03092"}},"official":{"repos":["yule-BUAA/MergeLLM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-assisted-generation-of-difficult-math","slug":"ai-assisted-generation-of-difficult-math","title":"AI-Assisted Generation of Difficult Math Questions","date":"2024-07-30","arxiv_id":"2407.21009","repositories_listed":1,"syntology":null},{"url":"/paper/physics-of-language-models-part-2-1-grade","slug":"physics-of-language-models-part-2-1-grade","title":"Physics of Language Models: Part 2.1, Grade-School Math and the Hidden Reasoning Process","date":"2024-07-29","arxiv_id":"2407.20311","repositories_listed":1,"syntology":null},{"url":"/paper/lora-pro-are-low-rank-adapters-properly","slug":"lora-pro-are-low-rank-adapters-properly","title":"LoRA-Pro: Are Low-Rank Adapters Properly Optimized?","date":"2024-07-25","arxiv_id":"2407.18242","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lora-pro-are-low-rank-adapters-properly#ran","syntology_url":"https://syntology.ai/paper/2407.18242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18242"}},"official":{"repos":["mrflogs/LoRA-Pro"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-training-with-direct-preference","slug":"self-training-with-direct-preference","title":"Self-Training with Direct Preference Optimization Improves Chain-of-Thought Reasoning","date":"2024-07-25","arxiv_id":"2407.18248","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-training-with-direct-preference#ran","syntology_url":"https://syntology.ai/paper/2407.18248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18248"}},"official":{"repos":["tianduowang/dpo-st"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lean-github-compiling-github-lean","slug":"lean-github-compiling-github-lean","title":"LEAN-GitHub: Compiling GitHub LEAN repositories for a versatile LEAN prover","date":"2024-07-24","arxiv_id":"2407.17227","repositories_listed":1,"syntology":null},{"url":"/paper/toward-adaptive-reasoning-in-large-language","slug":"toward-adaptive-reasoning-in-large-language","title":"Toward Adaptive Reasoning in Large Language Models with Thought Rollback","date":"2024-07-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-llms-for-optimization-modeling#ran","syntology_url":"https://syntology.ai/paper/2407.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09887"}},"official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-single-transformer-for-scalable-vision","slug":"a-single-transformer-for-scalable-vision","title":"SOLO: A Single Transformer for Scalable Vision-Language Modeling","date":"2024-07-08","arxiv_id":"2407.06438","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-single-transformer-for-scalable-vision#ran","syntology_url":"https://syntology.ai/paper/2407.06438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06438"}},"official":{"repos":["yangyi-chen/solo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/logicvista-multimodal-llm-logical-reasoning","slug":"logicvista-multimodal-llm-logical-reasoning","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","date":"2024-07-06","arxiv_id":"2407.04973","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicvista-multimodal-llm-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.04973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04973"}},"official":{"repos":["yijia-xiao/logicvista"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/smart-vision-language-reasoners","slug":"smart-vision-language-reasoners","title":"Smart Vision-Language Reasoners","date":"2024-07-05","arxiv_id":"2407.04212","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/smart-vision-language-reasoners#ran","syntology_url":"https://syntology.ai/paper/2407.04212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04212"}},"official":{"repos":["smarter-vlm/smarter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dotamath-decomposition-of-thought-with-code","slug":"dotamath-decomposition-of-thought-with-code","title":"DotaMath: Decomposition of Thought with Code Assistance and Self-correction for Mathematical Reasoning","date":"2024-07-04","arxiv_id":"2407.04078","repositories_listed":1,"syntology":null},{"url":"/paper/theoremllama-transforming-general-purpose","slug":"theoremllama-transforming-general-purpose","title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","date":"2024-07-03","arxiv_id":"2407.03203","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/theoremllama-transforming-general-purpose#ran","syntology_url":"https://syntology.ai/paper/2407.03203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03203"}},"official":{"repos":["RickySkywalker/TheoremLlama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/integrate-the-essence-and-eliminate-the-dross","slug":"integrate-the-essence-and-eliminate-the-dross","title":"Integrate the Essence and Eliminate the Dross: Fine-Grained Self-Consistency for Free-Form Language Generation","date":"2024-07-02","arxiv_id":"2407.02056","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/integrate-the-essence-and-eliminate-the-dross#ran","syntology_url":"https://syntology.ai/paper/2407.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02056"}},"official":{"repos":["WangXinglin/FSC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/frog-evaluating-fuzzy-reasoning-of","slug":"frog-evaluating-fuzzy-reasoning-of","title":"FRoG: Evaluating Fuzzy Reasoning of Generalized Quantifiers in Large Language Models","date":"2024-07-01","arxiv_id":"2407.01046","repositories_listed":1,"syntology":null},{"url":"/paper/we-math-does-your-large-multimodal-model","slug":"we-math-does-your-large-multimodal-model","title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","date":"2024-07-01","arxiv_id":"2407.01284","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/we-math-does-your-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2407.01284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01284"}},"official":{"repos":["we-math/we-math"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-controlled-dpo-leveraging-stepwise-error#ran","syntology_url":"https://syntology.ai/paper/2407.00782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00782"}},"official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive","slug":"h-star-llm-driven-hybrid-sql-text-adaptive","title":"H-STAR: LLM-driven Hybrid SQL-Text Adaptive Reasoning on Tables","date":"2024-06-29","arxiv_id":"2407.05952","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive#ran","syntology_url":"https://syntology.ai/paper/2407.05952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05952"}},"official":{"repos":["nikhilsab/h-star"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/step-dpo-step-wise-preference-optimization#ran","syntology_url":"https://syntology.ai/paper/2406.18629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18629"}},"official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/math-llava-bootstrapping-mathematical","slug":"math-llava-bootstrapping-mathematical","title":"Math-LLaVA: Bootstrapping Mathematical Reasoning for Multimodal Large Language Models","date":"2024-06-25","arxiv_id":"2406.17294","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-llava-bootstrapping-mathematical#ran","syntology_url":"https://syntology.ai/paper/2406.17294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17294"}},"official":{"repos":["hzq950419/math-llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-reason-behind-good-or-bad-towards-a","slug":"the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","arxiv_id":"2406.14024","repositories_listed":1,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":18,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":17,"n_pointer_only":23,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-reason-behind-good-or-bad-towards-a#ran","syntology_url":"https://syntology.ai/paper/2406.14024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14024"}},"official":{"repos":["kbsdjames/math-minos"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathador-lm-a-dynamic-benchmark-for","slug":"mathador-lm-a-dynamic-benchmark-for","title":"Mathador-LM: A Dynamic Benchmark for Mathematical Reasoning on Large Language Models","date":"2024-06-18","arxiv_id":"2406.12572","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathador-lm-a-dynamic-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.12572","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12572"}},"official":{"repos":["ist-daslab/mathador-lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-coder-v2-breaking-the-barrier-of","slug":"deepseek-coder-v2-breaking-the-barrier-of","title":"DeepSeek-Coder-V2: Breaking the Barrier of Closed-Source Models in Code Intelligence","date":"2024-06-17","arxiv_id":"2406.11931","repositories_listed":1,"syntology":null},{"url":"/paper/learn-beyond-the-answer-training-language","slug":"learn-beyond-the-answer-training-language","title":"Learn Beyond The Answer: Training Language Models with Reflection for Mathematical Reasoning","date":"2024-06-17","arxiv_id":"2406.12050","repositories_listed":1,"syntology":null},{"url":"/paper/step-level-value-preference-optimization-for","slug":"step-level-value-preference-optimization-for","title":"Step-level Value Preference Optimization for Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10858","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-level-value-preference-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2406.10858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10858"}},"official":{"repos":["MARIO-Math-Reasoning/Super_MARIO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/living-in-the-moment-can-large-language","slug":"living-in-the-moment-can-large-language","title":"Living in the Moment: Can Large Language Models Grasp Co-Temporal Reasoning?","date":"2024-06-13","arxiv_id":"2406.09072","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/living-in-the-moment-can-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.09072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09072"}},"official":{"repos":["zhaochen0110/cotempqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-of-reasoning-efficient-training-of-llm","slug":"flow-of-reasoning-efficient-training-of-llm","title":"Flow of Reasoning:Training LLMs for Divergent Problem Solving with Minimal Examples","date":"2024-06-09","arxiv_id":"2406.05673","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-of-reasoning-efficient-training-of-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05673"}},"official":{"repos":["yu-fangxu/for"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llms-are-not-intelligent-thinkers-introducing","slug":"llms-are-not-intelligent-thinkers-introducing","title":"LLMs Are Not Intelligent Thinkers: Introducing Mathematical Topic Tree Benchmark for Comprehensive Evaluation of LLMs","date":"2024-06-07","arxiv_id":"2406.05194","repositories_listed":1,"syntology":null},{"url":"/paper/numcot-numerals-and-units-of-measurement-in","slug":"numcot-numerals-and-units-of-measurement-in","title":"NUMCoT: Numerals and Units of Measurement in Chain-of-Thought Reasoning using Large Language Models","date":"2024-06-05","arxiv_id":"2406.02864","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-mathematical-reasoning-of-large","slug":"evaluating-mathematical-reasoning-of-large","title":"Evaluating Mathematical Reasoning of Large Language Models: A Focus on Error Identification and Correction","date":"2024-06-02","arxiv_id":"2406.00755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-mathematical-reasoning-of-large#ran","syntology_url":"https://syntology.ai/paper/2406.00755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00755"}},"official":{"repos":["littlecirc1e/eic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reflectioncoder-learning-from-reflection","slug":"reflectioncoder-learning-from-reflection","title":"ReflectionCoder: Learning from Reflection Sequence for Enhanced One-off Code Generation","date":"2024-05-27","arxiv_id":"2405.17057","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reflectioncoder-learning-from-reflection#ran","syntology_url":"https://syntology.ai/paper/2405.17057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17057"}},"official":{"repos":["sensellm/reflectioncoder"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/stride-a-tool-assisted-llm-agent-framework","slug":"stride-a-tool-assisted-llm-agent-framework","title":"STRIDE: A Tool-Assisted LLM Agent Framework for Strategic and Interactive Decision-Making","date":"2024-05-25","arxiv_id":"2405.16376","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stride-a-tool-assisted-llm-agent-framework#ran","syntology_url":"https://syntology.ai/paper/2405.16376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16376"}},"official":{"repos":["cyrilli/stride"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/intelligent-go-explore-standing-on-the","slug":"intelligent-go-explore-standing-on-the","title":"Intelligent Go-Explore: Standing on the Shoulders of Giant Foundation Models","date":"2024-05-24","arxiv_id":"2405.15143","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intelligent-go-explore-standing-on-the#ran","syntology_url":"https://syntology.ai/paper/2405.15143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15143"}},"official":{"repos":["conglu1997/intelligent-go-explore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vb-lora-extreme-parameter-efficient-fine","slug":"vb-lora-extreme-parameter-efficient-fine","title":"VB-LoRA: Extreme Parameter Efficient Fine-Tuning with Vector Banks","date":"2024-05-24","arxiv_id":"2405.15179","repositories_listed":1,"syntology":{"n":17,"n_ran":17,"n_constructed":0,"n_ran_checked":11,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":17,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vb-lora-extreme-parameter-efficient-fine#ran","syntology_url":"https://syntology.ai/paper/2405.15179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15179"}},"official":{"repos":["leo-yangli/vb-lora"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-solve-longer-math-word-problems","slug":"can-llms-solve-longer-math-word-problems","title":"Can LLMs Solve longer Math Word Problems Better?","date":"2024-05-23","arxiv_id":"2405.14804","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":5,"n_ran_checked":6,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-llms-solve-longer-math-word-problems#ran","syntology_url":"https://syntology.ai/paper/2405.14804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14804"}},"official":{"repos":["xinxu-ustc/coleg-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["community","found_in_text","official","unlocated"]}}},{"url":"/paper/jiuzhang3-0-efficiently-improving","slug":"jiuzhang3-0-efficiently-improving","title":"JiuZhang3.0: Efficiently Improving Mathematical Reasoning by Training Small Data Synthesis Models","date":"2024-05-23","arxiv_id":"2405.14365","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/jiuzhang3-0-efficiently-improving#ran","syntology_url":"https://syntology.ai/paper/2405.14365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14365"}},"official":{"repos":["rucaibox/jiuzhang3.0"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-volatility-for-out-of-distribution","slug":"trajectory-volatility-for-out-of-distribution","title":"Embedding Trajectory for Out-of-Distribution Detection in Mathematical Reasoning","date":"2024-05-22","arxiv_id":"2405.14039","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-volatility-for-out-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2405.14039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14039"}},"official":{"repos":["alsace08/ood-math-reasoning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mora-high-rank-updating-for-parameter","slug":"mora-high-rank-updating-for-parameter","title":"MoRA: High-Rank Updating for Parameter-Efficient Fine-Tuning","date":"2024-05-20","arxiv_id":"2405.12130","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mora-high-rank-updating-for-parameter#ran","syntology_url":"https://syntology.ai/paper/2405.12130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12130"}},"official":{"repos":["kongds/mora"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mumath-code-combining-tool-use-large-language","slug":"mumath-code-combining-tool-use-large-language","title":"MuMath-Code: Combining Tool-Use Large Language Models with Multi-perspective Data Augmentation for Mathematical Reasoning","date":"2024-05-13","arxiv_id":"2405.07551","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mumath-code-combining-tool-use-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07551"}},"official":null}},{"url":"/paper/visiongraph-leveraging-large-multimodal","slug":"visiongraph-leveraging-large-multimodal","title":"VisionGraph: Leveraging Large Multimodal Models for Graph Theory Problems in Visual Context","date":"2024-05-08","arxiv_id":"2405.04950","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visiongraph-leveraging-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2405.04950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04950"}},"official":{"repos":["hitsz-tmg/visiongraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alphamath-almost-zero-process-supervision","slug":"alphamath-almost-zero-process-supervision","title":"AlphaMath Almost Zero: Process Supervision without Process","date":"2024-05-06","arxiv_id":"2405.03553","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-compositional-deficiency-of","slug":"exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","arxiv_id":"2405.06680","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-the-compositional-deficiency-of#ran","syntology_url":"https://syntology.ai/paper/2405.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06680"}},"official":{"repos":["tongjingqi/MathTrap"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gold-geometry-problem-solver-with-natural","slug":"gold-geometry-problem-solver-with-natural","title":"GOLD: Geometry Problem Solver with Natural Language Description","date":"2024-05-01","arxiv_id":"2405.00494","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-benchmark-leakage-in-large","slug":"benchmarking-benchmark-leakage-in-large","title":"Benchmarking Benchmark Leakage in Large Language Models","date":"2024-04-29","arxiv_id":"2404.18824","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-benchmark-leakage-in-large#ran","syntology_url":"https://syntology.ai/paper/2404.18824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18824"}},"official":{"repos":["gair-nlp/benbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/calc-cmu-at-semeval-2024-task-7-pre-calc","slug":"calc-cmu-at-semeval-2024-task-7-pre-calc","title":"Pre-Calc: Learning to Use the Calculator Improves Numeracy in Language Models","date":"2024-04-22","arxiv_id":"2404.14355","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calc-cmu-at-semeval-2024-task-7-pre-calc#ran","syntology_url":"https://syntology.ai/paper/2404.14355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14355"}},"official":{"repos":["calc-cmu/pre-calc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-self-improvement-of-llms-via","slug":"toward-self-improvement-of-llms-via","title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","date":"2024-04-18","arxiv_id":"2404.12253","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":1,"n_ran_checked":12,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":18,"phrase":"14 ran (of which 1 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/toward-self-improvement-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2404.12253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12253"}},"official":{"repos":["yetianjhu/alphallm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":1,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/paraphrase-and-solve-exploring-and-exploiting","slug":"paraphrase-and-solve-exploring-and-exploiting","title":"Paraphrase and Solve: Exploring and Exploiting the Impact of Surface Form on Mathematical Reasoning in Large Language Models","date":"2024-04-17","arxiv_id":"2404.11500","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/paraphrase-and-solve-exploring-and-exploiting#ran","syntology_url":"https://syntology.ai/paper/2404.11500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11500"}},"official":{"repos":["yue-llm-pit/scop"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-explore-to-avoid-the-pit-improving-the","slug":"self-explore-to-avoid-the-pit-improving-the","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","date":"2024-04-16","arxiv_id":"2404.10346","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-explore-to-avoid-the-pit-improving-the#ran","syntology_url":"https://syntology.ai/paper/2404.10346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10346"}},"official":{"repos":["hbin0701/Self-Explore"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/compression-represents-intelligence-linearly","slug":"compression-represents-intelligence-linearly","title":"Compression Represents Intelligence Linearly","date":"2024-04-15","arxiv_id":"2404.09937","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compression-represents-intelligence-linearly#ran","syntology_url":"https://syntology.ai/paper/2404.09937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09937"}},"official":{"repos":["hkust-nlp/llm-compression-intelligence"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-human-evaluation-of-large","slug":"sample-efficient-human-evaluation-of-large","title":"Sample-Efficient Human Evaluation of Large Language Models via Maximum Discrepancy Competition","date":"2024-04-10","arxiv_id":"2404.08008","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-mathematical-reasoning-beyond","slug":"evaluating-mathematical-reasoning-beyond","title":"Evaluating Mathematical Reasoning Beyond Accuracy","date":"2024-04-08","arxiv_id":"2404.05692","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-mathematical-reasoning-beyond#ran","syntology_url":"https://syntology.ai/paper/2404.05692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05692"}},"official":{"repos":["gair-nlp/reasoneval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-master-math-investigating-large","slug":"can-llms-master-math-investigating-large","title":"Can LLMs Master Math? Investigating Large Language Models on Math Stack Exchange","date":"2024-03-30","arxiv_id":"2404.00344","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-llms-master-math-investigating-large#ran","syntology_url":"https://syntology.ai/paper/2404.00344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00344"}},"official":{"repos":["gipplab/llm-investig-mathstackexchange"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"24c1f8ea6b361fccf236087ce2ebdde6dca8397c83831d4a57875945abea4a3d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}