{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/gsm8k/papers/2","list_of":"/task/gsm8k","task":"GSM8K","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":439,"counts":{"archive_papers_tagged":439,"with_a_code_link":209,"where_syntology_ran_a_sample":116,"not_listed_spam_title":0,"listed":439,"listed_where_code_ran":116,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":20,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":20,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/gsm8k","prev":"/task/gsm8k","next":"/task/gsm8k/papers/3","papers":[{"url":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critical-tokens-matter-token-level#ran","syntology_url":"https://syntology.ai/paper/2411.19943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19943"}},"official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-optimization-for-reasoning-with","slug":"preference-optimization-for-reasoning-with","title":"Preference Optimization for Reasoning with Pseudo Feedback","date":"2024-11-25","arxiv_id":"2411.16345","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preference-optimization-for-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16345"}},"official":null}},{"url":"/paper/what-do-learning-dynamics-reveal-about","slug":"what-do-learning-dynamics-reveal-about","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","date":"2024-11-12","arxiv_id":"2411.07681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-do-learning-dynamics-reveal-about#ran","syntology_url":"https://syntology.ai/paper/2411.07681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07681"}},"official":{"repos":["katiekang1998/reasoning_generalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utmath-math-evaluation-with-unit-test-via","slug":"utmath-math-evaluation-with-unit-test-via","title":"UTMath: Math Evaluation with Unit Test via Reasoning-to-Coding Thoughts","date":"2024-11-11","arxiv_id":"2411.07240","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-are-hidden-reasoners","slug":"language-models-are-hidden-reasoners","title":"Language Models are Hidden Reasoners: Unlocking Latent Reasoning Capabilities via Self-Rewarding","date":"2024-11-06","arxiv_id":"2411.04282","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/language-models-are-hidden-reasoners#ran","syntology_url":"https://syntology.ai/paper/2411.04282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04282"}},"official":{"repos":["salesforceairesearch/latro"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-done-rite-robust-invariant","slug":"lora-done-rite-robust-invariant","title":"LoRA Done RITE: Robust Invariant Transformation Equilibration for LoRA Optimization","date":"2024-10-27","arxiv_id":"2410.20625","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lora-done-rite-robust-invariant#ran","syntology_url":"https://syntology.ai/paper/2410.20625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20625"}},"official":{"repos":["gkevinyen5418/LoRA-RITE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-up-masked-diffusion-models-on-text","slug":"scaling-up-masked-diffusion-models-on-text","title":"Scaling up Masked Diffusion Models on Text","date":"2024-10-24","arxiv_id":"2410.18514","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-up-masked-diffusion-models-on-text#ran","syntology_url":"https://syntology.ai/paper/2410.18514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18514"}},"official":{"repos":["ml-gsai/smdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/math-neurosurgery-isolating-language-models","slug":"math-neurosurgery-isolating-language-models","title":"Math Neurosurgery: Isolating Language Models' Math Reasoning Abilities Using Only Forward Passes","date":"2024-10-22","arxiv_id":"2410.16930","repositories_listed":1,"syntology":null},{"url":"/paper/smart-self-learning-meta-strategy-agent-for","slug":"smart-self-learning-meta-strategy-agent-for","title":"SMART: Self-learning Meta-strategy Agent for Reasoning Tasks","date":"2024-10-21","arxiv_id":"2410.16128","repositories_listed":1,"syntology":null},{"url":"/paper/sbi-rag-enhancing-math-word-problem-solving","slug":"sbi-rag-enhancing-math-word-problem-solving","title":"SBI-RAG: Enhancing Math Word Problem Solving for Students through Schema-Based Instruction and Retrieval-Augmented Generation","date":"2024-10-17","arxiv_id":"2410.13293","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-votes-count-programs-as-verifiers","slug":"not-all-votes-count-programs-as-verifiers","title":"Not All Votes Count! Programs as Verifiers Improve Self-Consistency of Language Models for Math Reasoning","date":"2024-10-16","arxiv_id":"2410.12608","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/one-language-many-gaps-evaluating-dialect#ran","syntology_url":"https://syntology.ai/paper/2410.11005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11005"}},"official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/coral-order-agnostic-language-modeling-for","slug":"coral-order-agnostic-language-modeling-for","title":"COrAL: Order-Agnostic Language Modeling for Efficient Iterative Refinement","date":"2024-10-12","arxiv_id":"2410.09675","repositories_listed":1,"syntology":null},{"url":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coevolving-with-the-other-you-fine-tuning-llm#ran","syntology_url":"https://syntology.ai/paper/2410.06101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06101"}},"official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-topla-efficient-llm-ensemble-by","slug":"llm-topla-efficient-llm-ensemble-by","title":"LLM-TOPLA: Efficient LLM Ensemble by Maximising Diversity","date":"2024-10-04","arxiv_id":"2410.03953","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llm-topla-efficient-llm-ensemble-by#ran","syntology_url":"https://syntology.ai/paper/2410.03953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03953"}},"official":{"repos":["git-disl/llm-topla"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vineppo-unlocking-rl-potential-for-llm","slug":"vineppo-unlocking-rl-potential-for-llm","title":"VinePPO: Unlocking RL Potential For LLM Reasoning Through Refined Credit Assignment","date":"2024-10-02","arxiv_id":"2410.01679","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vineppo-unlocking-rl-potential-for-llm#ran","syntology_url":"https://syntology.ai/paper/2410.01679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01679"}},"official":{"repos":["mcgill-nlp/vineppo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scheherazade-evaluating-chain-of-thought-math","slug":"scheherazade-evaluating-chain-of-thought-math","title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","date":"2024-09-30","arxiv_id":"2410.00151","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scheherazade-evaluating-chain-of-thought-math#ran","syntology_url":"https://syntology.ai/paper/2410.00151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00151"}},"official":{"repos":["yoshikitakashima/scheherazade-code-data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-symbolic-collaborative-distillation","slug":"neural-symbolic-collaborative-distillation","title":"Neural-Symbolic Collaborative Distillation: Advancing Small Language Models for Complex Reasoning Tasks","date":"2024-09-20","arxiv_id":"2409.13203","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-symbolic-collaborative-distillation#ran","syntology_url":"https://syntology.ai/paper/2409.13203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13203"}},"official":{"repos":["xnhyacinth/nesycd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/logicpro-improving-complex-logical-reasoning","slug":"logicpro-improving-complex-logical-reasoning","title":"LogicPro: Improving Complex Logical Reasoning via Program-Guided Learning","date":"2024-09-19","arxiv_id":"2409.12929","repositories_listed":1,"syntology":null},{"url":"/paper/improving-llm-reasoning-with-multi-agent-tree","slug":"improving-llm-reasoning-with-multi-agent-tree","title":"Improving LLM Reasoning with Multi-Agent Tree-of-Thought Validator Agent","date":"2024-09-17","arxiv_id":"2409.11527","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-llm-reasoning-with-multi-agent-tree#ran","syntology_url":"https://syntology.ai/paper/2409.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11527"}},"official":{"repos":["secureaiautonomylab/ma-tot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sorsa-singular-values-and-orthonormal","slug":"sorsa-singular-values-and-orthonormal","title":"SORSA: Singular Values and Orthonormal Regularized Singular Vectors Adaptation of Large Language Models","date":"2024-08-21","arxiv_id":"2409.00055","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sorsa-singular-values-and-orthonormal#ran","syntology_url":"https://syntology.ai/paper/2409.00055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00055"}},"official":{"repos":["Gunale0926/SORSA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-language-model-math-reasoning-via","slug":"evaluating-language-model-math-reasoning-via","title":"Mathfish: Evaluating Language Model Math Reasoning via Grounding in Educational Curricula","date":"2024-08-08","arxiv_id":"2408.04226","repositories_listed":1,"syntology":null},{"url":"/paper/physics-of-language-models-part-2-1-grade","slug":"physics-of-language-models-part-2-1-grade","title":"Physics of Language Models: Part 2.1, Grade-School Math and the Hidden Reasoning Process","date":"2024-07-29","arxiv_id":"2407.20311","repositories_listed":1,"syntology":null},{"url":"/paper/learning-goal-conditioned-representations-for","slug":"learning-goal-conditioned-representations-for","title":"Learning Goal-Conditioned Representations for Language Reward Models","date":"2024-07-18","arxiv_id":"2407.13887","repositories_listed":1,"syntology":null},{"url":"/paper/weak-to-strong-reasoning","slug":"weak-to-strong-reasoning","title":"Weak-to-Strong Reasoning","date":"2024-07-18","arxiv_id":"2407.13647","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/weak-to-strong-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13647"}},"official":{"repos":["gair-nlp/weak-to-strong-reasoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-ga-low-rank-adaptation-with-gradient","slug":"lora-ga-low-rank-adaptation-with-gradient","title":"LoRA-GA: Low-Rank Adaptation with Gradient Approximation","date":"2024-07-06","arxiv_id":"2407.05000","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lora-ga-low-rank-adaptation-with-gradient#ran","syntology_url":"https://syntology.ai/paper/2407.05000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05000"}},"official":{"repos":["outsider565/lora-ga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dotamath-decomposition-of-thought-with-code","slug":"dotamath-decomposition-of-thought-with-code","title":"DotaMath: Decomposition of Thought with Code Assistance and Self-correction for Mathematical Reasoning","date":"2024-07-04","arxiv_id":"2407.04078","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-metabench-a-sparse-benchmark-to","slug":"texttt-metabench-a-sparse-benchmark-to","title":"$\\texttt{metabench}$ -- A Sparse Benchmark to Measure General Ability in Large Language Models","date":"2024-07-04","arxiv_id":"2407.12844","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/texttt-metabench-a-sparse-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2407.12844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12844"}},"official":{"repos":["adkipnis/metabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-controlled-dpo-leveraging-stepwise-error#ran","syntology_url":"https://syntology.ai/paper/2407.00782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00782"}},"official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/step-dpo-step-wise-preference-optimization#ran","syntology_url":"https://syntology.ai/paper/2406.18629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18629"}},"official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/varbench-robust-language-model-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.17681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17681"}},"official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-decontamination-reusing-leaked","slug":"inference-time-decontamination-reusing-leaked","title":"Inference-Time Decontamination: Reusing Leaked Benchmarks for Large Language Model Evaluation","date":"2024-06-20","arxiv_id":"2406.13990","repositories_listed":1,"syntology":null},{"url":"/paper/the-reason-behind-good-or-bad-towards-a","slug":"the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","arxiv_id":"2406.14024","repositories_listed":1,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":18,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":17,"n_pointer_only":23,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-reason-behind-good-or-bad-towards-a#ran","syntology_url":"https://syntology.ai/paper/2406.14024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14024"}},"official":{"repos":["kbsdjames/math-minos"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-reason-in-the-wild-with-programs","slug":"can-llms-reason-in-the-wild-with-programs","title":"Can LLMs Reason in the Wild with Programs?","date":"2024-06-19","arxiv_id":"2406.13764","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prompting-taxonomy-a-universal","slug":"hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","arxiv_id":"2406.12644","repositories_listed":1,"syntology":null},{"url":"/paper/della-merging-reducing-interference-in-model","slug":"della-merging-reducing-interference-in-model","title":"DELLA-Merging: Reducing Interference in Model Merging through Magnitude-Based Sampling","date":"2024-06-17","arxiv_id":"2406.11617","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/della-merging-reducing-interference-in-model#ran","syntology_url":"https://syntology.ai/paper/2406.11617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11617"}},"official":{"repos":["declare-lab/della"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sharelora-parameter-efficient-and-robust#ran","syntology_url":"https://syntology.ai/paper/2406.10785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10785"}},"official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-instruction-evolving-for-large","slug":"automatic-instruction-evolving-for-large","title":"Automatic Instruction Evolving for Large Language Models","date":"2024-06-02","arxiv_id":"2406.00770","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-instruction-evolving-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.00770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00770"}},"official":null}},{"url":"/paper/gkt-a-novel-guidance-based-knowledge-transfer","slug":"gkt-a-novel-guidance-based-knowledge-transfer","title":"GKT: A Novel Guidance-Based Knowledge Transfer Framework For Efficient Cloud-edge Collaboration LLM Deployment","date":"2024-05-30","arxiv_id":"2405.19635","repositories_listed":1,"syntology":null},{"url":"/paper/lora-xs-low-rank-adaptation-with-extremely","slug":"lora-xs-low-rank-adaptation-with-extremely","title":"LoRA-XS: Low-Rank Adaptation with Extremely Small Number of Parameters","date":"2024-05-27","arxiv_id":"2405.17604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-xs-low-rank-adaptation-with-extremely#ran","syntology_url":"https://syntology.ai/paper/2405.17604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17604"}},"official":{"repos":["mohammadrezabanaei/lora-xs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-explicit-cot-to-implicit-cot-learning-to","slug":"from-explicit-cot-to-implicit-cot-learning-to","title":"From Explicit CoT to Implicit CoT: Learning to Internalize CoT Step by Step","date":"2024-05-23","arxiv_id":"2405.14838","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-explicit-cot-to-implicit-cot-learning-to#ran","syntology_url":"https://syntology.ai/paper/2405.14838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14838"}},"official":{"repos":["da03/internalize_cot_step_by_step"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-moe-and-dense-speed-accuracy","slug":"revisiting-moe-and-dense-speed-accuracy","title":"Revisiting MoE and Dense Speed-Accuracy Comparisons for LLM Training","date":"2024-05-23","arxiv_id":"2405.15052","repositories_listed":1,"syntology":null},{"url":"/paper/unchosen-experts-can-contribute-too","slug":"unchosen-experts-can-contribute-too","title":"Unchosen Experts Can Contribute Too: Unleashing MoE Models' Power by Self-Contrast","date":"2024-05-23","arxiv_id":"2405.14507","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unchosen-experts-can-contribute-too#ran","syntology_url":"https://syntology.ai/paper/2405.14507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14507"}},"official":{"repos":["davidfanzz/scmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zipcache-accurate-and-efficient-kv-cache","slug":"zipcache-accurate-and-efficient-kv-cache","title":"ZipCache: Accurate and Efficient KV Cache Quantization with Salient Token Identification","date":"2024-05-23","arxiv_id":"2405.14256","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zipcache-accurate-and-efficient-kv-cache#ran","syntology_url":"https://syntology.ai/paper/2405.14256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14256"}},"official":null}},{"url":"/paper/mathbench-evaluating-the-theory-and","slug":"mathbench-evaluating-the-theory-and","title":"MathBench: Evaluating the Theory and Application Proficiency of LLMs with a Hierarchical Mathematics Benchmark","date":"2024-05-20","arxiv_id":"2405.12209","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-choice-questions-are-efficient-and","slug":"multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","arxiv_id":"2405.11966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-choice-questions-are-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2405.11966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11966"}},"official":{"repos":["geralt-targaryen/mc-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mumath-code-combining-tool-use-large-language","slug":"mumath-code-combining-tool-use-large-language","title":"MuMath-Code: Combining Tool-Use Large Language Models with Multi-perspective Data Augmentation for Mathematical Reasoning","date":"2024-05-13","arxiv_id":"2405.07551","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mumath-code-combining-tool-use-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07551"}},"official":null}},{"url":"/paper/exploring-the-compositional-deficiency-of","slug":"exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","arxiv_id":"2405.06680","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-the-compositional-deficiency-of#ran","syntology_url":"https://syntology.ai/paper/2405.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06680"}},"official":{"repos":["tongjingqi/MathTrap"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/markovian-agents-for-truthful-language","slug":"markovian-agents-for-truthful-language","title":"Markovian Transformers for Informative Language Modeling","date":"2024-04-29","arxiv_id":"2404.18988","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":14,"n_pointer_only":16,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/markovian-agents-for-truthful-language#ran","syntology_url":"https://syntology.ai/paper/2404.18988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18988"}},"official":{"repos":["scottviteri/markoviantraining"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/layer-skip-enabling-early-exit-inference-and#ran","syntology_url":"https://syntology.ai/paper/2404.16710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16710"}},"official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/achieving-97-on-gsm8k-deeply-understanding","slug":"achieving-97-on-gsm8k-deeply-understanding","title":"Achieving >97% on GSM8K: Deeply Understanding the Problems Makes LLMs Better Solvers for Math Word Problems","date":"2024-04-23","arxiv_id":"2404.14963","repositories_listed":1,"syntology":null},{"url":"/paper/toward-self-improvement-of-llms-via","slug":"toward-self-improvement-of-llms-via","title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","date":"2024-04-18","arxiv_id":"2404.12253","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":1,"n_ran_checked":12,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":18,"phrase":"14 ran (of which 1 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/toward-self-improvement-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2404.12253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12253"}},"official":{"repos":["yetianjhu/alphallm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":1,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/self-explore-to-avoid-the-pit-improving-the","slug":"self-explore-to-avoid-the-pit-improving-the","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","date":"2024-04-16","arxiv_id":"2404.10346","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-explore-to-avoid-the-pit-improving-the#ran","syntology_url":"https://syntology.ai/paper/2404.10346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10346"}},"official":{"repos":["hbin0701/Self-Explore"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pissa-principal-singular-values-and-singular","slug":"pissa-principal-singular-values-and-singular","title":"PiSSA: Principal Singular Values and Singular Vectors Adaptation of Large Language Models","date":"2024-04-03","arxiv_id":"2404.02948","repositories_listed":1,"syntology":null},{"url":"/paper/don-t-trust-verify-grounding-llm-quantitative","slug":"don-t-trust-verify-grounding-llm-quantitative","title":"Don't Trust: Verify -- Grounding LLM Quantitative Reasoning with Autoformalization","date":"2024-03-26","arxiv_id":"2403.18120","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/don-t-trust-verify-grounding-llm-quantitative#ran","syntology_url":"https://syntology.ai/paper/2403.18120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18120"}},"official":{"repos":["jinpz/dtv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lisa-layerwise-importance-sampling-for-memory","slug":"lisa-layerwise-importance-sampling-for-memory","title":"LISA: Layerwise Importance Sampling for Memory-Efficient Large Language Model Fine-Tuning","date":"2024-03-26","arxiv_id":"2403.17919","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/lisa-layerwise-importance-sampling-for-memory#ran","syntology_url":"https://syntology.ai/paper/2403.17919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17919"}},"official":{"repos":["optimalscale/lmflow"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/llm2llm-boosting-llms-with-novel-iterative","slug":"llm2llm-boosting-llms-with-novel-iterative","title":"LLM2LLM: Boosting LLMs with Novel Iterative Data Enhancement","date":"2024-03-22","arxiv_id":"2403.15042","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm2llm-boosting-llms-with-novel-iterative#ran","syntology_url":"https://syntology.ai/paper/2403.15042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15042"}},"official":{"repos":["squeezeailab/llm2llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llmlingua-2-data-distillation-for-efficient","slug":"llmlingua-2-data-distillation-for-efficient","title":"LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression","date":"2024-03-19","arxiv_id":"2403.12968","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-contrastive","slug":"large-language-models-are-contrastive","title":"Large Language Models are Contrastive Reasoners","date":"2024-03-13","arxiv_id":"2403.08211","repositories_listed":1,"syntology":null},{"url":"/paper/mathscale-scaling-instruction-tuning-for","slug":"mathscale-scaling-instruction-tuning-for","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","date":"2024-03-05","arxiv_id":"2403.02884","repositories_listed":1,"syntology":null},{"url":"/paper/masked-thought-simply-masking-partial","slug":"masked-thought-simply-masking-partial","title":"Masked Thought: Simply Masking Partial Reasoning Steps Can Improve Mathematical Reasoning Learning of Language Models","date":"2024-03-04","arxiv_id":"2403.02178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/masked-thought-simply-masking-partial#ran","syntology_url":"https://syntology.ai/paper/2403.02178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02178"}},"official":{"repos":["changyuchen347/maskedthought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gsm-plus-a-comprehensive-benchmark-for","slug":"gsm-plus-a-comprehensive-benchmark-for","title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","date":"2024-02-29","arxiv_id":"2402.19255","repositories_listed":1,"syntology":null},{"url":"/paper/keeping-llms-aligned-after-fine-tuning-the","slug":"keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","arxiv_id":"2402.18540","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/keeping-llms-aligned-after-fine-tuning-the#ran","syntology_url":"https://syntology.ai/paper/2402.18540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18540"}},"official":{"repos":["vfleaking/ptst"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/distillation-contrastive-decoding-improving","slug":"distillation-contrastive-decoding-improving","title":"Distillation Contrastive Decoding: Improving LLMs Reasoning with Contrastive Decoding and Distillation","date":"2024-02-21","arxiv_id":"2402.14874","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/distillation-contrastive-decoding-improving#ran","syntology_url":"https://syntology.ai/paper/2402.14874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14874"}},"official":{"repos":["pphuc25/distil-cd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reformatted-alignment","slug":"reformatted-alignment","title":"Reformatted Alignment","date":"2024-02-19","arxiv_id":"2402.12219","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reformatted-alignment#ran","syntology_url":"https://syntology.ai/paper/2402.12219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12219"}},"official":{"repos":["gair-nlp/realign"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-as-science-tutors","slug":"language-models-as-science-tutors","title":"Language Models as Science Tutors","date":"2024-02-16","arxiv_id":"2402.11111","repositories_listed":1,"syntology":null},{"url":"/paper/openmathinstruct-1-a-1-8-million-math","slug":"openmathinstruct-1-a-1-8-million-math","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","date":"2024-02-15","arxiv_id":"2402.10176","repositories_listed":1,"syntology":null},{"url":"/paper/internlm-math-open-math-large-language-models","slug":"internlm-math-open-math-large-language-models","title":"InternLM-Math: Open Math Large Language Models Toward Verifiable Reasoning","date":"2024-02-09","arxiv_id":"2402.06332","repositories_listed":1,"syntology":null},{"url":"/paper/in-context-principle-learning-from-mistakes","slug":"in-context-principle-learning-from-mistakes","title":"In-Context Principle Learning from Mistakes","date":"2024-02-08","arxiv_id":"2402.05403","repositories_listed":1,"syntology":null},{"url":"/paper/training-large-language-models-for-reasoning","slug":"training-large-language-models-for-reasoning","title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.05808","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/training-large-language-models-for-reasoning#ran","syntology_url":"https://syntology.ai/paper/2402.05808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05808"}},"official":{"repos":["woooodyy/llm-reverse-curriculum-rl"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/superclue-math6-graded-multi-step-math","slug":"superclue-math6-graded-multi-step-math","title":"SuperCLUE-Math6: Graded Multi-Step Math Reasoning Benchmark for LLMs in Chinese","date":"2024-01-22","arxiv_id":"2401.11819","repositories_listed":1,"syntology":null},{"url":"/paper/over-reasoning-and-redundant-calculation-of","slug":"over-reasoning-and-redundant-calculation-of","title":"Over-Reasoning and Redundant Calculation of Large Language Models","date":"2024-01-21","arxiv_id":"2401.11467","repositories_listed":1,"syntology":null},{"url":"/paper/escape-sky-high-cost-early-stopping-self","slug":"escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","arxiv_id":"2401.10480","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/escape-sky-high-cost-early-stopping-self#ran","syntology_url":"https://syntology.ai/paper/2401.10480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10480"}},"official":{"repos":["yiwei98/esc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reft-reasoning-with-reinforced-fine-tuning","slug":"reft-reasoning-with-reinforced-fine-tuning","title":"ReFT: Reasoning with Reinforced Fine-Tuning","date":"2024-01-17","arxiv_id":"2401.08967","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/reft-reasoning-with-reinforced-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2401.08967","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08967"}},"official":{"repos":["lqtrung1998/mwp_reft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speak-like-a-native-prompting-large-language","slug":"speak-like-a-native-prompting-large-language","title":"AlignedCoT: Prompting Large Language Models via Native-Speaking Demonstrations","date":"2023-11-22","arxiv_id":"2311.13538","repositories_listed":1,"syntology":null},{"url":"/paper/meta-prompting-for-agi-systems","slug":"meta-prompting-for-agi-systems","title":"Meta Prompting for AI Systems","date":"2023-11-20","arxiv_id":"2311.11482","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-prompting-for-agi-systems#ran","syntology_url":"https://syntology.ai/paper/2311.11482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.11482"}},"official":{"repos":["meta-prompting/meta-prompting"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/token-level-adaptation-of-lora-adapters-for","slug":"token-level-adaptation-of-lora-adapters-for","title":"Token-Level Adaptation of LoRA Adapters for Downstream Task Generalization","date":"2023-11-17","arxiv_id":"2311.10847","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-symbolic-integration-brings-causal-and","slug":"neuro-symbolic-integration-brings-causal-and","title":"Neuro-Symbolic Integration Brings Causal and Reliable Reasoning Proofs","date":"2023-11-16","arxiv_id":"2311.09802","repositories_listed":1,"syntology":null},{"url":"/paper/outcome-supervised-verifiers-for-planning-in","slug":"outcome-supervised-verifiers-for-planning-in","title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","date":"2023-11-16","arxiv_id":"2311.09724","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/outcome-supervised-verifiers-for-planning-in#ran","syntology_url":"https://syntology.ai/paper/2311.09724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09724"}},"official":{"repos":["freedomintelligence/ovm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-mistakes-makes-llm-better","slug":"learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","arxiv_id":"2310.20689","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-mistakes-makes-llm-better#ran","syntology_url":"https://syntology.ai/paper/2310.20689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20689"}},"official":{"repos":["microsoft/lema"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skymath-technical-report","slug":"skymath-technical-report","title":"SkyMath: Technical Report","date":"2023-10-25","arxiv_id":"2310.16713","repositories_listed":1,"syntology":null},{"url":"/paper/trace-a-comprehensive-benchmark-for-continual","slug":"trace-a-comprehensive-benchmark-for-continual","title":"TRACE: A Comprehensive Benchmark for Continual Learning in Large Language Models","date":"2023-10-10","arxiv_id":"2310.06762","repositories_listed":1,"syntology":null},{"url":"/paper/llmlingua-compressing-prompts-for-accelerated","slug":"llmlingua-compressing-prompts-for-accelerated","title":"LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models","date":"2023-10-09","arxiv_id":"2310.05736","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llmlingua-compressing-prompts-for-accelerated#ran","syntology_url":"https://syntology.ai/paper/2310.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05736"}},"official":{"repos":["microsoft/LLMLingua"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/query-and-response-augmentation-cannot-help","slug":"query-and-response-augmentation-cannot-help","title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","date":"2023-10-09","arxiv_id":"2310.05506","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder-seamless-code-integration-in-llms","slug":"mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","arxiv_id":"2310.03731","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-seamless-code-integration-in-llms#ran","syntology_url":"https://syntology.ai/paper/2310.03731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03731"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fill-in-the-blank-exploring-and-enhancing-llm","slug":"fill-in-the-blank-exploring-and-enhancing-llm","title":"Fill in the Blank: Exploring and Enhancing LLM Capabilities for Backward Reasoning in Math Word Problems","date":"2023-10-03","arxiv_id":"2310.01991","repositories_listed":1,"syntology":null},{"url":"/paper/metamath-bootstrap-your-own-mathematical","slug":"metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","arxiv_id":"2309.12284","repositories_listed":1,"syntology":{"n":22,"n_ran":15,"n_constructed":0,"n_ran_checked":1,"n_instrument":14,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/metamath-bootstrap-your-own-mathematical#ran","syntology_url":"https://syntology.ai/paper/2309.12284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12284"}},"official":{"repos":["meta-math/MetaMath"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/design-of-chain-of-thought-in-math-problem","slug":"design-of-chain-of-thought-in-math-problem","title":"Design of Chain-of-Thought in Math Problem Solving","date":"2023-09-20","arxiv_id":"2309.11054","repositories_listed":1,"syntology":null},{"url":"/paper/echoprompt-instructing-the-model-to-rephrase","slug":"echoprompt-instructing-the-model-to-rephrase","title":"EchoPrompt: Instructing the Model to Rephrase Queries for Improved In-context Learning","date":"2023-09-16","arxiv_id":"2309.10687","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-equation-as-a-better-intermediate","slug":"exploring-equation-as-a-better-intermediate","title":"Exploring Equation as a Better Intermediate Meaning Representation for Numerical Reasoning","date":"2023-08-21","arxiv_id":"2308.10585","repositories_listed":1,"syntology":null},{"url":"/paper/wizardmath-empowering-mathematical-reasoning","slug":"wizardmath-empowering-mathematical-reasoning","title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","date":"2023-08-18","arxiv_id":"2308.09583","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":16,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/wizardmath-empowering-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2308.09583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09583"}},"official":null}},{"url":"/paper/scaling-relationship-on-learning-mathematical","slug":"scaling-relationship-on-learning-mathematical","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","date":"2023-08-03","arxiv_id":"2308.01825","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-relationship-on-learning-mathematical#ran","syntology_url":"https://syntology.ai/paper/2308.01825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01825"}},"official":{"repos":["ofa-sys/gsm8k-screl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selfcheck-using-llms-to-zero-shot-check-their","slug":"selfcheck-using-llms-to-zero-shot-check-their","title":"SelfCheck: Using LLMs to Zero-Shot Check Their Own Step-by-Step Reasoning","date":"2023-08-01","arxiv_id":"2308.00436","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/selfcheck-using-llms-to-zero-shot-check-their#ran","syntology_url":"https://syntology.ai/paper/2308.00436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00436"}},"official":{"repos":["ningmiao/selfcheck"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-mixed-policy-to-improve-performance-of","slug":"a-mixed-policy-to-improve-performance-of","title":"A mixed policy to improve performance of language models on math problems","date":"2023-07-17","arxiv_id":"2307.08767","repositories_listed":1,"syntology":null},{"url":"/paper/2305-15017","slug":"2305-15017","title":"Calc-X and Calcformers: Empowering Arithmetical Chain-of-Thought through Interaction with Symbolic Systems","date":"2023-05-24","arxiv_id":"2305.15017","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2305-15017#ran","syntology_url":"https://syntology.ai/paper/2305.15017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15017"}},"official":{"repos":["prompteus/calc-x"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminator-guided-multi-step-reasoning","slug":"discriminator-guided-multi-step-reasoning","title":"GRACE: Discriminator-Guided Chain-of-Thought Reasoning","date":"2023-05-24","arxiv_id":"2305.14934","repositories_listed":1,"syntology":null}],"record_sha256":"58b35c95f7d2ad2378df5c9c593f5b186532ab75ca0f3a7c970f6eca6a50fa83","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}