{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/gsm8k/papers/ran/1","list_of":"/task/gsm8k","task":"GSM8K","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":116,"counts":{"archive_papers_tagged":439,"with_a_code_link":209,"where_syntology_ran_a_sample":116,"not_listed_spam_title":0,"listed":439,"listed_where_code_ran":116,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":20,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":20,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/gsm8k/papers/ran/1","prev":null,"next":"/task/gsm8k/papers/ran/2","papers":[{"url":"/paper/any4-learned-4-bit-numeric-representation-for","slug":"any4-learned-4-bit-numeric-representation-for","title":"any4: Learned 4-bit Numeric Representation for LLMs","date":"2025-07-07","arxiv_id":"2507.04610","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/any4-learned-4-bit-numeric-representation-for#ran","syntology_url":"https://syntology.ai/paper/2507.04610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.04610"}},"official":{"repos":["facebookresearch/any4"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/commvq-commutative-vector-quantization-for-kv","slug":"commvq-commutative-vector-quantization-for-kv","title":"CommVQ: Commutative Vector Quantization for KV Cache Compression","date":"2025-06-23","arxiv_id":"2506.18879","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/commvq-commutative-vector-quantization-for-kv#ran","syntology_url":"https://syntology.ai/paper/2506.18879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18879"}},"official":{"repos":["umass-embodied-agi/commvq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plan-for-speed-dilated-scheduling-for-masked","slug":"plan-for-speed-dilated-scheduling-for-masked","title":"Plan for Speed -- Dilated Scheduling for Masked Diffusion Language Models","date":"2025-06-23","arxiv_id":"2506.19037","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plan-for-speed-dilated-scheduling-for-masked#ran","syntology_url":"https://syntology.ai/paper/2506.19037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.19037"}},"official":null}},{"url":"/paper/discriminative-policy-optimization-for-token","slug":"discriminative-policy-optimization-for-token","title":"Discriminative Policy Optimization for Token-Level Reward Models","date":"2025-05-29","arxiv_id":"2505.23363","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discriminative-policy-optimization-for-token#ran","syntology_url":"https://syntology.ai/paper/2505.23363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23363"}},"official":{"repos":["homzer/q-rm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/segment-policy-optimization-effective-segment","slug":"segment-policy-optimization-effective-segment","title":"Segment Policy Optimization: Effective Segment-Level Credit Assignment in RL for Large Language Models","date":"2025-05-29","arxiv_id":"2505.23564","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/segment-policy-optimization-effective-segment#ran","syntology_url":"https://syntology.ai/paper/2505.23564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23564"}},"official":{"repos":["aiframeresearch/spo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adactrl-towards-adaptive-and-controllable","slug":"adactrl-towards-adaptive-and-controllable","title":"AdaCtrl: Towards Adaptive and Controllable Reasoning via Difficulty-Aware Budgeting","date":"2025-05-24","arxiv_id":"2505.18822","repositories_listed":1,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/adactrl-towards-adaptive-and-controllable#ran","syntology_url":"https://syntology.ai/paper/2505.18822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18822"}},"official":{"repos":["joeying1019/adactrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/let-llms-break-free-from-overthinking-via","slug":"let-llms-break-free-from-overthinking-via","title":"Let LLMs Break Free from Overthinking via Self-Braking Tuning","date":"2025-05-20","arxiv_id":"2505.14604","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-llms-break-free-from-overthinking-via#ran","syntology_url":"https://syntology.ai/paper/2505.14604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14604"}},"official":{"repos":["ccai-lab/self-braking-tuning","zju-real/self-braking-tuning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seek-in-the-dark-reasoning-via-test-time","slug":"seek-in-the-dark-reasoning-via-test-time","title":"Seek in the Dark: Reasoning via Test-Time Instance-Level Policy Gradient in Latent Space","date":"2025-05-19","arxiv_id":"2505.13308","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/seek-in-the-dark-reasoning-via-test-time#ran","syntology_url":"https://syntology.ai/paper/2505.13308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13308"}},"official":{"repos":["bigai-nlco/latentseek"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/thinkless-llm-learns-when-to-think","slug":"thinkless-llm-learns-when-to-think","title":"Thinkless: LLM Learns When to Think","date":"2025-05-19","arxiv_id":"2505.13379","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/thinkless-llm-learns-when-to-think#ran","syntology_url":"https://syntology.ai/paper/2505.13379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13379"}},"official":{"repos":["vainf/thinkless"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/data-whisperer-efficient-data-selection-for","slug":"data-whisperer-efficient-data-selection-for","title":"Data Whisperer: Efficient Data Selection for Task-Specific LLM Fine-Tuning via Few-Shot In-Context Learning","date":"2025-05-18","arxiv_id":"2505.12212","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/data-whisperer-efficient-data-selection-for#ran","syntology_url":"https://syntology.ai/paper/2505.12212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12212"}},"official":{"repos":["gszfwsb/Data-Whisperer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-data-rl-task-definition-is-all-you","slug":"synthetic-data-rl-task-definition-is-all-you","title":"Synthetic Data RL: Task Definition Is All You Need","date":"2025-05-18","arxiv_id":"2505.17063","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/synthetic-data-rl-task-definition-is-all-you#ran","syntology_url":"https://syntology.ai/paper/2505.17063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17063"}},"official":{"repos":["gydpku/data_synthesis_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reasoning-for-llms-through","slug":"efficient-reasoning-for-llms-through","title":"Efficient Reasoning for LLMs through Speculative Chain-of-Thought","date":"2025-04-27","arxiv_id":"2504.19095","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-reasoning-for-llms-through#ran","syntology_url":"https://syntology.ai/paper/2504.19095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19095"}},"official":{"repos":["jikai0wang/speculative_cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-early-exit-in-reasoning-models","slug":"dynamic-early-exit-in-reasoning-models","title":"Dynamic Early Exit in Reasoning Models","date":"2025-04-22","arxiv_id":"2504.15895","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-early-exit-in-reasoning-models#ran","syntology_url":"https://syntology.ai/paper/2504.15895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15895"}},"official":{"repos":["iie-ycx/deer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-vision-language-models-are-unsupervised","slug":"large-vision-language-models-are-unsupervised","title":"Large (Vision) Language Models are Unsupervised In-Context Learners","date":"2025-04-03","arxiv_id":"2504.02349","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/large-vision-language-models-are-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2504.02349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02349"}},"official":{"repos":["mlbio-epfl/joint-inference"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-based-adaptive-weighting-for-self","slug":"entropy-based-adaptive-weighting-for-self","title":"Entropy-Based Adaptive Weighting for Self-Training","date":"2025-03-31","arxiv_id":"2503.23913","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-based-adaptive-weighting-for-self#ran","syntology_url":"https://syntology.ai/paper/2503.23913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23913"}},"official":{"repos":["mandyyyyii/east"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cppo-accelerating-the-training-of-group","slug":"cppo-accelerating-the-training-of-group","title":"CPPO: Accelerating the Training of Group Relative Policy Optimization-Based Reasoning Models","date":"2025-03-28","arxiv_id":"2503.22342","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cppo-accelerating-the-training-of-group#ran","syntology_url":"https://syntology.ai/paper/2503.22342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22342"}},"official":{"repos":["lzhxmu/cppo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/safemerge-preserving-safety-alignment-in-fine","slug":"safemerge-preserving-safety-alignment-in-fine","title":"SafeMERGE: Preserving Safety Alignment in Fine-Tuned Large Language Models via Selective Layer-Wise Model Merging","date":"2025-03-21","arxiv_id":"2503.17239","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safemerge-preserving-safety-alignment-in-fine#ran","syntology_url":"https://syntology.ai/paper/2503.17239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.17239"}},"official":{"repos":["aladinD/SafeMERGE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-hierarchical-multi-step-reward-models","slug":"towards-hierarchical-multi-step-reward-models","title":"Towards Hierarchical Multi-Step Reward Models for Enhanced Reasoning in Large Language Models","date":"2025-03-16","arxiv_id":"2503.13551","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-hierarchical-multi-step-reward-models#ran","syntology_url":"https://syntology.ai/paper/2503.13551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13551"}},"official":{"repos":["tengwang0318/hierarchial_reward_model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/promptcot-synthesizing-olympiad-level","slug":"promptcot-synthesizing-olympiad-level","title":"PromptCoT: Synthesizing Olympiad-level Problems for Mathematical Reasoning in Large Language Models","date":"2025-03-04","arxiv_id":"2503.02324","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/promptcot-synthesizing-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2503.02324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02324"}},"official":{"repos":["zhaoxlpku/promptcot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-training-elicits-concise-reasoning-in","slug":"self-training-elicits-concise-reasoning-in","title":"Self-Training Elicits Concise Reasoning in Large Language Models","date":"2025-02-27","arxiv_id":"2502.20122","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-training-elicits-concise-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2502.20122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20122"}},"official":{"repos":["tergelmunkhbat/concise-reasoning"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finereason-evaluating-and-improving-llms","slug":"finereason-evaluating-and-improving-llms","title":"FINEREASON: Evaluating and Improving LLMs' Deliberate Reasoning through Reflective Puzzle Solving","date":"2025-02-27","arxiv_id":"2502.20238","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/finereason-evaluating-and-improving-llms#ran","syntology_url":"https://syntology.ai/paper/2502.20238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20238"}},"official":{"repos":["DAMO-NLP-SG/FineReason"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/big-math-a-large-scale-high-quality-math","slug":"big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","arxiv_id":"2502.17387","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/big-math-a-large-scale-high-quality-math#ran","syntology_url":"https://syntology.ai/paper/2502.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17387"}},"official":{"repos":["synthlabsai/big-math"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earlier-tokens-contribute-more-learning#ran","syntology_url":"https://syntology.ai/paper/2502.14340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14340"}},"official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/treecut-a-synthetic-unanswerable-math-word","slug":"treecut-a-synthetic-unanswerable-math-word","title":"TreeCut: A Synthetic Unanswerable Math Word Problem Dataset for LLM Hallucination Evaluation","date":"2025-02-19","arxiv_id":"2502.13442","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treecut-a-synthetic-unanswerable-math-word#ran","syntology_url":"https://syntology.ai/paper/2502.13442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13442"}},"official":{"repos":["j-bagel/treecut-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sift-grounding-llm-reasoning-in-contexts-via","slug":"sift-grounding-llm-reasoning-in-contexts-via","title":"SIFT: Grounding LLM Reasoning in Contexts via Stickers","date":"2025-02-19","arxiv_id":"2502.14922","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sift-grounding-llm-reasoning-in-contexts-via#ran","syntology_url":"https://syntology.ai/paper/2502.14922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14922"}},"official":{"repos":["zhijie-group/sift"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/smart-self-aware-agent-for-tool-overuse","slug":"smart-self-aware-agent-for-tool-overuse","title":"SMART: Self-Aware Agent for Tool Overuse Mitigation","date":"2025-02-17","arxiv_id":"2502.11435","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smart-self-aware-agent-for-tool-overuse#ran","syntology_url":"https://syntology.ai/paper/2502.11435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11435"}},"official":{"repos":["qiancheng0/open-smartagent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokenskip-controllable-chain-of-thought","slug":"tokenskip-controllable-chain-of-thought","title":"TokenSkip: Controllable Chain-of-Thought Compression in LLMs","date":"2025-02-17","arxiv_id":"2502.12067","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tokenskip-controllable-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2502.12067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12067"}},"official":{"repos":["hemingkx/tokenskip"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-get-lost-in-the-trees-streamlining-llm","slug":"don-t-get-lost-in-the-trees-streamlining-llm","title":"Don't Get Lost in the Trees: Streamlining LLM Reasoning by Overcoming Tree Search Exploration Pitfalls","date":"2025-02-16","arxiv_id":"2502.11183","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/don-t-get-lost-in-the-trees-streamlining-llm#ran","syntology_url":"https://syntology.ai/paper/2502.11183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11183"}},"official":{"repos":["soistesimmer/fetch"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-process-reward-model","slug":"entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","arxiv_id":"2412.11006","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-process-reward-model#ran","syntology_url":"https://syntology.ai/paper/2412.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11006"}},"official":{"repos":["hanningzhang/er-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/greater-gradients-over-reasoning-makes","slug":"greater-gradients-over-reasoning-makes","title":"GReaTer: Gradients over Reasoning Makes Smaller Language Models Strong Prompt Optimizers","date":"2024-12-12","arxiv_id":"2412.09722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/greater-gradients-over-reasoning-makes#ran","syntology_url":"https://syntology.ai/paper/2412.09722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09722"}},"official":{"repos":["psunlpgroup/greater"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/processbench-identifying-process-errors-in","slug":"processbench-identifying-process-errors-in","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","date":"2024-12-09","arxiv_id":"2412.06559","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/processbench-identifying-process-errors-in#ran","syntology_url":"https://syntology.ai/paper/2412.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06559"}},"official":{"repos":["qwenlm/processbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-correctly-do-semantic-backpropagation","slug":"how-to-correctly-do-semantic-backpropagation","title":"How to Correctly do Semantic Backpropagation on Language-based Agentic Systems","date":"2024-12-04","arxiv_id":"2412.03624","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/how-to-correctly-do-semantic-backpropagation#ran","syntology_url":"https://syntology.ai/paper/2412.03624","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03624"}},"official":{"repos":["hishamalyahya/semantic_backprop"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critical-tokens-matter-token-level#ran","syntology_url":"https://syntology.ai/paper/2411.19943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19943"}},"official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-optimization-for-reasoning-with","slug":"preference-optimization-for-reasoning-with","title":"Preference Optimization for Reasoning with Pseudo Feedback","date":"2024-11-25","arxiv_id":"2411.16345","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preference-optimization-for-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16345"}},"official":null}},{"url":"/paper/what-do-learning-dynamics-reveal-about","slug":"what-do-learning-dynamics-reveal-about","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","date":"2024-11-12","arxiv_id":"2411.07681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-do-learning-dynamics-reveal-about#ran","syntology_url":"https://syntology.ai/paper/2411.07681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07681"}},"official":{"repos":["katiekang1998/reasoning_generalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-are-hidden-reasoners","slug":"language-models-are-hidden-reasoners","title":"Language Models are Hidden Reasoners: Unlocking Latent Reasoning Capabilities via Self-Rewarding","date":"2024-11-06","arxiv_id":"2411.04282","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/language-models-are-hidden-reasoners#ran","syntology_url":"https://syntology.ai/paper/2411.04282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04282"}},"official":{"repos":["salesforceairesearch/latro"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-done-rite-robust-invariant","slug":"lora-done-rite-robust-invariant","title":"LoRA Done RITE: Robust Invariant Transformation Equilibration for LoRA Optimization","date":"2024-10-27","arxiv_id":"2410.20625","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lora-done-rite-robust-invariant#ran","syntology_url":"https://syntology.ai/paper/2410.20625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20625"}},"official":{"repos":["gkevinyen5418/LoRA-RITE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-up-masked-diffusion-models-on-text","slug":"scaling-up-masked-diffusion-models-on-text","title":"Scaling up Masked Diffusion Models on Text","date":"2024-10-24","arxiv_id":"2410.18514","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-up-masked-diffusion-models-on-text#ran","syntology_url":"https://syntology.ai/paper/2410.18514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18514"}},"official":{"repos":["ml-gsai/smdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/one-language-many-gaps-evaluating-dialect#ran","syntology_url":"https://syntology.ai/paper/2410.11005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11005"}},"official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/omni-math-a-universal-olympiad-level","slug":"omni-math-a-universal-olympiad-level","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","date":"2024-10-10","arxiv_id":"2410.07985","repositories_listed":2,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":19,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/omni-math-a-universal-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2410.07985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07985"}},"official":{"repos":["kbsdjames/omni-math","kbsdjames/omni-math-rule"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coevolving-with-the-other-you-fine-tuning-llm#ran","syntology_url":"https://syntology.ai/paper/2410.06101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06101"}},"official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-topla-efficient-llm-ensemble-by","slug":"llm-topla-efficient-llm-ensemble-by","title":"LLM-TOPLA: Efficient LLM Ensemble by Maximising Diversity","date":"2024-10-04","arxiv_id":"2410.03953","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llm-topla-efficient-llm-ensemble-by#ran","syntology_url":"https://syntology.ai/paper/2410.03953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03953"}},"official":{"repos":["git-disl/llm-topla"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vineppo-unlocking-rl-potential-for-llm","slug":"vineppo-unlocking-rl-potential-for-llm","title":"VinePPO: Unlocking RL Potential For LLM Reasoning Through Refined Credit Assignment","date":"2024-10-02","arxiv_id":"2410.01679","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vineppo-unlocking-rl-potential-for-llm#ran","syntology_url":"https://syntology.ai/paper/2410.01679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01679"}},"official":{"repos":["mcgill-nlp/vineppo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scheherazade-evaluating-chain-of-thought-math","slug":"scheherazade-evaluating-chain-of-thought-math","title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","date":"2024-09-30","arxiv_id":"2410.00151","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scheherazade-evaluating-chain-of-thought-math#ran","syntology_url":"https://syntology.ai/paper/2410.00151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00151"}},"official":{"repos":["yoshikitakashima/scheherazade-code-data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-symbolic-collaborative-distillation","slug":"neural-symbolic-collaborative-distillation","title":"Neural-Symbolic Collaborative Distillation: Advancing Small Language Models for Complex Reasoning Tasks","date":"2024-09-20","arxiv_id":"2409.13203","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-symbolic-collaborative-distillation#ran","syntology_url":"https://syntology.ai/paper/2409.13203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13203"}},"official":{"repos":["xnhyacinth/nesycd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-llm-reasoning-with-multi-agent-tree","slug":"improving-llm-reasoning-with-multi-agent-tree","title":"Improving LLM Reasoning with Multi-Agent Tree-of-Thought Validator Agent","date":"2024-09-17","arxiv_id":"2409.11527","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-llm-reasoning-with-multi-agent-tree#ran","syntology_url":"https://syntology.ai/paper/2409.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11527"}},"official":{"repos":["secureaiautonomylab/ma-tot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sorsa-singular-values-and-orthonormal","slug":"sorsa-singular-values-and-orthonormal","title":"SORSA: Singular Values and Orthonormal Regularized Singular Vectors Adaptation of Large Language Models","date":"2024-08-21","arxiv_id":"2409.00055","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sorsa-singular-values-and-orthonormal#ran","syntology_url":"https://syntology.ai/paper/2409.00055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00055"}},"official":{"repos":["Gunale0926/SORSA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-reasoning-makes-smaller-llms-stronger","slug":"mutual-reasoning-makes-smaller-llms-stronger","title":"Mutual Reasoning Makes Smaller LLMs Stronger Problem-Solvers","date":"2024-08-12","arxiv_id":"2408.06195","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-reasoning-makes-smaller-llms-stronger#ran","syntology_url":"https://syntology.ai/paper/2408.06195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06195"}},"official":{"repos":["zhentingqi/rstar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-monkeys-scaling-inference","slug":"large-language-monkeys-scaling-inference","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","date":"2024-07-31","arxiv_id":"2407.21787","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-monkeys-scaling-inference#ran","syntology_url":"https://syntology.ai/paper/2407.21787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21787"}},"official":{"repos":["scalingintelligence/large_language_monkeys"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/weak-to-strong-reasoning","slug":"weak-to-strong-reasoning","title":"Weak-to-Strong Reasoning","date":"2024-07-18","arxiv_id":"2407.13647","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/weak-to-strong-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13647"}},"official":{"repos":["gair-nlp/weak-to-strong-reasoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-ga-low-rank-adaptation-with-gradient","slug":"lora-ga-low-rank-adaptation-with-gradient","title":"LoRA-GA: Low-Rank Adaptation with Gradient Approximation","date":"2024-07-06","arxiv_id":"2407.05000","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lora-ga-low-rank-adaptation-with-gradient#ran","syntology_url":"https://syntology.ai/paper/2407.05000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05000"}},"official":{"repos":["outsider565/lora-ga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-metabench-a-sparse-benchmark-to","slug":"texttt-metabench-a-sparse-benchmark-to","title":"$\\texttt{metabench}$ -- A Sparse Benchmark to Measure General Ability in Large Language Models","date":"2024-07-04","arxiv_id":"2407.12844","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/texttt-metabench-a-sparse-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2407.12844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12844"}},"official":{"repos":["adkipnis/metabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-controlled-dpo-leveraging-stepwise-error#ran","syntology_url":"https://syntology.ai/paper/2407.00782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00782"}},"official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/step-dpo-step-wise-preference-optimization#ran","syntology_url":"https://syntology.ai/paper/2406.18629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18629"}},"official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/varbench-robust-language-model-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.17681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17681"}},"official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-reason-behind-good-or-bad-towards-a","slug":"the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","arxiv_id":"2406.14024","repositories_listed":1,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":18,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":17,"n_pointer_only":23,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-reason-behind-good-or-bad-towards-a#ran","syntology_url":"https://syntology.ai/paper/2406.14024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14024"}},"official":{"repos":["kbsdjames/math-minos"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chatglm-a-family-of-large-language-models","slug":"chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":21,"n_constructed":0,"n_ran_checked":20,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/chatglm-a-family-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.12793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12793"}},"official":{"repos":["thudm/chatglm-6b"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/della-merging-reducing-interference-in-model","slug":"della-merging-reducing-interference-in-model","title":"DELLA-Merging: Reducing Interference in Model Merging through Magnitude-Based Sampling","date":"2024-06-17","arxiv_id":"2406.11617","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/della-merging-reducing-interference-in-model#ran","syntology_url":"https://syntology.ai/paper/2406.11617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11617"}},"official":{"repos":["declare-lab/della"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sharelora-parameter-efficient-and-robust#ran","syntology_url":"https://syntology.ai/paper/2406.10785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10785"}},"official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-instruction-evolving-for-large","slug":"automatic-instruction-evolving-for-large","title":"Automatic Instruction Evolving for Large Language Models","date":"2024-06-02","arxiv_id":"2406.00770","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-instruction-evolving-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.00770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00770"}},"official":null}},{"url":"/paper/lora-xs-low-rank-adaptation-with-extremely","slug":"lora-xs-low-rank-adaptation-with-extremely","title":"LoRA-XS: Low-Rank Adaptation with Extremely Small Number of Parameters","date":"2024-05-27","arxiv_id":"2405.17604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-xs-low-rank-adaptation-with-extremely#ran","syntology_url":"https://syntology.ai/paper/2405.17604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17604"}},"official":{"repos":["mohammadrezabanaei/lora-xs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zipcache-accurate-and-efficient-kv-cache","slug":"zipcache-accurate-and-efficient-kv-cache","title":"ZipCache: Accurate and Efficient KV Cache Quantization with Salient Token Identification","date":"2024-05-23","arxiv_id":"2405.14256","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zipcache-accurate-and-efficient-kv-cache#ran","syntology_url":"https://syntology.ai/paper/2405.14256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14256"}},"official":null}},{"url":"/paper/unchosen-experts-can-contribute-too","slug":"unchosen-experts-can-contribute-too","title":"Unchosen Experts Can Contribute Too: Unleashing MoE Models' Power by Self-Contrast","date":"2024-05-23","arxiv_id":"2405.14507","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unchosen-experts-can-contribute-too#ran","syntology_url":"https://syntology.ai/paper/2405.14507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14507"}},"official":{"repos":["davidfanzz/scmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-explicit-cot-to-implicit-cot-learning-to","slug":"from-explicit-cot-to-implicit-cot-learning-to","title":"From Explicit CoT to Implicit CoT: Learning to Internalize CoT Step by Step","date":"2024-05-23","arxiv_id":"2405.14838","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-explicit-cot-to-implicit-cot-learning-to#ran","syntology_url":"https://syntology.ai/paper/2405.14838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14838"}},"official":{"repos":["da03/internalize_cot_step_by_step"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-choice-questions-are-efficient-and","slug":"multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","arxiv_id":"2405.11966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-choice-questions-are-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2405.11966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11966"}},"official":{"repos":["geralt-targaryen/mc-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mumath-code-combining-tool-use-large-language","slug":"mumath-code-combining-tool-use-large-language","title":"MuMath-Code: Combining Tool-Use Large Language Models with Multi-perspective Data Augmentation for Mathematical Reasoning","date":"2024-05-13","arxiv_id":"2405.07551","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mumath-code-combining-tool-use-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07551"}},"official":null}},{"url":"/paper/exploring-the-compositional-deficiency-of","slug":"exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","arxiv_id":"2405.06680","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-the-compositional-deficiency-of#ran","syntology_url":"https://syntology.ai/paper/2405.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06680"}},"official":{"repos":["tongjingqi/MathTrap"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/monte-carlo-tree-search-boosts-reasoning-via","slug":"monte-carlo-tree-search-boosts-reasoning-via","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","date":"2024-05-01","arxiv_id":"2405.00451","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monte-carlo-tree-search-boosts-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2405.00451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00451"}},"official":{"repos":["YuxiXie/MCTS-DPO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/markovian-agents-for-truthful-language","slug":"markovian-agents-for-truthful-language","title":"Markovian Transformers for Informative Language Modeling","date":"2024-04-29","arxiv_id":"2404.18988","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":14,"n_pointer_only":16,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/markovian-agents-for-truthful-language#ran","syntology_url":"https://syntology.ai/paper/2404.18988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18988"}},"official":{"repos":["scottviteri/markoviantraining"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/layer-skip-enabling-early-exit-inference-and#ran","syntology_url":"https://syntology.ai/paper/2404.16710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16710"}},"official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-self-improvement-of-llms-via","slug":"toward-self-improvement-of-llms-via","title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","date":"2024-04-18","arxiv_id":"2404.12253","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":1,"n_ran_checked":12,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":18,"phrase":"14 ran (of which 1 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/toward-self-improvement-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2404.12253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12253"}},"official":{"repos":["yetianjhu/alphallm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":1,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/self-explore-to-avoid-the-pit-improving-the","slug":"self-explore-to-avoid-the-pit-improving-the","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","date":"2024-04-16","arxiv_id":"2404.10346","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-explore-to-avoid-the-pit-improving-the#ran","syntology_url":"https://syntology.ai/paper/2404.10346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10346"}},"official":{"repos":["hbin0701/Self-Explore"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/don-t-trust-verify-grounding-llm-quantitative","slug":"don-t-trust-verify-grounding-llm-quantitative","title":"Don't Trust: Verify -- Grounding LLM Quantitative Reasoning with Autoformalization","date":"2024-03-26","arxiv_id":"2403.18120","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/don-t-trust-verify-grounding-llm-quantitative#ran","syntology_url":"https://syntology.ai/paper/2403.18120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18120"}},"official":{"repos":["jinpz/dtv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llm2llm-boosting-llms-with-novel-iterative","slug":"llm2llm-boosting-llms-with-novel-iterative","title":"LLM2LLM: Boosting LLMs with Novel Iterative Data Enhancement","date":"2024-03-22","arxiv_id":"2403.15042","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm2llm-boosting-llms-with-novel-iterative#ran","syntology_url":"https://syntology.ai/paper/2403.15042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15042"}},"official":{"repos":["squeezeailab/llm2llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quiet-star-language-models-can-teach","slug":"quiet-star-language-models-can-teach","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","date":"2024-03-14","arxiv_id":"2403.09629","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quiet-star-language-models-can-teach#ran","syntology_url":"https://syntology.ai/paper/2403.09629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09629"}},"official":{"repos":["ezelikman/quiet-star"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/common-7b-language-models-already-possess","slug":"common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","arxiv_id":"2403.04706","repositories_listed":2,"syntology":{"n":28,"n_ran":21,"n_constructed":0,"n_ran_checked":20,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":12,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/common-7b-language-models-already-possess#ran","syntology_url":"https://syntology.ai/paper/2403.04706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04706"}},"official":{"repos":["xwin-lm/xwin-lm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/masked-thought-simply-masking-partial","slug":"masked-thought-simply-masking-partial","title":"Masked Thought: Simply Masking Partial Reasoning Steps Can Improve Mathematical Reasoning Learning of Language Models","date":"2024-03-04","arxiv_id":"2403.02178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/masked-thought-simply-masking-partial#ran","syntology_url":"https://syntology.ai/paper/2403.02178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02178"}},"official":{"repos":["changyuchen347/maskedthought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/keeping-llms-aligned-after-fine-tuning-the","slug":"keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","arxiv_id":"2402.18540","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/keeping-llms-aligned-after-fine-tuning-the#ran","syntology_url":"https://syntology.ai/paper/2402.18540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18540"}},"official":{"repos":["vfleaking/ptst"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/distillation-contrastive-decoding-improving","slug":"distillation-contrastive-decoding-improving","title":"Distillation Contrastive Decoding: Improving LLMs Reasoning with Contrastive Decoding and Distillation","date":"2024-02-21","arxiv_id":"2402.14874","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/distillation-contrastive-decoding-improving#ran","syntology_url":"https://syntology.ai/paper/2402.14874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14874"}},"official":{"repos":["pphuc25/distil-cd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reformatted-alignment","slug":"reformatted-alignment","title":"Reformatted Alignment","date":"2024-02-19","arxiv_id":"2402.12219","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reformatted-alignment#ran","syntology_url":"https://syntology.ai/paper/2402.12219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12219"}},"official":{"repos":["gair-nlp/realign"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automathtext-autonomous-data-selection-with","slug":"automathtext-autonomous-data-selection-with","title":"Autonomous Data Selection with Zero-shot Generative Classifiers for Mathematical Texts","date":"2024-02-12","arxiv_id":"2402.07625","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automathtext-autonomous-data-selection-with#ran","syntology_url":"https://syntology.ai/paper/2402.07625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07625"}},"official":{"repos":["hiyouga/llama-factory","yifanzhang-pro/automathtext"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-large-language-models-for-reasoning","slug":"training-large-language-models-for-reasoning","title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.05808","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/training-large-language-models-for-reasoning#ran","syntology_url":"https://syntology.ai/paper/2402.05808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05808"}},"official":{"repos":["woooodyy/llm-reverse-curriculum-rl"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/escape-sky-high-cost-early-stopping-self","slug":"escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","arxiv_id":"2401.10480","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/escape-sky-high-cost-early-stopping-self#ran","syntology_url":"https://syntology.ai/paper/2401.10480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10480"}},"official":{"repos":["yiwei98/esc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reft-reasoning-with-reinforced-fine-tuning","slug":"reft-reasoning-with-reinforced-fine-tuning","title":"ReFT: Reasoning with Reinforced Fine-Tuning","date":"2024-01-17","arxiv_id":"2401.08967","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/reft-reasoning-with-reinforced-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2401.08967","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08967"}},"official":{"repos":["lqtrung1998/mwp_reft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mario-math-reasoning-with-code-interpreter","slug":"mario-math-reasoning-with-code-interpreter","title":"MARIO: MAth Reasoning with code Interpreter Output -- A Reproducible Pipeline","date":"2024-01-16","arxiv_id":"2401.08190","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mario-math-reasoning-with-code-interpreter#ran","syntology_url":"https://syntology.ai/paper/2401.08190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08190"}},"official":{"repos":["mario-math-reasoning/mario"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-prompting-for-agi-systems","slug":"meta-prompting-for-agi-systems","title":"Meta Prompting for AI Systems","date":"2023-11-20","arxiv_id":"2311.11482","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-prompting-for-agi-systems#ran","syntology_url":"https://syntology.ai/paper/2311.11482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.11482"}},"official":{"repos":["meta-prompting/meta-prompting"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/outcome-supervised-verifiers-for-planning-in","slug":"outcome-supervised-verifiers-for-planning-in","title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","date":"2023-11-16","arxiv_id":"2311.09724","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/outcome-supervised-verifiers-for-planning-in#ran","syntology_url":"https://syntology.ai/paper/2311.09724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09724"}},"official":{"repos":["freedomintelligence/ovm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-contamination-quiz-a-tool-to-detect-and","slug":"data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/data-contamination-quiz-a-tool-to-detect-and#ran","syntology_url":"https://syntology.ai/paper/2311.06233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06233"}},"official":{"repos":["shahriargolchin/dcq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-are-super-mario-absorbing","slug":"language-models-are-super-mario-absorbing","title":"Language Models are Super Mario: Absorbing Abilities from Homologous Models as a Free Lunch","date":"2023-11-06","arxiv_id":"2311.03099","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-are-super-mario-absorbing#ran","syntology_url":"https://syntology.ai/paper/2311.03099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03099"}},"official":{"repos":["yule-buaa/mergelm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/breaking-language-barriers-in-multilingual","slug":"breaking-language-barriers-in-multilingual","title":"Breaking Language Barriers in Multilingual Mathematical Reasoning: Insights and Observations","date":"2023-10-31","arxiv_id":"2310.20246","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/breaking-language-barriers-in-multilingual#ran","syntology_url":"https://syntology.ai/paper/2310.20246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20246"}},"official":{"repos":["microsoft/MathOctopus"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-mistakes-makes-llm-better","slug":"learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","arxiv_id":"2310.20689","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-mistakes-makes-llm-better#ran","syntology_url":"https://syntology.ai/paper/2310.20689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20689"}},"official":{"repos":["microsoft/lema"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llmlingua-compressing-prompts-for-accelerated","slug":"llmlingua-compressing-prompts-for-accelerated","title":"LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models","date":"2023-10-09","arxiv_id":"2310.05736","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llmlingua-compressing-prompts-for-accelerated#ran","syntology_url":"https://syntology.ai/paper/2310.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05736"}},"official":{"repos":["microsoft/LLMLingua"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/mathcoder-seamless-code-integration-in-llms","slug":"mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","arxiv_id":"2310.03731","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-seamless-code-integration-in-llms#ran","syntology_url":"https://syntology.ai/paper/2310.03731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03731"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/metamath-bootstrap-your-own-mathematical","slug":"metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","arxiv_id":"2309.12284","repositories_listed":1,"syntology":{"n":22,"n_ran":15,"n_constructed":0,"n_ran_checked":1,"n_instrument":14,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/metamath-bootstrap-your-own-mathematical#ran","syntology_url":"https://syntology.ai/paper/2309.12284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12284"}},"official":{"repos":["meta-math/MetaMath"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}}],"record_sha256":"4847fa8ba1ff4a5587accdb0cffe5e37d32d3ee8cee82da86579208f3f386994","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}