{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/ran/2","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":349,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math/papers/ran/1","prev":"/task/math/papers/ran/1","next":"/task/math/papers/ran/3","papers":[{"url":"/paper/exploring-the-limit-of-outcome-reward-for","slug":"exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","arxiv_id":"2502.06781","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-limit-of-outcome-reward-for#ran","syntology_url":"https://syntology.ai/paper/2502.06781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06781"}},"official":{"repos":["internlm/oreal"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/process-reinforcement-through-implicit","slug":"process-reinforcement-through-implicit","title":"Process Reinforcement through Implicit Rewards","date":"2025-02-03","arxiv_id":"2502.01456","repositories_listed":5,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/process-reinforcement-through-implicit#ran","syntology_url":"https://syntology.ai/paper/2502.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01456"}},"official":{"repos":["prime-rl/prime"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-probabilistic-inference-approach-to","slug":"a-probabilistic-inference-approach-to","title":"A Probabilistic Inference Approach to Inference-Time Scaling of LLMs using Particle-Based Monte Carlo Methods","date":"2025-02-03","arxiv_id":"2502.01618","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-probabilistic-inference-approach-to#ran","syntology_url":"https://syntology.ai/paper/2502.01618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01618"}},"official":{"repos":["Red-Hat-AI-Innovation-Team/its_hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ugphysics-a-comprehensive-benchmark-for","slug":"ugphysics-a-comprehensive-benchmark-for","title":"UGPhysics: A Comprehensive Benchmark for Undergraduate Physics Reasoning with Large Language Models","date":"2025-02-01","arxiv_id":"2502.00334","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ugphysics-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2502.00334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00334"}},"official":{"repos":["yanglabhkust/ugphysics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/s1-simple-test-time-scaling","slug":"s1-simple-test-time-scaling","title":"s1: Simple test-time scaling","date":"2025-01-31","arxiv_id":"2501.19393","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s1-simple-test-time-scaling#ran","syntology_url":"https://syntology.ai/paper/2501.19393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19393"}},"official":{"repos":["simplescaling/s1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critique-fine-tuning-learning-to-critique-is#ran","syntology_url":"https://syntology.ai/paper/2501.17703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17703"}},"official":null}},{"url":"/paper/leveraging-online-olympiad-level-math","slug":"leveraging-online-olympiad-level-math","title":"Leveraging Online Olympiad-Level Math Problems for LLMs Training and Contamination-Resistant Evaluation","date":"2025-01-24","arxiv_id":"2501.14275","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-online-olympiad-level-math#ran","syntology_url":"https://syntology.ai/paper/2501.14275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14275"}},"official":{"repos":["dsl-lab/aops"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","slug":"kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","arxiv_id":"2501.12599","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kimi-k1-5-scaling-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2501.12599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12599"}},"official":null}},{"url":"/paper/ursa-understanding-and-verifying-chain-of","slug":"ursa-understanding-and-verifying-chain-of","title":"URSA: Understanding and Verifying Chain-of-thought Reasoning in Multimodal Mathematics","date":"2025-01-08","arxiv_id":"2501.04686","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ursa-understanding-and-verifying-chain-of#ran","syntology_url":"https://syntology.ai/paper/2501.04686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04686"}},"official":{"repos":["URSA-MATH/URSA-MATH"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-adaptive-reasoning-in-large-language-1","slug":"toward-adaptive-reasoning-in-large-language-1","title":"Toward Adaptive Reasoning in Large Language Models with Thought Rollback","date":"2024-12-27","arxiv_id":"2412.19707","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toward-adaptive-reasoning-in-large-language-1#ran","syntology_url":"https://syntology.ai/paper/2412.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19707"}},"official":{"repos":["iQua/llmpebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-5-technical-report","slug":"qwen2-5-technical-report","title":"Qwen2.5 Technical Report","date":"2024-12-19","arxiv_id":"2412.15115","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/qwen2-5-technical-report#ran","syntology_url":"https://syntology.ai/paper/2412.15115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15115"}},"official":{"repos":["qwenlm/qwen2.5"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/entropy-regularized-process-reward-model","slug":"entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","arxiv_id":"2412.11006","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-process-reward-model#ran","syntology_url":"https://syntology.ai/paper/2412.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11006"}},"official":{"repos":["hanningzhang/er-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-context-enhanced-framework-for-sequential","slug":"a-context-enhanced-framework-for-sequential","title":"A Context-Enhanced Framework for Sequential Graph Reasoning","date":"2024-12-12","arxiv_id":"2412.09056","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-context-enhanced-framework-for-sequential#ran","syntology_url":"https://syntology.ai/paper/2412.09056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09056"}},"official":{"repos":["Ghost-st/CEF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/harp-a-challenging-human-annotated-math","slug":"harp-a-challenging-human-annotated-math","title":"HARP: A challenging human-annotated math reasoning benchmark","date":"2024-12-11","arxiv_id":"2412.08819","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harp-a-challenging-human-annotated-math#ran","syntology_url":"https://syntology.ai/paper/2412.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08819"}},"official":{"repos":["aadityasingh/harp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/processbench-identifying-process-errors-in","slug":"processbench-identifying-process-errors-in","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","date":"2024-12-09","arxiv_id":"2412.06559","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/processbench-identifying-process-errors-in#ran","syntology_url":"https://syntology.ai/paper/2412.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06559"}},"official":{"repos":["qwenlm/processbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/free-process-rewards-without-process-labels","slug":"free-process-rewards-without-process-labels","title":"Free Process Rewards without Process Labels","date":"2024-12-02","arxiv_id":"2412.01981","repositories_listed":2,"syntology":{"n":22,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/free-process-rewards-without-process-labels#ran","syntology_url":"https://syntology.ai/paper/2412.01981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01981"}},"official":{"repos":["lifan-yuan/implicitprm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critical-tokens-matter-token-level#ran","syntology_url":"https://syntology.ai/paper/2411.19943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19943"}},"official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-and-evaluating-language-models-with","slug":"training-and-evaluating-language-models-with","title":"Training and Evaluating Language Models with Template-based Data Generation","date":"2024-11-27","arxiv_id":"2411.18104","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-and-evaluating-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2411.18104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18104"}},"official":{"repos":["iiis-ai/templatemath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-optimization-for-reasoning-with","slug":"preference-optimization-for-reasoning-with","title":"Preference Optimization for Reasoning with Pseudo Feedback","date":"2024-11-25","arxiv_id":"2411.16345","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preference-optimization-for-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16345"}},"official":null}},{"url":"/paper/llama-moe-v2-exploring-sparsity-of-llama-from","slug":"llama-moe-v2-exploring-sparsity-of-llama-from","title":"LLaMA-MoE v2: Exploring Sparsity of LLaMA from Perspective of Mixture-of-Experts with Post-Training","date":"2024-11-24","arxiv_id":"2411.15708","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llama-moe-v2-exploring-sparsity-of-llama-from#ran","syntology_url":"https://syntology.ai/paper/2411.15708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15708"}},"official":{"repos":["opensparsellms/llama-moe-v2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unlocking-state-tracking-in-linear-rnns","slug":"unlocking-state-tracking-in-linear-rnns","title":"Unlocking State-Tracking in Linear RNNs Through Negative Eigenvalues","date":"2024-11-19","arxiv_id":"2411.12537","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unlocking-state-tracking-in-linear-rnns#ran","syntology_url":"https://syntology.ai/paper/2411.12537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12537"}},"official":{"repos":["automl/unlocking_state_tracking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-do-learning-dynamics-reveal-about","slug":"what-do-learning-dynamics-reveal-about","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","date":"2024-11-12","arxiv_id":"2411.07681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-do-learning-dynamics-reveal-about#ran","syntology_url":"https://syntology.ai/paper/2411.07681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07681"}},"official":{"repos":["katiekang1998/reasoning_generalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aioli-a-unified-optimization-framework-for","slug":"aioli-a-unified-optimization-framework-for","title":"Aioli: A Unified Optimization Framework for Language Model Data Mixing","date":"2024-11-08","arxiv_id":"2411.05735","repositories_listed":1,"syntology":{"n":31,"n_ran":23,"n_constructed":3,"n_ran_checked":19,"n_instrument":4,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"23 ran (of which 3 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 0 violated, 18 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/aioli-a-unified-optimization-framework-for#ran","syntology_url":"https://syntology.ai/paper/2411.05735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05735"}},"official":{"repos":["hazyresearch/aioli"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":3,"n_ran_no_instrument_failure":19,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/regress-don-t-guess-a-regression-like-loss-on","slug":"regress-don-t-guess-a-regression-like-loss-on","title":"Regress, Don't Guess -- A Regression-like Loss on Number Tokens for Language Models","date":"2024-11-04","arxiv_id":"2411.02083","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/regress-don-t-guess-a-regression-like-loss-on#ran","syntology_url":"https://syntology.ai/paper/2411.02083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02083"}},"official":{"repos":["tum-ai/number-token-loss"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stem-pom-evaluating-language-models-math","slug":"stem-pom-evaluating-language-models-math","title":"STEM-POM: Evaluating Language Models Math-Symbol Reasoning in Document Parsing","date":"2024-11-01","arxiv_id":"2411.00387","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stem-pom-evaluating-language-models-math#ran","syntology_url":"https://syntology.ai/paper/2411.00387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00387"}},"official":null}},{"url":"/paper/autoformalize-mathematical-statements-by","slug":"autoformalize-mathematical-statements-by","title":"Autoformalize Mathematical Statements by Symbolic Equivalence and Semantic Consistency","date":"2024-10-28","arxiv_id":"2410.20936","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/autoformalize-mathematical-statements-by#ran","syntology_url":"https://syntology.ai/paper/2410.20936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20936"}},"official":{"repos":["miracle-messi/isa-autoformal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flaming-hot-initiation-with-regular-execution","slug":"flaming-hot-initiation-with-regular-execution","title":"Flaming-hot Initiation with Regular Execution Sampling for Large Language Models","date":"2024-10-28","arxiv_id":"2410.21236","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flaming-hot-initiation-with-regular-execution#ran","syntology_url":"https://syntology.ai/paper/2410.21236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21236"}},"official":null}},{"url":"/paper/arithmetic-without-algorithms-language-models","slug":"arithmetic-without-algorithms-language-models","title":"Arithmetic Without Algorithms: Language Models Solve Math With a Bag of Heuristics","date":"2024-10-28","arxiv_id":"2410.21272","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/arithmetic-without-algorithms-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21272"}},"official":{"repos":["technion-cs-nlp/llm-arithmetic-heuristics"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/library-learning-doesn-t-the-curious-case-of","slug":"library-learning-doesn-t-the-curious-case-of","title":"Library Learning Doesn't: The Curious Case of the Single-Use \"Library\"","date":"2024-10-26","arxiv_id":"2410.20274","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/library-learning-doesn-t-the-curious-case-of#ran","syntology_url":"https://syntology.ai/paper/2410.20274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20274"}},"official":{"repos":["ikb-a/curious-case"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/scaling-up-masked-diffusion-models-on-text","slug":"scaling-up-masked-diffusion-models-on-text","title":"Scaling up Masked Diffusion Models on Text","date":"2024-10-24","arxiv_id":"2410.18514","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-up-masked-diffusion-models-on-text#ran","syntology_url":"https://syntology.ai/paper/2410.18514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18514"}},"official":{"repos":["ml-gsai/smdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-reasoning-capability-of-llms-via","slug":"unleashing-reasoning-capability-of-llms-via","title":"Unleashing Reasoning Capability of LLMs via Scalable Question Synthesis from Scratch","date":"2024-10-24","arxiv_id":"2410.18693","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unleashing-reasoning-capability-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.18693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18693"}},"official":{"repos":["yyding1/scalequest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/internlm2-5-stepprover-advancing-automated","slug":"internlm2-5-stepprover-advancing-automated","title":"InternLM2.5-StepProver: Advancing Automated Theorem Proving via Expert Iteration on Large-Scale LEAN Problems","date":"2024-10-21","arxiv_id":"2410.15700","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm2-5-stepprover-advancing-automated#ran","syntology_url":"https://syntology.ai/paper/2410.15700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15700"}},"official":{"repos":["internlm/internlm-math"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparative-study-on-reasoning-patterns-of","slug":"a-comparative-study-on-reasoning-patterns-of","title":"A Comparative Study on Reasoning Patterns of OpenAI's o1 Model","date":"2024-10-17","arxiv_id":"2410.13639","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-comparative-study-on-reasoning-patterns-of#ran","syntology_url":"https://syntology.ai/paper/2410.13639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13639"}},"official":{"repos":["open-source-o1/o1_reasoning_patterns_study"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/judgebench-a-benchmark-for-evaluating-llm","slug":"judgebench-a-benchmark-for-evaluating-llm","title":"JudgeBench: A Benchmark for Evaluating LLM-based Judges","date":"2024-10-16","arxiv_id":"2410.12784","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/judgebench-a-benchmark-for-evaluating-llm#ran","syntology_url":"https://syntology.ai/paper/2410.12784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12784"}},"official":{"repos":["ScalerLab/JudgeBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-soups-merging-loras-for-practical-skill","slug":"lora-soups-merging-loras-for-practical-skill","title":"LoRA Soups: Merging LoRAs for Practical Skill Composition Tasks","date":"2024-10-16","arxiv_id":"2410.13025","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-soups-merging-loras-for-practical-skill#ran","syntology_url":"https://syntology.ai/paper/2410.13025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13025"}},"official":{"repos":["aksh555/lora-soups"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/comat-chain-of-mathematically-annotated","slug":"comat-chain-of-mathematically-annotated","title":"CoMAT: Chain of Mathematically Annotated Thought Improves Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10336","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comat-chain-of-mathematically-annotated#ran","syntology_url":"https://syntology.ai/paper/2410.10336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10336"}},"official":{"repos":["joshuaongg21/comat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/one-language-many-gaps-evaluating-dialect#ran","syntology_url":"https://syntology.ai/paper/2410.11005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11005"}},"official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/openr-an-open-source-framework-for-advanced","slug":"openr-an-open-source-framework-for-advanced","title":"OpenR: An Open Source Framework for Advanced Reasoning with Large Language Models","date":"2024-10-12","arxiv_id":"2410.09671","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/openr-an-open-source-framework-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2410.09671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09671"}},"official":null}},{"url":"/paper/omni-math-a-universal-olympiad-level","slug":"omni-math-a-universal-olympiad-level","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","date":"2024-10-10","arxiv_id":"2410.07985","repositories_listed":2,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":19,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/omni-math-a-universal-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2410.07985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07985"}},"official":{"repos":["kbsdjames/omni-math","kbsdjames/omni-math-rule"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-inspired-integrated-prompting","slug":"teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","arxiv_id":"2410.08068","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-inspired-integrated-prompting#ran","syntology_url":"https://syntology.ai/paper/2410.08068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08068"}},"official":{"repos":["sallytan13/teaching-inspired-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathcoder2-better-math-reasoning-from","slug":"mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","arxiv_id":"2410.08196","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mathcoder2-better-math-reasoning-from#ran","syntology_url":"https://syntology.ai/paper/2410.08196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08196"}},"official":{"repos":["mathllm/mathcoder2"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vibecheck-discover-and-quantify-qualitative","slug":"vibecheck-discover-and-quantify-qualitative","title":"VibeCheck: Discover and Quantify Qualitative Differences in Large Language Models","date":"2024-10-10","arxiv_id":"2410.12851","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vibecheck-discover-and-quantify-qualitative#ran","syntology_url":"https://syntology.ai/paper/2410.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12851"}},"official":{"repos":["lisadunlap/vibecheck"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-geometry-of-concepts-sparse-autoencoder","slug":"the-geometry-of-concepts-sparse-autoencoder","title":"The Geometry of Concepts: Sparse Autoencoder Feature Structure","date":"2024-10-10","arxiv_id":"2410.19750","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-geometry-of-concepts-sparse-autoencoder#ran","syntology_url":"https://syntology.ai/paper/2410.19750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19750"}},"official":{"repos":["ejmichaud/feature-geometry"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dataenvgym-data-generation-agents-in-teacher","slug":"dataenvgym-data-generation-agents-in-teacher","title":"DataEnvGym: Data Generation Agents in Teacher Environments with Student Feedback","date":"2024-10-08","arxiv_id":"2410.06215","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dataenvgym-data-generation-agents-in-teacher#ran","syntology_url":"https://syntology.ai/paper/2410.06215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06215"}},"official":{"repos":["codezakh/dataenvgym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/openmathinstruct-2-accelerating-ai-for-math","slug":"openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","arxiv_id":"2410.01560","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/openmathinstruct-2-accelerating-ai-for-math#ran","syntology_url":"https://syntology.ai/paper/2410.01560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01560"}},"official":null}},{"url":"/paper/vineppo-unlocking-rl-potential-for-llm","slug":"vineppo-unlocking-rl-potential-for-llm","title":"VinePPO: Unlocking RL Potential For LLM Reasoning Through Refined Credit Assignment","date":"2024-10-02","arxiv_id":"2410.01679","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vineppo-unlocking-rl-potential-for-llm#ran","syntology_url":"https://syntology.ai/paper/2410.01679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01679"}},"official":{"repos":["mcgill-nlp/vineppo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scheherazade-evaluating-chain-of-thought-math","slug":"scheherazade-evaluating-chain-of-thought-math","title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","date":"2024-09-30","arxiv_id":"2410.00151","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scheherazade-evaluating-chain-of-thought-math#ran","syntology_url":"https://syntology.ai/paper/2410.00151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00151"}},"official":{"repos":["yoshikitakashima/scheherazade-code-data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/archon-an-architecture-search-framework-for","slug":"archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","arxiv_id":"2409.15254","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/archon-an-architecture-search-framework-for#ran","syntology_url":"https://syntology.ai/paper/2409.15254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15254"}},"official":{"repos":["scalingintelligence/archon"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2409-14082","slug":"2409-14082","title":"PTD-SQL: Partitioning and Targeted Drilling with LLMs in Text-to-SQL","date":"2024-09-21","arxiv_id":"2409.14082","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2409-14082#ran","syntology_url":"https://syntology.ai/paper/2409.14082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14082"}},"official":{"repos":["lrlbbzl/ptd-sql"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magicore-multi-agent-iterative-coarse-to-fine","slug":"magicore-multi-agent-iterative-coarse-to-fine","title":"MAgICoRe: Multi-Agent, Iterative, Coarse-to-Fine Refinement for Reasoning","date":"2024-09-18","arxiv_id":"2409.12147","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magicore-multi-agent-iterative-coarse-to-fine#ran","syntology_url":"https://syntology.ai/paper/2409.12147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12147"}},"official":{"repos":["dinobby/magicore"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps","slug":"to-cot-or-not-to-cot-chain-of-thought-helps","title":"To CoT or not to CoT? Chain-of-thought helps mainly on math and symbolic reasoning","date":"2024-09-18","arxiv_id":"2409.12183","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps#ran","syntology_url":"https://syntology.ai/paper/2409.12183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12183"}},"official":{"repos":["zayne-sprague/to-cot-or-not-to-cot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/qwen2-5-coder-technical-report","slug":"qwen2-5-coder-technical-report","title":"Qwen2.5-Coder Technical Report","date":"2024-09-18","arxiv_id":"2409.12186","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/qwen2-5-coder-technical-report#ran","syntology_url":"https://syntology.ai/paper/2409.12186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12186"}},"official":{"repos":["qwenlm/qwen2.5-coder"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sirius-contextual-sparsity-with-correction","slug":"sirius-contextual-sparsity-with-correction","title":"Sirius: Contextual Sparsity with Correction for Efficient LLMs","date":"2024-09-05","arxiv_id":"2409.03856","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sirius-contextual-sparsity-with-correction#ran","syntology_url":"https://syntology.ai/paper/2409.03856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03856"}},"official":{"repos":["infini-ai-lab/sirius"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multimath-bridging-visual-and-mathematical","slug":"multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","arxiv_id":"2409.00147","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":6,"n_pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 3 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimath-bridging-visual-and-mathematical#ran","syntology_url":"https://syntology.ai/paper/2409.00147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00147"}},"official":{"repos":["pengshuai-rin/multimath"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sorsa-singular-values-and-orthonormal","slug":"sorsa-singular-values-and-orthonormal","title":"SORSA: Singular Values and Orthonormal Regularized Singular Vectors Adaptation of Large Language Models","date":"2024-08-21","arxiv_id":"2409.00055","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sorsa-singular-values-and-orthonormal#ran","syntology_url":"https://syntology.ai/paper/2409.00055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00055"}},"official":{"repos":["Gunale0926/SORSA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/math-puma-progressive-upward-multimodal","slug":"math-puma-progressive-upward-multimodal","title":"Math-PUMA: Progressive Upward Multimodal Alignment to Enhance Mathematical Reasoning","date":"2024-08-16","arxiv_id":"2408.08640","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-puma-progressive-upward-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.08640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08640"}},"official":{"repos":["wwzhuang01/math-puma"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-and-modeling-correlations-in","slug":"bridging-and-modeling-correlations-in","title":"Bridging and Modeling Correlations in Pairwise Data for Direct Preference Optimization","date":"2024-08-14","arxiv_id":"2408.07471","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-and-modeling-correlations-in#ran","syntology_url":"https://syntology.ai/paper/2408.07471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07471"}},"official":{"repos":["YJiangcm/BMC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathscape-evaluating-mllms-in-multimodal-math","slug":"mathscape-evaluating-mllms-in-multimodal-math","title":"MathScape: Evaluating MLLMs in multimodal Math Scenarios through a Hierarchical Benchmark","date":"2024-08-14","arxiv_id":"2408.07543","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathscape-evaluating-mllms-in-multimodal-math#ran","syntology_url":"https://syntology.ai/paper/2408.07543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07543"}},"official":{"repos":["PKU-Baichuan-MLSystemLab/MathScape"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-reasoning-makes-smaller-llms-stronger","slug":"mutual-reasoning-makes-smaller-llms-stronger","title":"Mutual Reasoning Makes Smaller LLMs Stronger Problem-Solvers","date":"2024-08-12","arxiv_id":"2408.06195","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-reasoning-makes-smaller-llms-stronger#ran","syntology_url":"https://syntology.ai/paper/2408.06195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06195"}},"official":{"repos":["zhentingqi/rstar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00989","slug":"2408-00989","title":"On the Resilience of LLM-Based Multi-Agent Collaboration with Faulty Agents","date":"2024-08-02","arxiv_id":"2408.00989","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-00989#ran","syntology_url":"https://syntology.ai/paper/2408.00989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00989"}},"official":{"repos":["cuhk-arise/mas-resilience"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00724","slug":"2408-00724","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","date":"2024-08-01","arxiv_id":"2408.00724","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-00724#ran","syntology_url":"https://syntology.ai/paper/2408.00724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00724"}},"official":null}},{"url":"/paper/2408-00765","slug":"2408-00765","title":"MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities","date":"2024-08-01","arxiv_id":"2408.00765","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-00765#ran","syntology_url":"https://syntology.ai/paper/2408.00765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00765"}},"official":{"repos":["yuweihao/mm-vet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-monkeys-scaling-inference","slug":"large-language-monkeys-scaling-inference","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","date":"2024-07-31","arxiv_id":"2407.21787","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-monkeys-scaling-inference#ran","syntology_url":"https://syntology.ai/paper/2407.21787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21787"}},"official":{"repos":["scalingintelligence/large_language_monkeys"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathviz-e-a-case-study-in-domain-specialized","slug":"mathviz-e-a-case-study-in-domain-specialized","title":"MathViz-E: A Case-study in Domain-Specialized Tool-Using Agents","date":"2024-07-24","arxiv_id":"2407.17544","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mathviz-e-a-case-study-in-domain-specialized#ran","syntology_url":"https://syntology.ai/paper/2407.17544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17544"}},"official":{"repos":["emergenceai/mathviz-e"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-v-s-memorization-tracing","slug":"generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","repositories_listed":0,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/generalization-v-s-memorization-tracing#ran","syntology_url":"https://syntology.ai/paper/2407.14985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14985"}},"official":null}},{"url":"/paper/weak-to-strong-reasoning","slug":"weak-to-strong-reasoning","title":"Weak-to-Strong Reasoning","date":"2024-07-18","arxiv_id":"2407.13647","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/weak-to-strong-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13647"}},"official":{"repos":["gair-nlp/weak-to-strong-reasoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-llms-for-optimization-modeling#ran","syntology_url":"https://syntology.ai/paper/2407.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09887"}},"official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autobencher-creating-salient-novel-difficult","slug":"autobencher-creating-salient-novel-difficult","title":"AutoBencher: Creating Salient, Novel, Difficult Datasets for Language Models","date":"2024-07-11","arxiv_id":"2407.08351","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobencher-creating-salient-novel-difficult#ran","syntology_url":"https://syntology.ai/paper/2407.08351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08351"}},"official":{"repos":["XiangLi1999/AutoBencher"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smart-vision-language-reasoners","slug":"smart-vision-language-reasoners","title":"Smart Vision-Language Reasoners","date":"2024-07-05","arxiv_id":"2407.04212","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/smart-vision-language-reasoners#ran","syntology_url":"https://syntology.ai/paper/2407.04212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04212"}},"official":{"repos":["smarter-vlm/smarter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/helpful-assistant-or-fruitful-facilitator","slug":"helpful-assistant-or-fruitful-facilitator","title":"Helpful assistant or fruitful facilitator? Investigating how personas affect language model behavior","date":"2024-07-02","arxiv_id":"2407.02099","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/helpful-assistant-or-fruitful-facilitator#ran","syntology_url":"https://syntology.ai/paper/2407.02099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02099"}},"official":{"repos":["peluz/persona-behavior"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/eliminating-position-bias-of-language-models","slug":"eliminating-position-bias-of-language-models","title":"Eliminating Position Bias of Language Models: A Mechanistic Approach","date":"2024-07-01","arxiv_id":"2407.01100","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eliminating-position-bias-of-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.01100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01100"}},"official":{"repos":["wzq016/pine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/we-math-does-your-large-multimodal-model","slug":"we-math-does-your-large-multimodal-model","title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","date":"2024-07-01","arxiv_id":"2407.01284","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/we-math-does-your-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2407.01284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01284"}},"official":{"repos":["we-math/we-math"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-controlled-dpo-leveraging-stepwise-error#ran","syntology_url":"https://syntology.ai/paper/2407.00782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00782"}},"official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/livebench-a-challenging-contamination-free","slug":"livebench-a-challenging-contamination-free","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","date":"2024-06-27","arxiv_id":"2406.19314","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/livebench-a-challenging-contamination-free#ran","syntology_url":"https://syntology.ai/paper/2406.19314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19314"}},"official":{"repos":["livebench/livebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/divert-distractor-generation-with-variational","slug":"divert-distractor-generation-with-variational","title":"DiVERT: Distractor Generation with Variational Errors Represented as Text for Math Multiple-choice Questions","date":"2024-06-27","arxiv_id":"2406.19356","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/divert-distractor-generation-with-variational#ran","syntology_url":"https://syntology.ai/paper/2406.19356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19356"}},"official":{"repos":["umass-ml4ed/divert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mathodyssey-benchmarking-mathematical-problem","slug":"mathodyssey-benchmarking-mathematical-problem","title":"MathOdyssey: Benchmarking Mathematical Problem-Solving Skills in Large Language Models Using Odyssey Math Data","date":"2024-06-26","arxiv_id":"2406.18321","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathodyssey-benchmarking-mathematical-problem#ran","syntology_url":"https://syntology.ai/paper/2406.18321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18321"}},"official":null}},{"url":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/step-dpo-step-wise-preference-optimization#ran","syntology_url":"https://syntology.ai/paper/2406.18629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18629"}},"official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/math-llava-bootstrapping-mathematical","slug":"math-llava-bootstrapping-mathematical","title":"Math-LLaVA: Bootstrapping Mathematical Reasoning for Multimodal Large Language Models","date":"2024-06-25","arxiv_id":"2406.17294","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-llava-bootstrapping-mathematical#ran","syntology_url":"https://syntology.ai/paper/2406.17294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17294"}},"official":{"repos":["hzq950419/math-llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lottery-ticket-adaptation-mitigating","slug":"lottery-ticket-adaptation-mitigating","title":"Lottery Ticket Adaptation: Mitigating Destructive Interference in LLMs","date":"2024-06-24","arxiv_id":"2406.16797","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lottery-ticket-adaptation-mitigating#ran","syntology_url":"https://syntology.ai/paper/2406.16797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16797"}},"official":{"repos":["kiddyboots216/lottery-ticket-adaptation"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/citygpt-empowering-urban-spatial-cognition-of","slug":"citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","arxiv_id":"2406.13948","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citygpt-empowering-urban-spatial-cognition-of#ran","syntology_url":"https://syntology.ai/paper/2406.13948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13948"}},"official":{"repos":["tsinghua-fib-lab/citygpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-reason-behind-good-or-bad-towards-a","slug":"the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","arxiv_id":"2406.14024","repositories_listed":1,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":18,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":17,"n_pointer_only":23,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-reason-behind-good-or-bad-towards-a#ran","syntology_url":"https://syntology.ai/paper/2406.14024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14024"}},"official":{"repos":["kbsdjames/math-minos"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-infinite-long-prefix-in-transformer","slug":"toward-infinite-long-prefix-in-transformer","title":"Towards Infinite-Long Prefix in Transformer","date":"2024-06-20","arxiv_id":"2406.14036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toward-infinite-long-prefix-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2406.14036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14036"}},"official":{"repos":["christianyang37/chiwun"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptable-logical-control-for-large-language","slug":"adaptable-logical-control-for-large-language","title":"Adaptable Logical Control for Large Language Models","date":"2024-06-19","arxiv_id":"2406.13892","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/adaptable-logical-control-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.13892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13892"}},"official":{"repos":["joshuacnf/Ctrl-G"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/chatglm-a-family-of-large-language-models","slug":"chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":21,"n_constructed":0,"n_ran_checked":20,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/chatglm-a-family-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.12793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12793"}},"official":{"repos":["thudm/chatglm-6b"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dart-math-difficulty-aware-rejection-tuning-1","slug":"dart-math-difficulty-aware-rejection-tuning-1","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","date":"2024-06-18","arxiv_id":"2407.13690","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dart-math-difficulty-aware-rejection-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2407.13690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13690"}},"official":{"repos":["hkust-nlp/dart-math"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/geogpt4v-towards-geometric-multi-modal-large","slug":"geogpt4v-towards-geometric-multi-modal-large","title":"GeoGPT4V: Towards Geometric Multi-modal Large Language Models with Geometric Image Generation","date":"2024-06-17","arxiv_id":"2406.11503","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/geogpt4v-towards-geometric-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2406.11503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11503"}},"official":{"repos":["lanyu0303/geogpt4v_project"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/della-merging-reducing-interference-in-model","slug":"della-merging-reducing-interference-in-model","title":"DELLA-Merging: Reducing Interference in Model Merging through Magnitude-Based Sampling","date":"2024-06-17","arxiv_id":"2406.11617","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/della-merging-reducing-interference-in-model#ran","syntology_url":"https://syntology.ai/paper/2406.11617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11617"}},"official":{"repos":["declare-lab/della"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/step-level-value-preference-optimization-for","slug":"step-level-value-preference-optimization-for","title":"Step-level Value Preference Optimization for Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10858","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-level-value-preference-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2406.10858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10858"}},"official":{"repos":["MARIO-Math-Reasoning/Super_MARIO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/delta-come-training-free-delta-compression","slug":"delta-come-training-free-delta-compression","title":"Delta-CoMe: Training-Free Delta-Compression with Mixed-Precision for Large Language Models","date":"2024-06-13","arxiv_id":"2406.08903","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/delta-come-training-free-delta-compression#ran","syntology_url":"https://syntology.ai/paper/2406.08903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08903"}},"official":{"repos":["thunlp/delta-come"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/milora-harnessing-minor-singular-components","slug":"milora-harnessing-minor-singular-components","title":"MiLoRA: Harnessing Minor Singular Components for Parameter-Efficient LLM Finetuning","date":"2024-06-13","arxiv_id":"2406.09044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/milora-harnessing-minor-singular-components#ran","syntology_url":"https://syntology.ai/paper/2406.09044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09044"}},"official":{"repos":["graphpku/pissa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unpacking-dpo-and-ppo-disentangling-best","slug":"unpacking-dpo-and-ppo-disentangling-best","title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","date":"2024-06-13","arxiv_id":"2406.09279","repositories_listed":2,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":8,"n_instrument":7,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/unpacking-dpo-and-ppo-disentangling-best#ran","syntology_url":"https://syntology.ai/paper/2406.09279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09279"}},"official":{"repos":["allenai/open-instruct","hamishivi/easylm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/corda-context-oriented-decomposition","slug":"corda-context-oriented-decomposition","title":"CorDA: Context-Oriented Decomposition Adaptation of Large Language Models for Task-Aware Parameter-Efficient Fine-tuning","date":"2024-06-07","arxiv_id":"2406.05223","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/corda-context-oriented-decomposition#ran","syntology_url":"https://syntology.ai/paper/2406.05223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05223"}},"official":{"repos":["iboing/corda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dice-detecting-in-distribution-contamination","slug":"dice-detecting-in-distribution-contamination","title":"DICE: Detecting In-distribution Contamination in LLM's Fine-tuning Phase for Math Reasoning","date":"2024-06-06","arxiv_id":"2406.04197","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dice-detecting-in-distribution-contamination#ran","syntology_url":"https://syntology.ai/paper/2406.04197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04197"}},"official":{"repos":["thu-keg/dice"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/taia-large-language-models-are-out-of","slug":"taia-large-language-models-are-out-of","title":"TAIA: Large Language Models are Out-of-Distribution Data Learners","date":"2024-05-30","arxiv_id":"2405.20192","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/taia-large-language-models-are-out-of#ran","syntology_url":"https://syntology.ai/paper/2405.20192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20192"}},"official":{"repos":["pixas/TAIA_LLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-the-robustness-of-llms-on-math","slug":"investigating-the-robustness-of-llms-on-math","title":"Cutting Through the Noise: Boosting LLM Performance on Math Word Problems","date":"2024-05-30","arxiv_id":"2406.15444","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/investigating-the-robustness-of-llms-on-math#ran","syntology_url":"https://syntology.ai/paper/2406.15444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15444"}},"official":{"repos":["him1411/problemathic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/yuan-2-0-m32-mixture-of-experts-with","slug":"yuan-2-0-m32-mixture-of-experts-with","title":"Yuan 2.0-M32: Mixture of Experts with Attention Router","date":"2024-05-28","arxiv_id":"2405.17976","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yuan-2-0-m32-mixture-of-experts-with#ran","syntology_url":"https://syntology.ai/paper/2405.17976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17976"}},"official":{"repos":["ieit-yuan/yuan2.0-m32"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"076a156e3be94cad1e83de779aee2f94ab411cb1460b42ab2eebe0607966325b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}