{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/4","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":16,"rows_per_page":100,"rows":[301,400],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/3","next":"/task/math/papers/5","papers":[{"url":"/paper/goedel-prover-a-frontier-model-for-open","slug":"goedel-prover-a-frontier-model-for-open","title":"Goedel-Prover: A Frontier Model for Open-Source Automated Theorem Proving","date":"2025-02-11","arxiv_id":"2502.07640","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goedel-prover-a-frontier-model-for-open#ran","syntology_url":"https://syntology.ai/paper/2502.07640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07640"}},"official":{"repos":["Goedel-LM/Goedel-Prover"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-fine-tuning-when-scaling-test-time","slug":"rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","arxiv_id":"2502.07154","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-fine-tuning-when-scaling-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.07154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07154"}},"official":{"repos":["allanraventos/refine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-1b-llm-surpass-405b-llm-rethinking","slug":"can-1b-llm-surpass-405b-llm-rethinking","title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling","date":"2025-02-10","arxiv_id":"2502.06703","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/can-1b-llm-surpass-405b-llm-rethinking#ran","syntology_url":"https://syntology.ai/paper/2502.06703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06703"}},"official":{"repos":["RyanLiu112/compute-optimal-tts"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-limit-of-outcome-reward-for","slug":"exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","arxiv_id":"2502.06781","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-limit-of-outcome-reward-for#ran","syntology_url":"https://syntology.ai/paper/2502.06781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06781"}},"official":{"repos":["internlm/oreal"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gsm-infinite-how-do-your-llms-behave-over","slug":"gsm-infinite-how-do-your-llms-behave-over","title":"GSM-Infinite: How Do Your LLMs Behave over Infinitely Increasing Context Length and Reasoning Complexity?","date":"2025-02-07","arxiv_id":"2502.05252","repositories_listed":1,"syntology":null},{"url":"/paper/do-large-language-model-benchmarks-test","slug":"do-large-language-model-benchmarks-test","title":"Do Large Language Model Benchmarks Test Reliability?","date":"2025-02-05","arxiv_id":"2502.03461","repositories_listed":1,"syntology":null},{"url":"/paper/upweighting-easy-samples-in-fine-tuning","slug":"upweighting-easy-samples-in-fine-tuning","title":"Upweighting Easy Samples in Fine-Tuning Mitigates Forgetting","date":"2025-02-05","arxiv_id":"2502.02797","repositories_listed":1,"syntology":null},{"url":"/paper/a-probabilistic-inference-approach-to","slug":"a-probabilistic-inference-approach-to","title":"A Probabilistic Inference Approach to Inference-Time Scaling of LLMs using Particle-Based Monte Carlo Methods","date":"2025-02-03","arxiv_id":"2502.01618","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-probabilistic-inference-approach-to#ran","syntology_url":"https://syntology.ai/paper/2502.01618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01618"}},"official":{"repos":["Red-Hat-AI-Innovation-Team/its_hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ugphysics-a-comprehensive-benchmark-for","slug":"ugphysics-a-comprehensive-benchmark-for","title":"UGPhysics: A Comprehensive Benchmark for Undergraduate Physics Reasoning with Large Language Models","date":"2025-02-01","arxiv_id":"2502.00334","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ugphysics-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2502.00334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00334"}},"official":{"repos":["yanglabhkust/ugphysics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spend-wisely-maximizing-post-training-gains","slug":"spend-wisely-maximizing-post-training-gains","title":"Spend Wisely: Maximizing Post-Training Gains in Iterative Synthetic Data Boostrapping","date":"2025-01-31","arxiv_id":"2501.18962","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-neural-theorem-proving-via-fine","slug":"efficient-neural-theorem-proving-via-fine","title":"Efficient Neural Theorem Proving via Fine-grained Proof Structure Analysis","date":"2025-01-30","arxiv_id":"2501.18310","repositories_listed":1,"syntology":null},{"url":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critique-fine-tuning-learning-to-critique-is#ran","syntology_url":"https://syntology.ai/paper/2501.17703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17703"}},"official":null}},{"url":"/paper/pairwise-rm-perform-best-of-n-sampling-with","slug":"pairwise-rm-perform-best-of-n-sampling-with","title":"Pairwise RM: Perform Best-of-N Sampling with Knockout Tournament","date":"2025-01-22","arxiv_id":"2501.13007","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-language-model-reasoning-through","slug":"advancing-language-model-reasoning-through","title":"Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling","date":"2025-01-20","arxiv_id":"2501.11651","repositories_listed":1,"syntology":null},{"url":"/paper/control-llm-controlled-evolution-for","slug":"control-llm-controlled-evolution-for","title":"Control LLM: Controlled Evolution for Intelligence Retention in LLM","date":"2025-01-19","arxiv_id":"2501.10979","repositories_listed":1,"syntology":null},{"url":"/paper/language-representation-favored-zero-shot","slug":"language-representation-favored-zero-shot","title":"Language Representation Favored Zero-Shot Cross-Domain Cognitive Diagnosis","date":"2025-01-18","arxiv_id":"2501.13943","repositories_listed":1,"syntology":null},{"url":"/paper/arithmattack-evaluating-robustness-of-llms-to","slug":"arithmattack-evaluating-robustness-of-llms-to","title":"ArithmAttack: Evaluating Robustness of LLMs to Noisy Context in Math Problem Solving","date":"2025-01-14","arxiv_id":"2501.08203","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-label-refinement-matters-more-than","slug":"iterative-label-refinement-matters-more-than","title":"Iterative Label Refinement Matters More than Preference Optimization under Weak Supervision","date":"2025-01-14","arxiv_id":"2501.07886","repositories_listed":1,"syntology":null},{"url":"/paper/can-vision-language-models-evaluate","slug":"can-vision-language-models-evaluate","title":"Can Vision-Language Models Evaluate Handwritten Math?","date":"2025-01-13","arxiv_id":"2501.07244","repositories_listed":1,"syntology":null},{"url":"/paper/zno-eval-benchmarking-reasoning-capabilities","slug":"zno-eval-benchmarking-reasoning-capabilities","title":"ZNO-Eval: Benchmarking reasoning capabilities of large language models in Ukrainian","date":"2025-01-12","arxiv_id":"2501.06715","repositories_listed":1,"syntology":null},{"url":"/paper/open-eyes-then-reason-fine-grained-visual","slug":"open-eyes-then-reason-fine-grained-visual","title":"Open Eyes, Then Reason: Fine-grained Visual Mathematical Understanding in MLLMs","date":"2025-01-11","arxiv_id":"2501.06430","repositories_listed":1,"syntology":null},{"url":"/paper/stream-aligner-efficient-sentence-level","slug":"stream-aligner-efficient-sentence-level","title":"Stream Aligner: Efficient Sentence-Level Alignment via Distribution Induction","date":"2025-01-09","arxiv_id":"2501.05336","repositories_listed":1,"syntology":null},{"url":"/paper/ursa-understanding-and-verifying-chain-of","slug":"ursa-understanding-and-verifying-chain-of","title":"URSA: Understanding and Verifying Chain-of-thought Reasoning in Multimodal Mathematics","date":"2025-01-08","arxiv_id":"2501.04686","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ursa-understanding-and-verifying-chain-of#ran","syntology_url":"https://syntology.ai/paper/2501.04686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04686"}},"official":{"repos":["URSA-MATH/URSA-MATH"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-diversity-enhanced-knowledge-distillation","slug":"a-diversity-enhanced-knowledge-distillation","title":"A Diversity-Enhanced Knowledge Distillation Model for Practical Math Word Problem Solving","date":"2025-01-07","arxiv_id":"2501.03670","repositories_listed":1,"syntology":null},{"url":"/paper/booststep-boosting-mathematical-capability-of","slug":"booststep-boosting-mathematical-capability-of","title":"BoostStep: Boosting mathematical capability of Large Language Models via improved single-step reasoning","date":"2025-01-06","arxiv_id":"2501.03226","repositories_listed":1,"syntology":null},{"url":"/paper/a-probabilistic-model-for-node-classification","slug":"a-probabilistic-model-for-node-classification","title":"A Probabilistic Model for Node Classification in Directed Graphs","date":"2025-01-03","arxiv_id":"2501.01630","repositories_listed":1,"syntology":null},{"url":"/paper/cot-based-synthesizer-enhancing-llm","slug":"cot-based-synthesizer-enhancing-llm","title":"CoT-based Synthesizer: Enhancing LLM Performance through Answer Synthesis","date":"2025-01-03","arxiv_id":"2501.01668","repositories_listed":1,"syntology":null},{"url":"/paper/dive-diversified-iterative-self-improvement","slug":"dive-diversified-iterative-self-improvement","title":"DIVE: Diversified Iterative Self-Improvement","date":"2025-01-01","arxiv_id":"2501.00747","repositories_listed":1,"syntology":null},{"url":"/paper/toward-adaptive-reasoning-in-large-language-1","slug":"toward-adaptive-reasoning-in-large-language-1","title":"Toward Adaptive Reasoning in Large Language Models with Thought Rollback","date":"2024-12-27","arxiv_id":"2412.19707","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toward-adaptive-reasoning-in-large-language-1#ran","syntology_url":"https://syntology.ai/paper/2412.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19707"}},"official":{"repos":["iQua/llmpebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/carl-gt-evaluating-causal-reasoning","slug":"carl-gt-evaluating-causal-reasoning","title":"CARL-GT: Evaluating Causal Reasoning Capabilities of Large Language Models","date":"2024-12-23","arxiv_id":"2412.17970","repositories_listed":1,"syntology":null},{"url":"/paper/drt-o1-optimized-deep-reasoning-translation","slug":"drt-o1-optimized-deep-reasoning-translation","title":"DRT-o1: Optimized Deep Reasoning Translation via Long Chain-of-Thought","date":"2024-12-23","arxiv_id":"2412.17498","repositories_listed":1,"syntology":null},{"url":"/paper/template-driven-llm-paraphrased-framework-for","slug":"template-driven-llm-paraphrased-framework-for","title":"Template-Driven LLM-Paraphrased Framework for Tabular Math Word Problem Generation","date":"2024-12-20","arxiv_id":"2412.15594","repositories_listed":1,"syntology":null},{"url":"/paper/critical-questions-of-thought-steering-llm","slug":"critical-questions-of-thought-steering-llm","title":"Critical-Questions-of-Thought: Steering LLM reasoning with Argumentative Querying","date":"2024-12-19","arxiv_id":"2412.15177","repositories_listed":1,"syntology":null},{"url":"/paper/coinmath-harnessing-the-power-of-coding","slug":"coinmath-harnessing-the-power-of-coding","title":"CoinMath: Harnessing the Power of Coding Instruction for Math LLMs","date":"2024-12-16","arxiv_id":"2412.11699","repositories_listed":1,"syntology":null},{"url":"/paper/combining-large-language-models-with-tutoring","slug":"combining-large-language-models-with-tutoring","title":"Combining Large Language Models with Tutoring System Intelligence: A Case Study in Caregiver Homework Support","date":"2024-12-16","arxiv_id":"2412.11995","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-regularized-process-reward-model","slug":"entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","arxiv_id":"2412.11006","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-process-reward-model#ran","syntology_url":"https://syntology.ai/paper/2412.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11006"}},"official":{"repos":["hanningzhang/er-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-context-enhanced-framework-for-sequential","slug":"a-context-enhanced-framework-for-sequential","title":"A Context-Enhanced Framework for Sequential Graph Reasoning","date":"2024-12-12","arxiv_id":"2412.09056","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-context-enhanced-framework-for-sequential#ran","syntology_url":"https://syntology.ai/paper/2412.09056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09056"}},"official":{"repos":["Ghost-st/CEF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/harp-a-challenging-human-annotated-math","slug":"harp-a-challenging-human-annotated-math","title":"HARP: A challenging human-annotated math reasoning benchmark","date":"2024-12-11","arxiv_id":"2412.08819","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harp-a-challenging-human-annotated-math#ran","syntology_url":"https://syntology.ai/paper/2412.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08819"}},"official":{"repos":["aadityasingh/harp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-as-an-interviewer-beyond-static-testing","slug":"llm-as-an-interviewer-beyond-static-testing","title":"LLM-as-an-Interviewer: Beyond Static Testing Through Dynamic LLM Evaluation","date":"2024-12-10","arxiv_id":"2412.10424","repositories_listed":1,"syntology":null},{"url":"/paper/processbench-identifying-process-errors-in","slug":"processbench-identifying-process-errors-in","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","date":"2024-12-09","arxiv_id":"2412.06559","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/processbench-identifying-process-errors-in#ran","syntology_url":"https://syntology.ai/paper/2412.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06559"}},"official":{"repos":["qwenlm/processbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/u-math-a-university-level-benchmark-for","slug":"u-math-a-university-level-benchmark-for","title":"U-MATH: A University-Level Benchmark for Evaluating Mathematical Skills in LLMs","date":"2024-12-04","arxiv_id":"2412.03205","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-learning-based-calibration","slug":"unsupervised-learning-based-calibration","title":"Unsupervised learning-based calibration scheme for Rough Bergomi model","date":"2024-12-03","arxiv_id":"2412.02135","repositories_listed":1,"syntology":null},{"url":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critical-tokens-matter-token-level#ran","syntology_url":"https://syntology.ai/paper/2411.19943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19943"}},"official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-examples-high-level-automated","slug":"beyond-examples-high-level-automated","title":"Beyond Examples: High-level Automated Reasoning Paradigm in In-Context Learning via MCTS","date":"2024-11-27","arxiv_id":"2411.18478","repositories_listed":1,"syntology":null},{"url":"/paper/training-and-evaluating-language-models-with","slug":"training-and-evaluating-language-models-with","title":"Training and Evaluating Language Models with Template-based Data Generation","date":"2024-11-27","arxiv_id":"2411.18104","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-and-evaluating-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2411.18104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18104"}},"official":{"repos":["iiis-ai/templatemath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-optimization-for-reasoning-with","slug":"preference-optimization-for-reasoning-with","title":"Preference Optimization for Reasoning with Pseudo Feedback","date":"2024-11-25","arxiv_id":"2411.16345","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preference-optimization-for-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16345"}},"official":null}},{"url":"/paper/llama-moe-v2-exploring-sparsity-of-llama-from","slug":"llama-moe-v2-exploring-sparsity-of-llama-from","title":"LLaMA-MoE v2: Exploring Sparsity of LLaMA from Perspective of Mixture-of-Experts with Post-Training","date":"2024-11-24","arxiv_id":"2411.15708","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llama-moe-v2-exploring-sparsity-of-llama-from#ran","syntology_url":"https://syntology.ai/paper/2411.15708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15708"}},"official":{"repos":["opensparsellms/llama-moe-v2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-eval-a-hierarchical-benchmark-for-modern","slug":"mm-eval-a-hierarchical-benchmark-for-modern","title":"MM-Eval: A Hierarchical Benchmark for Modern Mongolian Evaluation in LLMs","date":"2024-11-14","arxiv_id":"2411.09492","repositories_listed":1,"syntology":null},{"url":"/paper/resolve-relational-reasoning-with-symbolic","slug":"resolve-relational-reasoning-with-symbolic","title":"RESOLVE: Relational Reasoning with Symbolic and Object-Level Features Using Vector Symbolic Processing","date":"2024-11-13","arxiv_id":"2411.08290","repositories_listed":1,"syntology":null},{"url":"/paper/problem-oriented-segmentation-and-retrieval","slug":"problem-oriented-segmentation-and-retrieval","title":"Problem-Oriented Segmentation and Retrieval: Case Study on Tutoring Conversations","date":"2024-11-12","arxiv_id":"2411.07598","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-learning-dynamics-reveal-about","slug":"what-do-learning-dynamics-reveal-about","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","date":"2024-11-12","arxiv_id":"2411.07681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-do-learning-dynamics-reveal-about#ran","syntology_url":"https://syntology.ai/paper/2411.07681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07681"}},"official":{"repos":["katiekang1998/reasoning_generalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utmath-math-evaluation-with-unit-test-via","slug":"utmath-math-evaluation-with-unit-test-via","title":"UTMath: Math Evaluation with Unit Test via Reasoning-to-Coding Thoughts","date":"2024-11-11","arxiv_id":"2411.07240","repositories_listed":1,"syntology":null},{"url":"/paper/aioli-a-unified-optimization-framework-for","slug":"aioli-a-unified-optimization-framework-for","title":"Aioli: A Unified Optimization Framework for Language Model Data Mixing","date":"2024-11-08","arxiv_id":"2411.05735","repositories_listed":1,"syntology":{"n":31,"n_ran":23,"n_constructed":3,"n_ran_checked":19,"n_instrument":4,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"23 ran (of which 3 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 0 violated, 18 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/aioli-a-unified-optimization-framework-for#ran","syntology_url":"https://syntology.ai/paper/2411.05735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05735"}},"official":{"repos":["hazyresearch/aioli"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":3,"n_ran_no_instrument_failure":19,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reasoning-improves-tool-use-in-large","slug":"meta-reasoning-improves-tool-use-in-large","title":"Meta-Reasoning Improves Tool Use in Large Language Models","date":"2024-11-07","arxiv_id":"2411.04535","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-label-semantics-and-meta-label","slug":"leveraging-label-semantics-and-meta-label","title":"Leveraging Label Semantics and Meta-Label Refinement for Multi-Label Question Classification","date":"2024-11-04","arxiv_id":"2411.01841","repositories_listed":1,"syntology":null},{"url":"/paper/regress-don-t-guess-a-regression-like-loss-on","slug":"regress-don-t-guess-a-regression-like-loss-on","title":"Regress, Don't Guess -- A Regression-like Loss on Number Tokens for Language Models","date":"2024-11-04","arxiv_id":"2411.02083","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/regress-don-t-guess-a-regression-like-loss-on#ran","syntology_url":"https://syntology.ai/paper/2411.02083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02083"}},"official":{"repos":["tum-ai/number-token-loss"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/arithmetic-without-algorithms-language-models","slug":"arithmetic-without-algorithms-language-models","title":"Arithmetic Without Algorithms: Language Models Solve Math With a Bag of Heuristics","date":"2024-10-28","arxiv_id":"2410.21272","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/arithmetic-without-algorithms-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21272"}},"official":{"repos":["technion-cs-nlp/llm-arithmetic-heuristics"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/autoformalize-mathematical-statements-by","slug":"autoformalize-mathematical-statements-by","title":"Autoformalize Mathematical Statements by Symbolic Equivalence and Semantic Consistency","date":"2024-10-28","arxiv_id":"2410.20936","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/autoformalize-mathematical-statements-by#ran","syntology_url":"https://syntology.ai/paper/2410.20936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20936"}},"official":{"repos":["miracle-messi/isa-autoformal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-through-complexity-what-makes-good","slug":"guiding-through-complexity-what-makes-good","title":"Guiding Through Complexity: What Makes Good Supervision for Hard Reasoning Tasks?","date":"2024-10-27","arxiv_id":"2410.20533","repositories_listed":1,"syntology":null},{"url":"/paper/library-learning-doesn-t-the-curious-case-of","slug":"library-learning-doesn-t-the-curious-case-of","title":"Library Learning Doesn't: The Curious Case of the Single-Use \"Library\"","date":"2024-10-26","arxiv_id":"2410.20274","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/library-learning-doesn-t-the-curious-case-of#ran","syntology_url":"https://syntology.ai/paper/2410.20274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20274"}},"official":{"repos":["ikb-a/curious-case"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/scaling-up-masked-diffusion-models-on-text","slug":"scaling-up-masked-diffusion-models-on-text","title":"Scaling up Masked Diffusion Models on Text","date":"2024-10-24","arxiv_id":"2410.18514","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-up-masked-diffusion-models-on-text#ran","syntology_url":"https://syntology.ai/paper/2410.18514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18514"}},"official":{"repos":["ml-gsai/smdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-reasoning-capability-of-llms-via","slug":"unleashing-reasoning-capability-of-llms-via","title":"Unleashing Reasoning Capability of LLMs via Scalable Question Synthesis from Scratch","date":"2024-10-24","arxiv_id":"2410.18693","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unleashing-reasoning-capability-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.18693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18693"}},"official":{"repos":["yyding1/scalequest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/language-model-non-myopic-generation-for","slug":"language-model-non-myopic-generation-for","title":"Non-myopic Generation of Language Models for Reasoning and Planning","date":"2024-10-22","arxiv_id":"2410.17195","repositories_listed":1,"syntology":null},{"url":"/paper/math-neurosurgery-isolating-language-models","slug":"math-neurosurgery-isolating-language-models","title":"Math Neurosurgery: Isolating Language Models' Math Reasoning Abilities Using Only Forward Passes","date":"2024-10-22","arxiv_id":"2410.16930","repositories_listed":1,"syntology":null},{"url":"/paper/internlm2-5-stepprover-advancing-automated","slug":"internlm2-5-stepprover-advancing-automated","title":"InternLM2.5-StepProver: Advancing Automated Theorem Proving via Expert Iteration on Large-Scale LEAN Problems","date":"2024-10-21","arxiv_id":"2410.15700","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm2-5-stepprover-advancing-automated#ran","syntology_url":"https://syntology.ai/paper/2410.15700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15700"}},"official":{"repos":["internlm/internlm-math"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparative-study-on-reasoning-patterns-of","slug":"a-comparative-study-on-reasoning-patterns-of","title":"A Comparative Study on Reasoning Patterns of OpenAI's o1 Model","date":"2024-10-17","arxiv_id":"2410.13639","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-comparative-study-on-reasoning-patterns-of#ran","syntology_url":"https://syntology.ai/paper/2410.13639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13639"}},"official":{"repos":["open-source-o1/o1_reasoning_patterns_study"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sbi-rag-enhancing-math-word-problem-solving","slug":"sbi-rag-enhancing-math-word-problem-solving","title":"SBI-RAG: Enhancing Math Word Problem Solving for Students through Schema-Based Instruction and Retrieval-Augmented Generation","date":"2024-10-17","arxiv_id":"2410.13293","repositories_listed":1,"syntology":null},{"url":"/paper/judgebench-a-benchmark-for-evaluating-llm","slug":"judgebench-a-benchmark-for-evaluating-llm","title":"JudgeBench: A Benchmark for Evaluating LLM-based Judges","date":"2024-10-16","arxiv_id":"2410.12784","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/judgebench-a-benchmark-for-evaluating-llm#ran","syntology_url":"https://syntology.ai/paper/2410.12784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12784"}},"official":{"repos":["ScalerLab/JudgeBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-soups-merging-loras-for-practical-skill","slug":"lora-soups-merging-loras-for-practical-skill","title":"LoRA Soups: Merging LoRAs for Practical Skill Composition Tasks","date":"2024-10-16","arxiv_id":"2410.13025","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-soups-merging-loras-for-practical-skill#ran","syntology_url":"https://syntology.ai/paper/2410.13025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13025"}},"official":{"repos":["aksh555/lora-soups"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/not-all-votes-count-programs-as-verifiers","slug":"not-all-votes-count-programs-as-verifiers","title":"Not All Votes Count! Programs as Verifiers Improve Self-Consistency of Language Models for Math Reasoning","date":"2024-10-16","arxiv_id":"2410.12608","repositories_listed":1,"syntology":null},{"url":"/paper/comat-chain-of-mathematically-annotated","slug":"comat-chain-of-mathematically-annotated","title":"CoMAT: Chain of Mathematically Annotated Thought Improves Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10336","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comat-chain-of-mathematically-annotated#ran","syntology_url":"https://syntology.ai/paper/2410.10336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10336"}},"official":{"repos":["joshuaongg21/comat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/one-language-many-gaps-evaluating-dialect#ran","syntology_url":"https://syntology.ai/paper/2410.11005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11005"}},"official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hardmath-a-benchmark-dataset-for-challenging","slug":"hardmath-a-benchmark-dataset-for-challenging","title":"HARDMath: A Benchmark Dataset for Challenging Problems in Applied Mathematics","date":"2024-10-13","arxiv_id":"2410.09988","repositories_listed":1,"syntology":null},{"url":"/paper/openr-an-open-source-framework-for-advanced","slug":"openr-an-open-source-framework-for-advanced","title":"OpenR: An Open Source Framework for Advanced Reasoning with Large Language Models","date":"2024-10-12","arxiv_id":"2410.09671","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/openr-an-open-source-framework-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2410.09671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09671"}},"official":null}},{"url":"/paper/mathcoder2-better-math-reasoning-from","slug":"mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","arxiv_id":"2410.08196","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mathcoder2-better-math-reasoning-from#ran","syntology_url":"https://syntology.ai/paper/2410.08196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08196"}},"official":{"repos":["mathllm/mathcoder2"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-inspired-integrated-prompting","slug":"teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","arxiv_id":"2410.08068","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-inspired-integrated-prompting#ran","syntology_url":"https://syntology.ai/paper/2410.08068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08068"}},"official":{"repos":["sallytan13/teaching-inspired-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-geometry-of-concepts-sparse-autoencoder","slug":"the-geometry-of-concepts-sparse-autoencoder","title":"The Geometry of Concepts: Sparse Autoencoder Feature Structure","date":"2024-10-10","arxiv_id":"2410.19750","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-geometry-of-concepts-sparse-autoencoder#ran","syntology_url":"https://syntology.ai/paper/2410.19750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19750"}},"official":{"repos":["ejmichaud/feature-geometry"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vibecheck-discover-and-quantify-qualitative","slug":"vibecheck-discover-and-quantify-qualitative","title":"VibeCheck: Discover and Quantify Qualitative Differences in Large Language Models","date":"2024-10-10","arxiv_id":"2410.12851","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vibecheck-discover-and-quantify-qualitative#ran","syntology_url":"https://syntology.ai/paper/2410.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12851"}},"official":{"repos":["lisadunlap/vibecheck"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dataenvgym-data-generation-agents-in-teacher","slug":"dataenvgym-data-generation-agents-in-teacher","title":"DataEnvGym: Data Generation Agents in Teacher Environments with Student Feedback","date":"2024-10-08","arxiv_id":"2410.06215","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dataenvgym-data-generation-agents-in-teacher#ran","syntology_url":"https://syntology.ai/paper/2410.06215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06215"}},"official":{"repos":["codezakh/dataenvgym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/give-me-a-hint-can-llms-take-a-hint-to-solve","slug":"give-me-a-hint-can-llms-take-a-hint-to-solve","title":"Give me a hint: Can LLMs take a hint to solve math problems?","date":"2024-10-08","arxiv_id":"2410.05915","repositories_listed":1,"syntology":null},{"url":"/paper/o1-replication-journey-a-strategic-progress","slug":"o1-replication-journey-a-strategic-progress","title":"O1 Replication Journey: A Strategic Progress Report -- Part 1","date":"2024-10-08","arxiv_id":"2410.18982","repositories_listed":1,"syntology":null},{"url":"/paper/steering-large-language-models-between-code","slug":"steering-large-language-models-between-code","title":"Steering Large Language Models between Code Execution and Textual Reasoning","date":"2024-10-04","arxiv_id":"2410.03524","repositories_listed":1,"syntology":null},{"url":"/paper/llama-slayer-8b-shallow-layers-hold-the-key","slug":"llama-slayer-8b-shallow-layers-hold-the-key","title":"Llama SLayer 8B: Shallow Layers Hold the Key to Knowledge Injection","date":"2024-10-03","arxiv_id":"2410.02330","repositories_listed":1,"syntology":null},{"url":"/paper/an-exploration-of-self-supervised-mutual","slug":"an-exploration-of-self-supervised-mutual","title":"An Exploration of Self-Supervised Mutual Information Alignment for Multi-Task Settings","date":"2024-10-02","arxiv_id":"2410.01704","repositories_listed":1,"syntology":null},{"url":"/paper/automated-knowledge-concept-annotation-and","slug":"automated-knowledge-concept-annotation-and","title":"Automated Knowledge Concept Annotation and Question Representation Learning for Knowledge Tracing","date":"2024-10-02","arxiv_id":"2410.01727","repositories_listed":1,"syntology":null},{"url":"/paper/laser-learning-to-adaptively-select-reward","slug":"laser-learning-to-adaptively-select-reward","title":"LASeR: Learning to Adaptively Select Reward Models with Multi-Armed Bandits","date":"2024-10-02","arxiv_id":"2410.01735","repositories_listed":1,"syntology":null},{"url":"/paper/mind-scramble-unveiling-large-language-model","slug":"mind-scramble-unveiling-large-language-model","title":"Mind Scramble: Unveiling Large Language Model Psychology Via Typoglycemia","date":"2024-10-02","arxiv_id":"2410.01677","repositories_listed":1,"syntology":null},{"url":"/paper/openmathinstruct-2-accelerating-ai-for-math","slug":"openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","arxiv_id":"2410.01560","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/openmathinstruct-2-accelerating-ai-for-math#ran","syntology_url":"https://syntology.ai/paper/2410.01560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01560"}},"official":null}},{"url":"/paper/vineppo-unlocking-rl-potential-for-llm","slug":"vineppo-unlocking-rl-potential-for-llm","title":"VinePPO: Unlocking RL Potential For LLM Reasoning Through Refined Credit Assignment","date":"2024-10-02","arxiv_id":"2410.01679","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vineppo-unlocking-rl-potential-for-llm#ran","syntology_url":"https://syntology.ai/paper/2410.01679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01679"}},"official":{"repos":["mcgill-nlp/vineppo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scheherazade-evaluating-chain-of-thought-math","slug":"scheherazade-evaluating-chain-of-thought-math","title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","date":"2024-09-30","arxiv_id":"2410.00151","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scheherazade-evaluating-chain-of-thought-math#ran","syntology_url":"https://syntology.ai/paper/2410.00151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00151"}},"official":{"repos":["yoshikitakashima/scheherazade-code-data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beats-optimizing-llm-mathematical","slug":"beats-optimizing-llm-mathematical","title":"BEATS: Optimizing LLM Mathematical Capabilities with BackVerify and Adaptive Disambiguate based Efficient Tree Search","date":"2024-09-26","arxiv_id":"2409.17972","repositories_listed":1,"syntology":null},{"url":"/paper/archon-an-architecture-search-framework-for","slug":"archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","arxiv_id":"2409.15254","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/archon-an-architecture-search-framework-for#ran","syntology_url":"https://syntology.ai/paper/2409.15254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15254"}},"official":{"repos":["scalingintelligence/archon"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2409-14082","slug":"2409-14082","title":"PTD-SQL: Partitioning and Targeted Drilling with LLMs in Text-to-SQL","date":"2024-09-21","arxiv_id":"2409.14082","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2409-14082#ran","syntology_url":"https://syntology.ai/paper/2409.14082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14082"}},"official":{"repos":["lrlbbzl/ptd-sql"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-accuracy-optimization-computer-vision","slug":"beyond-accuracy-optimization-computer-vision","title":"Beyond Accuracy Optimization: Computer Vision Losses for Large Language Model Fine-Tuning","date":"2024-09-20","arxiv_id":"2409.13641","repositories_listed":1,"syntology":null},{"url":"/paper/magicore-multi-agent-iterative-coarse-to-fine","slug":"magicore-multi-agent-iterative-coarse-to-fine","title":"MAgICoRe: Multi-Agent, Iterative, Coarse-to-Fine Refinement for Reasoning","date":"2024-09-18","arxiv_id":"2409.12147","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magicore-multi-agent-iterative-coarse-to-fine#ran","syntology_url":"https://syntology.ai/paper/2409.12147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12147"}},"official":{"repos":["dinobby/magicore"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps","slug":"to-cot-or-not-to-cot-chain-of-thought-helps","title":"To CoT or not to CoT? Chain-of-thought helps mainly on math and symbolic reasoning","date":"2024-09-18","arxiv_id":"2409.12183","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps#ran","syntology_url":"https://syntology.ai/paper/2409.12183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12183"}},"official":{"repos":["zayne-sprague/to-cot-or-not-to-cot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/diversify-and-conquer-diversity-centric-data","slug":"diversify-and-conquer-diversity-centric-data","title":"Diversify and Conquer: Diversity-Centric Data Selection with Iterative Refinement","date":"2024-09-17","arxiv_id":"2409.11378","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-graph-enhanced-exemplars-retrieval","slug":"reasoning-graph-enhanced-exemplars-retrieval","title":"Reasoning Graph Enhanced Exemplars Retrieval for In-Context Learning","date":"2024-09-17","arxiv_id":"2409.11147","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-datasets-in-words-statistical","slug":"explaining-datasets-in-words-statistical","title":"Explaining Datasets in Words: Statistical Models with Natural Language Parameters","date":"2024-09-13","arxiv_id":"2409.08466","repositories_listed":1,"syntology":null},{"url":"/paper/ai-for-mathematics-mathematical-formalized","slug":"ai-for-mathematics-mathematical-formalized","title":"Mathematical Formalized Problem Solving and Theorem Proving in Different Fields in Lean 4","date":"2024-09-09","arxiv_id":"2409.05977","repositories_listed":1,"syntology":null}],"record_sha256":"36588e732c0fa558d3181962ebc0c5f7898521defb12ca3c24fbb47e4811f478","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}