{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mathematical-reasoning/papers/7","list_of":"/task/mathematical-reasoning","task":"Mathematical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":9,"rows_per_page":100,"rows":[601,700],"of":805,"counts":{"archive_papers_tagged":805,"with_a_code_link":395,"where_syntology_ran_a_sample":197,"not_listed_spam_title":0,"listed":805,"listed_where_code_ran":197,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":159,"every_run_a_failure_of_syntologys_instrument":38,"listed_with_a_run_with_no_instrument_failure":159,"listed_every_run_a_failure_of_syntologys_instrument":38,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mathematical-reasoning","prev":"/task/mathematical-reasoning/papers/6","next":"/task/mathematical-reasoning/papers/8","papers":[{"url":null,"slug":"optimizing-alignment-with-less-leveraging","title":"Optimizing Alignment with Less: Leveraging Data Augmentation for Personalized Evaluation","date":"2024-12-10","arxiv_id":"2412.07429","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-positive-unlabeled-pu-and","title":"Applications of Positive Unlabeled (PU) and Negative Unlabeled (NU) Learning in Cybersecurity","date":"2024-12-09","arxiv_id":"2412.06203","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-symbolic-data-generation-for-math","title":"Neuro-Symbolic Data Generation for Math Reasoning","date":"2024-12-06","arxiv_id":"2412.04857","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-mathematical-reasoning-in-llms-with","title":"Enhancing Mathematical Reasoning in LLMs with Background Operators","date":"2024-12-05","arxiv_id":"2412.04110","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-pre-prompt-optimization-for","title":"Evolutionary Pre-Prompt Optimization for Mathematical Reasoning","date":"2024-12-05","arxiv_id":"2412.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-free-mitigation-of-language","title":"Training-Free Mitigation of Language Reasoning Degradation After Multimodal Instruction Tuning","date":"2024-12-04","arxiv_id":"2412.03467","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-physics-reasoning-in-large-language","title":"Improving Physics Reasoning in Large Language Models Using Mixture of Refinement Agents","date":"2024-12-01","arxiv_id":"2412.00821","repositories_listed":0,"syntology":null},{"url":null,"slug":"mars-po-multi-agent-reasoning-system","title":"Mars-PO: Multi-Agent Reasoning System Preference Optimization","date":"2024-11-28","arxiv_id":"2411.19039","repositories_listed":0,"syntology":null},{"url":null,"slug":"matata-a-weak-supervised-mathematical-tool","title":"MATATA: Weakly Supervised End-to-End MAthematical Tool-Augmented Reasoning for Tabular Applications","date":"2024-11-28","arxiv_id":"2411.18915","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llm-reasoning-via-critique-models","title":"Enhancing LLM Reasoning via Critique Models with Test-Time and Training-Time Supervision","date":"2024-11-25","arxiv_id":"2411.16579","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-mathematical-reasoning-capabilities","title":"Improving Mathematical Reasoning Capabilities of Small Language Models via Feedback-Driven Distillation","date":"2024-11-22","arxiv_id":"2411.14698","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-combinatorial","title":"Large Language Models for Combinatorial Optimization of Design Structure Matrix","date":"2024-11-19","arxiv_id":"2411.12571","repositories_listed":0,"syntology":null},{"url":null,"slug":"lynx-enabling-efficient-moe-inference-through","title":"Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection","date":"2024-11-13","arxiv_id":"2411.08982","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-with-1","title":"Benchmarking Large Language Models with Integer Sequence Generation Tasks","date":"2024-11-07","arxiv_id":"2411.04372","repositories_listed":0,"syntology":null},{"url":"/paper/frontiermath-a-benchmark-for-evaluating","slug":"frontiermath-a-benchmark-for-evaluating","title":"FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI","date":"2024-11-07","arxiv_id":"2411.04872","repositories_listed":0,"syntology":null},{"url":null,"slug":"kwai-star-transform-llms-into-state","title":"Kwai-STaR: Transform LLMs into State-Transition Reasoners","date":"2024-11-07","arxiv_id":"2411.04799","repositories_listed":0,"syntology":null},{"url":"/paper/stem-pom-evaluating-language-models-math","slug":"stem-pom-evaluating-language-models-math","title":"STEM-POM: Evaluating Language Models Math-Symbol Reasoning in Document Parsing","date":"2024-11-01","arxiv_id":"2411.00387","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stem-pom-evaluating-language-models-math#ran","syntology_url":"https://syntology.ai/paper/2411.00387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00387"}},"official":null}},{"url":null,"slug":"visaidmath-benchmarking-visual-aided","title":"VisAidMath: Benchmarking Visual-Aided Mathematical Reasoning","date":"2024-10-30","arxiv_id":"2410.22995","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamath-a-dynamic-visual-benchmark-for","title":"DynaMath: A Dynamic Visual Benchmark for Evaluating Mathematical Reasoning Robustness of Vision Language Models","date":"2024-10-29","arxiv_id":"2411.00836","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-dpo-improving-llm-mathematical-reasoning","title":"Flow-DPO: Improving LLM Mathematical Reasoning through Online Multi-Agent Learning","date":"2024-10-29","arxiv_id":"2410.22304","repositories_listed":0,"syntology":null},{"url":null,"slug":"gflownet-fine-tuning-for-diverse-correct","title":"GFlowNet Fine-tuning for Diverse Correct Solutions in Mathematical Reasoning Tasks","date":"2024-10-26","arxiv_id":"2410.20147","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-small-scale-large-language-models","title":"Improving Small-Scale Large Language Models Function Calling for Reasoning Tasks","date":"2024-10-24","arxiv_id":"2410.18890","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasonagain-using-extractable-symbolic","title":"ReasonAgain: Using Extractable Symbolic Programs to Evaluate Mathematical Reasoning","date":"2024-10-24","arxiv_id":"2410.19056","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-chain-of-thought-for-efficient","title":"Markov Chain of Thought for Efficient Mathematical Reasoning","date":"2024-10-23","arxiv_id":"2410.17635","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-language-models-invent-algorithms","title":"Can Large Language Models Invent Algorithms to Improve Themselves?","date":"2024-10-21","arxiv_id":"2410.15639","repositories_listed":0,"syntology":null},{"url":null,"slug":"keep-guessing-when-considering-inference","title":"Keep Guessing? When Considering Inference Scaling, Mind the Baselines","date":"2024-10-20","arxiv_id":"2410.15466","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-large-language-models-truly-grasp","title":"Do Large Language Models Truly Grasp Mathematics? An Empirical Exploration From Cognitive Psychology","date":"2024-10-19","arxiv_id":"2410.14979","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-guided-reasoning-improving-mathematical","title":"Step Guided Reasoning: Improving Mathematical Reasoning using Guidance Generation and Step Reasoning","date":"2024-10-18","arxiv_id":"2410.19817","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaswitch-adaptive-switching-between-small","title":"AdaSwitch: Adaptive Switching between Small and Large Agents for Effective Cloud-Local Collaborative Learning","date":"2024-10-17","arxiv_id":"2410.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-numerical-precision-affects-mathematical","title":"How Numerical Precision Affects Mathematical Reasoning Capabilities of LLMs","date":"2024-10-17","arxiv_id":"2410.13857","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-mathematical-reasoning-in-llms-by","title":"Enhancing Mathematical Reasoning in LLMs by Stepwise Correction","date":"2024-10-16","arxiv_id":"2410.12934","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-math-informed-synthetic-dialogues-for","title":"MIND: Math Informed syNthetic Dialogues for Pretraining LLMs","date":"2024-10-15","arxiv_id":"2410.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-in-context-learning-in-llms-via","title":"Augmenting In-Context-Learning in LLMs via Automatic Data Labeling and Refinement","date":"2024-10-14","arxiv_id":"2410.10348","repositories_listed":0,"syntology":null},{"url":null,"slug":"embedding-self-correction-as-an-inherent","title":"Embedding Self-Correction as an Inherent Ability in Large Language Models for Enhanced Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10735","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-search-space-with-diverse-prompting","title":"Expanding Search Space with Diverse Prompting Agents: An Efficient Sampling Approach for LLM Mathematical Reasoning","date":"2024-10-13","arxiv_id":"2410.09780","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-systematic-survey-on-large-language-models","title":"A Systematic Survey on Large Language Models for Algorithm Design","date":"2024-10-11","arxiv_id":"2410.14716","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-of-thought-elicits-stronger","title":"Diversity of Thought Elicits Stronger Reasoning Capabilities in Multi-Agent Debate Frameworks","date":"2024-10-10","arxiv_id":"2410.12853","repositories_listed":0,"syntology":null},{"url":null,"slug":"tpo-aligning-large-language-models-with-multi","title":"TPO: Aligning Large Language Models with Multi-branch & Multi-step Preference Trees","date":"2024-10-10","arxiv_id":"2410.12854","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifierq-enhancing-llm-test-time-compute","title":"VerifierQ: Enhancing LLM Test Time Compute with Q-Learning-based Verifiers","date":"2024-10-10","arxiv_id":"2410.08048","repositories_listed":0,"syntology":null},{"url":null,"slug":"herald-a-natural-language-annotated-lean-4","title":"Herald: A Natural Language Annotated Lean 4 Dataset","date":"2024-10-09","arxiv_id":"2410.10878","repositories_listed":0,"syntology":null},{"url":null,"slug":"positionid-llms-can-control-lengths-copy-and","title":"PositionID: LLMs can Control Lengths, Copy and Paste with Explicit Positional Awareness","date":"2024-10-09","arxiv_id":"2410.07035","repositories_listed":0,"syntology":null},{"url":null,"slug":"subtle-errors-matter-preference-learning-via","title":"Subtle Errors Matter: Preference Learning via Error-injected Self-editing","date":"2024-10-09","arxiv_id":"2410.06638","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-self-improvement-of-llms-via-mcts","title":"Towards Self-Improvement of LLMs via MCTS: Leveraging Stepwise Knowledge with Curriculum Preference Learning","date":"2024-10-09","arxiv_id":"2410.06508","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-captioning-task-specific-prompting-for","title":"Beyond Captioning: Task-Specific Prompting for Improved VLM Performance in Mathematical Reasoning","date":"2024-10-08","arxiv_id":"2410.05928","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-hallucination-detection-and-2","title":"FG-PRM: Fine-grained Hallucination Detection and Mitigation in Language Model Mathematical Reasoning","date":"2024-10-08","arxiv_id":"2410.06304","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathhay-an-automated-benchmark-for-long","title":"MathHay: An Automated Benchmark for Long-Context Mathematical Reasoning in LLMs","date":"2024-10-07","arxiv_id":"2410.04698","repositories_listed":0,"syntology":null},{"url":null,"slug":"errorradar-benchmarking-complex-mathematical","title":"ErrorRadar: Benchmarking Complex Mathematical Reasoning of Multimodal Large Language Models Via Error Detection","date":"2024-10-06","arxiv_id":"2410.04509","repositories_listed":0,"syntology":null},{"url":null,"slug":"codepmp-scalable-preference-model-pretraining","title":"CodePMP: Scalable Preference Model Pretraining for Large Language Model Reasoning","date":"2024-10-03","arxiv_id":"2410.02229","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphic-a-graph-based-in-context-example","title":"GraphIC: A Graph-Based In-Context Example Retrieval Model for Multi-Step Reasoning","date":"2024-10-03","arxiv_id":"2410.02203","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-reward-models-for","title":"Evaluating Robustness of Reward Models for Mathematical Reasoning","date":"2024-10-02","arxiv_id":"2410.01729","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-swapping-for-zero-shot-cross-lingual","title":"Layer Swapping for Zero-Shot Cross-Lingual Transfer in Large Language Models","date":"2024-10-02","arxiv_id":"2410.01335","repositories_listed":0,"syntology":null},{"url":null,"slug":"metamath-integrating-natural-language-and","title":"INC-Math: Integrating Natural Language and Code for Enhanced Mathematical Reasoning in Large Language Models","date":"2024-09-28","arxiv_id":"2409.19381","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-openai-o1-opportunities-and","title":"Evaluation of OpenAI o1: Opportunities and Challenges of AGI","date":"2024-09-27","arxiv_id":"2409.18486","repositories_listed":0,"syntology":null},{"url":null,"slug":"hm3-hierarchical-multi-objective-model","title":"HM3: Hierarchical Multi-Objective Model Merging for Pretrained Models","date":"2024-09-27","arxiv_id":"2409.18893","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-superficial-alignment","title":"Revisiting the Superficial Alignment Hypothesis","date":"2024-09-27","arxiv_id":"2410.03717","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-sciq-an-educational-chatbot-for","title":"LLaMa-SciQ: An Educational Chatbot for Answering Science MCQ","date":"2024-09-25","arxiv_id":"2409.16779","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlmath-controllable-data-generation","title":"ControlMath: Controllable Data Generation Promotes Math Generalist Models","date":"2024-09-20","arxiv_id":"2409.15376","repositories_listed":0,"syntology":null},{"url":null,"slug":"infimm-webmath-40b-advancing-multimodal-pre","title":"InfiMM-WebMath-40B: Advancing Multimodal Pre-Training for Enhanced Mathematical Reasoning","date":"2024-09-19","arxiv_id":"2409.12568","repositories_listed":0,"syntology":null},{"url":"/paper/qwen2-5-math-technical-report-toward","slug":"qwen2-5-math-technical-report-toward","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","date":"2024-09-18","arxiv_id":"2409.12122","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-inference-with-large-language-model-a","title":"Causal Inference with Large Language Model: A Survey","date":"2024-09-15","arxiv_id":"2409.09822","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpl-critical-planning-step-learning-boosts","title":"CPL: Critical Plan Step Learning Boosts LLM Generalization in Reasoning Tasks","date":"2024-09-13","arxiv_id":"2409.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"expediting-and-elevating-large-language-model","title":"Expediting and Elevating Large Language Model Reasoning via Hidden Chain-of-Thought Decoding","date":"2024-09-13","arxiv_id":"2409.08561","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathglm-vision-solving-mathematical-problems","title":"MathGLM-Vision: Solving Mathematical Problems with Multi-Modal Large Language Model","date":"2024-09-10","arxiv_id":"2409.13729","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-calculation-to-adjudication-examining","title":"From Calculation to Adjudication: Examining LLM judges on Mathematical Reasoning Tasks","date":"2024-09-06","arxiv_id":"2409.04168","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-math-agents-with-multi-turn","title":"Building Math Agents with Multi-Turn Iterative Preference Learning","date":"2024-09-04","arxiv_id":"2409.02392","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-3-c-math-spontaneous-step-level-self","title":"S$^3$c-Math: Spontaneous Step-level Self-correction Makes Large Language Models Better Mathematical Reasoners","date":"2024-09-03","arxiv_id":"2409.01524","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-contrastive-reasoning-with-lightweight","title":"Logic Contrastive Reasoning with Lightweight Large Language Model for Math Word Problems","date":"2024-08-29","arxiv_id":"2409.00131","repositories_listed":0,"syntology":null},{"url":null,"slug":"autogeo-automating-geometric-image-dataset","title":"AutoGeo: Automating Geometric Image Dataset Creation for Enhanced Geometry Understanding","date":"2024-08-28","arxiv_id":"2409.09039","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-lossless-speculative-decoding-via","title":"Boosting Lossless Speculative Decoding via Feature Sampling and Partial Alignment Distillation","date":"2024-08-28","arxiv_id":"2408.15562","repositories_listed":0,"syntology":null},{"url":null,"slug":"siam-self-improving-code-assisted","title":"SIaM: Self-Improving Code-Assisted Mathematical Reasoning of Large Language Models","date":"2024-08-28","arxiv_id":"2408.15565","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-consistency-prefix-enhancement-for","title":"Path-Consistency: Prefix Enhancement for Efficient Inference in LLM","date":"2024-08-25","arxiv_id":"2409.01281","repositories_listed":0,"syntology":null},{"url":null,"slug":"tangram-a-challenging-benchmark-for-geometric","title":"Tangram: Benchmark for Evaluating Geometric Element Recognition in Large Multimodal Models","date":"2024-08-25","arxiv_id":"2408.13854","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-tool-integration-application-for-math","title":"Multi-tool Integration Application for Math Reasoning Using Large Language Model","date":"2024-08-22","arxiv_id":"2408.12148","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-generative-diffusion-for-universal","title":"Taming Generative Diffusion Prior for Universal Blind Image Restoration","date":"2024-08-21","arxiv_id":"2408.11287","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-evaluating-large-language-models-on","title":"SarcasmBench: Towards Evaluating Large Language Models on Sarcasm Understanding","date":"2024-08-21","arxiv_id":"2408.11319","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-distillation-from-strong-to-weak","title":"Concept Distillation from Strong to Weak Models via Hypotheses-to-Theories Prompting","date":"2024-08-18","arxiv_id":"2408.09365","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01779","title":"MathLearner: A Large Language Model Agent Framework for Learning to Solve Mathematical Problems","date":"2024-08-03","arxiv_id":"2408.01779","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-numerical-estimation-and","title":"Optimizing Numerical Estimation and Operational Efficiency in the Legal Domain through Large Language Models","date":"2024-07-26","arxiv_id":"2407.19041","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-evaluation-of-large-language-2","title":"A Comprehensive Evaluation of Large Language Models on Temporal Event Forecasting","date":"2024-07-16","arxiv_id":"2407.11638","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-reasoning-beyond-natural-language","title":"Reliable Reasoning Beyond Natural Language","date":"2024-07-16","arxiv_id":"2407.11373","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-and-prompt-optimization-two-great","title":"Fine-Tuning and Prompt Optimization: Two Great Steps that Work Better Together","date":"2024-07-15","arxiv_id":"2407.10930","repositories_listed":0,"syntology":null},{"url":null,"slug":"key-point-driven-mathematical-reasoning","title":"Key-Point-Driven Mathematical Reasoning Distillation of Large Language Model","date":"2024-07-14","arxiv_id":"2407.10167","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-supervised-value-models-for-enhancing","title":"Token-Supervised Value Models for Enhancing Mathematical Reasoning Capabilities of Large Language Models","date":"2024-07-12","arxiv_id":"2407.12863","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-your-model-really-a-good-math-reasoner","title":"Is Your Model Really A Good Math Reasoner? Evaluating Mathematical Reasoning with Checklist","date":"2024-07-11","arxiv_id":"2407.08733","repositories_listed":0,"syntology":null},{"url":null,"slug":"skywork-math-data-scaling-laws-for","title":"Skywork-Math: Data Scaling Laws for Mathematical Reasoning in Large Language Models -- The Story Goes On","date":"2024-07-11","arxiv_id":"2407.08348","repositories_listed":0,"syntology":null},{"url":null,"slug":"progress-or-regress-self-improvement-reversal","title":"Progress or Regress? Self-Improvement Reversal in Post-training","date":"2024-07-06","arxiv_id":"2407.05013","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-quantization-affect-multilingual","title":"How Does Quantization Affect Multilingual LLMs?","date":"2024-07-03","arxiv_id":"2407.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"litesearch-efficacious-tree-search-for-llm","title":"LiteSearch: Efficacious Tree Search for LLM","date":"2024-06-29","arxiv_id":"2407.00320","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-as-instructors-learning-from-errors","title":"LLMs-as-Instructors: Learning from Errors Toward Automating Model Improvement","date":"2024-06-29","arxiv_id":"2407.00497","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-rlaif-for-code-generation-with-api","title":"Applying RLAIF for Code Generation with API-usage in Lightweight LLMs","date":"2024-06-28","arxiv_id":"2406.20060","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-qiyas-benchmark-measuring-chatgpt","title":"The Qiyas Benchmark: Measuring ChatGPT Mathematical and Language Understanding in Arabic","date":"2024-06-28","arxiv_id":"2407.00146","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomaly-detection-of-tabular-data-using-llms","title":"Anomaly Detection of Tabular Data Using LLMs","date":"2024-06-24","arxiv_id":"2406.16308","repositories_listed":0,"syntology":null},{"url":null,"slug":"losing-visual-needles-in-image-haystacks","title":"Losing Visual Needles in Image Haystacks: Vision Language Models are Easily Distracted in Short and Long Contexts","date":"2024-06-24","arxiv_id":"2406.16851","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-large-vision-and-language-models","title":"Evaluating Large Vision-and-Language Models on Children's Mathematical Olympiads","date":"2024-06-22","arxiv_id":"2406.15736","repositories_listed":0,"syntology":null},{"url":null,"slug":"codegemma-open-code-models-based-on-gemma","title":"CodeGemma: Open Code Models Based on Gemma","date":"2024-06-17","arxiv_id":"2406.11409","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-the-achilles-heel-evaluating-llms","title":"Exposing the Achilles' Heel: Evaluating LLMs Ability to Handle Mistakes in Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10834","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-or-simply-next-token-prediction-a","title":"MMLU-SR: A Benchmark for Stress-Testing Reasoning Capability of Large Language Models","date":"2024-06-15","arxiv_id":"2406.15468","repositories_listed":0,"syntology":null},{"url":null,"slug":"me-switch-a-memory-efficient-expert-switching","title":"ME-Switch: A Memory-Efficient Expert Switching Framework for Large Language Models","date":"2024-06-13","arxiv_id":"2406.09041","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-assessment-of-mathematical","title":"Robustness Assessment of Mathematical Reasoning in the Presence of Missing and Contradictory Conditions","date":"2024-06-07","arxiv_id":"2406.05055","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-emergent-symbolic-reasoning","title":"Assessing the Emergent Symbolic Reasoning Abilities of Llama Large Language Models","date":"2024-06-05","arxiv_id":"2406.06588","repositories_listed":0,"syntology":null}],"record_sha256":"1902b537f03326d08226df36003df925b8c04f54c47a7bb49ee5cde121773b43","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}