{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/12","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":16,"rows_per_page":100,"rows":[1101,1200],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/11","next":"/task/math/papers/13","papers":[{"url":null,"slug":"fine-grained-hallucination-detection-and-2","title":"FG-PRM: Fine-grained Hallucination Detection and Mitigation in Language Model Mathematical Reasoning","date":"2024-10-08","arxiv_id":"2410.06304","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-pontryagin-s-maximum-principle-all-you","title":"Solving Functional Optimization with Deep Networks and Variational Principles","date":"2024-10-08","arxiv_id":"2410.06277","repositories_listed":0,"syntology":null},{"url":null,"slug":"fplsa-learning-semantic-structures-in","title":"fPLSA: Learning Semantic Structures in Document Collections Using Foundation Models","date":"2024-10-07","arxiv_id":"2410.05481","repositories_listed":0,"syntology":null},{"url":null,"slug":"intriguing-properties-of-large-language-and","title":"Intriguing Properties of Large Language and Vision Models","date":"2024-10-07","arxiv_id":"2410.04751","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-paths-optimization-learning-to","title":"Reasoning Paths Optimization: Learning to Reason and Explore From Diverse Paths","date":"2024-10-07","arxiv_id":"2410.10858","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-based-data-selection-for-large-language","title":"Rule-based Data Selection for Large Language Models","date":"2024-10-07","arxiv_id":"2410.04715","repositories_listed":0,"syntology":null},{"url":null,"slug":"bloomwise-enhancing-problem-solving","title":"BloomWise: Enhancing Problem-Solving capabilities of Large Language Models using Bloom's-Taxonomy-Inspired Prompts","date":"2024-10-05","arxiv_id":"2410.04094","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-llm-reasoning-through-scaling","title":"Improving LLM Reasoning through Scaling Inference Computation with Collaborative Verification","date":"2024-10-05","arxiv_id":"2410.05318","repositories_listed":0,"syntology":null},{"url":null,"slug":"deliberate-reasoning-for-llms-as-structure","title":"Deliberate Reasoning for LLMs as Structure-aware Planning with Accurate World Model","date":"2024-10-04","arxiv_id":"2410.03136","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-inference-time-compute-llms-can","title":"Adaptive Inference-Time Compute: LLMs Can Predict if They Can Do Better, Even Mid-Generation","date":"2024-10-03","arxiv_id":"2410.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"codepmp-scalable-preference-model-pretraining","title":"CodePMP: Scalable Preference Model Pretraining for Large Language Model Reasoning","date":"2024-10-03","arxiv_id":"2410.02229","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometry-is-all-you-need-a-unified-taxonomy","title":"Geometry is All You Need: A Unified Taxonomy of Matrix and Tensor Factorization for Compression of Generative Language Models","date":"2024-10-03","arxiv_id":"2410.03040","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-further-elicit-reasoning-in-llms","title":"Can We Further Elicit Reasoning in LLMs? Critic-Guided Planning with Retrieval-Augmentation for Solving Challenging Tasks","date":"2024-10-02","arxiv_id":"2410.01428","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-knowledge-tracing-for-personalized","title":"Deep Knowledge Tracing for Personalized Adaptive Learning at Historically Black Colleges and Universities","date":"2024-10-02","arxiv_id":"2410.13876","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-reward-models-for","title":"Evaluating Robustness of Reward Models for Mathematical Reasoning","date":"2024-10-02","arxiv_id":"2410.01729","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-swapping-for-zero-shot-cross-lingual","title":"Layer Swapping for Zero-Shot Cross-Lingual Transfer in Large Language Models","date":"2024-10-02","arxiv_id":"2410.01335","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-all-llm-reasoners-are-created-equal","title":"Not All LLM Reasoners Are Created Equal","date":"2024-10-02","arxiv_id":"2410.01748","repositories_listed":0,"syntology":null},{"url":null,"slug":"personamath-enhancing-math-reasoning-through","title":"PersonaMath: Enhancing Math Reasoning through Persona-Driven Data Augmentation","date":"2024-10-02","arxiv_id":"2410.01504","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-by-step-reasoning-for-math-problems-via","title":"Step-by-Step Reasoning for Math Problems via Twisted Sequential Monte Carlo","date":"2024-10-02","arxiv_id":"2410.01920","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-adaptive-zero-shot-chain-of-thought","title":"Instance-adaptive Zero-shot Chain-of-Thought Prompting","date":"2024-09-30","arxiv_id":"2409.20441","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-perfect-blend-redefining-rlhf-with","title":"The Perfect Blend: Redefining RLHF with Mixture of Judges","date":"2024-09-30","arxiv_id":"2409.20370","repositories_listed":0,"syntology":null},{"url":null,"slug":"metamath-integrating-natural-language-and","title":"INC-Math: Integrating Natural Language and Code for Enhanced Mathematical Reasoning in Large Language Models","date":"2024-09-28","arxiv_id":"2409.19381","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-inductive-bias-of-stacking-towards","title":"On the Inductive Bias of Stacking Towards Improving Reasoning","date":"2024-09-27","arxiv_id":"2409.19044","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-superficial-alignment","title":"Revisiting the Superficial Alignment Hypothesis","date":"2024-09-27","arxiv_id":"2410.03717","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-love-edge-cases-in-formative-math","title":"Learning to Love Edge Cases in Formative Math Assessment: Using the AMMORE Dataset and Chain-of-Thought Prompting to Improve Grading Accuracy","date":"2024-09-26","arxiv_id":"2409.17904","repositories_listed":0,"syntology":null},{"url":null,"slug":"democratizing-signal-processing-and-machine","title":"Democratizing Signal Processing and Machine Learning: Math Learning Equity for Elementary and Middle School Students","date":"2024-09-25","arxiv_id":"2409.17304","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-sciq-an-educational-chatbot-for","title":"LLaMa-SciQ: An Educational Chatbot for Answering Science MCQ","date":"2024-09-25","arxiv_id":"2409.16779","repositories_listed":0,"syntology":null},{"url":null,"slug":"models-can-and-should-embrace-the","title":"Models Can and Should Embrace the Communicative Nature of Human-Generated Math","date":"2024-09-25","arxiv_id":"2409.17005","repositories_listed":0,"syntology":null},{"url":null,"slug":"pmss-pretrained-matrices-skeleton-selection","title":"PMSS: Pretrained Matrices Skeleton Selection for LLM Fine-tuning","date":"2024-09-25","arxiv_id":"2409.16722","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlmath-controllable-data-generation","title":"ControlMath: Controllable Data Generation Promotes Math Generalist Models","date":"2024-09-20","arxiv_id":"2409.15376","repositories_listed":0,"syntology":null},{"url":null,"slug":"infimm-webmath-40b-advancing-multimodal-pre","title":"InfiMM-WebMath-40B: Advancing Multimodal Pre-Training for Enhanced Mathematical Reasoning","date":"2024-09-19","arxiv_id":"2409.12568","repositories_listed":0,"syntology":null},{"url":null,"slug":"grin-gradient-informed-moe","title":"GRIN: GRadient-INformed MoE","date":"2024-09-18","arxiv_id":"2409.12136","repositories_listed":0,"syntology":null},{"url":"/paper/qwen2-5-math-technical-report-toward","slug":"qwen2-5-math-technical-report-toward","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","date":"2024-09-18","arxiv_id":"2409.12122","repositories_listed":0,"syntology":null},{"url":null,"slug":"nvlm-open-frontier-class-multimodal-llms","title":"NVLM: Open Frontier-Class Multimodal LLMs","date":"2024-09-17","arxiv_id":"2409.11402","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-takes-the-sat-tracing-changes-in-test","title":"GPT takes the SAT: Tracing changes in Test Difficulty and Math Performance of Students","date":"2024-09-16","arxiv_id":"2409.10750","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpl-critical-planning-step-learning-boosts","title":"CPL: Critical Plan Step Learning Boosts LLM Generalization in Reasoning Tasks","date":"2024-09-13","arxiv_id":"2409.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"cracking-the-code-multi-domain-llm-evaluation","title":"Cracking the Code: Multi-domain LLM Evaluation on Real-World Professional Exams in Indonesia","date":"2024-09-13","arxiv_id":"2409.08564","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-with-preference-optimization-is-all","title":"Alignment with Preference Optimization Is All You Need for LLM Safety","date":"2024-09-12","arxiv_id":"2409.07772","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-tagging-with-large-language-model","title":"Knowledge Tagging with Large Language Model based Multi-Agent System","date":"2024-09-12","arxiv_id":"2409.08406","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-unstructured-text-data-for","title":"Leveraging Unstructured Text Data for Federated Instruction Tuning of Large Language Models","date":"2024-09-11","arxiv_id":"2409.07136","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-practice-of-post-training-on-llama-3-70b","title":"A Practice of Post-Training on Llama-3 70B with Optimal Selection of Additional Language Mixture Ratio","date":"2024-09-10","arxiv_id":"2409.06624","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-math-agents-with-multi-turn","title":"Building Math Agents with Multi-Turn Iterative Preference Learning","date":"2024-09-04","arxiv_id":"2409.02392","repositories_listed":0,"syntology":null},{"url":null,"slug":"deconfounded-causality-aware-parameter","title":"Deconfounded Causality-aware Parameter-Efficient Fine-Tuning for Problem-Solving Improvement of LLMs","date":"2024-09-04","arxiv_id":"2409.02686","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-baking","title":"Prompt Baking","date":"2024-09-04","arxiv_id":"2409.13697","repositories_listed":0,"syntology":null},{"url":null,"slug":"waveletgpt-wavelets-meet-large-language","title":"Wavelet GPT: Wavelet Inspired Large Language Models","date":"2024-09-04","arxiv_id":"2409.12924","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-3-c-math-spontaneous-step-level-self","title":"S$^3$c-Math: Spontaneous Step-level Self-correction Makes Large Language Models Better Mathematical Reasoners","date":"2024-09-03","arxiv_id":"2409.01524","repositories_listed":0,"syntology":null},{"url":null,"slug":"critic-cot-boosting-the-reasoning-abilities","title":"Critic-CoT: Boosting the reasoning abilities of large language model via Chain-of-thoughts Critic","date":"2024-08-29","arxiv_id":"2408.16326","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropic-distribution-matching-in-supervised","title":"Entropic Distribution Matching in Supervised Fine-tuning of LLMs: Less Overfitting and Better Diversity","date":"2024-08-29","arxiv_id":"2408.16673","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-contrastive-reasoning-with-lightweight","title":"Logic Contrastive Reasoning with Lightweight Large Language Model for Math Word Problems","date":"2024-08-29","arxiv_id":"2409.00131","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-of-language-models-part-2-2-how-to","title":"Physics of Language Models: Part 2.2, How to Learn From Mistakes on Grade-School Math Problems","date":"2024-08-29","arxiv_id":"2408.16293","repositories_listed":0,"syntology":null},{"url":null,"slug":"siam-self-improving-code-assisted","title":"SIaM: Self-Improving Code-Assisted Mathematical Reasoning of Large Language Models","date":"2024-08-28","arxiv_id":"2408.15565","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-verifiers-reward-modeling-as-next","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","date":"2024-08-27","arxiv_id":"2408.15240","repositories_listed":0,"syntology":null},{"url":null,"slug":"students-perceived-roles-opportunities-and","title":"Students' Perceived Roles, Opportunities, and Challenges of a Generative AI-powered Teachable Agent: A Case of Middle School Math Class","date":"2024-08-26","arxiv_id":"2409.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-tool-integration-application-for-math","title":"Multi-tool Integration Application for Math Reasoning Using Large Language Model","date":"2024-08-22","arxiv_id":"2408.12148","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathematical-information-retrieval-search-and","title":"Mathematical Information Retrieval: Search and Question Answering","date":"2024-08-21","arxiv_id":"2408.11646","repositories_listed":0,"syntology":null},{"url":null,"slug":"qpo-query-dependent-prompt-optimization-via","title":"QPO: Query-dependent Prompt Optimization via Multi-Loop Offline Reinforcement Learning","date":"2024-08-20","arxiv_id":"2408.10504","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-phoc-spatial-region-configurations","title":"A Study of PHOC Spatial Region Configurations for Math Formula Retrieval","date":"2024-08-17","arxiv_id":"2408.09283","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-might-not-care-what-you","title":"Large Language Models Might Not Care What You Are Saying: Prompt Format Beats Descriptions","date":"2024-08-16","arxiv_id":"2408.08780","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-reasoning-emerge-examining-the","title":"Does Reasoning Emerge? Examining the Probabilities of Causation in Large Language Models","date":"2024-08-15","arxiv_id":"2408.08210","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-perspective-on-large-language-models","title":"A Perspective on Large Language Models, Intelligent Machines, and Knowledge Acquisition","date":"2024-08-13","arxiv_id":"2408.06598","repositories_listed":0,"syntology":null},{"url":null,"slug":"p3-a-policy-driven-pace-adaptive-and","title":"P3: A Policy-Driven, Pace-Adaptive, and Diversity-Promoted Framework for data pruning in LLM Training","date":"2024-08-10","arxiv_id":"2408.05541","repositories_listed":0,"syntology":null},{"url":null,"slug":"examining-the-behavior-of-llm-architectures","title":"Examining the Behavior of LLM Architectures Within the Framework of Standardized National Exams in Brazil","date":"2024-08-09","arxiv_id":"2408.05035","repositories_listed":0,"syntology":null},{"url":null,"slug":"altcanvas-a-tile-based-image-editor-with","title":"AltCanvas: A Tile-Based Image Editor with Generative AI for Blind or Visually Impaired People","date":"2024-08-05","arxiv_id":"2408.10240","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-logic-of-political-survival-revisited","title":"The Logic of Political Survival Revisited: Consequences of Elite Uncertainty Under Authoritarian Rule","date":"2024-08-04","arxiv_id":"2408.01887","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-effective-and-efficient-continual-pre","title":"Towards Effective and Efficient Continual Pre-training of Large Language Models","date":"2024-07-26","arxiv_id":"2407.18743","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-introspection-teaching-language","title":"Recursive Introspection: Teaching Language Model Agents How to Self-Improve","date":"2024-07-25","arxiv_id":"2407.18219","repositories_listed":0,"syntology":null},{"url":"/paper/generalization-v-s-memorization-tracing","slug":"generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","repositories_listed":0,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/generalization-v-s-memorization-tracing#ran","syntology_url":"https://syntology.ai/paper/2407.14985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14985"}},"official":null}},{"url":null,"slug":"a-llm-benchmark-based-on-the-minecraft","title":"A LLM Benchmark based on the Minecraft Builder Dialog Agent Task","date":"2024-07-17","arxiv_id":"2407.12734","repositories_listed":0,"syntology":null},{"url":null,"slug":"ccoe-a-compact-llm-with-collaboration-of","title":"CCoE: A Compact LLM with Collaboration of Experts","date":"2024-07-16","arxiv_id":"2407.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-with-large-language-models-a-survey","title":"Reasoning with Large Language Models, a Survey","date":"2024-07-16","arxiv_id":"2407.11511","repositories_listed":0,"syntology":null},{"url":null,"slug":"telecomgpt-a-framework-to-build-telecom","title":"TelecomGPT: A Framework to Build Telecom-Specfic Large Language Models","date":"2024-07-12","arxiv_id":"2407.09424","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-supervised-value-models-for-enhancing","title":"Token-Supervised Value Models for Enhancing Mathematical Reasoning Capabilities of Large Language Models","date":"2024-07-12","arxiv_id":"2407.12863","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-your-model-really-a-good-math-reasoner","title":"Is Your Model Really A Good Math Reasoner? Evaluating Mathematical Reasoning with Checklist","date":"2024-07-11","arxiv_id":"2407.08733","repositories_listed":0,"syntology":null},{"url":null,"slug":"skywork-math-data-scaling-laws-for","title":"Skywork-Math: Data Scaling Laws for Mathematical Reasoning in Large Language Models -- The Story Goes On","date":"2024-07-11","arxiv_id":"2407.08348","repositories_listed":0,"syntology":null},{"url":null,"slug":"convnlp-image-based-ai-text-detection","title":"ConvNLP: Image-based AI Text Detection","date":"2024-07-09","arxiv_id":"2407.07225","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-process-verification-for-large","title":"Advancing Process Verification for Large Language Models via Tree-Based Preference Learning","date":"2024-06-29","arxiv_id":"2407.00390","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmmath-a-chinese-multi-modal-math-skill","title":"CMMaTH: A Chinese Multi-modal Math Skill Evaluation Benchmark for Foundation Models","date":"2024-06-28","arxiv_id":"2407.12023","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalebio-scalable-bilevel-optimization-for","title":"ScaleBiO: Scalable Bilevel Optimization for LLM Data Reweighting","date":"2024-06-28","arxiv_id":"2406.19976","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-in-domain-data-augmentation","title":"Task Oriented In-Domain Data Augmentation","date":"2024-06-24","arxiv_id":"2406.16694","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ai-for-enhancing-active-learning","title":"Generative AI for Enhancing Active Learning in Education: A Comparative Study of GPT-3.5 and GPT-4 in Crafting Customized Test Questions","date":"2024-06-20","arxiv_id":"2406.13903","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-improving-multi-step-reasoning-for-llms","title":"Q*: Improving Multi-step Reasoning for LLMs with Deliberative Planning","date":"2024-06-20","arxiv_id":"2406.14283","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-tagging-system-on-math-questions","title":"Knowledge Tagging System on Math Questions via LLMs with Flexible Demonstration Retriever","date":"2024-06-19","arxiv_id":"2406.13885","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-the-labyrinth-evaluating-and","title":"Navigating the Labyrinth: Evaluating and Enhancing LLMs' Ability to Reason About Search Problems","date":"2024-06-18","arxiv_id":"2406.12172","repositories_listed":0,"syntology":null},{"url":null,"slug":"program-synthesis-benchmark-for-visual","title":"Program Synthesis Benchmark for Visual Programming in XLogoOnline Environment","date":"2024-06-17","arxiv_id":"2406.11334","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-moe-towards-compositional-large-language","title":"Self-MoE: Towards Compositional Large Language Models with Self-Specialized Experts","date":"2024-06-17","arxiv_id":"2406.12034","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-the-achilles-heel-evaluating-llms","title":"Exposing the Achilles' Heel: Evaluating LLMs Ability to Handle Mistakes in Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10834","repositories_listed":0,"syntology":null},{"url":null,"slug":"clst-cold-start-mitigation-in-knowledge","title":"CLST: Cold-Start Mitigation in Knowledge Tracing by Aligning a Generative Language Model as a Students' Knowledge Tracer","date":"2024-06-13","arxiv_id":"2406.10296","repositories_listed":0,"syntology":null},{"url":null,"slug":"remi-a-dataset-for-reasoning-with-multiple","title":"ReMI: A Dataset for Reasoning with Multiple Images","date":"2024-06-13","arxiv_id":"2406.09175","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-i-understand-what-i-create-self-knowledge","title":"Can I understand what I create? Self-Knowledge Evaluation of Large Language Models","date":"2024-06-10","arxiv_id":"2406.06140","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-learning-about-ai-performance","title":"Human Learning about AI","date":"2024-06-08","arxiv_id":"2406.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-core-periphery-perspective-ranking","title":"A multi-core periphery perspective: Ranking via relative centrality","date":"2024-06-06","arxiv_id":"2406.04487","repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-mathematical-reasoning-in-language","title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","date":"2024-06-05","arxiv_id":"2406.06592","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-cpt-law-domain-specific-continual-pre","title":"D-CPT Law: Domain-specific Continual Pre-Training Scaling Law for Large Language Models","date":"2024-06-03","arxiv_id":"2406.01375","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-pretraining-improves-entity-tracking","title":"Code Pretraining Improves Entity Tracking Abilities of Language Models","date":"2024-05-31","arxiv_id":"2405.21068","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-meets-consensus-unleashing","title":"Divide-and-Conquer Meets Consensus: Unleashing the Power of Functions in Code Generation","date":"2024-05-30","arxiv_id":"2405.20092","repositories_listed":0,"syntology":null},{"url":null,"slug":"arithmetic-reasoning-with-llm-prolog","title":"Arithmetic Reasoning with LLM: Prolog Generation & Permutation","date":"2024-05-28","arxiv_id":"2405.17893","repositories_listed":0,"syntology":null},{"url":null,"slug":"mindstar-enhancing-math-reasoning-in-pre","title":"MindStar: Enhancing Math Reasoning in Pre-trained LLMs at Inference Time","date":"2024-05-25","arxiv_id":"2405.16265","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-beyond-pattern-matching-assaying","title":"Learning Beyond Pattern Matching? Assaying Mathematical Understanding in LLMs","date":"2024-05-24","arxiv_id":"2405.15485","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-can-self-correct-with","title":"Large Language Models Can Self-Correct with Key Condition Verification","date":"2024-05-23","arxiv_id":"2405.14092","repositories_listed":0,"syntology":null},{"url":null,"slug":"turing-tests-for-an-ai-scientist","title":"\"Turing Tests\" For An AI Scientist","date":"2024-05-22","arxiv_id":"2405.13352","repositories_listed":0,"syntology":null}],"record_sha256":"41f04b59b128caab05d53e83041e43f5c1ab2d0adc140d2cfe3b55cf4756591f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}