{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/arithmetic-reasoning/papers/2","list_of":"/task/arithmetic-reasoning","task":"Arithmetic Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,175],"of":175,"counts":{"archive_papers_tagged":175,"with_a_code_link":112,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":175,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":53,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":53,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/arithmetic-reasoning","prev":"/task/arithmetic-reasoning","next":null,"papers":[{"url":"/paper/do-deep-neural-networks-capture","slug":"do-deep-neural-networks-capture","title":"Do Deep Neural Networks Capture Compositionality in Arithmetic Reasoning?","date":"2023-02-15","arxiv_id":"2302.07866","repositories_listed":1,"syntology":null},{"url":"/paper/is-chatgpt-a-general-purpose-natural-language","slug":"is-chatgpt-a-general-purpose-natural-language","title":"Is ChatGPT a General-Purpose Natural Language Processing Task Solver?","date":"2023-02-08","arxiv_id":"2302.06476","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-can-be-easily","slug":"large-language-models-can-be-easily","title":"Large Language Models Can Be Easily Distracted by Irrelevant Context","date":"2023-01-31","arxiv_id":"2302.00093","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-reasoners-with-self","slug":"large-language-models-are-reasoners-with-self","title":"Large Language Models are Better Reasoners with Self-Verification","date":"2022-12-19","arxiv_id":"2212.09561","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/large-language-models-are-reasoners-with-self#ran","syntology_url":"https://syntology.ai/paper/2212.09561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09561"}},"official":{"repos":["WENGSYX/Self-Verification"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/overcoming-barriers-to-skill-injection-in","slug":"overcoming-barriers-to-skill-injection-in","title":"Overcoming Barriers to Skill Injection in Language Modeling: Case Study in Arithmetic","date":"2022-11-03","arxiv_id":"2211.02098","repositories_listed":1,"syntology":null},{"url":"/paper/solving-math-word-problem-via-cooperative","slug":"solving-math-word-problem-via-cooperative","title":"Solving Math Word Problems via Cooperative Reasoning induced Language Models","date":"2022-10-28","arxiv_id":"2210.16257","repositories_listed":1,"syntology":null},{"url":"/paper/opencqa-open-ended-question-answering-with","slug":"opencqa-open-ended-question-answering-with","title":"OpenCQA: Open-ended Question Answering with Charts","date":"2022-10-12","arxiv_id":"2210.06628","repositories_listed":1,"syntology":null},{"url":"/paper/solving-quantitative-reasoning-problems-with","slug":"solving-quantitative-reasoning-problems-with","title":"Solving Quantitative Reasoning Problems with Language Models","date":"2022-06-29","arxiv_id":"2206.14858","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-self-sampled-correct-and","slug":"learning-from-self-sampled-correct-and","title":"Learning Math Reasoning from Self-Sampled Correct and Partially-Correct Solutions","date":"2022-05-28","arxiv_id":"2205.14318","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-from-self-sampled-correct-and#ran","syntology_url":"https://syntology.ai/paper/2205.14318","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14318"}},"official":{"repos":["microsoft/tracecodegen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/least-to-most-prompting-enables-complex","slug":"least-to-most-prompting-enables-complex","title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","date":"2022-05-21","arxiv_id":"2205.10625","repositories_listed":1,"syntology":null},{"url":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/iconqa-a-new-benchmark-for-abstract-diagram#ran","syntology_url":"https://syntology.ai/paper/2110.13214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13214"}},"official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/inter-gps-interpretable-geometry-problem","slug":"inter-gps-interpretable-geometry-problem","title":"Inter-GPS: Interpretable Geometry Problem Solving with Formal Language and Symbolic Reasoning","date":"2021-05-10","arxiv_id":"2105.04165","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inter-gps-interpretable-geometry-problem#ran","syntology_url":"https://syntology.ai/paper/2105.04165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04165"}},"official":{"repos":["lupantech/InterGPS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":null,"slug":"finlmm-r1-enhancing-financial-reasoning-in","title":"FinLMM-R1: Enhancing Financial Reasoning in LMM through Scalable Data and Reward Design","date":"2025-06-16","arxiv_id":"2506.13066","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-at-criticality-in-large-language","title":"Learning-at-Criticality in Large Language Models for Quantum Field Theory and Beyond","date":"2025-06-04","arxiv_id":"2506.03703","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualsphinx-large-scale-synthetic-vision","title":"VisualSphinx: Large-Scale Synthetic Vision Logic Puzzles for RL","date":"2025-05-29","arxiv_id":"2505.23977","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-flashback-adaptation-for-forgetting","title":"Joint Flashback Adaptation for Forgetting-Resistant Instruction Tuning","date":"2025-05-21","arxiv_id":"2505.15467","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenization-constraints-in-llms-a-study-of","title":"Tokenization Constraints in LLMs: A Study of Symbolic and Arithmetic Reasoning Limits","date":"2025-05-20","arxiv_id":"2505.14178","repositories_listed":0,"syntology":null},{"url":null,"slug":"fact-consistency-evaluation-of-text-to-sql","title":"Fact-Consistency Evaluation of Text-to-SQL Generation for Business Intelligence Using Exaone 3.5","date":"2025-04-30","arxiv_id":"2505.00060","repositories_listed":0,"syntology":null},{"url":null,"slug":"thoughtprobe-classifier-guided-thought-space","title":"ThoughtProbe: Classifier-Guided Thought Space Exploration Leveraging LLM Intrinsic Reasoning","date":"2025-04-09","arxiv_id":"2504.06650","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-language-model-may-think-too-rigidly","title":"Your Language Model May Think Too Rigidly: Achieving Reasoning Consistency with Symmetry-Enhanced Training","date":"2025-02-25","arxiv_id":"2502.17800","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-lottery-llm-hypothesis-rethinking-what","title":"The Lottery LLM Hypothesis, Rethinking What Abilities Should LLM Compression Preserve?","date":"2025-02-24","arxiv_id":"2502.17535","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-time-computations-for-llm-reasoning","title":"Inference-Time Computations for LLM Reasoning and Planning: A Benchmark and Insights","date":"2025-02-18","arxiv_id":"2502.12521","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-representational-dissociation-of-language","title":"On Representational Dissociation of Language and Arithmetic in Large Language Models","date":"2025-02-17","arxiv_id":"2502.11932","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-vision-language-models-struggle-with","title":"Why Vision Language Models Struggle with Visual Arithmetic? Towards Enhanced Chart and Geometry Understanding","date":"2025-02-17","arxiv_id":"2502.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-maintain-fundamental-abilities-under","title":"Can LLMs Maintain Fundamental Abilities under KV Cache Compression?","date":"2025-02-04","arxiv_id":"2502.01941","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloq-enhancing-fine-tuning-of-quantized-llms","title":"CLoQ: Enhancing Fine-Tuning of Quantized LLMs via Calibrated LoRA Initialization","date":"2025-01-30","arxiv_id":"2501.18475","repositories_listed":0,"syntology":null},{"url":null,"slug":"sft-memorizes-rl-generalizes-a-comparative","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","date":"2025-01-28","arxiv_id":"2501.17161","repositories_listed":0,"syntology":null},{"url":null,"slug":"dota-weight-decomposed-tensor-adaptation-for","title":"DoTA: Weight-Decomposed Tensor Adaptation for Large Language Models","date":"2024-12-30","arxiv_id":"2412.20891","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-intrinsic-self-correction-enhancement","title":"Towards Intrinsic Self-Correction Enhancement in Monte Carlo Tree Search Boosted Reasoning via Iterative Preference Learning","date":"2024-12-23","arxiv_id":"2412.17397","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-we-build-local-large-language-models-an","title":"Why We Build Local Large Language Models: An Observational Analysis from 35 Japanese and Multilingual LLMs","date":"2024-12-19","arxiv_id":"2412.14471","repositories_listed":0,"syntology":null},{"url":null,"slug":"hint-marginalization-for-improved-reasoning","title":"Hint Marginalization for Improved Reasoning in Large Language Models","date":"2024-12-17","arxiv_id":"2412.13292","repositories_listed":0,"syntology":null},{"url":null,"slug":"galore-boosting-low-rank-adaptation-for-llms","title":"GaLore$+$: Boosting Low-Rank Adaptation for LLMs with Cross-Head Projection","date":"2024-12-15","arxiv_id":"2412.19820","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-2-ft-efficient-scalable-and-generalizable","title":"S$^{2}$FT: Efficient, Scalable and Generalizable LLM Fine-tuning by Structured Sparsity","date":"2024-12-09","arxiv_id":"2412.06289","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-to-talk-or-talk-to-think-when-llms-come","title":"Think-to-Talk or Talk-to-Think? When LLMs Come Up with an Answer in Multi-Step Arithmetic Reasoning","date":"2024-12-02","arxiv_id":"2412.01113","repositories_listed":0,"syntology":null},{"url":null,"slug":"perft-parameter-efficient-routed-fine-tuning","title":"PERFT: Parameter-Efficient Routed Fine-Tuning for Mixture-of-Expert Model","date":"2024-11-12","arxiv_id":"2411.08212","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-beyond-size-dynamic-prompting-for-more","title":"Think Beyond Size: Adaptive Prompting for More Effective Reasoning","date":"2024-10-10","arxiv_id":"2410.08130","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-structured-thinking-in-language","title":"Unlocking Structured Thinking in Language Models with Cognitive Prompting","date":"2024-10-03","arxiv_id":"2410.02953","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-language-models-are-equation-reasoners","title":"Small Language Models are Equation Reasoners","date":"2024-09-19","arxiv_id":"2409.12393","repositories_listed":0,"syntology":null},{"url":null,"slug":"relating-the-seemingly-unrelated-principled","title":"Relating the Seemingly Unrelated: Principled Understanding of Generalization for Generative Models in Arithmetic Reasoning Tasks","date":"2024-07-25","arxiv_id":"2407.17963","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00802","title":"Leveraging LLM Reasoning Enhances Personalized Recommender Systems","date":"2024-07-22","arxiv_id":"2408.00802","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-and-prompt-optimization-two-great","title":"Fine-Tuning and Prompt Optimization: Two Great Steps that Work Better Together","date":"2024-07-15","arxiv_id":"2407.10930","repositories_listed":0,"syntology":null},{"url":null,"slug":"arithmetic-reasoning-with-llm-prolog","title":"Arithmetic Reasoning with LLM: Prolog Generation & Permutation","date":"2024-05-28","arxiv_id":"2405.17893","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-can-self-correct-with","title":"Large Language Models Can Self-Correct with Key Condition Verification","date":"2024-05-23","arxiv_id":"2405.14092","repositories_listed":0,"syntology":null},{"url":null,"slug":"skin-in-the-game-decision-making-via-multi","title":"Skin-in-the-Game: Decision Making via Multi-Stakeholder Alignment in LLMs","date":"2024-05-21","arxiv_id":"2405.12933","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-high-sparsity-foundational-llama","title":"Enabling High-Sparsity Foundational Llama Models with Efficient Pretraining and Deployment","date":"2024-05-06","arxiv_id":"2405.03594","repositories_listed":0,"syntology":null},{"url":"/paper/the-claude-3-model-family-opus-sonnet-haiku","slug":"the-claude-3-model-family-opus-sonnet-haiku","title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","date":"2024-03-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"symba-symbolic-backward-chaining-for-multi","title":"SymBa: Symbolic Backward Chaining for Structured Natural Language Reasoning","date":"2024-02-20","arxiv_id":"2402.12806","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-llms-mathematical-reasoning-in","title":"Evaluating LLMs' Mathematical Reasoning in Financial Document Question Answering","date":"2024-02-17","arxiv_id":"2402.11194","repositories_listed":0,"syntology":null},{"url":"/paper/orca-math-unlocking-the-potential-of-slms-in","slug":"orca-math-unlocking-the-potential-of-slms-in","title":"Orca-Math: Unlocking the potential of SLMs in Grade School Math","date":"2024-02-16","arxiv_id":"2402.14830","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-group-and-symmetry-principles-in","title":"Exploring Group and Symmetry Principles in Large Language Models","date":"2024-02-09","arxiv_id":"2402.06120","repositories_listed":0,"syntology":null},{"url":"/paper/the-unreasonable-effectiveness-of-eccentric","slug":"the-unreasonable-effectiveness-of-eccentric","title":"The Unreasonable Effectiveness of Eccentric Automatic Prompts","date":"2024-02-09","arxiv_id":"2402.10949","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-gender-bias-in-large-language","title":"Evaluating Gender Bias in Large Language Models via Chain-of-Thought Prompting","date":"2024-01-28","arxiv_id":"2401.15585","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-are-null-shot-learners","title":"Large Language Models are Null-Shot Learners","date":"2024-01-16","arxiv_id":"2401.08273","repositories_listed":0,"syntology":null},{"url":"/paper/boosting-llm-reasoning-push-the-limits-of-few","slug":"boosting-llm-reasoning-push-the-limits-of-few","title":"Fewer is More: Boosting LLM Reasoning with Reinforced Context Pruning","date":"2023-12-14","arxiv_id":"2312.08901","repositories_listed":0,"syntology":null},{"url":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","slug":"tinygsm-achieving-80-on-gsm8k-with-small","title":"TinyGSM: achieving >80% on GSM8k with small language models","date":"2023-12-14","arxiv_id":"2312.09241","repositories_listed":0,"syntology":null},{"url":"/paper/orca-2-teaching-small-language-models-how-to","slug":"orca-2-teaching-small-language-models-how-to","title":"Orca 2: Teaching Small Language Models How to Reason","date":"2023-11-18","arxiv_id":"2311.11045","repositories_listed":0,"syntology":null},{"url":"/paper/the-art-of-llm-refinement-ask-refine-and","slug":"the-art-of-llm-refinement-ask-refine-and","title":"The ART of LLM Refinement: Ask, Refine, and Trust","date":"2023-11-14","arxiv_id":"2311.07961","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-sketching-for-large-language-models","title":"Prompt Sketching for Large Language Models","date":"2023-11-08","arxiv_id":"2311.04954","repositories_listed":0,"syntology":null},{"url":"/paper/kwaiyiimath-technical-report","slug":"kwaiyiimath-technical-report","title":"KwaiYiiMath: Technical Report","date":"2023-10-11","arxiv_id":"2310.07488","repositories_listed":0,"syntology":null},{"url":"/paper/model-card-and-evaluations-for-claude-models","slug":"model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"gkd-generalized-knowledge-distillation-for","title":"On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes","date":"2023-06-23","arxiv_id":"2306.13649","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversigate-a-comprehensive-framework-for","title":"DiversiGATE: A Comprehensive Framework for Reliable Large Language Models","date":"2023-06-22","arxiv_id":"2306.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-prompting-a-neural-symbolic-method-for","title":"Code Prompting: a Neural Symbolic Method for Complex Reasoning in Large Language Models","date":"2023-05-29","arxiv_id":"2305.18507","repositories_listed":0,"syntology":null},{"url":null,"slug":"rcot-detecting-and-rectifying-factual","title":"RCOT: Detecting and Rectifying Factual Inconsistency in Reasoning by Reversing Chain-of-Thought","date":"2023-05-19","arxiv_id":"2305.11499","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfzcot-a-self-prompt-zero-shot-cot-from","title":"Hint of Thought prompting: an explainable and zero-shot approach to reasoning tasks with LLMs","date":"2023-05-19","arxiv_id":"2305.11461","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evaluation-guided-beam-search-for-1","title":"Self-Evaluation Guided Beam Search for Reasoning","date":"2023-05-01","arxiv_id":"2305.00633","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-do-you-need-chain-of-thought-prompting","title":"When do you need Chain-of-Thought Prompting for ChatGPT?","date":"2023-04-06","arxiv_id":"2304.03262","repositories_listed":0,"syntology":null},{"url":"/paper/solving-math-word-problems-with-process-and","slug":"solving-math-word-problems-with-process-and","title":"Solving math word problems with process- and outcome-based feedback","date":"2022-11-25","arxiv_id":"2211.14275","repositories_listed":0,"syntology":null},{"url":"/paper/composing-ensembles-of-pre-trained-models-via","slug":"composing-ensembles-of-pre-trained-models-via","title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","date":"2022-10-20","arxiv_id":"2210.11522","repositories_listed":0,"syntology":null},{"url":"/paper/large-language-models-can-self-improve","slug":"large-language-models-can-self-improve","title":"Large Language Models Can Self-Improve","date":"2022-10-20","arxiv_id":"2210.11610","repositories_listed":0,"syntology":null},{"url":"/paper/transcending-scaling-laws-with-0-1-extra","slug":"transcending-scaling-laws-with-0-1-extra","title":"Transcending Scaling Laws with 0.1% Extra Compute","date":"2022-10-20","arxiv_id":"2210.11399","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-symbolic-recursive-machine-for","title":"Neural-Symbolic Recursive Machine for Systematic Generalization","date":"2022-10-04","arxiv_id":"2210.01603","repositories_listed":0,"syntology":null},{"url":"/paper/0-1-deep-neural-networks-via-block-coordinate","slug":"0-1-deep-neural-networks-via-block-coordinate","title":"0/1 Deep Neural Networks via Block Coordinate Descent","date":"2022-06-19","arxiv_id":"2206.09379","repositories_listed":0,"syntology":null},{"url":"/paper/on-the-advance-of-making-language-models","slug":"on-the-advance-of-making-language-models","title":"Making Large Language Models Better Reasoners with Step-Aware Verifier","date":"2022-06-06","arxiv_id":"2206.02336","repositories_listed":0,"syntology":null},{"url":"/paper/numglue-a-suite-of-fundamental-yet","slug":"numglue-a-suite-of-fundamental-yet","title":"NumGLUE: A Suite of Fundamental yet Challenging Mathematical Reasoning Tasks","date":"2022-04-12","arxiv_id":"2204.05660","repositories_listed":0,"syntology":null}],"record_sha256":"974423218d1aa348fdab05cfcba7db946fbc40b3dea9321d03ab381c102735a4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}