{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/5","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":16,"rows_per_page":100,"rows":[401,500],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/4","next":"/task/math/papers/6","papers":[{"url":"/paper/sirius-contextual-sparsity-with-correction","slug":"sirius-contextual-sparsity-with-correction","title":"Sirius: Contextual Sparsity with Correction for Efficient LLMs","date":"2024-09-05","arxiv_id":"2409.03856","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sirius-contextual-sparsity-with-correction#ran","syntology_url":"https://syntology.ai/paper/2409.03856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03856"}},"official":{"repos":["infini-ai-lab/sirius"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/more-is-more-addition-bias-in-large-language","slug":"more-is-more-addition-bias-in-large-language","title":"More is More: Addition Bias in Large Language Models","date":"2024-09-04","arxiv_id":"2409.02569","repositories_listed":1,"syntology":null},{"url":"/paper/general-ocr-theory-towards-ocr-2-0-via-a","slug":"general-ocr-theory-towards-ocr-2-0-via-a","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","date":"2024-09-03","arxiv_id":"2409.01704","repositories_listed":1,"syntology":null},{"url":"/paper/multimath-bridging-visual-and-mathematical","slug":"multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","arxiv_id":"2409.00147","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":6,"n_pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 3 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimath-bridging-visual-and-mathematical#ran","syntology_url":"https://syntology.ai/paper/2409.00147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00147"}},"official":{"repos":["pengshuai-rin/multimath"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-math-problems-hard-for","slug":"what-makes-math-problems-hard-for","title":"What makes math problems hard for reinforcement learning: a case study","date":"2024-08-27","arxiv_id":"2408.15332","repositories_listed":1,"syntology":null},{"url":"/paper/sorsa-singular-values-and-orthonormal","slug":"sorsa-singular-values-and-orthonormal","title":"SORSA: Singular Values and Orthonormal Regularized Singular Vectors Adaptation of Large Language Models","date":"2024-08-21","arxiv_id":"2409.00055","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sorsa-singular-values-and-orthonormal#ran","syntology_url":"https://syntology.ai/paper/2409.00055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00055"}},"official":{"repos":["Gunale0926/SORSA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-for-math","slug":"benchmarking-large-language-models-for-math","title":"Benchmarking Large Language Models for Math Reasoning Tasks","date":"2024-08-20","arxiv_id":"2408.10839","repositories_listed":1,"syntology":null},{"url":"/paper/math-puma-progressive-upward-multimodal","slug":"math-puma-progressive-upward-multimodal","title":"Math-PUMA: Progressive Upward Multimodal Alignment to Enhance Mathematical Reasoning","date":"2024-08-16","arxiv_id":"2408.08640","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-puma-progressive-upward-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.08640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08640"}},"official":{"repos":["wwzhuang01/math-puma"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-web-crawled-data-for-high-quality","slug":"leveraging-web-crawled-data-for-high-quality","title":"Leveraging Web-Crawled Data for High-Quality Fine-Tuning","date":"2024-08-15","arxiv_id":"2408.08003","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-and-modeling-correlations-in","slug":"bridging-and-modeling-correlations-in","title":"Bridging and Modeling Correlations in Pairwise Data for Direct Preference Optimization","date":"2024-08-14","arxiv_id":"2408.07471","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-and-modeling-correlations-in#ran","syntology_url":"https://syntology.ai/paper/2408.07471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07471"}},"official":{"repos":["YJiangcm/BMC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathscape-evaluating-mllms-in-multimodal-math","slug":"mathscape-evaluating-mllms-in-multimodal-math","title":"MathScape: Evaluating MLLMs in multimodal Math Scenarios through a Hierarchical Benchmark","date":"2024-08-14","arxiv_id":"2408.07543","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathscape-evaluating-mllms-in-multimodal-math#ran","syntology_url":"https://syntology.ai/paper/2408.07543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07543"}},"official":{"repos":["PKU-Baichuan-MLSystemLab/MathScape"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-language-model-math-reasoning-via","slug":"evaluating-language-model-math-reasoning-via","title":"Mathfish: Evaluating Language Model Math Reasoning via Grounding in Educational Curricula","date":"2024-08-08","arxiv_id":"2408.04226","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00989","slug":"2408-00989","title":"On the Resilience of LLM-Based Multi-Agent Collaboration with Faulty Agents","date":"2024-08-02","arxiv_id":"2408.00989","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-00989#ran","syntology_url":"https://syntology.ai/paper/2408.00989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00989"}},"official":{"repos":["cuhk-arise/mas-resilience"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00765","slug":"2408-00765","title":"MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities","date":"2024-08-01","arxiv_id":"2408.00765","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-00765#ran","syntology_url":"https://syntology.ai/paper/2408.00765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00765"}},"official":{"repos":["yuweihao/mm-vet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-assisted-generation-of-difficult-math","slug":"ai-assisted-generation-of-difficult-math","title":"AI-Assisted Generation of Difficult Math Questions","date":"2024-07-30","arxiv_id":"2407.21009","repositories_listed":1,"syntology":null},{"url":"/paper/physics-of-language-models-part-2-1-grade","slug":"physics-of-language-models-part-2-1-grade","title":"Physics of Language Models: Part 2.1, Grade-School Math and the Hidden Reasoning Process","date":"2024-07-29","arxiv_id":"2407.20311","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-large-language-models-with-socratic","slug":"boosting-large-language-models-with-socratic","title":"Boosting Large Language Models with Socratic Method for Conversational Mathematics Teaching","date":"2024-07-24","arxiv_id":"2407.17349","repositories_listed":1,"syntology":null},{"url":"/paper/lean-github-compiling-github-lean","slug":"lean-github-compiling-github-lean","title":"LEAN-GitHub: Compiling GitHub LEAN repositories for a versatile LEAN prover","date":"2024-07-24","arxiv_id":"2407.17227","repositories_listed":1,"syntology":null},{"url":"/paper/mathviz-e-a-case-study-in-domain-specialized","slug":"mathviz-e-a-case-study-in-domain-specialized","title":"MathViz-E: A Case-study in Domain-Specialized Tool-Using Agents","date":"2024-07-24","arxiv_id":"2407.17544","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mathviz-e-a-case-study-in-domain-specialized#ran","syntology_url":"https://syntology.ai/paper/2407.17544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17544"}},"official":{"repos":["emergenceai/mathviz-e"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nerva-a-truly-sparse-implementation-of-neural","slug":"nerva-a-truly-sparse-implementation-of-neural","title":"Nerva: a Truly Sparse Implementation of Neural Networks","date":"2024-07-24","arxiv_id":"2407.17437","repositories_listed":1,"syntology":null},{"url":"/paper/taskgen-a-task-based-memory-infused-agentic","slug":"taskgen-a-task-based-memory-infused-agentic","title":"TaskGen: A Task-Based, Memory-Infused Agentic Framework using StrictJSON","date":"2024-07-22","arxiv_id":"2407.15734","repositories_listed":1,"syntology":null},{"url":"/paper/toward-adaptive-reasoning-in-large-language","slug":"toward-adaptive-reasoning-in-large-language","title":"Toward Adaptive Reasoning in Large Language Models with Thought Rollback","date":"2024-07-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-goal-conditioned-representations-for","slug":"learning-goal-conditioned-representations-for","title":"Learning Goal-Conditioned Representations for Language Reward Models","date":"2024-07-18","arxiv_id":"2407.13887","repositories_listed":1,"syntology":null},{"url":"/paper/prover-verifier-games-improve-legibility-of","slug":"prover-verifier-games-improve-legibility-of","title":"Prover-Verifier Games improve legibility of LLM outputs","date":"2024-07-18","arxiv_id":"2407.13692","repositories_listed":1,"syntology":null},{"url":"/paper/weak-to-strong-reasoning","slug":"weak-to-strong-reasoning","title":"Weak-to-Strong Reasoning","date":"2024-07-18","arxiv_id":"2407.13647","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/weak-to-strong-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13647"}},"official":{"repos":["gair-nlp/weak-to-strong-reasoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/turkishmmlu-measuring-massive-multitask","slug":"turkishmmlu-measuring-massive-multitask","title":"TurkishMMLU: Measuring Massive Multitask Language Understanding in Turkish","date":"2024-07-17","arxiv_id":"2407.12402","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-llms-for-optimization-modeling#ran","syntology_url":"https://syntology.ai/paper/2407.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09887"}},"official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stepwise-verification-and-remediation-of","slug":"stepwise-verification-and-remediation-of","title":"Stepwise Verification and Remediation of Student Reasoning Errors with Large Language Model Tutors","date":"2024-07-12","arxiv_id":"2407.09136","repositories_listed":1,"syntology":null},{"url":"/paper/autobencher-creating-salient-novel-difficult","slug":"autobencher-creating-salient-novel-difficult","title":"AutoBencher: Creating Salient, Novel, Difficult Datasets for Language Models","date":"2024-07-11","arxiv_id":"2407.08351","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobencher-creating-salient-novel-difficult#ran","syntology_url":"https://syntology.ai/paper/2407.08351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08351"}},"official":{"repos":["XiangLi1999/AutoBencher"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who-is-better-at-math-jenny-or-jingzhen","slug":"who-is-better-at-math-jenny-or-jingzhen","title":"Who is better at math, Jenny or Jingzhen? Uncovering Stereotypes in Large Language Models","date":"2024-07-09","arxiv_id":"2407.06917","repositories_listed":1,"syntology":null},{"url":"/paper/solving-for-x-and-beyond-can-large-language","slug":"solving-for-x-and-beyond-can-large-language","title":"Solving for X and Beyond: Can Large Language Models Solve Complex Math Problems with More-Than-Two Unknowns?","date":"2024-07-06","arxiv_id":"2407.05134","repositories_listed":1,"syntology":null},{"url":"/paper/smart-vision-language-reasoners","slug":"smart-vision-language-reasoners","title":"Smart Vision-Language Reasoners","date":"2024-07-05","arxiv_id":"2407.04212","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/smart-vision-language-reasoners#ran","syntology_url":"https://syntology.ai/paper/2407.04212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04212"}},"official":{"repos":["smarter-vlm/smarter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dotamath-decomposition-of-thought-with-code","slug":"dotamath-decomposition-of-thought-with-code","title":"DotaMath: Decomposition of Thought with Code Assistance and Self-correction for Mathematical Reasoning","date":"2024-07-04","arxiv_id":"2407.04078","repositories_listed":1,"syntology":null},{"url":"/paper/helpful-assistant-or-fruitful-facilitator","slug":"helpful-assistant-or-fruitful-facilitator","title":"Helpful assistant or fruitful facilitator? Investigating how personas affect language model behavior","date":"2024-07-02","arxiv_id":"2407.02099","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/helpful-assistant-or-fruitful-facilitator#ran","syntology_url":"https://syntology.ai/paper/2407.02099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02099"}},"official":{"repos":["peluz/persona-behavior"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/eliminating-position-bias-of-language-models","slug":"eliminating-position-bias-of-language-models","title":"Eliminating Position Bias of Language Models: A Mechanistic Approach","date":"2024-07-01","arxiv_id":"2407.01100","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eliminating-position-bias-of-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.01100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01100"}},"official":{"repos":["wzq016/pine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/we-math-does-your-large-multimodal-model","slug":"we-math-does-your-large-multimodal-model","title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","date":"2024-07-01","arxiv_id":"2407.01284","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/we-math-does-your-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2407.01284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01284"}},"official":{"repos":["we-math/we-math"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-controlled-dpo-leveraging-stepwise-error#ran","syntology_url":"https://syntology.ai/paper/2407.00782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00782"}},"official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/divert-distractor-generation-with-variational","slug":"divert-distractor-generation-with-variational","title":"DiVERT: Distractor Generation with Variational Errors Represented as Text for Math Multiple-choice Questions","date":"2024-06-27","arxiv_id":"2406.19356","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/divert-distractor-generation-with-variational#ran","syntology_url":"https://syntology.ai/paper/2406.19356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19356"}},"official":{"repos":["umass-ml4ed/divert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/livebench-a-challenging-contamination-free","slug":"livebench-a-challenging-contamination-free","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","date":"2024-06-27","arxiv_id":"2406.19314","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/livebench-a-challenging-contamination-free#ran","syntology_url":"https://syntology.ai/paper/2406.19314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19314"}},"official":{"repos":["livebench/livebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/step-dpo-step-wise-preference-optimization#ran","syntology_url":"https://syntology.ai/paper/2406.18629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18629"}},"official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/math-llava-bootstrapping-mathematical","slug":"math-llava-bootstrapping-mathematical","title":"Math-LLaVA: Bootstrapping Mathematical Reasoning for Multimodal Large Language Models","date":"2024-06-25","arxiv_id":"2406.17294","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-llava-bootstrapping-mathematical#ran","syntology_url":"https://syntology.ai/paper/2406.17294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17294"}},"official":{"repos":["hzq950419/math-llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lottery-ticket-adaptation-mitigating","slug":"lottery-ticket-adaptation-mitigating","title":"Lottery Ticket Adaptation: Mitigating Destructive Interference in LLMs","date":"2024-06-24","arxiv_id":"2406.16797","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lottery-ticket-adaptation-mitigating#ran","syntology_url":"https://syntology.ai/paper/2406.16797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16797"}},"official":{"repos":["kiddyboots216/lottery-ticket-adaptation"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/citygpt-empowering-urban-spatial-cognition-of","slug":"citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","arxiv_id":"2406.13948","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citygpt-empowering-urban-spatial-cognition-of#ran","syntology_url":"https://syntology.ai/paper/2406.13948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13948"}},"official":{"repos":["tsinghua-fib-lab/citygpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-on-incorrect-synthetic-data-scales-the","slug":"rl-on-incorrect-synthetic-data-scales-the","title":"RL on Incorrect Synthetic Data Scales the Efficiency of LLM Math Reasoning by Eight-Fold","date":"2024-06-20","arxiv_id":"2406.14532","repositories_listed":1,"syntology":null},{"url":"/paper/the-reason-behind-good-or-bad-towards-a","slug":"the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","arxiv_id":"2406.14024","repositories_listed":1,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":18,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":17,"n_pointer_only":23,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-reason-behind-good-or-bad-towards-a#ran","syntology_url":"https://syntology.ai/paper/2406.14024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14024"}},"official":{"repos":["kbsdjames/math-minos"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-infinite-long-prefix-in-transformer","slug":"toward-infinite-long-prefix-in-transformer","title":"Towards Infinite-Long Prefix in Transformer","date":"2024-06-20","arxiv_id":"2406.14036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toward-infinite-long-prefix-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2406.14036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14036"}},"official":{"repos":["christianyang37/chiwun"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptable-logical-control-for-large-language","slug":"adaptable-logical-control-for-large-language","title":"Adaptable Logical Control for Large Language Models","date":"2024-06-19","arxiv_id":"2406.13892","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/adaptable-logical-control-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.13892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13892"}},"official":{"repos":["joshuacnf/Ctrl-G"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-reason-in-the-wild-with-programs","slug":"can-llms-reason-in-the-wild-with-programs","title":"Can LLMs Reason in the Wild with Programs?","date":"2024-06-19","arxiv_id":"2406.13764","repositories_listed":1,"syntology":null},{"url":"/paper/dart-math-difficulty-aware-rejection-tuning-1","slug":"dart-math-difficulty-aware-rejection-tuning-1","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","date":"2024-06-18","arxiv_id":"2407.13690","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dart-math-difficulty-aware-rejection-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2407.13690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13690"}},"official":{"repos":["hkust-nlp/dart-math"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-prompting-taxonomy-a-universal","slug":"hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","arxiv_id":"2406.12644","repositories_listed":1,"syntology":null},{"url":"/paper/deepseek-coder-v2-breaking-the-barrier-of","slug":"deepseek-coder-v2-breaking-the-barrier-of","title":"DeepSeek-Coder-V2: Breaking the Barrier of Closed-Source Models in Code Intelligence","date":"2024-06-17","arxiv_id":"2406.11931","repositories_listed":1,"syntology":null},{"url":"/paper/della-merging-reducing-interference-in-model","slug":"della-merging-reducing-interference-in-model","title":"DELLA-Merging: Reducing Interference in Model Merging through Magnitude-Based Sampling","date":"2024-06-17","arxiv_id":"2406.11617","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/della-merging-reducing-interference-in-model#ran","syntology_url":"https://syntology.ai/paper/2406.11617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11617"}},"official":{"repos":["declare-lab/della"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/geogpt4v-towards-geometric-multi-modal-large","slug":"geogpt4v-towards-geometric-multi-modal-large","title":"GeoGPT4V: Towards Geometric Multi-modal Large Language Models with Geometric Image Generation","date":"2024-06-17","arxiv_id":"2406.11503","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/geogpt4v-towards-geometric-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2406.11503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11503"}},"official":{"repos":["lanyu0303/geogpt4v_project"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-level-value-preference-optimization-for","slug":"step-level-value-preference-optimization-for","title":"Step-level Value Preference Optimization for Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10858","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-level-value-preference-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2406.10858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10858"}},"official":{"repos":["MARIO-Math-Reasoning/Super_MARIO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/delta-come-training-free-delta-compression","slug":"delta-come-training-free-delta-compression","title":"Delta-CoMe: Training-Free Delta-Compression with Mixed-Precision for Large Language Models","date":"2024-06-13","arxiv_id":"2406.08903","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/delta-come-training-free-delta-compression#ran","syntology_url":"https://syntology.ai/paper/2406.08903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08903"}},"official":{"repos":["thunlp/delta-come"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/milora-harnessing-minor-singular-components","slug":"milora-harnessing-minor-singular-components","title":"MiLoRA: Harnessing Minor Singular Components for Parameter-Efficient LLM Finetuning","date":"2024-06-13","arxiv_id":"2406.09044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/milora-harnessing-minor-singular-components#ran","syntology_url":"https://syntology.ai/paper/2406.09044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09044"}},"official":{"repos":["graphpku/pissa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-sketchpad-sketching-as-a-visual-chain","slug":"visual-sketchpad-sketching-as-a-visual-chain","title":"Visual Sketchpad: Sketching as a Visual Chain of Thought for Multimodal Language Models","date":"2024-06-13","arxiv_id":"2406.09403","repositories_listed":1,"syntology":null},{"url":"/paper/collective-constitutional-ai-aligning-a","slug":"collective-constitutional-ai-aligning-a","title":"Collective Constitutional AI: Aligning a Language Model with Public Input","date":"2024-06-12","arxiv_id":"2406.07814","repositories_listed":1,"syntology":null},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/corda-context-oriented-decomposition","slug":"corda-context-oriented-decomposition","title":"CorDA: Context-Oriented Decomposition Adaptation of Large Language Models for Task-Aware Parameter-Efficient Fine-tuning","date":"2024-06-07","arxiv_id":"2406.05223","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/corda-context-oriented-decomposition#ran","syntology_url":"https://syntology.ai/paper/2406.05223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05223"}},"official":{"repos":["iboing/corda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dice-detecting-in-distribution-contamination","slug":"dice-detecting-in-distribution-contamination","title":"DICE: Detecting In-distribution Contamination in LLM's Fine-tuning Phase for Math Reasoning","date":"2024-06-06","arxiv_id":"2406.04197","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dice-detecting-in-distribution-contamination#ran","syntology_url":"https://syntology.ai/paper/2406.04197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04197"}},"official":{"repos":["thu-keg/dice"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lean-workbook-a-large-scale-lean-problem-set","slug":"lean-workbook-a-large-scale-lean-problem-set","title":"Lean Workbook: A large-scale Lean problem set formalized from natural language math problems","date":"2024-06-06","arxiv_id":"2406.03847","repositories_listed":1,"syntology":null},{"url":"/paper/numcot-numerals-and-units-of-measurement-in","slug":"numcot-numerals-and-units-of-measurement-in","title":"NUMCoT: Numerals and Units of Measurement in Chain-of-Thought Reasoning using Large Language Models","date":"2024-06-05","arxiv_id":"2406.02864","repositories_listed":1,"syntology":null},{"url":"/paper/mcot-multilingual-instruction-tuning-for","slug":"mcot-multilingual-instruction-tuning-for","title":"mCoT: Multilingual Instruction Tuning for Reasoning Consistency in Language Models","date":"2024-06-04","arxiv_id":"2406.02301","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-robustness-of-llms-on-math","slug":"investigating-the-robustness-of-llms-on-math","title":"Cutting Through the Noise: Boosting LLM Performance on Math Word Problems","date":"2024-05-30","arxiv_id":"2406.15444","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/investigating-the-robustness-of-llms-on-math#ran","syntology_url":"https://syntology.ai/paper/2406.15444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15444"}},"official":{"repos":["him1411/problemathic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/taia-large-language-models-are-out-of","slug":"taia-large-language-models-are-out-of","title":"TAIA: Large Language Models are Out-of-Distribution Data Learners","date":"2024-05-30","arxiv_id":"2405.20192","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/taia-large-language-models-are-out-of#ran","syntology_url":"https://syntology.ai/paper/2405.20192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20192"}},"official":{"repos":["pixas/TAIA_LLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/yuan-2-0-m32-mixture-of-experts-with","slug":"yuan-2-0-m32-mixture-of-experts-with","title":"Yuan 2.0-M32: Mixture of Experts with Attention Router","date":"2024-05-28","arxiv_id":"2405.17976","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yuan-2-0-m32-mixture-of-experts-with#ran","syntology_url":"https://syntology.ai/paper/2405.17976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17976"}},"official":{"repos":["ieit-yuan/yuan2.0-m32"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autoformalizing-euclidean-geometry","slug":"autoformalizing-euclidean-geometry","title":"Autoformalizing Euclidean Geometry","date":"2024-05-27","arxiv_id":"2405.17216","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":4,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autoformalizing-euclidean-geometry#ran","syntology_url":"https://syntology.ai/paper/2405.17216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17216"}},"official":{"repos":["loganrjmurphy/leaneuclid"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-xs-low-rank-adaptation-with-extremely","slug":"lora-xs-low-rank-adaptation-with-extremely","title":"LoRA-XS: Low-Rank Adaptation with Extremely Small Number of Parameters","date":"2024-05-27","arxiv_id":"2405.17604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-xs-low-rank-adaptation-with-extremely#ran","syntology_url":"https://syntology.ai/paper/2405.17604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17604"}},"official":{"repos":["mohammadrezabanaei/lora-xs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meteor-mamba-based-traversal-of-rationale-for","slug":"meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","arxiv_id":"2405.15574","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meteor-mamba-based-traversal-of-rationale-for#ran","syntology_url":"https://syntology.ai/paper/2405.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15574"}},"official":{"repos":["byungkwanlee/meteor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-solve-longer-math-word-problems","slug":"can-llms-solve-longer-math-word-problems","title":"Can LLMs Solve longer Math Word Problems Better?","date":"2024-05-23","arxiv_id":"2405.14804","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":5,"n_ran_checked":6,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-llms-solve-longer-math-word-problems#ran","syntology_url":"https://syntology.ai/paper/2405.14804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14804"}},"official":{"repos":["xinxu-ustc/coleg-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["community","found_in_text","official","unlocated"]}}},{"url":"/paper/jiuzhang3-0-efficiently-improving","slug":"jiuzhang3-0-efficiently-improving","title":"JiuZhang3.0: Efficiently Improving Mathematical Reasoning by Training Small Data Synthesis Models","date":"2024-05-23","arxiv_id":"2405.14365","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/jiuzhang3-0-efficiently-improving#ran","syntology_url":"https://syntology.ai/paper/2405.14365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14365"}},"official":{"repos":["rucaibox/jiuzhang3.0"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mathbench-evaluating-the-theory-and","slug":"mathbench-evaluating-the-theory-and","title":"MathBench: Evaluating the Theory and Application Proficiency of LLMs with a Hierarchical Mathematics Benchmark","date":"2024-05-20","arxiv_id":"2405.12209","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-choice-questions-are-efficient-and","slug":"multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","arxiv_id":"2405.11966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-choice-questions-are-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2405.11966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11966"}},"official":{"repos":["geralt-targaryen/mc-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-safety-realignment-framework-via-subspace","slug":"a-safety-realignment-framework-via-subspace","title":"A safety realignment framework via subspace-oriented model fusion for large language models","date":"2024-05-15","arxiv_id":"2405.09055","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-safety-realignment-framework-via-subspace#ran","syntology_url":"https://syntology.ai/paper/2405.09055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09055"}},"official":{"repos":["xinykou/safety_realignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mumath-code-combining-tool-use-large-language","slug":"mumath-code-combining-tool-use-large-language","title":"MuMath-Code: Combining Tool-Use Large Language Models with Multi-perspective Data Augmentation for Mathematical Reasoning","date":"2024-05-13","arxiv_id":"2405.07551","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mumath-code-combining-tool-use-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07551"}},"official":null}},{"url":"/paper/tanq-an-open-domain-dataset-of-table-answered","slug":"tanq-an-open-domain-dataset-of-table-answered","title":"TANQ: An open domain dataset of table answered questions","date":"2024-05-13","arxiv_id":"2405.07765","repositories_listed":1,"syntology":null},{"url":"/paper/can-large-language-models-replicate-its","slug":"can-large-language-models-replicate-its","title":"Can Large Language Models Replicate ITS Feedback on Open-Ended Math Questions?","date":"2024-05-10","arxiv_id":"2405.06414","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-geometry-problems-via","slug":"learning-to-solve-geometry-problems-via","title":"Learning to Solve Geometry Problems via Simulating Human Dual-Reasoning Process","date":"2024-05-10","arxiv_id":"2405.06232","repositories_listed":1,"syntology":null},{"url":"/paper/visiongraph-leveraging-large-multimodal","slug":"visiongraph-leveraging-large-multimodal","title":"VisionGraph: Leveraging Large Multimodal Models for Graph Theory Problems in Visual Context","date":"2024-05-08","arxiv_id":"2405.04950","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visiongraph-leveraging-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2405.04950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04950"}},"official":{"repos":["hitsz-tmg/visiongraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-compositional-deficiency-of","slug":"exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","arxiv_id":"2405.06680","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-the-compositional-deficiency-of#ran","syntology_url":"https://syntology.ai/paper/2405.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06680"}},"official":{"repos":["tongjingqi/MathTrap"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gold-geometry-problem-solver-with-natural","slug":"gold-geometry-problem-solver-with-natural","title":"GOLD: Geometry Problem Solver with Natural Language Description","date":"2024-05-01","arxiv_id":"2405.00494","repositories_listed":1,"syntology":null},{"url":"/paper/pecc-problem-extraction-and-coding-challenges","slug":"pecc-problem-extraction-and-coding-challenges","title":"PECC: Problem Extraction and Coding Challenges","date":"2024-04-29","arxiv_id":"2404.18766","repositories_listed":1,"syntology":null},{"url":"/paper/small-language-models-need-strong-verifiers","slug":"small-language-models-need-strong-verifiers","title":"Small Language Models Need Strong Verifiers to Self-Correct Reasoning","date":"2024-04-26","arxiv_id":"2404.17140","repositories_listed":1,"syntology":null},{"url":"/paper/ai-coders-are-among-us-rethinking-programming","slug":"ai-coders-are-among-us-rethinking-programming","title":"AI Coders Are Among Us: Rethinking Programming Language Grammar Towards Efficient Code Generation","date":"2024-04-25","arxiv_id":"2404.16333","repositories_listed":1,"syntology":null},{"url":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/layer-skip-enabling-early-exit-inference-and#ran","syntology_url":"https://syntology.ai/paper/2404.16710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16710"}},"official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/achieving-97-on-gsm8k-deeply-understanding","slug":"achieving-97-on-gsm8k-deeply-understanding","title":"Achieving >97% on GSM8K: Deeply Understanding the Problems Makes LLMs Better Solvers for Math Word Problems","date":"2024-04-23","arxiv_id":"2404.14963","repositories_listed":1,"syntology":null},{"url":"/paper/toward-self-improvement-of-llms-via","slug":"toward-self-improvement-of-llms-via","title":"Toward Self-Improvement of LLMs via Imagination, Searching, and Criticizing","date":"2024-04-18","arxiv_id":"2404.12253","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":1,"n_ran_checked":12,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":18,"phrase":"14 ran (of which 1 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/toward-self-improvement-of-llms-via#ran","syntology_url":"https://syntology.ai/paper/2404.12253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12253"}},"official":{"repos":["yetianjhu/alphallm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":1,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/self-explore-to-avoid-the-pit-improving-the","slug":"self-explore-to-avoid-the-pit-improving-the","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","date":"2024-04-16","arxiv_id":"2404.10346","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-explore-to-avoid-the-pit-improving-the#ran","syntology_url":"https://syntology.ai/paper/2404.10346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10346"}},"official":{"repos":["hbin0701/Self-Explore"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/evaluating-mathematical-reasoning-beyond","slug":"evaluating-mathematical-reasoning-beyond","title":"Evaluating Mathematical Reasoning Beyond Accuracy","date":"2024-04-08","arxiv_id":"2404.05692","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-mathematical-reasoning-beyond#ran","syntology_url":"https://syntology.ai/paper/2404.05692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05692"}},"official":{"repos":["gair-nlp/reasoneval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-geometric-problem-solving-a","slug":"advancing-geometric-problem-solving-a","title":"MM-MATH: Advancing Multimodal Math Evaluation with Process Evaluation and Fine-grained Classification","date":"2024-04-07","arxiv_id":"2404.05091","repositories_listed":1,"syntology":null},{"url":"/paper/macm-utilizing-a-multi-agent-system-for","slug":"macm-utilizing-a-multi-agent-system-for","title":"MACM: Utilizing a Multi-Agent System for Condition Mining in Solving Complex Mathematical Problems","date":"2024-04-06","arxiv_id":"2404.04735","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/macm-utilizing-a-multi-agent-system-for#ran","syntology_url":"https://syntology.ai/paper/2404.04735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04735"}},"official":{"repos":["bin123apple/macm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/badam-a-memory-efficient-full-parameter","slug":"badam-a-memory-efficient-full-parameter","title":"BAdam: A Memory Efficient Full Parameter Optimization Method for Large Language Models","date":"2024-04-03","arxiv_id":"2404.02827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/badam-a-memory-efficient-full-parameter#ran","syntology_url":"https://syntology.ai/paper/2404.02827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02827"}},"official":{"repos":["ledzy/badam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-for-2","slug":"benchmarking-large-language-models-for-2","title":"Benchmarking Large Language Models for Persian: A Preliminary Study Focusing on ChatGPT","date":"2024-04-03","arxiv_id":"2404.02403","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-automated-distractor-generation-for","slug":"exploring-automated-distractor-generation-for","title":"Exploring Automated Distractor Generation for Math Multiple-choice Questions via Large Language Models","date":"2024-04-02","arxiv_id":"2404.02124","repositories_listed":1,"syntology":null},{"url":"/paper/self-demos-eliciting-out-of-demonstration","slug":"self-demos-eliciting-out-of-demonstration","title":"Self-Demos: Eliciting Out-of-Demonstration Generalizability in Large Language Models","date":"2024-04-01","arxiv_id":"2404.00884","repositories_listed":1,"syntology":null},{"url":"/paper/what-s-in-your-safe-data-identifying-benign","slug":"what-s-in-your-safe-data-identifying-benign","title":"What is in Your Safe Data? Identifying Benign Data that Breaks Safety","date":"2024-04-01","arxiv_id":"2404.01099","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/what-s-in-your-safe-data-identifying-benign#ran","syntology_url":"https://syntology.ai/paper/2404.01099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01099"}},"official":{"repos":["princeton-nlp/benign-data-breaks-safety"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-master-math-investigating-large","slug":"can-llms-master-math-investigating-large","title":"Can LLMs Master Math? Investigating Large Language Models on Math Stack Exchange","date":"2024-03-30","arxiv_id":"2404.00344","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-llms-master-math-investigating-large#ran","syntology_url":"https://syntology.ai/paper/2404.00344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00344"}},"official":{"repos":["gipplab/llm-investig-mathstackexchange"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"692334e9ae2e4e9db74458955cfd59a3c8d838c038d8326e64629d8c264be3ff","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}