{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/math/papers/6","list_of":"/task/math","task":"Math","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":16,"rows_per_page":100,"rows":[501,600],"of":1596,"counts":{"archive_papers_tagged":1596,"with_a_code_link":765,"where_syntology_ran_a_sample":349,"not_listed_spam_title":0,"listed":1596,"listed_where_code_ran":349,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/math","prev":"/task/math/papers/5","next":"/task/math/papers/7","papers":[{"url":"/paper/scaling-up-ridge-regression-for-brain","slug":"scaling-up-ridge-regression-for-brain","title":"Scaling up ridge regression for brain encoding in a massive individual fMRI dataset","date":"2024-03-28","arxiv_id":"2403.19421","repositories_listed":1,"syntology":null},{"url":"/paper/don-t-trust-verify-grounding-llm-quantitative","slug":"don-t-trust-verify-grounding-llm-quantitative","title":"Don't Trust: Verify -- Grounding LLM Quantitative Reasoning with Autoformalization","date":"2024-03-26","arxiv_id":"2403.18120","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/don-t-trust-verify-grounding-llm-quantitative#ran","syntology_url":"https://syntology.ai/paper/2403.18120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18120"}},"official":{"repos":["jinpz/dtv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-optimization-of-model-merging","slug":"evolutionary-optimization-of-model-merging","title":"Evolutionary Optimization of Model Merging Recipes","date":"2024-03-19","arxiv_id":"2403.13187","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evolutionary-optimization-of-model-merging#ran","syntology_url":"https://syntology.ai/paper/2403.13187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13187"}},"official":{"repos":["sakanaai/evolutionary-model-merge"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/instructing-large-language-models-to-identify","slug":"instructing-large-language-models-to-identify","title":"Instructing Large Language Models to Identify and Ignore Irrelevant Conditions","date":"2024-03-19","arxiv_id":"2403.12744","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-and-secure-dnn-inference-on","slug":"memory-efficient-and-secure-dnn-inference-on","title":"Memory-Efficient and Secure DNN Inference on TrustZone-enabled Consumer IoT Devices","date":"2024-03-19","arxiv_id":"2403.12568","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-math-word-problems-challenging-for","slug":"what-makes-math-word-problems-challenging-for","title":"What Makes Math Word Problems Challenging for LLMs?","date":"2024-03-17","arxiv_id":"2403.11369","repositories_listed":1,"syntology":null},{"url":"/paper/easy-to-hard-generalization-scalable","slug":"easy-to-hard-generalization-scalable","title":"Easy-to-Hard Generalization: Scalable Alignment Beyond Human Supervision","date":"2024-03-14","arxiv_id":"2403.09472","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/easy-to-hard-generalization-scalable#ran","syntology_url":"https://syntology.ai/paper/2403.09472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09472"}},"official":{"repos":["edward-sun/easy-to-hard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/incorporating-graph-attention-mechanism-into-1","slug":"incorporating-graph-attention-mechanism-into-1","title":"Incorporating Graph Attention Mechanism into Geometric Problem Solving Based on Deep Reinforcement Learning","date":"2024-03-14","arxiv_id":"2403.14690","repositories_listed":1,"syntology":null},{"url":"/paper/the-first-to-know-how-token-distributions","slug":"the-first-to-know-how-token-distributions","title":"The First to Know: How Token Distributions Reveal Hidden Knowledge in Large Vision-Language Models?","date":"2024-03-14","arxiv_id":"2403.09037","repositories_listed":1,"syntology":null},{"url":"/paper/branch-train-mix-mixing-expert-llms-into-a","slug":"branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","arxiv_id":"2403.07816","repositories_listed":1,"syntology":null},{"url":"/paper/smalltolarge-s2l-scalable-data-selection-for","slug":"smalltolarge-s2l-scalable-data-selection-for","title":"SmallToLarge (S2L): Scalable Data Selection for Fine-tuning Large Language Models by Summarizing Training Trajectories of Small Models","date":"2024-03-12","arxiv_id":"2403.07384","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smalltolarge-s2l-scalable-data-selection-for#ran","syntology_url":"https://syntology.ai/paper/2403.07384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07384"}},"official":{"repos":["bigml-cs-ucla/s2l"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-hallucination-in-large-language","slug":"benchmarking-hallucination-in-large-language","title":"Benchmarking Hallucination in Large Language Models based on Unanswerable Math Word Problem","date":"2024-03-06","arxiv_id":"2403.03558","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-and-optimizing-educational-content","slug":"evaluating-and-optimizing-educational-content","title":"Evaluating and Optimizing Educational Content with Large Language Model Judgments","date":"2024-03-05","arxiv_id":"2403.02795","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/evaluating-and-optimizing-educational-content#ran","syntology_url":"https://syntology.ai/paper/2403.02795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02795"}},"official":{"repos":["StanfordAI4HI/ed-expert-simulator"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/mathscale-scaling-instruction-tuning-for","slug":"mathscale-scaling-instruction-tuning-for","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","date":"2024-03-05","arxiv_id":"2403.02884","repositories_listed":1,"syntology":null},{"url":"/paper/brilla-ai-ai-contestant-for-the-national","slug":"brilla-ai-ai-contestant-for-the-national","title":"Brilla AI: AI Contestant for the National Science and Maths Quiz","date":"2024-03-04","arxiv_id":"2403.01699","repositories_listed":1,"syntology":null},{"url":"/paper/masked-thought-simply-masking-partial","slug":"masked-thought-simply-masking-partial","title":"Masked Thought: Simply Masking Partial Reasoning Steps Can Improve Mathematical Reasoning Learning of Language Models","date":"2024-03-04","arxiv_id":"2403.02178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/masked-thought-simply-masking-partial#ran","syntology_url":"https://syntology.ai/paper/2403.02178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02178"}},"official":{"repos":["changyuchen347/maskedthought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-validity-of-automatically","slug":"improving-the-validity-of-automatically","title":"Improving the Validity of Automatically Generated Feedback via Reinforcement Learning","date":"2024-03-02","arxiv_id":"2403.01304","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-the-validity-of-automatically#ran","syntology_url":"https://syntology.ai/paper/2403.01304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01304"}},"official":{"repos":["umass-ml4ed/feedback-gen-dpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/functional-benchmarks-for-robust-evaluation","slug":"functional-benchmarks-for-robust-evaluation","title":"Functional Benchmarks for Robust Evaluation of Reasoning Performance, and the Reasoning Gap","date":"2024-02-29","arxiv_id":"2402.19450","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/functional-benchmarks-for-robust-evaluation#ran","syntology_url":"https://syntology.ai/paper/2402.19450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19450"}},"official":{"repos":["consequentai/fneval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gsm-plus-a-comprehensive-benchmark-for","slug":"gsm-plus-a-comprehensive-benchmark-for","title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","date":"2024-02-29","arxiv_id":"2402.19255","repositories_listed":1,"syntology":null},{"url":"/paper/data-interpreter-an-llm-agent-for-data","slug":"data-interpreter-an-llm-agent-for-data","title":"Data Interpreter: An LLM Agent For Data Science","date":"2024-02-28","arxiv_id":"2402.18679","repositories_listed":1,"syntology":null},{"url":"/paper/case-based-or-rule-based-how-do-transformers","slug":"case-based-or-rule-based-how-do-transformers","title":"Case-Based or Rule-Based: How Do Transformers Do the Math?","date":"2024-02-27","arxiv_id":"2402.17709","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/case-based-or-rule-based-how-do-transformers#ran","syntology_url":"https://syntology.ai/paper/2402.17709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17709"}},"official":{"repos":["graphpku/case_or_rule"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-resistant-math-word-problem-generation","slug":"llm-resistant-math-word-problem-generation","title":"Adversarial Math Word Problem Generation","date":"2024-02-27","arxiv_id":"2402.17916","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/llm-resistant-math-word-problem-generation#ran","syntology_url":"https://syntology.ai/paper/2402.17916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17916"}},"official":{"repos":["ruoyuxie/adversarial_mwps_generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/mathsensei-a-tool-augmented-large-language","slug":"mathsensei-a-tool-augmented-large-language","title":"MATHSENSEI: A Tool-Augmented Large Language Model for Mathematical Reasoning","date":"2024-02-27","arxiv_id":"2402.17231","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-humans-write-code-large-models-do-it","slug":"how-do-humans-write-code-large-models-do-it","title":"How Do Humans Write Code? Large Models Do It the Same Way Too","date":"2024-02-24","arxiv_id":"2402.15729","repositories_listed":1,"syntology":null},{"url":"/paper/stepwise-self-consistent-mathematical","slug":"stepwise-self-consistent-mathematical","title":"Stepwise Self-Consistent Mathematical Reasoning with Large Language Models","date":"2024-02-24","arxiv_id":"2402.17786","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stepwise-self-consistent-mathematical#ran","syntology_url":"https://syntology.ai/paper/2402.17786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17786"}},"official":{"repos":["zhao-zilong/ssc-cot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conceptmath-a-bilingual-concept-wise","slug":"conceptmath-a-bilingual-concept-wise","title":"ConceptMath: A Bilingual Concept-wise Benchmark for Measuring Mathematical Reasoning of Large Language Models","date":"2024-02-22","arxiv_id":"2402.14660","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conceptmath-a-bilingual-concept-wise#ran","syntology_url":"https://syntology.ai/paper/2402.14660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14660"}},"official":{"repos":["conceptmath/conceptmath"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-multimodal-mathematical-reasoning","slug":"measuring-multimodal-mathematical-reasoning","title":"Measuring Multimodal Mathematical Reasoning with MATH-Vision Dataset","date":"2024-02-22","arxiv_id":"2402.14804","repositories_listed":1,"syntology":null},{"url":"/paper/moelora-contrastive-learning-guided-mixture","slug":"moelora-contrastive-learning-guided-mixture","title":"MoELoRA: Contrastive Learning Guided Mixture of Experts on Parameter-Efficient Fine-Tuning for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12851","repositories_listed":1,"syntology":null},{"url":"/paper/reformatted-alignment","slug":"reformatted-alignment","title":"Reformatted Alignment","date":"2024-02-19","arxiv_id":"2402.12219","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reformatted-alignment#ran","syntology_url":"https://syntology.ai/paper/2402.12219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12219"}},"official":{"repos":["gair-nlp/realign"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-as-science-tutors","slug":"language-models-as-science-tutors","title":"Language Models as Science Tutors","date":"2024-02-16","arxiv_id":"2402.11111","repositories_listed":1,"syntology":null},{"url":"/paper/geoeval-benchmark-for-evaluating-llms-and","slug":"geoeval-benchmark-for-evaluating-llms-and","title":"GeoEval: Benchmark for Evaluating LLMs and Multi-Modal Models on Geometry Problem-Solving","date":"2024-02-15","arxiv_id":"2402.10104","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/geoeval-benchmark-for-evaluating-llms-and#ran","syntology_url":"https://syntology.ai/paper/2402.10104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10104"}},"official":{"repos":["geoeval/geoeval"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/openmathinstruct-1-a-1-8-million-math","slug":"openmathinstruct-1-a-1-8-million-math","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","date":"2024-02-15","arxiv_id":"2402.10176","repositories_listed":1,"syntology":null},{"url":"/paper/mustard-mastering-uniform-synthesis-of","slug":"mustard-mastering-uniform-synthesis-of","title":"MUSTARD: Mastering Uniform Synthesis of Theorem and Proof Data","date":"2024-02-14","arxiv_id":"2402.08957","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mustard-mastering-uniform-synthesis-of#ran","syntology_url":"https://syntology.ai/paper/2402.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08957"}},"official":{"repos":["eleanor-h/mustard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-the-authoring-of-autotutors-with","slug":"scaling-the-authoring-of-autotutors-with","title":"AutoTutor meets Large Language Models: A Language Model Tutor with Rich Pedagogy and Guardrails","date":"2024-02-14","arxiv_id":"2402.09216","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/scaling-the-authoring-of-autotutors-with#ran","syntology_url":"https://syntology.ai/paper/2402.09216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09216"}},"official":{"repos":["eth-lre/mwptutor"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-of-thoughts-chain-of-thought","slug":"diffusion-of-thoughts-chain-of-thought","title":"Diffusion of Thoughts: Chain-of-Thought Reasoning in Diffusion Language Models","date":"2024-02-12","arxiv_id":"2402.07754","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":13,"n_pointer_only":17,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffusion-of-thoughts-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2402.07754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07754"}},"official":{"repos":["hkunlp/diffusion-of-thoughts"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/internlm-math-open-math-large-language-models","slug":"internlm-math-open-math-large-language-models","title":"InternLM-Math: Open Math Large Language Models Toward Verifiable Reasoning","date":"2024-02-09","arxiv_id":"2402.06332","repositories_listed":1,"syntology":null},{"url":"/paper/in-context-principle-learning-from-mistakes","slug":"in-context-principle-learning-from-mistakes","title":"In-Context Principle Learning from Mistakes","date":"2024-02-08","arxiv_id":"2402.05403","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-reasoning-ability-of","slug":"understanding-the-reasoning-ability-of","title":"Understanding Reasoning Ability of Language Models From the Perspective of Reasoning Paths Aggregation","date":"2024-02-05","arxiv_id":"2402.03268","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/understanding-the-reasoning-ability-of#ran","syntology_url":"https://syntology.ai/paper/2402.03268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03268"}},"official":{"repos":["wangxinyilinda/lm_random_walk"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/magdi-structured-distillation-of-multi-agent","slug":"magdi-structured-distillation-of-multi-agent","title":"MAGDi: Structured Distillation of Multi-Agent Interaction Graphs Improves Reasoning in Smaller Language Models","date":"2024-02-02","arxiv_id":"2402.01620","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magdi-structured-distillation-of-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2402.01620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01620"}},"official":{"repos":["dinobby/magdi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taxonomy-of-mathematical-plagiarism","slug":"taxonomy-of-mathematical-plagiarism","title":"Taxonomy of Mathematical Plagiarism","date":"2024-01-30","arxiv_id":"2401.16969","repositories_listed":1,"syntology":null},{"url":"/paper/regal-refactoring-programs-to-discover","slug":"regal-refactoring-programs-to-discover","title":"ReGAL: Refactoring Programs to Discover Generalizable Abstractions","date":"2024-01-29","arxiv_id":"2401.16467","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regal-refactoring-programs-to-discover#ran","syntology_url":"https://syntology.ai/paper/2401.16467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.16467"}},"official":{"repos":["esteng/regal_program_learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-ai-assistants-know-what-they-don-t-know","slug":"can-ai-assistants-know-what-they-don-t-know","title":"Can AI Assistants Know What They Don't Know?","date":"2024-01-24","arxiv_id":"2401.13275","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-ai-assistants-know-what-they-don-t-know#ran","syntology_url":"https://syntology.ai/paper/2401.13275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13275"}},"official":{"repos":["openmoss/say-i-dont-know"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/trove-inducing-verifiable-and-efficient","slug":"trove-inducing-verifiable-and-efficient","title":"TroVE: Inducing Verifiable and Efficient Toolboxes for Solving Programmatic Tasks","date":"2024-01-23","arxiv_id":"2401.12869","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trove-inducing-verifiable-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2401.12869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12869"}},"official":{"repos":["zorazrw/trove"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/superclue-math6-graded-multi-step-math","slug":"superclue-math6-graded-multi-step-math","title":"SuperCLUE-Math6: Graded Multi-Step Math Reasoning Benchmark for LLMs in Chinese","date":"2024-01-22","arxiv_id":"2401.11819","repositories_listed":1,"syntology":null},{"url":"/paper/over-reasoning-and-redundant-calculation-of","slug":"over-reasoning-and-redundant-calculation-of","title":"Over-Reasoning and Redundant Calculation of Large Language Models","date":"2024-01-21","arxiv_id":"2401.11467","repositories_listed":1,"syntology":null},{"url":"/paper/escape-sky-high-cost-early-stopping-self","slug":"escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","arxiv_id":"2401.10480","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/escape-sky-high-cost-early-stopping-self#ran","syntology_url":"https://syntology.ai/paper/2401.10480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10480"}},"official":{"repos":["yiwei98/esc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/augmenting-math-word-problems-via-iterative","slug":"augmenting-math-word-problems-via-iterative","title":"Augmenting Math Word Problems via Iterative Question Composing","date":"2024-01-17","arxiv_id":"2401.09003","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-math-word-problems-via-iterative#ran","syntology_url":"https://syntology.ai/paper/2401.09003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09003"}},"official":{"repos":["iiis-ai/iterativequestioncomposing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-neurosymbolic","slug":"large-language-models-are-neurosymbolic","title":"Large Language Models Are Neurosymbolic Reasoners","date":"2024-01-17","arxiv_id":"2401.09334","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-neurosymbolic#ran","syntology_url":"https://syntology.ai/paper/2401.09334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09334"}},"official":{"repos":["hyintell/llmsymbolic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reft-reasoning-with-reinforced-fine-tuning","slug":"reft-reasoning-with-reinforced-fine-tuning","title":"ReFT: Reasoning with Reinforced Fine-Tuning","date":"2024-01-17","arxiv_id":"2401.08967","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/reft-reasoning-with-reinforced-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2401.08967","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08967"}},"official":{"repos":["lqtrung1998/mwp_reft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sciglm-training-scientific-language-models","slug":"sciglm-training-scientific-language-models","title":"SciInstruct: a Self-Reflective Instruction Annotated Dataset for Training Scientific Language Models","date":"2024-01-15","arxiv_id":"2401.07950","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sciglm-training-scientific-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.07950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07950"}},"official":{"repos":["thudm/sciglm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-benefits-of-a-concise-chain-of-thought-on#ran","syntology_url":"https://syntology.ai/paper/2401.05618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05618"}},"official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-understand-numbers-at-least","slug":"language-models-understand-numbers-at-least","title":"Language Models Encode the Value of Numbers Linearly","date":"2024-01-08","arxiv_id":"2401.03735","repositories_listed":1,"syntology":null},{"url":"/paper/llama-pro-progressive-llama-with-block","slug":"llama-pro-progressive-llama-with-block","title":"LLaMA Pro: Progressive LLaMA with Block Expansion","date":"2024-01-04","arxiv_id":"2401.02415","repositories_listed":1,"syntology":null},{"url":"/paper/generative-ai-for-math-part-i-mathpile-a","slug":"generative-ai-for-math-part-i-mathpile-a","title":"MathPile: A Billion-Token-Scale Pretraining Corpus for Math","date":"2023-12-28","arxiv_id":"2312.17120","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generative-ai-for-math-part-i-mathpile-a#ran","syntology_url":"https://syntology.ai/paper/2312.17120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17120"}},"official":{"repos":["gair-nlp/mathpile"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-in-depth-look-at-gemini-s-language#ran","syntology_url":"https://syntology.ai/paper/2312.11444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11444"}},"official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-complex-mathematical-reasoning-via","slug":"modeling-complex-mathematical-reasoning-via","title":"Modeling Complex Mathematical Reasoning via Large Language Model based MathAgent","date":"2023-12-14","arxiv_id":"2312.08926","repositories_listed":1,"syntology":null},{"url":"/paper/get-an-a-in-math-progressive-rectification","slug":"get-an-a-in-math-progressive-rectification","title":"Get an A in Math: Progressive Rectification Prompting","date":"2023-12-11","arxiv_id":"2312.06867","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/get-an-a-in-math-progressive-rectification#ran","syntology_url":"https://syntology.ai/paper/2312.06867","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06867"}},"official":{"repos":["wzy6642/PRP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-surface-probing-llama-across-scales","slug":"beyond-surface-probing-llama-across-scales","title":"Is Bigger and Deeper Always Better? Probing LLaMA Across Scales and Layers","date":"2023-12-07","arxiv_id":"2312.04333","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-surface-probing-llama-across-scales#ran","syntology_url":"https://syntology.ai/paper/2312.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04333"}},"official":{"repos":["nuochenpku/llama_analysis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-as-a-math-questioner-evaluating","slug":"chatgpt-as-a-math-questioner-evaluating","title":"ChatGPT as a Math Questioner? Evaluating ChatGPT on Generating Pre-university Math Questions","date":"2023-12-04","arxiv_id":"2312.01661","repositories_listed":1,"syntology":null},{"url":"/paper/eliciting-latent-knowledge-from-quirky","slug":"eliciting-latent-knowledge-from-quirky","title":"Eliciting Latent Knowledge from Quirky Language Models","date":"2023-12-02","arxiv_id":"2312.01037","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/eliciting-latent-knowledge-from-quirky#ran","syntology_url":"https://syntology.ai/paper/2312.01037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01037"}},"official":{"repos":["eleutherai/elk-generalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reds-resource-efficient-deep-subnetworks-for","slug":"reds-resource-efficient-deep-subnetworks-for","title":"REDS: Resource-Efficient Deep Subnetworks for Dynamic Resource Constraints","date":"2023-11-22","arxiv_id":"2311.13349","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reds-resource-efficient-deep-subnetworks-for#ran","syntology_url":"https://syntology.ai/paper/2311.13349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13349"}},"official":{"repos":["FraCorti/Deep_Subnetworks_for_Dynamic_Resource_Constraints"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-prompting-for-agi-systems","slug":"meta-prompting-for-agi-systems","title":"Meta Prompting for AI Systems","date":"2023-11-20","arxiv_id":"2311.11482","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-prompting-for-agi-systems#ran","syntology_url":"https://syntology.ai/paper/2311.11482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.11482"}},"official":{"repos":["meta-prompting/meta-prompting"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/system-2-attention-is-something-you-might","slug":"system-2-attention-is-something-you-might","title":"System 2 Attention (is something you might need too)","date":"2023-11-20","arxiv_id":"2311.11829","repositories_listed":1,"syntology":null},{"url":"/paper/docmath-eval-evaluating-numerical-reasoning","slug":"docmath-eval-evaluating-numerical-reasoning","title":"DocMath-Eval: Evaluating Math Reasoning Capabilities of LLMs in Understanding Long and Specialized Documents","date":"2023-11-16","arxiv_id":"2311.09805","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/docmath-eval-evaluating-numerical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2311.09805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09805"}},"official":{"repos":["yale-nlp/docmath-eval"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledgemath-knowledge-intensive-math-word","slug":"knowledgemath-knowledge-intensive-math-word","title":"FinanceMath: Knowledge-Intensive Math Reasoning in Finance Domains","date":"2023-11-16","arxiv_id":"2311.09797","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledgemath-knowledge-intensive-math-word#ran","syntology_url":"https://syntology.ai/paper/2311.09797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09797"}},"official":{"repos":["yale-nlp/knowledgemath"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/strategyllm-large-language-models-as-strategy","slug":"strategyllm-large-language-models-as-strategy","title":"StrategyLLM: Large Language Models as Strategy Generators, Executors, Optimizers, and Evaluators for Problem Solving","date":"2023-11-15","arxiv_id":"2311.08803","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/strategyllm-large-language-models-as-strategy#ran","syntology_url":"https://syntology.ai/paper/2311.08803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08803"}},"official":{"repos":["gao-xiao-bai/strategyllm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-reasoning-in-large-language-models","slug":"towards-reasoning-in-large-language-models","title":"Towards Reasoning in Large Language Models via Multi-Agent Peer Review Collaboration","date":"2023-11-14","arxiv_id":"2311.08152","repositories_listed":1,"syntology":null},{"url":"/paper/veritymath-advancing-mathematical-reasoning","slug":"veritymath-advancing-mathematical-reasoning","title":"VerityMath: Advancing Mathematical Reasoning by Self-Verification Through Unit Consistency","date":"2023-11-13","arxiv_id":"2311.07172","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/veritymath-advancing-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2311.07172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07172"}},"official":{"repos":["vernontoh/veritymath"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conic10k-a-challenging-math-problem","slug":"conic10k-a-challenging-math-problem","title":"Conic10K: A Challenging Math Problem Understanding and Reasoning Dataset","date":"2023-11-09","arxiv_id":"2311.05113","repositories_listed":1,"syntology":null},{"url":"/paper/bias-runs-deep-implicit-reasoning-biases-in","slug":"bias-runs-deep-implicit-reasoning-biases-in","title":"Bias Runs Deep: Implicit Reasoning Biases in Persona-Assigned LLMs","date":"2023-11-08","arxiv_id":"2311.04892","repositories_listed":1,"syntology":null},{"url":"/paper/locating-cross-task-sequence-continuation","slug":"locating-cross-task-sequence-continuation","title":"Towards Interpretable Sequence Continuation: Analyzing Shared Circuits in Large Language Models","date":"2023-11-07","arxiv_id":"2311.04131","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/locating-cross-task-sequence-continuation#ran","syntology_url":"https://syntology.ai/paper/2311.04131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04131"}},"official":{"repos":["apartresearch/seqcont_circuits"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/athena-mathematical-reasoning-with-thought","slug":"athena-mathematical-reasoning-with-thought","title":"ATHENA: Mathematical Reasoning with Thought Expansion","date":"2023-11-02","arxiv_id":"2311.01036","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-chain-of-thought-reasoning-via","slug":"implicit-chain-of-thought-reasoning-via","title":"Implicit Chain of Thought Reasoning via Knowledge Distillation","date":"2023-11-02","arxiv_id":"2311.01460","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-chain-of-thought-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2311.01460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01460"}},"official":{"repos":["da03/implicit_chain_of_thought"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-the-creative-mind-language-model","slug":"unleashing-the-creative-mind-language-model","title":"Unleashing the Creative Mind: Language Model As Hierarchical Policy For Improved Exploration on Challenging Problem Solving","date":"2023-11-01","arxiv_id":"2311.00694","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-mistakes-makes-llm-better","slug":"learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","arxiv_id":"2310.20689","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-mistakes-makes-llm-better#ran","syntology_url":"https://syntology.ai/paper/2310.20689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20689"}},"official":{"repos":["microsoft/lema"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-evaluation-exploring-the","slug":"collaborative-evaluation-exploring-the","title":"Exploring the Reliability of Large Language Models as Customized Evaluators for Diverse NLP Tasks","date":"2023-10-30","arxiv_id":"2310.19740","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/collaborative-evaluation-exploring-the#ran","syntology_url":"https://syntology.ai/paper/2310.19740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19740"}},"official":{"repos":["qtli/coeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/an-early-evaluation-of-gpt-4v-ision","slug":"an-early-evaluation-of-gpt-4v-ision","title":"An Early Evaluation of GPT-4V(ision)","date":"2023-10-25","arxiv_id":"2310.16534","repositories_listed":1,"syntology":null},{"url":"/paper/expression-syntax-information-bottleneck-for","slug":"expression-syntax-information-bottleneck-for","title":"Expression Syntax Information Bottleneck for Math Word Problems","date":"2023-10-24","arxiv_id":"2310.15664","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expression-syntax-information-bottleneck-for#ran","syntology_url":"https://syntology.ai/paper/2310.15664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15664"}},"official":{"repos":["menik1126/math_esib"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/plan-verify-and-switch-integrated-reasoning","slug":"plan-verify-and-switch-integrated-reasoning","title":"Plan, Verify and Switch: Integrated Reasoning with Diverse X-of-Thoughts","date":"2023-10-23","arxiv_id":"2310.14628","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plan-verify-and-switch-integrated-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.14628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14628"}},"official":{"repos":["tengxiaoliu/xot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/we-are-who-we-cite-bridges-of-influence","slug":"we-are-who-we-cite-bridges-of-influence","title":"We are Who We Cite: Bridges of Influence Between Natural Language Processing and Other Academic Fields","date":"2023-10-23","arxiv_id":"2310.14870","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-language-models-to-self-improve","slug":"teaching-language-models-to-self-improve","title":"Teaching Language Models to Self-Improve through Interactive Demonstrations","date":"2023-10-20","arxiv_id":"2310.13522","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-language-models-to-self-improve#ran","syntology_url":"https://syntology.ai/paper/2310.13522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13522"}},"official":{"repos":["jasonyux/tripost"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/step-by-step-remediation-of-students","slug":"step-by-step-remediation-of-students","title":"Bridging the Novice-Expert Gap via Models of Decision-Making: A Case Study on Remediating Math Mistakes","date":"2023-10-16","arxiv_id":"2310.10648","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-by-step-remediation-of-students#ran","syntology_url":"https://syntology.ai/paper/2310.10648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10648"}},"official":{"repos":["rosewang2008/bridge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-expression-tree-decoding-strategy-for","slug":"an-expression-tree-decoding-strategy-for","title":"An Expression Tree Decoding Strategy for Mathematical Equation Generation","date":"2023-10-14","arxiv_id":"2310.09619","repositories_listed":1,"syntology":null},{"url":"/paper/solving-math-word-problems-with-reexamination","slug":"solving-math-word-problems-with-reexamination","title":"Solving Math Word Problems with Reexamination","date":"2023-10-14","arxiv_id":"2310.09590","repositories_listed":1,"syntology":null},{"url":"/paper/syntax-error-free-and-generalizable-tool-use","slug":"syntax-error-free-and-generalizable-tool-use","title":"Don't Fine-Tune, Decode: Syntax Error-Free Tool Use via Constrained Decoding","date":"2023-10-10","arxiv_id":"2310.07075","repositories_listed":1,"syntology":null},{"url":"/paper/query-and-response-augmentation-cannot-help","slug":"query-and-response-augmentation-cannot-help","title":"MuggleMath: Assessing the Impact of Query and Response Augmentation on Math Reasoning","date":"2023-10-09","arxiv_id":"2310.05506","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder-seamless-code-integration-in-llms","slug":"mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","arxiv_id":"2310.03731","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-seamless-code-integration-in-llms#ran","syntology_url":"https://syntology.ai/paper/2310.03731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03731"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-rise-of-open-science-tracking-the","slug":"the-rise-of-open-science-tracking-the","title":"The Rise of Open Science: Tracking the Evolution and Perceived Value of Data and Methods Link-Sharing Practices","date":"2023-10-04","arxiv_id":"2310.03193","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fill-in-the-blank-exploring-and-enhancing-llm","slug":"fill-in-the-blank-exploring-and-enhancing-llm","title":"Fill in the Blank: Exploring and Enhancing LLM Capabilities for Backward Reasoning in Math Word Problems","date":"2023-10-03","arxiv_id":"2310.01991","repositories_listed":1,"syntology":null},{"url":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instance-needs-more-care-rewriting-prompts#ran","syntology_url":"https://syntology.ai/paper/2310.02107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02107"}},"official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/felm-benchmarking-factuality-evaluation-of","slug":"felm-benchmarking-factuality-evaluation-of","title":"FELM: Benchmarking Factuality Evaluation of Large Language Models","date":"2023-10-01","arxiv_id":"2310.00741","repositories_listed":1,"syntology":null},{"url":"/paper/tora-a-tool-integrated-reasoning-agent-for","slug":"tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","arxiv_id":"2309.17452","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tora-a-tool-integrated-reasoning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2309.17452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17452"}},"official":{"repos":["microsoft/tora"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nlpbench-evaluating-large-language-models-on","slug":"nlpbench-evaluating-large-language-models-on","title":"NLPBench: Evaluating Large Language Models on Solving NLP Problems","date":"2023-09-27","arxiv_id":"2309.15630","repositories_listed":1,"syntology":null},{"url":"/paper/metamath-bootstrap-your-own-mathematical","slug":"metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","arxiv_id":"2309.12284","repositories_listed":1,"syntology":{"n":22,"n_ran":15,"n_constructed":0,"n_ran_checked":1,"n_instrument":14,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/metamath-bootstrap-your-own-mathematical#ran","syntology_url":"https://syntology.ai/paper/2309.12284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12284"}},"official":{"repos":["meta-math/MetaMath"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/design-of-chain-of-thought-in-math-problem","slug":"design-of-chain-of-thought-in-math-problem","title":"Design of Chain-of-Thought in Math Problem Solving","date":"2023-09-20","arxiv_id":"2309.11054","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-embedded-programs-for-hybrid","slug":"natural-language-embedded-programs-for-hybrid","title":"Natural Language Embedded Programs for Hybrid Language Symbolic Reasoning","date":"2023-09-19","arxiv_id":"2309.10814","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/natural-language-embedded-programs-for-hybrid#ran","syntology_url":"https://syntology.ai/paper/2309.10814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10814"}},"official":{"repos":["luohongyin/langcode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/mammoth-building-math-generalist-models","slug":"mammoth-building-math-generalist-models","title":"MAmmoTH: Building Math Generalist Models through Hybrid Instruction Tuning","date":"2023-09-11","arxiv_id":"2309.05653","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammoth-building-math-generalist-models#ran","syntology_url":"https://syntology.ai/paper/2309.05653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05653"}},"official":null}}],"record_sha256":"19805e8ef18a3251cc9d644ecb450a37869ec72b77215c5d36116e94b82a3e59","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}