{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/logical-reasoning/papers/2","list_of":"/task/logical-reasoning","task":"Logical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":8,"rows_per_page":100,"rows":[101,200],"of":747,"counts":{"archive_papers_tagged":747,"with_a_code_link":330,"where_syntology_ran_a_sample":113,"not_listed_spam_title":0,"listed":747,"listed_where_code_ran":113,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/logical-reasoning","prev":"/task/logical-reasoning","next":"/task/logical-reasoning/papers/3","papers":[{"url":"/paper/a-survey-on-large-language-model-acceleration","slug":"a-survey-on-large-language-model-acceleration","title":"A Survey on Large Language Model Acceleration based on KV Cache Management","date":"2024-12-27","arxiv_id":"2412.19442","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-large-language-models-for-1","slug":"harnessing-large-language-models-for-1","title":"Harnessing Large Language Models for Knowledge Graph Question Answering via Adaptive Multi-Aspect Retrieval-Augmentation","date":"2024-12-24","arxiv_id":"2412.18537","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2412.18537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18537"}},"official":{"repos":["Applied-Machine-Learning-Lab/AMAR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphere-a-hierarchical-evaluation-on-spatial","slug":"sphere-a-hierarchical-evaluation-on-spatial","title":"SPHERE: A Hierarchical Evaluation on Spatial Perception and Reasoning for Vision-Language Models","date":"2024-12-17","arxiv_id":"2412.12693","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphere-a-hierarchical-evaluation-on-spatial#ran","syntology_url":"https://syntology.ai/paper/2412.12693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12693"}},"official":{"repos":["zwenyu/SPHERE-VLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wisead-knowledge-augmented-end-to-end","slug":"wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","arxiv_id":"2412.09951","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wisead-knowledge-augmented-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2412.09951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09951"}},"official":{"repos":["wyddmw/WiseAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rulearena-a-benchmark-for-rule-guided","slug":"rulearena-a-benchmark-for-rule-guided","title":"RuleArena: A Benchmark for Rule-Guided Reasoning with LLMs in Real-World Scenarios","date":"2024-12-12","arxiv_id":"2412.08972","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rulearena-a-benchmark-for-rule-guided#ran","syntology_url":"https://syntology.ai/paper/2412.08972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08972"}},"official":{"repos":["skyriver-2000/rulearena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/flashrnn-optimizing-traditional-rnns-on","slug":"flashrnn-optimizing-traditional-rnns-on","title":"FlashRNN: Optimizing Traditional RNNs on Modern Hardware","date":"2024-12-10","arxiv_id":"2412.07752","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flashrnn-optimizing-traditional-rnns-on#ran","syntology_url":"https://syntology.ai/paper/2412.07752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07752"}},"official":{"repos":["nx-ai/flashrnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-large-language-models-to-reason-in-a","slug":"training-large-language-models-to-reason-in-a","title":"Training Large Language Models to Reason in a Continuous Latent Space","date":"2024-12-09","arxiv_id":"2412.06769","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-large-language-models-to-reason-in-a#ran","syntology_url":"https://syntology.ai/paper/2412.06769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06769"}},"official":{"repos":["facebookresearch/coconut"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/who-speaks-next-multi-party-ai-discussion","slug":"who-speaks-next-multi-party-ai-discussion","title":"Who Speaks Next? Multi-party AI Discussion Leveraging the Systematics of Turn-taking in Murder Mystery Games","date":"2024-12-06","arxiv_id":"2412.04937","repositories_listed":1,"syntology":null},{"url":"/paper/clusterkv-manipulating-llm-kv-cache-in","slug":"clusterkv-manipulating-llm-kv-cache-in","title":"ClusterKV: Manipulating LLM KV Cache in Semantic Space for Recallable Compression","date":"2024-12-04","arxiv_id":"2412.03213","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clusterkv-manipulating-llm-kv-cache-in#ran","syntology_url":"https://syntology.ai/paper/2412.03213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03213"}},"official":{"repos":["sjtu-zhao-lab/clusterkv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-for-long-horizon-planning-via-neuro","slug":"learning-for-long-horizon-planning-via-neuro","title":"Learning for Long-Horizon Planning via Neuro-Symbolic Abductive Imitation","date":"2024-11-27","arxiv_id":"2411.18201","repositories_listed":1,"syntology":null},{"url":"/paper/object-centric-proto-symbolic-behavioural","slug":"object-centric-proto-symbolic-behavioural","title":"Object-centric proto-symbolic behavioural reasoning from pixels","date":"2024-11-26","arxiv_id":"2411.17438","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-creativity-and-deception-in-large","slug":"evaluating-creativity-and-deception-in-large","title":"Evaluating Creativity and Deception in Large Language Models: A Simulation Framework for Multi-Agent Balderdash","date":"2024-11-15","arxiv_id":"2411.10422","repositories_listed":1,"syntology":null},{"url":"/paper/the-llm-language-network-a-neuroscientific","slug":"the-llm-language-network-a-neuroscientific","title":"The LLM Language Network: A Neuroscientific Approach for Identifying Causally Task-Relevant Units","date":"2024-11-04","arxiv_id":"2411.02280","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-llm-language-network-a-neuroscientific#ran","syntology_url":"https://syntology.ai/paper/2411.02280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02280"}},"official":{"repos":["bkhmsi/llm-localization"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/logicity-advancing-neuro-symbolic-ai-with","slug":"logicity-advancing-neuro-symbolic-ai-with","title":"LogiCity: Advancing Neuro-Symbolic AI with Abstract Urban Simulation","date":"2024-11-01","arxiv_id":"2411.00773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/logicity-advancing-neuro-symbolic-ai-with#ran","syntology_url":"https://syntology.ai/paper/2411.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00773"}},"official":{"repos":["Jaraxxus-Me/LogiCity"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-llms-for-hypothetical-deduction-in","slug":"leveraging-llms-for-hypothetical-deduction-in","title":"Leveraging LLMs for Hypothetical Deduction in Logical Inference: A Neuro-Symbolic Approach","date":"2024-10-29","arxiv_id":"2410.21779","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-symbolic-learning-yielding-logical-1","slug":"neuro-symbolic-learning-yielding-logical-1","title":"Neuro-symbolic Learning Yielding Logical Constraints","date":"2024-10-28","arxiv_id":"2410.20957","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/neuro-symbolic-learning-yielding-logical-1#ran","syntology_url":"https://syntology.ai/paper/2410.20957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20957"}},"official":{"repos":["lizn-zn/nesy-programming"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/medlogic-aqa-enhancing-medical-question","slug":"medlogic-aqa-enhancing-medical-question","title":"MedLogic-AQA: Enhancing Medical Question Answering with Abstractive Models Focusing on Logical Structures","date":"2024-10-20","arxiv_id":"2410.15463","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-autoregressive-llm-knowledge-of","slug":"uncovering-autoregressive-llm-knowledge-of","title":"Uncovering Autoregressive LLM Knowledge of Thematic Fit in Event Representation","date":"2024-10-19","arxiv_id":"2410.15173","repositories_listed":1,"syntology":null},{"url":"/paper/from-babbling-to-fluency-evaluating-the","slug":"from-babbling-to-fluency-evaluating-the","title":"From Babbling to Fluency: Evaluating the Evolution of Language Models in Terms of Human Language Acquisition","date":"2024-10-17","arxiv_id":"2410.13259","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-curriculum-expert-iteration-for","slug":"automatic-curriculum-expert-iteration-for","title":"Automatic Curriculum Expert Iteration for Reliable LLM Reasoning","date":"2024-10-10","arxiv_id":"2410.07627","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-curriculum-expert-iteration-for#ran","syntology_url":"https://syntology.ai/paper/2410.07627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07627"}},"official":{"repos":["salesforceairesearch/auto-cei"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/divide-and-translate-compositional-first","slug":"divide-and-translate-compositional-first","title":"Divide and Translate: Compositional First-Order Logic Translation and Verification for Complex Logical Reasoning","date":"2024-10-10","arxiv_id":"2410.08047","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divide-and-translate-compositional-first#ran","syntology_url":"https://syntology.ai/paper/2410.08047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08047"}},"official":{"repos":["Hyun-Ryu/clover"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hlm-cite-hybrid-language-model-workflow-for","slug":"hlm-cite-hybrid-language-model-workflow-for","title":"HLM-Cite: Hybrid Language Model Workflow for Text-based Scientific Citation Prediction","date":"2024-10-10","arxiv_id":"2410.09112","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hlm-cite-hybrid-language-model-workflow-for#ran","syntology_url":"https://syntology.ai/paper/2410.09112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09112"}},"official":{"repos":["tsinghua-fib-lab/H-LM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/which-programming-language-and-what-features","slug":"which-programming-language-and-what-features","title":"Which Programming Language and What Features at Pre-training Stage Affect Downstream Logical Inference Performance?","date":"2024-10-09","arxiv_id":"2410.06735","repositories_listed":1,"syntology":null},{"url":"/paper/turtlebench-evaluating-top-language-models","slug":"turtlebench-evaluating-top-language-models","title":"TurtleBench: Evaluating Top Language Models via Real-World Yes/No Puzzles","date":"2024-10-07","arxiv_id":"2410.05262","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/turtlebench-evaluating-top-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05262"}},"official":{"repos":["mazzzystar/TurtleBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpret-your-decision-logical-reasoning","slug":"interpret-your-decision-logical-reasoning","title":"Interpret Your Decision: Logical Reasoning Regularization for Generalization in Visual Classification","date":"2024-10-06","arxiv_id":"2410.04492","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interpret-your-decision-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2410.04492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04492"}},"official":{"repos":["zhaorui-tan/L-Reg_NeurIPS24"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhance-reasoning-by-learning-from-mistakes","slug":"enhance-reasoning-by-learning-from-mistakes","title":"Learning from Committee: Reasoning Distillation from a Mixture of Teachers with Peer-Review","date":"2024-10-04","arxiv_id":"2410.03663","repositories_listed":1,"syntology":null},{"url":"/paper/babelbench-an-omni-benchmark-for-code-driven","slug":"babelbench-an-omni-benchmark-for-code-driven","title":"BabelBench: An Omni Benchmark for Code-Driven Analysis of Multimodal and Multistructured Data","date":"2024-10-01","arxiv_id":"2410.00773","repositories_listed":1,"syntology":null},{"url":"/paper/rationalyst-pre-training-process-supervision","slug":"rationalyst-pre-training-process-supervision","title":"RATIONALYST: Pre-training Process-Supervision for Improving Reasoning","date":"2024-10-01","arxiv_id":"2410.01044","repositories_listed":1,"syntology":null},{"url":"/paper/logic-of-thought-injecting-logic-into","slug":"logic-of-thought-injecting-logic-into","title":"Logic-of-Thought: Injecting Logic into Contexts for Full Reasoning in Large Language Models","date":"2024-09-26","arxiv_id":"2409.17539","repositories_listed":1,"syntology":null},{"url":"/paper/ltntorch-pytorch-implementation-of-logic","slug":"ltntorch-pytorch-implementation-of-logic","title":"LTNtorch: PyTorch Implementation of Logic Tensor Networks","date":"2024-09-24","arxiv_id":"2409.16045","repositories_listed":1,"syntology":null},{"url":"/paper/strategies-for-improving-nl-to-fol","slug":"strategies-for-improving-nl-to-fol","title":"Strategies for Improving NL-to-FOL Translation with LLMs: Data Generation, Incremental Fine-Tuning, and Verification","date":"2024-09-24","arxiv_id":"2409.16461","repositories_listed":1,"syntology":null},{"url":"/paper/thought-path-contrastive-learning-via-premise","slug":"thought-path-contrastive-learning-via-premise","title":"Thought-Path Contrastive Learning via Premise-Oriented Data Augmentation for Logical Reading Comprehension","date":"2024-09-22","arxiv_id":"2409.14495","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-logical-reasoning-in-large-language-1","slug":"enhancing-logical-reasoning-in-large-language-1","title":"Enhancing Logical Reasoning in Large Language Models through Graph-based Synthetic Data","date":"2024-09-19","arxiv_id":"2409.12437","repositories_listed":1,"syntology":null},{"url":"/paper/logicpro-improving-complex-logical-reasoning","slug":"logicpro-improving-complex-logical-reasoning","title":"LogicPro: Improving Complex Logical Reasoning via Program-Guided Learning","date":"2024-09-19","arxiv_id":"2409.12929","repositories_listed":1,"syntology":null},{"url":"/paper/vprochart-answering-chart-question-through","slug":"vprochart-answering-chart-question-through","title":"VProChart: Answering Chart Question through Visual Perception Alignment Agent and Programmatic Solution Reasoning","date":"2024-09-03","arxiv_id":"2409.01667","repositories_listed":1,"syntology":null},{"url":"/paper/logicgame-benchmarking-rule-based-reasoning","slug":"logicgame-benchmarking-rule-based-reasoning","title":"LogicGame: Benchmarking Rule-Based Reasoning Abilities of Large Language Models","date":"2024-08-28","arxiv_id":"2408.15778","repositories_listed":1,"syntology":null},{"url":"/paper/checkwhy-causal-fact-verification-via","slug":"checkwhy-causal-fact-verification-via","title":"CHECKWHY: Causal Fact Verification via Argument Structure","date":"2024-08-20","arxiv_id":"2408.10918","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-reasoning-biases-in-large-language","slug":"exploring-reasoning-biases-in-large-language","title":"Exploring Reasoning Biases in Large Language Models Through Syllogism: Insights from the NeuBAROCO Dataset","date":"2024-08-08","arxiv_id":"2408.04403","repositories_listed":1,"syntology":null},{"url":"/paper/step-by-step-reasoning-to-solve-grid-puzzles","slug":"step-by-step-reasoning-to-solve-grid-puzzles","title":"Step-by-Step Reasoning to Solve Grid Puzzles: Where do LLMs Falter?","date":"2024-07-20","arxiv_id":"2407.14790","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-large-language-models-for-nano","slug":"leveraging-large-language-models-for-nano","title":"Leveraging large language models for nano synthesis mechanism explanation: solid foundations or mere conjectures?","date":"2024-07-12","arxiv_id":"2407.08922","repositories_listed":1,"syntology":null},{"url":"/paper/hypergraph-multi-modal-large-language-model","slug":"hypergraph-multi-modal-large-language-model","title":"Hypergraph Multi-modal Large Language Model: Exploiting EEG and Eye-tracking Modalities to Evaluate Heterogeneous Responses for Video Understanding","date":"2024-07-11","arxiv_id":"2407.08150","repositories_listed":1,"syntology":null},{"url":"/paper/r-2-guard-robust-reasoning-enabled-llm","slug":"r-2-guard-robust-reasoning-enabled-llm","title":"$R^2$-Guard: Robust Reasoning Enabled LLM Guardrail via Knowledge-Enhanced Logical Reasoning","date":"2024-07-08","arxiv_id":"2407.05557","repositories_listed":1,"syntology":null},{"url":"/paper/elecbench-a-power-dispatch-evaluation","slug":"elecbench-a-power-dispatch-evaluation","title":"ElecBench: a Power Dispatch Evaluation Benchmark for Large Language Models","date":"2024-07-07","arxiv_id":"2407.05365","repositories_listed":1,"syntology":null},{"url":"/paper/logicvista-multimodal-llm-logical-reasoning","slug":"logicvista-multimodal-llm-logical-reasoning","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","date":"2024-07-06","arxiv_id":"2407.04973","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicvista-multimodal-llm-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.04973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04973"}},"official":{"repos":["yijia-xiao/logicvista"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/puzzles-a-benchmark-for-neural-algorithmic","slug":"puzzles-a-benchmark-for-neural-algorithmic","title":"PUZZLES: A Benchmark for Neural Algorithmic Reasoning","date":"2024-06-29","arxiv_id":"2407.00401","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puzzles-a-benchmark-for-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2407.00401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00401"}},"official":{"repos":["eth-disco/rlp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-cross-lingual","slug":"large-language-models-are-cross-lingual","title":"Large Language Models Are Cross-Lingual Knowledge-Free Reasoners","date":"2024-06-24","arxiv_id":"2406.16655","repositories_listed":1,"syntology":null},{"url":"/paper/multi-logieval-towards-evaluating-multi-step","slug":"multi-logieval-towards-evaluating-multi-step","title":"Multi-LogiEval: Towards Evaluating Multi-Step Logical Reasoning Ability of Large Language Models","date":"2024-06-24","arxiv_id":"2406.17169","repositories_listed":1,"syntology":null},{"url":"/paper/liar-liar-logical-mire-a-benchmark-for","slug":"liar-liar-logical-mire-a-benchmark-for","title":"Liar, Liar, Logical Mire: A Benchmark for Suppositional Reasoning in Large Language Models","date":"2024-06-18","arxiv_id":"2406.12546","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/liar-liar-logical-mire-a-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.12546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12546"}},"official":{"repos":["mainlp/TruthQuest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/videovista-a-versatile-benchmark-for-video","slug":"videovista-a-versatile-benchmark-for-video","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","date":"2024-06-17","arxiv_id":"2406.11303","repositories_listed":1,"syntology":null},{"url":"/paper/a-peek-into-token-bias-large-language-models","slug":"a-peek-into-token-bias-large-language-models","title":"A Peek into Token Bias: Large Language Models Are Not Yet Genuine Reasoners","date":"2024-06-16","arxiv_id":"2406.11050","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-peek-into-token-bias-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.11050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11050"}},"official":{"repos":["bowen-upenn/llm_token_bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ontology-embedding-a-survey-of-methods","slug":"ontology-embedding-a-survey-of-methods","title":"Ontology Embedding: A Survey of Methods, Applications and Resources","date":"2024-06-16","arxiv_id":"2406.10964","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-synthetic-logical-reasoning-datasets","slug":"scaling-synthetic-logical-reasoning-datasets","title":"Scaling Synthetic Logical Reasoning Datasets with Context-Sensitive Declarative Grammars","date":"2024-06-16","arxiv_id":"2406.11035","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-chatgpt-4-vision-on-brazil-s","slug":"evaluating-chatgpt-4-vision-on-brazil-s","title":"Evaluating ChatGPT-4 Vision on Brazil's National Undergraduate Computer Science Exam","date":"2024-06-14","arxiv_id":"2406.09671","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-preference-optimization-improving","slug":"chain-of-preference-optimization-improving","title":"Chain of Preference Optimization: Improving Chain-of-Thought Reasoning in LLMs","date":"2024-06-13","arxiv_id":"2406.09136","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-preference-optimization-improving#ran","syntology_url":"https://syntology.ai/paper/2406.09136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09136"}},"official":{"repos":["sail-sg/cpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-thinking-and-perceptual-analysis-of-deep","slug":"dual-thinking-and-perceptual-analysis-of-deep","title":"Dual Thinking and Logical Processing -- Are Multi-modal Large Language Models Closing the Gap with Human Vision ?","date":"2024-06-11","arxiv_id":"2406.06967","repositories_listed":1,"syntology":null},{"url":"/paper/improving-multi-hop-logical-reasoning-in","slug":"improving-multi-hop-logical-reasoning-in","title":"Improving Multi-hop Logical Reasoning in Knowledge Graphs with Context-Aware Query Representation Learning","date":"2024-06-11","arxiv_id":"2406.07034","repositories_listed":1,"syntology":null},{"url":"/paper/limited-out-of-context-knowledge-reasoning-in","slug":"limited-out-of-context-knowledge-reasoning-in","title":"Large Language Models are Limited in Out-of-Context Knowledge Reasoning","date":"2024-06-11","arxiv_id":"2406.07393","repositories_listed":1,"syntology":null},{"url":"/paper/lingoly-a-benchmark-of-olympiad-level","slug":"lingoly-a-benchmark-of-olympiad-level","title":"LINGOLY: A Benchmark of Olympiad-Level Linguistic Reasoning Puzzles in Low-Resource and Extinct Languages","date":"2024-06-10","arxiv_id":"2406.06196","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lingoly-a-benchmark-of-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2406.06196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06196"}},"official":{"repos":["am-bean/lingOly"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-of-reasoning-efficient-training-of-llm","slug":"flow-of-reasoning-efficient-training-of-llm","title":"Flow of Reasoning:Training LLMs for Divergent Problem Solving with Minimal Examples","date":"2024-06-09","arxiv_id":"2406.05673","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-of-reasoning-efficient-training-of-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05673"}},"official":{"repos":["yu-fangxu/for"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/logicode-an-llm-driven-framework-for-logical","slug":"logicode-an-llm-driven-framework-for-logical","title":"LogiCode: an LLM-Driven Framework for Logical Anomaly Detection","date":"2024-06-07","arxiv_id":"2406.04687","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"0 ran · 5 unverified","sample_list":"/paper/logicode-an-llm-driven-framework-for-logical#ran","syntology_url":"https://syntology.ai/paper/2406.04687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04687"}},"official":{"repos":["22strongestme/LOCO-Annotations"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/evaluating-the-world-model-implicit-in-a","slug":"evaluating-the-world-model-implicit-in-a","title":"Evaluating the World Model Implicit in a Generative Model","date":"2024-06-06","arxiv_id":"2406.03689","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-the-world-model-implicit-in-a#ran","syntology_url":"https://syntology.ai/paper/2406.03689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03689"}},"official":{"repos":["keyonvafa/world-model-evaluation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-hardness-of-probabilistic","slug":"on-the-hardness-of-probabilistic","title":"On the Hardness of Probabilistic Neurosymbolic Learning","date":"2024-06-06","arxiv_id":"2406.04472","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-hardness-of-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2406.04472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04472"}},"official":{"repos":["jjcmoon/hardness-nesy"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-closer-look-at-logical-reasoning-with-llms","slug":"a-closer-look-at-logical-reasoning-with-llms","title":"A Closer Look at Logical Reasoning with LLMs: The Choice of Tool Matters","date":"2024-06-01","arxiv_id":"2406.00284","repositories_listed":1,"syntology":null},{"url":"/paper/easy-problems-that-llms-get-wrong","slug":"easy-problems-that-llms-get-wrong","title":"Easy Problems That LLMs Get Wrong","date":"2024-05-30","arxiv_id":"2405.19616","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/easy-problems-that-llms-get-wrong#ran","syntology_url":"https://syntology.ai/paper/2405.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19616"}},"official":{"repos":["autogenai/easy-problems-that-llms-get-wrong"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/faithful-logical-reasoning-via-symbolic-chain","slug":"faithful-logical-reasoning-via-symbolic-chain","title":"Faithful Logical Reasoning via Symbolic Chain-of-Thought","date":"2024-05-28","arxiv_id":"2405.18357","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/faithful-logical-reasoning-via-symbolic-chain#ran","syntology_url":"https://syntology.ai/paper/2405.18357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18357"}},"official":{"repos":["aiden0526/symbcot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continuously-learning-adapting-and-improving","slug":"continuously-learning-adapting-and-improving","title":"Continuously Learning, Adapting, and Improving: A Dual-Process Approach to Autonomous Driving","date":"2024-05-24","arxiv_id":"2405.15324","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-theoretical-understanding-of-the-2","slug":"towards-a-theoretical-understanding-of-the-2","title":"Towards a Theoretical Understanding of the 'Reversal Curse' via Training Dynamics","date":"2024-05-07","arxiv_id":"2405.04669","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-a-theoretical-understanding-of-the-2#ran","syntology_url":"https://syntology.ai/paper/2405.04669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04669"}},"official":{"repos":["marlo-z/reversal_curse_analysis"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-knowledge-graphs-provided-by-humans","slug":"aligning-knowledge-graphs-provided-by-humans","title":"Aligning Knowledge Graphs Provided by Humans and Generated from Neural Networks in Specific Tasks","date":"2024-04-23","arxiv_id":"2404.16884","repositories_listed":1,"syntology":null},{"url":"/paper/towards-systematic-evaluation-of-logical","slug":"towards-systematic-evaluation-of-logical","title":"LogicBench: Towards Systematic Evaluation of Logical Reasoning Ability of Large Language Models","date":"2024-04-23","arxiv_id":"2404.15522","repositories_listed":1,"syntology":null},{"url":"/paper/macm-utilizing-a-multi-agent-system-for","slug":"macm-utilizing-a-multi-agent-system-for","title":"MACM: Utilizing a Multi-Agent System for Condition Mining in Solving Complex Mathematical Problems","date":"2024-04-06","arxiv_id":"2404.04735","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/macm-utilizing-a-multi-agent-system-for#ran","syntology_url":"https://syntology.ai/paper/2404.04735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04735"}},"official":{"repos":["bin123apple/macm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-llm-reasoning-generalists-with#ran","syntology_url":"https://syntology.ai/paper/2404.02078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02078"}},"official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-model-guided-interpretable-video","slug":"language-model-guided-interpretable-video","title":"Language Model Guided Interpretable Video Action Reasoning","date":"2024-04-02","arxiv_id":"2404.01591","repositories_listed":1,"syntology":null},{"url":"/paper/classifying-conspiratorial-narratives-at","slug":"classifying-conspiratorial-narratives-at","title":"Classifying Conspiratorial Narratives At Scale: False Alarms and Erroneous Connections","date":"2024-03-29","arxiv_id":"2404.00141","repositories_listed":1,"syntology":null},{"url":"/paper/leanreasoner-boosting-complex-logical","slug":"leanreasoner-boosting-complex-logical","title":"LeanReasoner: Boosting Complex Logical Reasoning with Lean","date":"2024-03-20","arxiv_id":"2403.13312","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leanreasoner-boosting-complex-logical#ran","syntology_url":"https://syntology.ai/paper/2403.13312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13312"}},"official":{"repos":["some-random/theorem-proving-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transforming-competition-into-collaboration","slug":"transforming-competition-into-collaboration","title":"Transforming Competition into Collaboration: The Revolutionary Role of Multi-Agent Systems and Language Models in Modern Organizations","date":"2024-03-12","arxiv_id":"2403.07769","repositories_listed":1,"syntology":null},{"url":"/paper/hint-before-solving-prompting-guiding-llms-to","slug":"hint-before-solving-prompting-guiding-llms-to","title":"Hint-before-Solving Prompting: Guiding LLMs to Effectively Utilize Encoded Knowledge","date":"2024-02-22","arxiv_id":"2402.14310","repositories_listed":1,"syntology":null},{"url":"/paper/simplot-enhancing-chart-question-answering-by","slug":"simplot-enhancing-chart-question-answering-by","title":"SIMPLOT: Enhancing Chart Question Answering by Distilling Essentials","date":"2024-02-22","arxiv_id":"2405.00021","repositories_listed":1,"syntology":null},{"url":"/paper/omgeval-an-open-multilingual-generative","slug":"omgeval-an-open-multilingual-generative","title":"OMGEval: An Open Multilingual Generative Evaluation Benchmark for Large Language Models","date":"2024-02-21","arxiv_id":"2402.13524","repositories_listed":1,"syntology":null},{"url":"/paper/science-checker-reloaded-a-bidirectional","slug":"science-checker-reloaded-a-bidirectional","title":"Science Checker Reloaded: A Bidirectional Paradigm for Transparency and Logical Reasoning","date":"2024-02-21","arxiv_id":"2402.13897","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-logical-message-passing","slug":"conditional-logical-message-passing","title":"Conditional Logical Message Passing Transformer for Complex Query Answering","date":"2024-02-20","arxiv_id":"2402.12954","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-reason-with-rules-logic-scaffolding","slug":"can-llms-reason-with-rules-logic-scaffolding","title":"Can LLMs Reason with Rules? Logic Scaffolding for Stress-Testing and Improving LLMs","date":"2024-02-18","arxiv_id":"2402.11442","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llms-reason-with-rules-logic-scaffolding#ran","syntology_url":"https://syntology.ai/paper/2402.11442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11442"}},"official":{"repos":["siyuanwangw/ulogic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-quantified-boolean-bayesian-network","slug":"the-quantified-boolean-bayesian-network","title":"The Quantified Boolean Bayesian Network: Theory and Experiments with a Logical Graphical Model","date":"2024-02-09","arxiv_id":"2402.06557","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-and-modal-reasoning-in-large","slug":"conditional-and-modal-reasoning-in-large","title":"Conditional and Modal Reasoning in Large Language Models","date":"2024-01-30","arxiv_id":"2401.17169","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-document-level-relation-extraction","slug":"revisiting-document-level-relation-extraction","title":"Revisiting Document-Level Relation Extraction with Context-Guided Link Prediction","date":"2024-01-22","arxiv_id":"2401.11800","repositories_listed":1,"syntology":null},{"url":"/paper/langbridge-multilingual-reasoning-without","slug":"langbridge-multilingual-reasoning-without","title":"LangBridge: Multilingual Reasoning Without Multilingual Supervision","date":"2024-01-19","arxiv_id":"2401.10695","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/langbridge-multilingual-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2401.10695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10695"}},"official":{"repos":["kaistAI/LangBridge"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-b-b-a-triggering-logical-reasoning-failures#ran","syntology_url":"https://syntology.ai/paper/2401.00757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00757"}},"official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/abductive-logical-reasoning-on-knowledge","slug":"abductive-logical-reasoning-on-knowledge","title":"Advancing Abductive Reasoning in Knowledge Graphs through Complex Logical Hypothesis Generation","date":"2023-12-25","arxiv_id":"2312.15643","repositories_listed":1,"syntology":null},{"url":"/paper/teilp-time-prediction-over-knowledge-graphs","slug":"teilp-time-prediction-over-knowledge-graphs","title":"TEILP: Time Prediction over Knowledge Graphs via Logical Reasoning","date":"2023-12-25","arxiv_id":"2312.15816","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/teilp-time-prediction-over-knowledge-graphs#ran","syntology_url":"https://syntology.ai/paper/2312.15816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15816"}},"official":null}},{"url":"/paper/empowering-few-shot-recommender-systems-with","slug":"empowering-few-shot-recommender-systems-with","title":"Empowering Few-Shot Recommender Systems with Large Language Models -- Enhanced Representations","date":"2023-12-21","arxiv_id":"2312.13557","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-inter-session-intentions-via","slug":"understanding-inter-session-intentions-via","title":"Understanding Inter-Session Intentions via Complex Logical Reasoning","date":"2023-12-21","arxiv_id":"2312.13866","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-logical-reasoning-capabilities-of","slug":"assessing-logical-reasoning-capabilities-of","title":"Assessing Logical Reasoning Capabilities of Encoder-Only Transformer Models","date":"2023-12-18","arxiv_id":"2312.11720","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-complex-mathematical-reasoning-via","slug":"modeling-complex-mathematical-reasoning-via","title":"Modeling Complex Mathematical Reasoning via Large Language Model based MathAgent","date":"2023-12-14","arxiv_id":"2312.08926","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-large-language-models-llms-succumb-to","slug":"not-all-large-language-models-llms-succumb-to","title":"Exploring the Reversal Curse and Other Deductive Logical Reasoning in BERT and GPT-Based Large Language Models","date":"2023-12-06","arxiv_id":"2312.03633","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-think-outside-the-box-exploring-leap-of","slug":"let-s-think-outside-the-box-exploring-leap-of","title":"Let's Think Outside the Box: Exploring Leap-of-Thought in Large Language Models with Creative Humor Generation","date":"2023-12-05","arxiv_id":"2312.02439","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-symbolic-integration-brings-causal-and","slug":"neuro-symbolic-integration-brings-causal-and","title":"Neuro-Symbolic Integration Brings Causal and Reliable Reasoning Proofs","date":"2023-11-16","arxiv_id":"2311.09802","repositories_listed":1,"syntology":null},{"url":"/paper/a-closer-look-at-the-self-verification","slug":"a-closer-look-at-the-self-verification","title":"A Closer Look at the Self-Verification Abilities of Large Language Models in Logical Reasoning","date":"2023-11-14","arxiv_id":"2311.07954","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-logical-puzzle-solving-in-large","slug":"assessing-logical-puzzle-solving-in-large","title":"Assessing Logical Puzzle Solving in Large Language Models: Insights from a Minesweeper Case Study","date":"2023-11-13","arxiv_id":"2311.07387","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/assessing-logical-puzzle-solving-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.07387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07387"}},"official":{"repos":["yinghao-li/minesweeper-for-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-images-for-intuitively-reasoning","slug":"chain-of-images-for-intuitively-reasoning","title":"Chain of Images for Intuitively Reasoning","date":"2023-11-09","arxiv_id":"2311.09241","repositories_listed":1,"syntology":null},{"url":"/paper/rule-learning-as-machine-translation-using","slug":"rule-learning-as-machine-translation-using","title":"Rule Learning as Machine Translation using the Atomic Knowledge Bank","date":"2023-11-05","arxiv_id":"2311.02765","repositories_listed":1,"syntology":null}],"record_sha256":"e0274972eed3a21b8760c58fd2b61e1141f597fdb56ee3152a5a01ba6454ea79","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}