{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/logical-reasoning/papers/4","list_of":"/task/logical-reasoning","task":"Logical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":8,"rows_per_page":100,"rows":[301,400],"of":747,"counts":{"archive_papers_tagged":747,"with_a_code_link":330,"where_syntology_ran_a_sample":113,"not_listed_spam_title":0,"listed":747,"listed_where_code_ran":113,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/logical-reasoning","prev":"/task/logical-reasoning/papers/3","next":"/task/logical-reasoning/papers/5","papers":[{"url":"/paper/reasoning-like-program-executors-1","slug":"reasoning-like-program-executors-1","title":"Reasoning Like Program Executors","date":"2022-01-27","arxiv_id":"2201.11473","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-based-multilingual-language-model-1","slug":"knowledge-based-multilingual-language-model-1","title":"Enhancing Multilingual Language Model with Massive Multilingual Knowledge Triples","date":"2021-11-22","arxiv_id":"2111.10962","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-entity-representation-model-for","slug":"probabilistic-entity-representation-model-for","title":"Probabilistic Entity Representation Model for Reasoning over Knowledge Graphs","date":"2021-10-26","arxiv_id":"2110.13522","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/probabilistic-entity-representation-model-for#ran","syntology_url":"https://syntology.ai/paper/2110.13522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13522"}},"official":{"repos":["akirato/perm-gaussiankg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/one-step-abductive-multi-target-learning-with","slug":"one-step-abductive-multi-target-learning-with","title":"One-Step Abductive Multi-Target Learning with Diverse Noisy Samples and Its Application to Tumour Segmentation for Breast Cancer","date":"2021-10-20","arxiv_id":"2110.10325","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-explainable-phrasal","slug":"weakly-supervised-explainable-phrasal","title":"Weakly Supervised Explainable Phrasal Reasoning with Neural Fuzzy Logic","date":"2021-09-18","arxiv_id":"2109.08927","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/weakly-supervised-explainable-phrasal#ran","syntology_url":"https://syntology.ai/paper/2109.08927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.08927"}},"official":{"repos":["manga-uofa/epr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-adversarial-learning-with","slug":"counterfactual-adversarial-learning-with","title":"Counterfactual Adversarial Learning with Representation Interpolation","date":"2021-09-10","arxiv_id":"2109.04746","repositories_listed":1,"syntology":null},{"url":"/paper/integration-of-data-and-theory-for","slug":"integration-of-data-and-theory-for","title":"AI Descartes: Combining Data and Theory for Derivable Scientific Discovery","date":"2021-09-03","arxiv_id":"2109.01634","repositories_listed":1,"syntology":null},{"url":"/paper/from-lsat-the-progress-and-challenges-of","slug":"from-lsat-the-progress-and-challenges-of","title":"From LSAT: The Progress and Challenges of Complex Reasoning","date":"2021-08-02","arxiv_id":"2108.00648","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/from-lsat-the-progress-and-challenges-of#ran","syntology_url":"https://syntology.ai/paper/2108.00648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.00648"}},"official":null}},{"url":"/paper/reasoning-with-transformer-based-models-deep","slug":"reasoning-with-transformer-based-models-deep","title":"Reasoning with Transformer-based Models: Deep Learning, but Shallow Reasoning","date":"2021-06-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/techniques-for-symbol-grounding-with-satnet","slug":"techniques-for-symbol-grounding-with-satnet","title":"Techniques for Symbol Grounding with SATNet","date":"2021-06-16","arxiv_id":"2106.11072","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/techniques-for-symbol-grounding-with-satnet#ran","syntology_url":"https://syntology.ai/paper/2106.11072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11072"}},"official":{"repos":["SeverTopan/SATNet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/query-embedding-on-hyper-relational-knowledge","slug":"query-embedding-on-hyper-relational-knowledge","title":"Query Embedding on Hyper-relational Knowledge Graphs","date":"2021-06-15","arxiv_id":"2106.08166","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/query-embedding-on-hyper-relational-knowledge#ran","syntology_url":"https://syntology.ai/paper/2106.08166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08166"}},"official":{"repos":["DimitrisAlivas/StarQE"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sire-separate-intra-and-inter-sentential","slug":"sire-separate-intra-and-inter-sentential","title":"SIRE: Separate Intra- and Inter-sentential Reasoning for Document-level Relation Extraction","date":"2021-06-03","arxiv_id":"2106.01709","repositories_listed":1,"syntology":null},{"url":"/paper/volta-at-semeval-2021-task-9-statement","slug":"volta-at-semeval-2021-task-9-statement","title":"Volta at SemEval-2021 Task 9: Statement Verification and Evidence Finding with Tables using TAPAS and Transfer Learning","date":"2021-06-01","arxiv_id":"2106.00248","repositories_listed":1,"syntology":null},{"url":"/paper/neurallog-natural-language-inference-with","slug":"neurallog-natural-language-inference-with","title":"NeuralLog: Natural Language Inference with Joint Neural and Logical Reasoning","date":"2021-05-29","arxiv_id":"2105.14167","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neurallog-natural-language-inference-with#ran","syntology_url":"https://syntology.ai/paper/2105.14167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14167"}},"official":{"repos":["eric11eca/NeuralLog"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-sufficient-explanations","slug":"probabilistic-sufficient-explanations","title":"Probabilistic Sufficient Explanations","date":"2021-05-21","arxiv_id":"2105.10118","repositories_listed":1,"syntology":null},{"url":"/paper/context-transformer-with-stacked-pointer","slug":"context-transformer-with-stacked-pointer","title":"Context Transformer with Stacked Pointer Networks for Conversational Question Answering over Knowledge Graphs","date":"2021-03-13","arxiv_id":"2103.07766","repositories_listed":1,"syntology":null},{"url":"/paper/neural-sequence-to-grid-module-for-learning","slug":"neural-sequence-to-grid-module-for-learning","title":"Neural Sequence-to-grid Module for Learning Symbolic Rules","date":"2021-01-13","arxiv_id":"2101.04921","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-inference-in-context","slug":"natural-language-inference-in-context","title":"Natural Language Inference in Context -- Investigating Contextual Reasoning over Long Texts","date":"2020-11-10","arxiv_id":"2011.04864","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-language-inference-in-context#ran","syntology_url":"https://syntology.ai/paper/2011.04864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.04864"}},"official":{"repos":["csitfun/ConTRoL-dataset"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reco-a-large-scale-chinese-reading","slug":"reco-a-large-scale-chinese-reading","title":"ReCO: A Large Scale Chinese Reading Comprehension Dataset on Opinion","date":"2020-06-22","arxiv_id":"2006.12146","repositories_listed":1,"syntology":null},{"url":"/paper/logic-and-the-2-simplicial-transformer-1","slug":"logic-and-the-2-simplicial-transformer-1","title":"Logic and the 2-Simplicial Transformer","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-robustness-for-sensing-reasoning","slug":"end-to-end-robustness-for-sensing-reasoning","title":"Improving Certified Robustness via Statistical Learning with Logical Reasoning","date":"2020-02-28","arxiv_id":"2003.00120","repositories_listed":1,"syntology":null},{"url":"/paper/reclor-a-reading-comprehension-dataset-1","slug":"reclor-a-reading-comprehension-dataset-1","title":"ReClor: A Reading Comprehension Dataset Requiring Logical Reasoning","date":"2020-02-11","arxiv_id":"2002.04326","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/reclor-a-reading-comprehension-dataset-1#ran","syntology_url":"https://syntology.ai/paper/2002.04326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.04326"}},"official":{"repos":["yuweihao/reclor"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/generating-programmatic-referring-expressions","slug":"generating-programmatic-referring-expressions","title":"Generating Programmatic Referring Expressions via Program Synthesis","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bridging-machine-learning-and-logical","slug":"bridging-machine-learning-and-logical","title":"Bridging Machine Learning and Logical Reasoning by Abductive Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/quantum-embedding-of-knowledge-for-reasoning","slug":"quantum-embedding-of-knowledge-for-reasoning","title":"Quantum Embedding of Knowledge for Reasoning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/action-grammars-a-cognitive-model-for","slug":"action-grammars-a-cognitive-model-for","title":"Semantic RL with Action Grammars: Data-Efficient Learning of Hierarchical Task Abstractions","date":"2019-07-29","arxiv_id":"1907.12477","repositories_listed":1,"syntology":null},{"url":"/paper/declarative-question-answering-over-knowledge","slug":"declarative-question-answering-over-knowledge","title":"Declarative Question Answering over Knowledge Bases containing Natural Language Text with Answer Set Programming","date":"2019-05-01","arxiv_id":"1905.00198","repositories_listed":1,"syntology":null},{"url":"/paper/deeplogic-towards-end-to-end-differentiable","slug":"deeplogic-towards-end-to-end-differentiable","title":"DeepLogic: Towards End-to-End Differentiable Logical Reasoning","date":"2018-05-18","arxiv_id":"1805.07433","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deeplogic-towards-end-to-end-differentiable#ran","syntology_url":"https://syntology.ai/paper/1805.07433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07433"}},"official":{"repos":["nuric/deeplogic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gotaxon-representing-the-evolution-of","slug":"gotaxon-representing-the-evolution-of","title":"GOTaxon: Representing the evolution of biological functions in the Gene Ontology","date":"2018-02-16","arxiv_id":"1802.06004","repositories_listed":1,"syntology":null},{"url":"/paper/can-recursive-neural-tensor-networks-learn","slug":"can-recursive-neural-tensor-networks-learn","title":"Can recursive neural tensor networks learn logical reasoning?","date":"2013-12-21","arxiv_id":"1312.6192","repositories_listed":1,"syntology":null},{"url":null,"slug":"fevo-financial-knowledge-expansion-and","title":"FEVO: Financial Knowledge Expansion and Reasoning Evolution for Large Language Models","date":"2025-07-08","arxiv_id":"2507.06057","repositories_listed":0,"syntology":null},{"url":null,"slug":"mico-multi-image-contrast-for-reinforcement","title":"MiCo: Multi-image Contrast for Reinforcement Visual Reasoning","date":"2025-06-27","arxiv_id":"2506.22434","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-jepa-learning-discrete-token","title":"Discrete JEPA: Learning Discrete Token Representations without Reconstruction","date":"2025-06-17","arxiv_id":"2506.14373","repositories_listed":0,"syntology":null},{"url":null,"slug":"capo-reinforcing-consistent-reasoning-in","title":"CAPO: Reinforcing Consistent Reasoning in Medical Decision-Making","date":"2025-06-15","arxiv_id":"2506.12849","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-r1-chain-of-thought-reasoning-and","title":"Motion-R1: Chain-of-Thought Reasoning and Reinforcement Learning for Human Motion Generation","date":"2025-06-12","arxiv_id":"2506.10353","repositories_listed":0,"syntology":null},{"url":null,"slug":"telemath-a-benchmark-for-large-language","title":"TeleMath: A Benchmark for Large Language Models in Telecom Mathematical Problem Solving","date":"2025-06-12","arxiv_id":"2506.10674","repositories_listed":0,"syntology":null},{"url":null,"slug":"ttt-bench-a-benchmark-for-evaluating","title":"TTT-Bench: A Benchmark for Evaluating Reasoning Ability with Simple and Novel Tic-Tac-Toe-style Games","date":"2025-06-11","arxiv_id":"2506.10209","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-02690","title":"Towards Geometry Problem Solving in the Large Model Era: A Survey","date":"2025-06-03","arxiv_id":"2506.02690","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-chain-of-thought-enables-parallel","title":"Continuous Chain of Thought Enables Parallel Exploration and Reasoning","date":"2025-05-29","arxiv_id":"2505.23648","repositories_listed":0,"syntology":null},{"url":null,"slug":"infi-mmr-curriculum-based-unlocking","title":"Infi-MMR: Curriculum-based Unlocking Multimodal Reasoning via Phased Reinforcement Learning in Multimodal Small Language Models","date":"2025-05-29","arxiv_id":"2505.23091","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualsphinx-large-scale-synthetic-vision","title":"VisualSphinx: Large-Scale Synthetic Vision Logic Puzzles for RL","date":"2025-05-29","arxiv_id":"2505.23977","repositories_listed":0,"syntology":null},{"url":null,"slug":"mme-reasoning-a-comprehensive-benchmark-for","title":"MME-Reasoning: A Comprehensive Benchmark for Logical Reasoning in MLLMs","date":"2025-05-27","arxiv_id":"2505.21327","repositories_listed":0,"syntology":null},{"url":null,"slug":"cp-router-an-uncertainty-aware-router-between","title":"CP-Router: An Uncertainty-Aware Router Between LLM and LRM","date":"2025-05-26","arxiv_id":"2505.19970","repositories_listed":0,"syntology":null},{"url":null,"slug":"enigmata-scaling-logical-reasoning-in-large","title":"Enigmata: Scaling Logical Reasoning in Large Language Models with Synthetic Verifiable Puzzles","date":"2025-05-26","arxiv_id":"2505.19914","repositories_listed":0,"syntology":null},{"url":null,"slug":"interleaved-reasoning-for-large-language","title":"Interleaved Reasoning for Large Language Models via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19640","repositories_listed":0,"syntology":null},{"url":null,"slug":"marco-meta-reflection-with-cross-referencing","title":"MARCO: Meta-Reflection with Cross-Referencing for Code Reasoning","date":"2025-05-23","arxiv_id":"2505.17481","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-in-neurosymbolic-ai","title":"Reasoning in Neurosymbolic AI","date":"2025-05-22","arxiv_id":"2505.20313","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-competent-ai-for-fundamental-analysis","title":"Towards Competent AI for Fundamental Analysis in Finance: A Benchmark Dataset and Evaluation","date":"2025-05-22","arxiv_id":"2506.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-gap-bridging-thought-leap-for","title":"Mind the Gap: Bridging Thought Leap for Improved Chain-of-Thought Tuning","date":"2025-05-20","arxiv_id":"2505.14684","repositories_listed":0,"syntology":null},{"url":null,"slug":"satbench-benchmarking-llms-logical-reasoning","title":"SATBench: Benchmarking LLMs' Logical Reasoning via Automated Puzzle Generation from SAT Formulas","date":"2025-05-20","arxiv_id":"2505.14615","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-abductive-learning","title":"Curriculum Abductive Learning","date":"2025-05-18","arxiv_id":"2505.12275","repositories_listed":0,"syntology":null},{"url":null,"slug":"system-prompt-poisoning-persistent-attacks-on","title":"System Prompt Poisoning: Persistent Attacks on Large Language Models Beyond User Injection","date":"2025-05-10","arxiv_id":"2505.06493","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-symbolic-persistent-macro-actions","title":"Learning Symbolic Persistent Macro-Actions for POMDP Solving Over Time","date":"2025-05-06","arxiv_id":"2505.03668","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypertree-planning-enhancing-llm-reasoning","title":"HyperTree Planning: Enhancing LLM Reasoning via Hierarchical Thinking","date":"2025-05-05","arxiv_id":"2505.02322","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-report-on-the-llms-evaluating-the-high","title":"A Report on the llms evaluating the high school questions","date":"2025-04-30","arxiv_id":"2505.00057","repositories_listed":0,"syntology":null},{"url":null,"slug":"lr-iad-mask-free-industrial-anomaly-detection","title":"LR-IAD:Mask-Free Industrial Anomaly Detection with Logical Reasoning","date":"2025-04-28","arxiv_id":"2504.19524","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyrag-integrating-polyviews-into-retrieval","title":"POLYRAG: Integrating Polyviews into Retrieval-Augmented Generation for Medical Applications","date":"2025-04-21","arxiv_id":"2504.14917","repositories_listed":0,"syntology":null},{"url":null,"slug":"hf4rec-human-like-feedback-driven","title":"HF4Rec: Human-Like Feedback-Driven Optimization Framework for Explainable Recommendation","date":"2025-04-19","arxiv_id":"2504.14147","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-awareness-and-interpretability-of","title":"Context-Awareness and Interpretability of Rare Occurrences for Discovery and Formalization of Critical Failure Modes","date":"2025-04-18","arxiv_id":"2504.16117","repositories_listed":0,"syntology":null},{"url":null,"slug":"logictree-structured-proof-exploration-for","title":"LogicTree: Structured Proof Exploration for Coherent and Rigorous Logical Reasoning with Large Language Models","date":"2025-04-18","arxiv_id":"2504.14089","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-retrieval-for-operational","title":"Multi-Stage Retrieval for Operational Technology Cybersecurity Compliance Using Large Language Models: A Railway Casestudy","date":"2025-04-18","arxiv_id":"2504.14044","repositories_listed":0,"syntology":null},{"url":null,"slug":"lad-reasoner-tiny-multimodal-models-are-good","title":"LAD-Reasoner: Tiny Multimodal Models are Good Reasoners for Logical Anomaly Detection","date":"2025-04-17","arxiv_id":"2504.12749","repositories_listed":0,"syntology":null},{"url":null,"slug":"d1-scaling-reasoning-in-diffusion-large","title":"d1: Scaling Reasoning in Diffusion Large Language Models via Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"medisee-reasoning-based-pixel-level","title":"MediSee: Reasoning-based Pixel-level Perception in Medical Images","date":"2025-04-15","arxiv_id":"2504.11008","repositories_listed":0,"syntology":null},{"url":null,"slug":"puzzlebench-a-fully-dynamic-evaluation","title":"PuzzleBench: A Fully Dynamic Evaluation Framework for Large Multimodal Models on Puzzle Solving","date":"2025-04-15","arxiv_id":"2504.10885","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualpuzzles-decoupling-multimodal-reasoning","title":"VisualPuzzles: Decoupling Multimodal Reasoning Evaluation from Domain Knowledge","date":"2025-04-14","arxiv_id":"2504.10342","repositories_listed":0,"syntology":null},{"url":null,"slug":"movsam-a-single-image-moving-object","title":"MovSAM: A Single-image Moving Object Segmentation Framework Based on Deep Thinking","date":"2025-04-09","arxiv_id":"2504.06863","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-failure-of-language-models-in","title":"Provable Failure of Language Models in Learning Majority Boolean Logic via Gradient Descent","date":"2025-04-07","arxiv_id":"2504.04702","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-models-know-when-they-re-right","title":"Reasoning Models Know When They're Right: Probing Hidden States for Self-Verification","date":"2025-04-07","arxiv_id":"2504.05419","repositories_listed":0,"syntology":null},{"url":null,"slug":"have-large-language-models-learned-to-reason","title":"Have Large Language Models Learned to Reason? A Characterization via 3-SAT Phase Transition","date":"2025-04-04","arxiv_id":"2504.03930","repositories_listed":0,"syntology":null},{"url":null,"slug":"vgrp-bench-visual-grid-reasoning-puzzle","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","date":"2025-03-29","arxiv_id":"2503.23064","repositories_listed":0,"syntology":null},{"url":null,"slug":"negation-a-pink-elephant-in-the-large","title":"Negation: A Pink Elephant in the Large Language Models' Room?","date":"2025-03-28","arxiv_id":"2503.22395","repositories_listed":0,"syntology":null},{"url":null,"slug":"shieldagent-shielding-agents-via-verifiable","title":"ShieldAgent: Shielding Agents via Verifiable Safety Policy Reasoning","date":"2025-03-26","arxiv_id":"2503.22738","repositories_listed":0,"syntology":null},{"url":null,"slug":"rosetta-pl-propositional-logic-as-a-benchmark","title":"Rosetta-PL: Propositional Logic as a Benchmark for Large Language Model Reasoning","date":"2025-03-25","arxiv_id":"2505.00001","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-neuro-symbolic-artificial","title":"A Study on Neuro-Symbolic Artificial Intelligence: Healthcare Perspectives","date":"2025-03-23","arxiv_id":"2503.18213","repositories_listed":0,"syntology":null},{"url":null,"slug":"g-i-dle-generative-inference-via-distribution","title":"(G)I-DLE: Generative Inference via Distribution-preserving Logit Exclusion with KL Divergence Minimization for Constrained Decoding","date":"2025-03-23","arxiv_id":"2503.18050","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-retrieval-systems-with-inference","title":"Enhancing Retrieval Systems with Inference-Time Logical Reasoning","date":"2025-03-22","arxiv_id":"2503.17860","repositories_listed":0,"syntology":null},{"url":null,"slug":"lamour-leveraging-language-models-for-out-of","title":"LaMOuR: Leveraging Language Models for Out-of-Distribution Recovery in Reinforcement Learning","date":"2025-03-21","arxiv_id":"2503.17125","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-technology-and-humanities-evaluating","title":"Bridging Technology and Humanities: Evaluating the Impact of Large Language Models on Social Sciences Research with DeepSeek-R1","date":"2025-03-20","arxiv_id":"2503.16304","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-chaos-to-order-the-atomic-reasoner","title":"From Chaos to Order: The Atomic Reasoner Framework for Fine-grained Reasoning in Large Language Models","date":"2025-03-20","arxiv_id":"2503.15944","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-but-vulnerable-benchmarking-and","title":"Efficient but Vulnerable: Benchmarking and Defending LLM Batch Prompting Attack","date":"2025-03-18","arxiv_id":"2503.15551","repositories_listed":0,"syntology":null},{"url":null,"slug":"3daxisprompt-promoting-the-3d-grounding-and","title":"3DAxisPrompt: Promoting the 3D Grounding and Reasoning in GPT-4o","date":"2025-03-17","arxiv_id":"2503.13185","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-is-all-you-need-for-video","title":"Reasoning is All You Need for Video Generalization: A Counterfactual Benchmark with Sub-question Evaluation","date":"2025-03-12","arxiv_id":"2503.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-reasoning-era-a-survey-of-long-chain","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","date":"2025-03-12","arxiv_id":"2503.09567","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-superior-quantization-accuracy-a","title":"Towards Superior Quantization Accuracy: A Layer-sensitive Approach","date":"2025-03-09","arxiv_id":"2503.06518","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-society-of-hivemind-multi-agent","title":"The Society of HiveMind: Multi-Agent Optimization of Foundation Model Swarms to Unlock the Potential of Collective Intelligence","date":"2025-03-07","arxiv_id":"2503.05473","repositories_listed":0,"syntology":null},{"url":null,"slug":"db-explore-automated-database-exploration-and","title":"DB-Explore: Automated Database Exploration and Instruction Synthesis for Text-to-SQL","date":"2025-03-06","arxiv_id":"2503.04959","repositories_listed":0,"syntology":null},{"url":null,"slug":"helpsteer3-human-annotated-feedback-and-edit","title":"HelpSteer3: Human-Annotated Feedback and Edit Data to Empower Inference-Time Scaling in Open-Ended General-Domain Tasks","date":"2025-03-06","arxiv_id":"2503.04378","repositories_listed":0,"syntology":null},{"url":null,"slug":"psy-insight-explainable-multi-turn-bilingual","title":"Psy-Insight: Explainable Multi-turn Bilingual Dataset for Mental Health Counseling","date":"2025-03-05","arxiv_id":"2503.03607","repositories_listed":0,"syntology":null},{"url":null,"slug":"hot-highlighted-chain-of-thought-for","title":"HoT: Highlighted Chain of Thought for Referencing Supporting Facts from Inputs","date":"2025-03-03","arxiv_id":"2503.02003","repositories_listed":0,"syntology":null},{"url":null,"slug":"order-doesn-t-matter-but-reasoning-does","title":"Order Doesn't Matter, But Reasoning Does: Training LLMs with Order-Centric Augmentation","date":"2025-02-27","arxiv_id":"2502.19907","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversal-blessing-thinking-backward-may","title":"Reversal Blessing: Thinking Backward May Outpace Thinking Forward in Multi-choice Questions","date":"2025-02-25","arxiv_id":"2502.18435","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-image-generation-guided-by","title":"Autoregressive Image Generation Guided by Chains of Thought","date":"2025-02-24","arxiv_id":"2502.16965","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-haystacks-probing-llms-long-context","title":"Logic Haystacks: Probing LLMs Long-Context Logical Reasoning (Without Easily Identifiable Unrelated Padding)","date":"2025-02-24","arxiv_id":"2502.17169","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-llms-reason-the-intermediate-language","title":"Intermediate Languages Matter: Formal Choice Drives Neurosymbolic LLM Reasoning","date":"2025-02-24","arxiv_id":"2502.17216","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-logical-consistency-in","title":"Quantifying Logical Consistency in Transformers via Query-Key Alignment","date":"2025-02-24","arxiv_id":"2502.17017","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-llms-with-logical-reasoning-a","title":"Empowering LLMs with Logical Reasoning: A Comprehensive Survey","date":"2025-02-21","arxiv_id":"2502.15652","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-features-that-shape-perceived","title":"Identifying Features that Shape Perceived Consciousness in Large Language Model-based AI: A Quantitative Study of Human Responses","date":"2025-02-21","arxiv_id":"2502.15365","repositories_listed":0,"syntology":null},{"url":null,"slug":"triangulating-llm-progress-through-benchmarks","title":"Triangulating LLM Progress through Benchmarks, Games, and Cognitive Tests","date":"2025-02-20","arxiv_id":"2502.14359","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mousetrap-fooling-large-reasoning-models","title":"A Mousetrap: Fooling Large Reasoning Models for Jailbreak with Chain of Iterative Chaos","date":"2025-02-19","arxiv_id":"2502.15806","repositories_listed":0,"syntology":null}],"record_sha256":"31a7fc7be2bb59645bfd7cd177f4fcf9de01ca91f77641adf0e27a652dc1229e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}