{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/logical-reasoning/papers/ran/1","list_of":"/task/logical-reasoning","task":"Logical Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":113,"counts":{"archive_papers_tagged":747,"with_a_code_link":330,"where_syntology_ran_a_sample":113,"not_listed_spam_title":0,"listed":747,"listed_where_code_ran":113,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/logical-reasoning/papers/ran/1","prev":null,"next":"/task/logical-reasoning/papers/ran/2","papers":[{"url":"/paper/soundmind-rl-incentivized-logic-reasoning-for","slug":"soundmind-rl-incentivized-logic-reasoning-for","title":"SoundMind: RL-Incentivized Logic Reasoning for Audio-Language Models","date":"2025-06-15","arxiv_id":"2506.12935","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/soundmind-rl-incentivized-logic-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2506.12935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12935"}},"official":{"repos":["xid32/soundmind"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/infigui-r1-advancing-multimodal-gui-agents","slug":"infigui-r1-advancing-multimodal-gui-agents","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","date":"2025-04-19","arxiv_id":"2504.14239","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/infigui-r1-advancing-multimodal-gui-agents#ran","syntology_url":"https://syntology.ai/paper/2504.14239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14239"}},"official":{"repos":["reallm-labs/infigui-r1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/envisioning-beyond-the-pixels-benchmarking","slug":"envisioning-beyond-the-pixels-benchmarking","title":"Envisioning Beyond the Pixels: Benchmarking Reasoning-Informed Visual Editing","date":"2025-04-03","arxiv_id":"2504.02826","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envisioning-beyond-the-pixels-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2504.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02826"}},"official":{"repos":["phoenixz810/risebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-effect-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.24376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24376"}},"official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/questbench-can-llms-ask-the-right-question-to","slug":"questbench-can-llms-ask-the-right-question-to","title":"QuestBench: Can LLMs ask the right question to acquire information in reasoning tasks?","date":"2025-03-28","arxiv_id":"2503.22674","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/questbench-can-llms-ask-the-right-question-to#ran","syntology_url":"https://syntology.ai/paper/2503.22674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22674"}},"official":{"repos":["google-deepmind/questbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lmm-r1-empowering-3b-lmms-with-strong","slug":"lmm-r1-empowering-3b-lmms-with-strong","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","date":"2025-03-10","arxiv_id":"2503.07536","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lmm-r1-empowering-3b-lmms-with-strong#ran","syntology_url":"https://syntology.ai/paper/2503.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07536"}},"official":null}},{"url":"/paper/textgames-learning-to-self-play-text-based","slug":"textgames-learning-to-self-play-text-based","title":"TextGames: Learning to Self-Play Text-Based Puzzle Games via Language Model Reasoning","date":"2025-02-25","arxiv_id":"2502.18431","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/textgames-learning-to-self-play-text-based#ran","syntology_url":"https://syntology.ai/paper/2502.18431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18431"}},"official":{"repos":["fhudi/textgames"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/large-language-models-meet-symbolic-provers","slug":"large-language-models-meet-symbolic-provers","title":"Large Language Models Meet Symbolic Provers for Logical Reasoning Evaluation","date":"2025-02-10","arxiv_id":"2502.06563","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-meet-symbolic-provers#ran","syntology_url":"https://syntology.ai/paper/2502.06563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06563"}},"official":{"repos":["opendatalab/provergen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sedareval-automated-evaluation-using-self","slug":"sedareval-automated-evaluation-using-self","title":"SedarEval: Automated Evaluation using Self-Adaptive Rubrics","date":"2025-01-26","arxiv_id":"2501.15595","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sedareval-automated-evaluation-using-self#ran","syntology_url":"https://syntology.ai/paper/2501.15595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15595"}},"official":{"repos":["wwn1233/sedareval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pike-rag-specialized-knowledge-and-rationale","slug":"pike-rag-specialized-knowledge-and-rationale","title":"PIKE-RAG: sPecIalized KnowledgE and Rationale Augmented Generation","date":"2025-01-20","arxiv_id":"2501.11551","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pike-rag-specialized-knowledge-and-rationale#ran","syntology_url":"https://syntology.ai/paper/2501.11551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11551"}},"official":{"repos":["microsoft/pike-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leapvad-a-leap-in-autonomous-driving-via","slug":"leapvad-a-leap-in-autonomous-driving-via","title":"LeapVAD: A Leap in Autonomous Driving via Cognitive Perception and Dual-Process Thinking","date":"2025-01-14","arxiv_id":"2501.08168","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leapvad-a-leap-in-autonomous-driving-via#ran","syntology_url":"https://syntology.ai/paper/2501.08168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08168"}},"official":null}},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-large-language-models-for-1","slug":"harnessing-large-language-models-for-1","title":"Harnessing Large Language Models for Knowledge Graph Question Answering via Adaptive Multi-Aspect Retrieval-Augmentation","date":"2024-12-24","arxiv_id":"2412.18537","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2412.18537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18537"}},"official":{"repos":["Applied-Machine-Learning-Lab/AMAR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphere-a-hierarchical-evaluation-on-spatial","slug":"sphere-a-hierarchical-evaluation-on-spatial","title":"SPHERE: A Hierarchical Evaluation on Spatial Perception and Reasoning for Vision-Language Models","date":"2024-12-17","arxiv_id":"2412.12693","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphere-a-hierarchical-evaluation-on-spatial#ran","syntology_url":"https://syntology.ai/paper/2412.12693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12693"}},"official":{"repos":["zwenyu/SPHERE-VLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wisead-knowledge-augmented-end-to-end","slug":"wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","arxiv_id":"2412.09951","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wisead-knowledge-augmented-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2412.09951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09951"}},"official":{"repos":["wyddmw/WiseAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rulearena-a-benchmark-for-rule-guided","slug":"rulearena-a-benchmark-for-rule-guided","title":"RuleArena: A Benchmark for Rule-Guided Reasoning with LLMs in Real-World Scenarios","date":"2024-12-12","arxiv_id":"2412.08972","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rulearena-a-benchmark-for-rule-guided#ran","syntology_url":"https://syntology.ai/paper/2412.08972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08972"}},"official":{"repos":["skyriver-2000/rulearena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/flashrnn-optimizing-traditional-rnns-on","slug":"flashrnn-optimizing-traditional-rnns-on","title":"FlashRNN: Optimizing Traditional RNNs on Modern Hardware","date":"2024-12-10","arxiv_id":"2412.07752","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flashrnn-optimizing-traditional-rnns-on#ran","syntology_url":"https://syntology.ai/paper/2412.07752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07752"}},"official":{"repos":["nx-ai/flashrnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-large-language-models-to-reason-in-a","slug":"training-large-language-models-to-reason-in-a","title":"Training Large Language Models to Reason in a Continuous Latent Space","date":"2024-12-09","arxiv_id":"2412.06769","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-large-language-models-to-reason-in-a#ran","syntology_url":"https://syntology.ai/paper/2412.06769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06769"}},"official":{"repos":["facebookresearch/coconut"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clusterkv-manipulating-llm-kv-cache-in","slug":"clusterkv-manipulating-llm-kv-cache-in","title":"ClusterKV: Manipulating LLM KV Cache in Semantic Space for Recallable Compression","date":"2024-12-04","arxiv_id":"2412.03213","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clusterkv-manipulating-llm-kv-cache-in#ran","syntology_url":"https://syntology.ai/paper/2412.03213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03213"}},"official":{"repos":["sjtu-zhao-lab/clusterkv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-o1-let-vision-language-models-reason","slug":"llava-o1-let-vision-language-models-reason","title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","date":"2024-11-15","arxiv_id":"2411.10440","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-o1-let-vision-language-models-reason#ran","syntology_url":"https://syntology.ai/paper/2411.10440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.10440"}},"official":{"repos":["PKU-YuanGroup/LLaVA-CoT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-llm-language-network-a-neuroscientific","slug":"the-llm-language-network-a-neuroscientific","title":"The LLM Language Network: A Neuroscientific Approach for Identifying Causally Task-Relevant Units","date":"2024-11-04","arxiv_id":"2411.02280","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-llm-language-network-a-neuroscientific#ran","syntology_url":"https://syntology.ai/paper/2411.02280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02280"}},"official":{"repos":["bkhmsi/llm-localization"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/logicity-advancing-neuro-symbolic-ai-with","slug":"logicity-advancing-neuro-symbolic-ai-with","title":"LogiCity: Advancing Neuro-Symbolic AI with Abstract Urban Simulation","date":"2024-11-01","arxiv_id":"2411.00773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/logicity-advancing-neuro-symbolic-ai-with#ran","syntology_url":"https://syntology.ai/paper/2411.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00773"}},"official":{"repos":["Jaraxxus-Me/LogiCity"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-curriculum-expert-iteration-for","slug":"automatic-curriculum-expert-iteration-for","title":"Automatic Curriculum Expert Iteration for Reliable LLM Reasoning","date":"2024-10-10","arxiv_id":"2410.07627","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-curriculum-expert-iteration-for#ran","syntology_url":"https://syntology.ai/paper/2410.07627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07627"}},"official":{"repos":["salesforceairesearch/auto-cei"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/divide-and-translate-compositional-first","slug":"divide-and-translate-compositional-first","title":"Divide and Translate: Compositional First-Order Logic Translation and Verification for Complex Logical Reasoning","date":"2024-10-10","arxiv_id":"2410.08047","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divide-and-translate-compositional-first#ran","syntology_url":"https://syntology.ai/paper/2410.08047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08047"}},"official":{"repos":["Hyun-Ryu/clover"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hlm-cite-hybrid-language-model-workflow-for","slug":"hlm-cite-hybrid-language-model-workflow-for","title":"HLM-Cite: Hybrid Language Model Workflow for Text-based Scientific Citation Prediction","date":"2024-10-10","arxiv_id":"2410.09112","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hlm-cite-hybrid-language-model-workflow-for#ran","syntology_url":"https://syntology.ai/paper/2410.09112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09112"}},"official":{"repos":["tsinghua-fib-lab/H-LM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/turtlebench-evaluating-top-language-models","slug":"turtlebench-evaluating-top-language-models","title":"TurtleBench: Evaluating Top Language Models via Real-World Yes/No Puzzles","date":"2024-10-07","arxiv_id":"2410.05262","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/turtlebench-evaluating-top-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05262"}},"official":{"repos":["mazzzystar/TurtleBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpret-your-decision-logical-reasoning","slug":"interpret-your-decision-logical-reasoning","title":"Interpret Your Decision: Logical Reasoning Regularization for Generalization in Visual Classification","date":"2024-10-06","arxiv_id":"2410.04492","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interpret-your-decision-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2410.04492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04492"}},"official":{"repos":["zhaorui-tan/L-Reg_NeurIPS24"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logicvista-multimodal-llm-logical-reasoning","slug":"logicvista-multimodal-llm-logical-reasoning","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","date":"2024-07-06","arxiv_id":"2407.04973","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicvista-multimodal-llm-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.04973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04973"}},"official":{"repos":["yijia-xiao/logicvista"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/puzzles-a-benchmark-for-neural-algorithmic","slug":"puzzles-a-benchmark-for-neural-algorithmic","title":"PUZZLES: A Benchmark for Neural Algorithmic Reasoning","date":"2024-06-29","arxiv_id":"2407.00401","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puzzles-a-benchmark-for-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2407.00401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00401"}},"official":{"repos":["eth-disco/rlp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-synthetic-data-creation-with","slug":"scaling-synthetic-data-creation-with","title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","date":"2024-06-28","arxiv_id":"2406.20094","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-synthetic-data-creation-with#ran","syntology_url":"https://syntology.ai/paper/2406.20094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20094"}},"official":{"repos":["tencent-ailab/persona-hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/liar-liar-logical-mire-a-benchmark-for","slug":"liar-liar-logical-mire-a-benchmark-for","title":"Liar, Liar, Logical Mire: A Benchmark for Suppositional Reasoning in Large Language Models","date":"2024-06-18","arxiv_id":"2406.12546","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/liar-liar-logical-mire-a-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.12546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12546"}},"official":{"repos":["mainlp/TruthQuest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-peek-into-token-bias-large-language-models","slug":"a-peek-into-token-bias-large-language-models","title":"A Peek into Token Bias: Large Language Models Are Not Yet Genuine Reasoners","date":"2024-06-16","arxiv_id":"2406.11050","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-peek-into-token-bias-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.11050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11050"}},"official":{"repos":["bowen-upenn/llm_token_bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-preference-optimization-improving","slug":"chain-of-preference-optimization-improving","title":"Chain of Preference Optimization: Improving Chain-of-Thought Reasoning in LLMs","date":"2024-06-13","arxiv_id":"2406.09136","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-preference-optimization-improving#ran","syntology_url":"https://syntology.ai/paper/2406.09136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09136"}},"official":{"repos":["sail-sg/cpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lingoly-a-benchmark-of-olympiad-level","slug":"lingoly-a-benchmark-of-olympiad-level","title":"LINGOLY: A Benchmark of Olympiad-Level Linguistic Reasoning Puzzles in Low-Resource and Extinct Languages","date":"2024-06-10","arxiv_id":"2406.06196","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lingoly-a-benchmark-of-olympiad-level#ran","syntology_url":"https://syntology.ai/paper/2406.06196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06196"}},"official":{"repos":["am-bean/lingOly"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-of-reasoning-efficient-training-of-llm","slug":"flow-of-reasoning-efficient-training-of-llm","title":"Flow of Reasoning:Training LLMs for Divergent Problem Solving with Minimal Examples","date":"2024-06-09","arxiv_id":"2406.05673","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-of-reasoning-efficient-training-of-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05673"}},"official":{"repos":["yu-fangxu/for"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-world-model-implicit-in-a","slug":"evaluating-the-world-model-implicit-in-a","title":"Evaluating the World Model Implicit in a Generative Model","date":"2024-06-06","arxiv_id":"2406.03689","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-the-world-model-implicit-in-a#ran","syntology_url":"https://syntology.ai/paper/2406.03689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03689"}},"official":{"repos":["keyonvafa/world-model-evaluation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-hardness-of-probabilistic","slug":"on-the-hardness-of-probabilistic","title":"On the Hardness of Probabilistic Neurosymbolic Learning","date":"2024-06-06","arxiv_id":"2406.04472","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-hardness-of-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2406.04472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04472"}},"official":{"repos":["jjcmoon/hardness-nesy"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/easy-problems-that-llms-get-wrong","slug":"easy-problems-that-llms-get-wrong","title":"Easy Problems That LLMs Get Wrong","date":"2024-05-30","arxiv_id":"2405.19616","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/easy-problems-that-llms-get-wrong#ran","syntology_url":"https://syntology.ai/paper/2405.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19616"}},"official":{"repos":["autogenai/easy-problems-that-llms-get-wrong"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/faithful-logical-reasoning-via-symbolic-chain","slug":"faithful-logical-reasoning-via-symbolic-chain","title":"Faithful Logical Reasoning via Symbolic Chain-of-Thought","date":"2024-05-28","arxiv_id":"2405.18357","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/faithful-logical-reasoning-via-symbolic-chain#ran","syntology_url":"https://syntology.ai/paper/2405.18357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18357"}},"official":{"repos":["aiden0526/symbcot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-theoretical-understanding-of-the-2","slug":"towards-a-theoretical-understanding-of-the-2","title":"Towards a Theoretical Understanding of the 'Reversal Curse' via Training Dynamics","date":"2024-05-07","arxiv_id":"2405.04669","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-a-theoretical-understanding-of-the-2#ran","syntology_url":"https://syntology.ai/paper/2405.04669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04669"}},"official":{"repos":["marlo-z/reversal_curse_analysis"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-llm-reasoning-generalists-with#ran","syntology_url":"https://syntology.ai/paper/2404.02078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02078"}},"official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leanreasoner-boosting-complex-logical","slug":"leanreasoner-boosting-complex-logical","title":"LeanReasoner: Boosting Complex Logical Reasoning with Lean","date":"2024-03-20","arxiv_id":"2403.13312","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leanreasoner-boosting-complex-logical#ran","syntology_url":"https://syntology.ai/paper/2403.13312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13312"}},"official":{"repos":["some-random/theorem-proving-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-reason-with-rules-logic-scaffolding","slug":"can-llms-reason-with-rules-logic-scaffolding","title":"Can LLMs Reason with Rules? Logic Scaffolding for Stress-Testing and Improving LLMs","date":"2024-02-18","arxiv_id":"2402.11442","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llms-reason-with-rules-logic-scaffolding#ran","syntology_url":"https://syntology.ai/paper/2402.11442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11442"}},"official":{"repos":["siyuanwangw/ulogic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-fusion-of-large-language-models","slug":"knowledge-fusion-of-large-language-models","title":"Knowledge Fusion of Large Language Models","date":"2024-01-19","arxiv_id":"2401.10491","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-fusion-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.10491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10491"}},"official":{"repos":["fanqiwan/fusellm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/langbridge-multilingual-reasoning-without","slug":"langbridge-multilingual-reasoning-without","title":"LangBridge: Multilingual Reasoning Without Multilingual Supervision","date":"2024-01-19","arxiv_id":"2401.10695","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/langbridge-multilingual-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2401.10695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10695"}},"official":{"repos":["kaistAI/LangBridge"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-b-b-a-triggering-logical-reasoning-failures#ran","syntology_url":"https://syntology.ai/paper/2401.00757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00757"}},"official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/teilp-time-prediction-over-knowledge-graphs","slug":"teilp-time-prediction-over-knowledge-graphs","title":"TEILP: Time Prediction over Knowledge Graphs via Logical Reasoning","date":"2023-12-25","arxiv_id":"2312.15816","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/teilp-time-prediction-over-knowledge-graphs#ran","syntology_url":"https://syntology.ai/paper/2312.15816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15816"}},"official":null}},{"url":"/paper/efficiently-programming-large-language-models","slug":"efficiently-programming-large-language-models","title":"SGLang: Efficient Execution of Structured Language Model Programs","date":"2023-12-12","arxiv_id":"2312.07104","repositories_listed":2,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":9,"n_instrument":7,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficiently-programming-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2312.07104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07104"}},"official":{"repos":["sgl-project/sglang"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/mmmu-a-massive-multi-discipline-multimodal","slug":"mmmu-a-massive-multi-discipline-multimodal","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","date":"2023-11-27","arxiv_id":"2311.16502","repositories_listed":5,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmmu-a-massive-multi-discipline-multimodal#ran","syntology_url":"https://syntology.ai/paper/2311.16502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16502"}},"official":{"repos":["MMMU-Benchmark/MMMU"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/assessing-logical-puzzle-solving-in-large","slug":"assessing-logical-puzzle-solving-in-large","title":"Assessing Logical Puzzle Solving in Large Language Models: Insights from a Minesweeper Case Study","date":"2023-11-13","arxiv_id":"2311.07387","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/assessing-logical-puzzle-solving-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.07387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07387"}},"official":{"repos":["yinghao-li/minesweeper-for-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/plan-verify-and-switch-integrated-reasoning","slug":"plan-verify-and-switch-integrated-reasoning","title":"Plan, Verify and Switch: Integrated Reasoning with Diverse X-of-Thoughts","date":"2023-10-23","arxiv_id":"2310.14628","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plan-verify-and-switch-integrated-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.14628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14628"}},"official":{"repos":["tengxiaoliu/xot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/linc-a-neurosymbolic-approach-for-logical","slug":"linc-a-neurosymbolic-approach-for-logical","title":"LINC: A Neurosymbolic Approach for Logical Reasoning by Combining Language Models with First-Order Logic Provers","date":"2023-10-23","arxiv_id":"2310.15164","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/linc-a-neurosymbolic-approach-for-logical#ran","syntology_url":"https://syntology.ai/paper/2310.15164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15164"}},"official":{"repos":["benlipkin/linc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/bongard-openworld-few-shot-reasoning-for-free","slug":"bongard-openworld-few-shot-reasoning-for-free","title":"Bongard-OpenWorld: Few-Shot Reasoning for Free-form Visual Concepts in the Real World","date":"2023-10-16","arxiv_id":"2310.10207","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bongard-openworld-few-shot-reasoning-for-free#ran","syntology_url":"https://syntology.ai/paper/2310.10207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10207"}},"official":{"repos":["joyjayng/Bongard-OpenWorld"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-systematic-evaluation-of-large-language-1","slug":"a-systematic-evaluation-of-large-language-1","title":"Assessing and Enhancing the Robustness of Large Language Models with Task Structure Variations for Logical Reasoning","date":"2023-10-13","arxiv_id":"2310.09430","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-systematic-evaluation-of-large-language-1#ran","syntology_url":"https://syntology.ai/paper/2310.09430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09430"}},"official":{"repos":["strong-ai-lab/logical-and-abstract-reasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instance-needs-more-care-rewriting-prompts#ran","syntology_url":"https://syntology.ai/paper/2310.02107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02107"}},"official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2309-05936","slug":"2309-05936","title":"Do PLMs Know and Understand Ontological Knowledge?","date":"2023-09-12","arxiv_id":"2309.05936","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/2309-05936#ran","syntology_url":"https://syntology.ai/paper/2309.05936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05936"}},"official":{"repos":["vickywu1022/ontoprobe-plms"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lr-xfl-logical-reasoning-based-explainable","slug":"lr-xfl-logical-reasoning-based-explainable","title":"LR-XFL: Logical Reasoning-based Explainable Federated Learning","date":"2023-08-24","arxiv_id":"2308.12681","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lr-xfl-logical-reasoning-based-explainable#ran","syntology_url":"https://syntology.ai/paper/2308.12681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12681"}},"official":{"repos":["yanci87/lr-xfl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/deciphering-raw-data-in-neuro-symbolic","slug":"deciphering-raw-data-in-neuro-symbolic","title":"Deciphering Raw Data in Neuro-Symbolic Learning with Provable Guarantees","date":"2023-08-21","arxiv_id":"2308.10487","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/deciphering-raw-data-in-neuro-symbolic#ran","syntology_url":"https://syntology.ai/paper/2308.10487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10487"}},"official":{"repos":["abductivelearning/abl-tl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/lateval-an-interactive-llms-evaluation","slug":"lateval-an-interactive-llms-evaluation","title":"LatEval: An Interactive LLMs Evaluation Benchmark with Incomplete Information from Lateral Thinking Puzzles","date":"2023-08-21","arxiv_id":"2308.10855","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/lateval-an-interactive-llms-evaluation#ran","syntology_url":"https://syntology.ai/paper/2308.10855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10855"}},"official":{"repos":["thukelab/lateval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-deductive-reasoning-from-synthetic","slug":"learning-deductive-reasoning-from-synthetic","title":"Learning Deductive Reasoning from Synthetic Corpus based on Formal Logic","date":"2023-08-11","arxiv_id":"2308.07336","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-deductive-reasoning-from-synthetic#ran","syntology_url":"https://syntology.ai/paper/2308.07336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07336"}},"official":{"repos":["hitachi-nlp/fld"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/cumulative-reasoning-with-large-language","slug":"cumulative-reasoning-with-large-language","title":"Cumulative Reasoning with Large Language Models","date":"2023-08-08","arxiv_id":"2308.04371","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumulative-reasoning-with-large-language#ran","syntology_url":"https://syntology.ai/paper/2308.04371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04371"}},"official":{"repos":["iiis-ai/cumulative-reasoning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/collie-systematic-construction-of-constrained","slug":"collie-systematic-construction-of-constrained","title":"COLLIE: Systematic Construction of Constrained Text Generation Tasks","date":"2023-07-17","arxiv_id":"2307.08689","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collie-systematic-construction-of-constrained#ran","syntology_url":"https://syntology.ai/paper/2307.08689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08689"}},"official":{"repos":["princeton-nlp/Collie"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/text-efo-k-cqa-towards-knowledge-graph","slug":"text-efo-k-cqa-towards-knowledge-graph","title":"$\\text{EFO}_{k}$-CQA: Towards Knowledge Graph Complex Query Answering beyond Set Operation","date":"2023-07-15","arxiv_id":"2307.13701","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/text-efo-k-cqa-towards-knowledge-graph#ran","syntology_url":"https://syntology.ai/paper/2307.13701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.13701"}},"official":{"repos":["hkust-knowcomp/efok-cqa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/v-lol-a-diagnostic-dataset-for-visual-logical","slug":"v-lol-a-diagnostic-dataset-for-visual-logical","title":"V-LoL: A Diagnostic Dataset for Visual Logical Learning","date":"2023-06-13","arxiv_id":"2306.07743","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/v-lol-a-diagnostic-dataset-for-visual-logical#ran","syntology_url":"https://syntology.ai/paper/2306.07743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07743"}},"official":{"repos":["ml-research/vlol-dataset-gen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deductive-verification-of-chain-of-thought-1","slug":"deductive-verification-of-chain-of-thought-1","title":"Deductive Verification of Chain-of-Thought Reasoning","date":"2023-06-06","arxiv_id":"2306.03872","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deductive-verification-of-chain-of-thought-1#ran","syntology_url":"https://syntology.ai/paper/2306.03872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03872"}},"official":{"repos":["lz1oceani/verify_cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/certified-reasoning-with-language-models","slug":"certified-reasoning-with-language-models","title":"Certified Deductive Reasoning with Language Models","date":"2023-06-06","arxiv_id":"2306.04031","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/certified-reasoning-with-language-models#ran","syntology_url":"https://syntology.ai/paper/2306.04031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04031"}},"official":null}},{"url":"/paper/logicllm-exploring-self-supervised-logic","slug":"logicllm-exploring-self-supervised-logic","title":"Exploring Self-supervised Logic-enhanced Training for Large Language Models","date":"2023-05-23","arxiv_id":"2305.13718","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicllm-exploring-self-supervised-logic#ran","syntology_url":"https://syntology.ai/paper/2305.13718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13718"}},"official":{"repos":["sparkjiao/logicllm","sparkjiao/merit-v2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/logic-lm-empowering-large-language-models","slug":"logic-lm-empowering-large-language-models","title":"Logic-LM: Empowering Large Language Models with Symbolic Solvers for Faithful Logical Reasoning","date":"2023-05-20","arxiv_id":"2305.12295","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/logic-lm-empowering-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2305.12295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12295"}},"official":{"repos":["teacherpeterpan/logic-llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explicit-planning-helps-language-models-in","slug":"explicit-planning-helps-language-models-in","title":"Explicit Planning Helps Language Models in Logical Reasoning","date":"2023-03-28","arxiv_id":"2303.15714","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/explicit-planning-helps-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2303.15714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15714"}},"official":{"repos":["cindermond/explicit-planning-for-reasoning","cindermond/leap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-knowledge-transfer-with","slug":"weakly-supervised-knowledge-transfer-with","title":"Weakly Supervised Knowledge Transfer with Probabilistic Logical Reasoning for Object Detection","date":"2023-03-09","arxiv_id":"2303.05148","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/weakly-supervised-knowledge-transfer-with#ran","syntology_url":"https://syntology.ai/paper/2303.05148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05148"}},"official":{"repos":["molden/probkt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chatcad-interactive-computer-aided-diagnosis","slug":"chatcad-interactive-computer-aided-diagnosis","title":"ChatCAD: Interactive Computer-Aided Diagnosis on Medical Image using Large Language Models","date":"2023-02-14","arxiv_id":"2302.07257","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chatcad-interactive-computer-aided-diagnosis#ran","syntology_url":"https://syntology.ai/paper/2302.07257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07257"}},"official":{"repos":["zhaozh10/ChatCAD"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-second-thought-let-s-not-think-step-by","slug":"on-second-thought-let-s-not-think-step-by","title":"On Second Thought, Let's Not Think Step by Step! Bias and Toxicity in Zero-Shot Reasoning","date":"2022-12-15","arxiv_id":"2212.08061","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-second-thought-let-s-not-think-step-by#ran","syntology_url":"https://syntology.ai/paper/2212.08061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08061"}},"official":{"repos":["salt-nlp/chain-of-thought-bias"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unigeo-unifying-geometry-logical-reasoning","slug":"unigeo-unifying-geometry-logical-reasoning","title":"UniGeo: Unifying Geometry Logical Reasoning via Reformulating Mathematical Expression","date":"2022-12-06","arxiv_id":"2212.02746","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/unigeo-unifying-geometry-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2212.02746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02746"}},"official":{"repos":["chen-judge/unigeo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/logical-tasks-for-measuring-extrapolation-and","slug":"logical-tasks-for-measuring-extrapolation-and","title":"Logical Tasks for Measuring Extrapolation and Rule Comprehension","date":"2022-11-14","arxiv_id":"2211.07727","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/logical-tasks-for-measuring-extrapolation-and#ran","syntology_url":"https://syntology.ai/paper/2211.07727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07727"}},"official":{"repos":["ifujisawa/addition-experiment"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gammae-gamma-embeddings-for-logical-queries","slug":"gammae-gamma-embeddings-for-logical-queries","title":"GammaE: Gamma Embeddings for Logical Queries on Knowledge Graphs","date":"2022-10-27","arxiv_id":"2210.15578","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gammae-gamma-embeddings-for-logical-queries#ran","syntology_url":"https://syntology.ai/paper/2210.15578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15578"}},"official":{"repos":["dyang67/GammaE"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/inductive-logical-query-answering-in","slug":"inductive-logical-query-answering-in","title":"Inductive Logical Query Answering in Knowledge Graphs","date":"2022-10-13","arxiv_id":"2210.08008","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inductive-logical-query-answering-in#ran","syntology_url":"https://syntology.ai/paper/2210.08008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08008"}},"official":null}},{"url":"/paper/dynamic-prompt-learning-via-policy-gradient","slug":"dynamic-prompt-learning-via-policy-gradient","title":"Dynamic Prompt Learning via Policy Gradient for Semi-structured Mathematical Reasoning","date":"2022-09-29","arxiv_id":"2209.14610","repositories_listed":2,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/dynamic-prompt-learning-via-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/2209.14610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14610"}},"official":null}},{"url":"/paper/neural-methods-for-logical-reasoning-over-1","slug":"neural-methods-for-logical-reasoning-over-1","title":"Neural Methods for Logical Reasoning Over Knowledge Graphs","date":"2022-09-28","arxiv_id":"2209.14464","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-methods-for-logical-reasoning-over-1#ran","syntology_url":"https://syntology.ai/paper/2209.14464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14464"}},"official":{"repos":["amayuelas/NNKGReasoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tflex-temporal-feature-logic-embedding-1","slug":"tflex-temporal-feature-logic-embedding-1","title":"TFLEX: Temporal Feature-Logic Embedding Framework for Complex Reasoning over Temporal Knowledge Graph","date":"2022-05-28","arxiv_id":"2205.14307","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tflex-temporal-feature-logic-embedding-1#ran","syntology_url":"https://syntology.ai/paper/2205.14307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14307"}},"official":{"repos":["linxueyuanstdio/tflex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robustlr-evaluating-robustness-to-logical","slug":"robustlr-evaluating-robustness-to-logical","title":"RobustLR: Evaluating Robustness to Logical Perturbation in Deductive Reasoning","date":"2022-05-25","arxiv_id":"2205.12598","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/robustlr-evaluating-robustness-to-logical#ran","syntology_url":"https://syntology.ai/paper/2205.12598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12598"}},"official":{"repos":["ink-usc/robustlr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-zero-shot-reasoners","slug":"large-language-models-are-zero-shot-reasoners","title":"Large Language Models are Zero-Shot Reasoners","date":"2022-05-24","arxiv_id":"2205.11916","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-models-are-zero-shot-reasoners#ran","syntology_url":"https://syntology.ai/paper/2205.11916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11916"}},"official":{"repos":["kojima-takeshi188/zero_shot_cot"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/logical-reasoning-with-span-predictions-span","slug":"logical-reasoning-with-span-predictions-span","title":"Logical Reasoning with Span-Level Predictions for Interpretable and Robust NLI Models","date":"2022-05-23","arxiv_id":"2205.11432","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/logical-reasoning-with-span-predictions-span#ran","syntology_url":"https://syntology.ai/paper/2205.11432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11432"}},"official":{"repos":["joestacey/snli_logic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-paradox-of-learning-to-reason-from","slug":"on-the-paradox-of-learning-to-reason-from","title":"On the Paradox of Learning to Reason from Data","date":"2022-05-23","arxiv_id":"2205.11502","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-the-paradox-of-learning-to-reason-from#ran","syntology_url":"https://syntology.ai/paper/2205.11502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11502"}},"official":{"repos":["joshuacnf/paradox-learning2reason"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/table-based-fact-verification-with-self-1","slug":"table-based-fact-verification-with-self-1","title":"Table-based Fact Verification with Self-adaptive Mixture of Experts","date":"2022-04-19","arxiv_id":"2204.08753","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/table-based-fact-verification-with-self-1#ran","syntology_url":"https://syntology.ai/paper/2204.08753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08753"}},"official":{"repos":["thumlp/samoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/palm-scaling-language-modeling-with-pathways-1","slug":"palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","repositories_listed":7,"syntology":{"n":37,"n_ran":32,"n_constructed":16,"n_ran_checked":24,"n_instrument":8,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":21,"n_pointer_only":0,"phrase":"32 ran (of which 16 constructed an object rather than computing a result; 24 with no instrument failure: 2 honoured, 1 violated, 21 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/palm-scaling-language-modeling-with-pathways-1#ran","syntology_url":"https://syntology.ai/paper/2204.02311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02311"}},"official":null}},{"url":"/paper/training-compute-optimal-large-language","slug":"training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":3,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/training-compute-optimal-large-language#ran","syntology_url":"https://syntology.ai/paper/2203.15556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15556"}},"official":null}},{"url":"/paper/abductionrules-training-transformers-to-1","slug":"abductionrules-training-transformers-to-1","title":"AbductionRules: Training Transformers to Explain Unexpected Inputs","date":"2022-03-23","arxiv_id":"2203.12186","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/abductionrules-training-transformers-to-1#ran","syntology_url":"https://syntology.ai/paper/2203.12186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12186"}},"official":{"repos":["strong-ai-lab/abductionrules"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fairr-faithful-and-robust-deductive-reasoning-1","slug":"fairr-faithful-and-robust-deductive-reasoning-1","title":"FaiRR: Faithful and Robust Deductive Reasoning over Natural Language","date":"2022-03-19","arxiv_id":"2203.10261","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fairr-faithful-and-robust-deductive-reasoning-1#ran","syntology_url":"https://syntology.ai/paper/2203.10261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.10261"}},"official":{"repos":["ink-usc/fairr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adalogn-adaptive-logic-graph-network-for","slug":"adalogn-adaptive-logic-graph-network-for","title":"AdaLoGN: Adaptive Logic Graph Network for Reasoning-Based Machine Reading Comprehension","date":"2022-03-16","arxiv_id":"2203.08992","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adalogn-adaptive-logic-graph-network-for#ran","syntology_url":"https://syntology.ai/paper/2203.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08992"}},"official":{"repos":["nju-websoft/adalogn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-neuro-vector-symbolic-architecture-for","slug":"a-neuro-vector-symbolic-architecture-for","title":"A Neuro-vector-symbolic Architecture for Solving Raven's Progressive Matrices","date":"2022-03-09","arxiv_id":"2203.04571","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-neuro-vector-symbolic-architecture-for#ran","syntology_url":"https://syntology.ai/paper/2203.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.04571"}},"official":{"repos":["ibm/neuro-vector-symbolic-architectures"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/merit-meta-path-guided-contrastive-learning","slug":"merit-meta-path-guided-contrastive-learning","title":"MERIt: Meta-Path Guided Contrastive Learning for Logical Reasoning","date":"2022-03-01","arxiv_id":"2203.00357","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/merit-meta-path-guided-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2203.00357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.00357"}},"official":{"repos":["sparkjiao/merit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-algorithm-synthesis-with-recurrent","slug":"end-to-end-algorithm-synthesis-with-recurrent","title":"End-to-end Algorithm Synthesis with Recurrent Networks: Logical Extrapolation Without Overthinking","date":"2022-02-11","arxiv_id":"2202.05826","repositories_listed":1,"syntology":{"n":18,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":15,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/end-to-end-algorithm-synthesis-with-recurrent#ran","syntology_url":"https://syntology.ai/paper/2202.05826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05826"}},"official":{"repos":["aks2203/deep-thinking"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":15,"ran_from_kinds":["official"]}}},{"url":"/paper/vael-bridging-variational-autoencoders-and","slug":"vael-bridging-variational-autoencoders-and","title":"VAEL: Bridging Variational Autoencoders and Probabilistic Logic Programming","date":"2022-02-07","arxiv_id":"2202.04178","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vael-bridging-variational-autoencoders-and#ran","syntology_url":"https://syntology.ai/paper/2202.04178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04178"}},"official":{"repos":["elemisi/vael"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quantifying-adaptability-in-pre-trained","slug":"quantifying-adaptability-in-pre-trained","title":"Quantifying Adaptability in Pre-trained Language Models with 500 Tasks","date":"2021-12-06","arxiv_id":"2112.03204","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quantifying-adaptability-in-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2112.03204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03204"}},"official":{"repos":["belindal/taskbench500","facebookresearch/task_bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sqaler-scaling-question-answering-by","slug":"sqaler-scaling-question-answering-by","title":"SQALER: Scaling Question Answering by Decoupling Multi-Hop and Logical Reasoning","date":"2021-10-27","arxiv_id":"2110.14266","repositories_listed":0,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/sqaler-scaling-question-answering-by#ran","syntology_url":"https://syntology.ai/paper/2110.14266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14266"}},"official":null}},{"url":"/paper/probabilistic-entity-representation-model-for","slug":"probabilistic-entity-representation-model-for","title":"Probabilistic Entity Representation Model for Reasoning over Knowledge Graphs","date":"2021-10-26","arxiv_id":"2110.13522","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/probabilistic-entity-representation-model-for#ran","syntology_url":"https://syntology.ai/paper/2110.13522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13522"}},"official":{"repos":["akirato/perm-gaussiankg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/conditionalqa-a-complex-reading-comprehension","slug":"conditionalqa-a-complex-reading-comprehension","title":"ConditionalQA: A Complex Reading Comprehension Dataset with Conditional Answers","date":"2021-10-13","arxiv_id":"2110.06884","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conditionalqa-a-complex-reading-comprehension#ran","syntology_url":"https://syntology.ai/paper/2110.06884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06884"}},"official":{"repos":["haitian-sun/conditionalqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-explainable-phrasal","slug":"weakly-supervised-explainable-phrasal","title":"Weakly Supervised Explainable Phrasal Reasoning with Neural Fuzzy Logic","date":"2021-09-18","arxiv_id":"2109.08927","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/weakly-supervised-explainable-phrasal#ran","syntology_url":"https://syntology.ai/paper/2109.08927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.08927"}},"official":{"repos":["manga-uofa/epr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/from-lsat-the-progress-and-challenges-of","slug":"from-lsat-the-progress-and-challenges-of","title":"From LSAT: The Progress and Challenges of Complex Reasoning","date":"2021-08-02","arxiv_id":"2108.00648","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/from-lsat-the-progress-and-challenges-of#ran","syntology_url":"https://syntology.ai/paper/2108.00648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.00648"}},"official":null}}],"record_sha256":"80aa1354dba72ba47c7e96c8a5462f1ed64f36f8226a5965eed53ad46f2ef16d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}