{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/common-sense-reasoning/papers/2","list_of":"/task/common-sense-reasoning","task":"Common Sense Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":10,"rows_per_page":100,"rows":[101,200],"of":939,"counts":{"archive_papers_tagged":939,"with_a_code_link":325,"where_syntology_ran_a_sample":114,"not_listed_spam_title":0,"listed":939,"listed_where_code_ran":114,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":97,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":97,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/common-sense-reasoning","prev":"/task/common-sense-reasoning","next":"/task/common-sense-reasoning/papers/3","papers":[{"url":"/paper/global-local-tree-search-for-language-guided","slug":"global-local-tree-search-for-language-guided","title":"Global-Local Tree Search in VLMs for 3D Indoor Scene Generation","date":"2025-03-24","arxiv_id":"2503.18476","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/global-local-tree-search-for-language-guided#ran","syntology_url":"https://syntology.ai/paper/2503.18476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18476"}},"official":{"repos":["dw-dengwei/treesearchgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-fight-hallucinations-use-them","slug":"don-t-fight-hallucinations-use-them","title":"Don't Fight Hallucinations, Use Them: Estimating Image Realism using NLI over Atomic Facts","date":"2025-03-20","arxiv_id":"2503.15948","repositories_listed":1,"syntology":null},{"url":"/paper/cosmos-reason1-from-physical-common-sense-to","slug":"cosmos-reason1-from-physical-common-sense-to","title":"Cosmos-Reason1: From Physical Common Sense To Embodied Reasoning","date":"2025-03-18","arxiv_id":"2503.15558","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosmos-reason1-from-physical-common-sense-to#ran","syntology_url":"https://syntology.ai/paper/2503.15558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15558"}},"official":{"repos":["nvidia-cosmos/cosmos-reason1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alphadrive-unleashing-the-power-of-vlms-in","slug":"alphadrive-unleashing-the-power-of-vlms-in","title":"AlphaDrive: Unleashing the Power of VLMs in Autonomous Driving via Reinforcement Learning and Reasoning","date":"2025-03-10","arxiv_id":"2503.07608","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alphadrive-unleashing-the-power-of-vlms-in#ran","syntology_url":"https://syntology.ai/paper/2503.07608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07608"}},"official":{"repos":["hustvl/alphadrive"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-box-is-in-the-pen-evaluating-commonsense-1","slug":"the-box-is-in-the-pen-evaluating-commonsense-1","title":"The Box is in the Pen: Evaluating Commonsense Reasoning in Neural Machine Translation","date":"2025-03-05","arxiv_id":"2503.03308","repositories_listed":1,"syntology":null},{"url":"/paper/knowzrel-common-sense-knowledge-based-zero","slug":"knowzrel-common-sense-knowledge-based-zero","title":"KnowZRel: Common Sense Knowledge-based Zero-Shot Relationship Retrieval for Generalised Scene Graph Generation","date":"2025-02-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/predictaboard-benchmarking-llm-score","slug":"predictaboard-benchmarking-llm-score","title":"PredictaBoard: Benchmarking LLM Score Predictability","date":"2025-02-20","arxiv_id":"2502.14445","repositories_listed":1,"syntology":null},{"url":"/paper/titans-learning-to-memorize-at-test-time","slug":"titans-learning-to-memorize-at-test-time","title":"Titans: Learning to Memorize at Test Time","date":"2024-12-31","arxiv_id":"2501.00663","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-image-quality-assessment-for-robotic","slug":"embodied-image-quality-assessment-for-robotic","title":"Embodied Image Quality Assessment for Robotic Intelligence","date":"2024-12-25","arxiv_id":"2412.18774","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-grounded-planning-and-efficient","slug":"multi-modal-grounded-planning-and-efficient","title":"Multi-Modal Grounded Planning and Efficient Replanning For Learning Embodied Agents with A Few Examples","date":"2024-12-23","arxiv_id":"2412.17288","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-modal-grounded-planning-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2412.17288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.17288"}},"official":{"repos":["snumprlab/flare"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/quench-measuring-the-gap-between-indic-and","slug":"quench-measuring-the-gap-between-indic-and","title":"QUENCH: Measuring the gap between Indic and Non-Indic Contextual General Reasoning in LLMs","date":"2024-12-16","arxiv_id":"2412.11763","repositories_listed":1,"syntology":null},{"url":"/paper/a-surprisal-oracle-for-when-every-layer","slug":"a-surprisal-oracle-for-when-every-layer","title":"A surprisal oracle for when every layer counts","date":"2024-12-04","arxiv_id":"2412.03098","repositories_listed":1,"syntology":null},{"url":"/paper/citywalker-learning-embodied-urban-navigation","slug":"citywalker-learning-embodied-urban-navigation","title":"CityWalker: Learning Embodied Urban Navigation from Web-Scale Videos","date":"2024-11-26","arxiv_id":"2411.17820","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citywalker-learning-embodied-urban-navigation#ran","syntology_url":"https://syntology.ai/paper/2411.17820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17820"}},"official":{"repos":["ai4ce/CityWalker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-little-less-conversation-a-little-more","slug":"a-little-less-conversation-a-little-more","title":"A little less conversation, a little more action, please: Investigating the physical common-sense of LLMs in a 3D embodied environment","date":"2024-10-30","arxiv_id":"2410.23242","repositories_listed":1,"syntology":null},{"url":"/paper/language-agents-meet-causality-bridging-llms","slug":"language-agents-meet-causality-bridging-llms","title":"Language Agents Meet Causality -- Bridging LLMs and Causal World Models","date":"2024-10-25","arxiv_id":"2410.19923","repositories_listed":1,"syntology":null},{"url":"/paper/learning-low-level-causal-relations-using-a","slug":"learning-low-level-causal-relations-using-a","title":"Learning Low-Level Causal Relations using a Simulated Robotic Arm","date":"2024-10-10","arxiv_id":"2410.07751","repositories_listed":1,"syntology":null},{"url":"/paper/prefixquant-static-quantization-beats-dynamic","slug":"prefixquant-static-quantization-beats-dynamic","title":"PrefixQuant: Eliminating Outliers by Prefixed Tokens for Large Language Models Quantization","date":"2024-10-07","arxiv_id":"2410.05265","repositories_listed":1,"syntology":null},{"url":"/paper/a-hitchhikers-guide-to-fine-grained-face","slug":"a-hitchhikers-guide-to-fine-grained-face","title":"A Hitchhikers Guide to Fine-Grained Face Forgery Detection Using Common Sense Reasoning","date":"2024-10-01","arxiv_id":"2410.00485","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-hitchhikers-guide-to-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2410.00485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00485"}},"official":{"repos":["NickyFot/HitchhikersGuide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-essential-and-nonessential","slug":"revisiting-essential-and-nonessential","title":"Revisiting Essential and Nonessential Settings of Evidential Deep Learning","date":"2024-10-01","arxiv_id":"2410.00393","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/revisiting-essential-and-nonessential#ran","syntology_url":"https://syntology.ai/paper/2410.00393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00393"}},"official":{"repos":["mengyuanchen21/re-edl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-02769","slug":"2408-02769","title":"From Recognition to Prediction: Leveraging Sequence Reasoning for Action Anticipation","date":"2024-08-05","arxiv_id":"2408.02769","repositories_listed":1,"syntology":null},{"url":"/paper/model-surgery-modulating-llm-s-behavior-via","slug":"model-surgery-modulating-llm-s-behavior-via","title":"Model Surgery: Modulating LLM's Behavior Via Simple Parameter Editing","date":"2024-07-11","arxiv_id":"2407.08770","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-surgery-modulating-llm-s-behavior-via#ran","syntology_url":"https://syntology.ai/paper/2407.08770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08770"}},"official":{"repos":["lucywang720/model-surgery"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-sample-efficiency-of-reinforcement","slug":"improving-sample-efficiency-of-reinforcement","title":"Improving Sample Efficiency of Reinforcement Learning with Background Knowledge from Large Language Models","date":"2024-07-04","arxiv_id":"2407.03964","repositories_listed":1,"syntology":null},{"url":"/paper/regmix-data-mixture-as-regression-for","slug":"regmix-data-mixture-as-regression-for","title":"RegMix: Data Mixture as Regression for Language Model Pre-training","date":"2024-07-01","arxiv_id":"2407.01492","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regmix-data-mixture-as-regression-for#ran","syntology_url":"https://syntology.ai/paper/2407.01492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01492"}},"official":{"repos":["sail-sg/regmix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-and-analyzing-relationship","slug":"evaluating-and-analyzing-relationship","title":"Evaluating and Analyzing Relationship Hallucinations in Large Vision-Language Models","date":"2024-06-24","arxiv_id":"2406.16449","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-and-analyzing-relationship#ran","syntology_url":"https://syntology.ai/paper/2406.16449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16449"}},"official":{"repos":["mrwu-mac/R-Bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/human-ai-collectives-produce-the-most","slug":"human-ai-collectives-produce-the-most","title":"Human-AI collectives produce the most accurate differential diagnoses","date":"2024-06-21","arxiv_id":"2406.14981","repositories_listed":1,"syntology":null},{"url":"/paper/improving-visual-commonsense-in-language","slug":"improving-visual-commonsense-in-language","title":"Improving Visual Commonsense in Language Models via Multiple Image Generation","date":"2024-06-19","arxiv_id":"2406.13621","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prompting-taxonomy-a-universal","slug":"hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","arxiv_id":"2406.12644","repositories_listed":1,"syntology":null},{"url":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}},"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-of-video-datasets-for-grounded-event","slug":"a-survey-of-video-datasets-for-grounded-event","title":"A Survey of Video Datasets for Grounded Event Understanding","date":"2024-06-14","arxiv_id":"2406.09646","repositories_listed":1,"syntology":null},{"url":"/paper/bamo-at-semeval-2024-task-9-brainteaser-a","slug":"bamo-at-semeval-2024-task-9-brainteaser-a","title":"BAMO at SemEval-2024 Task 9: BRAINTEASER: A Novel Task Defying Common Sense","date":"2024-06-07","arxiv_id":"2406.04947","repositories_listed":1,"syntology":null},{"url":"/paper/do-language-models-understand-morality","slug":"do-language-models-understand-morality","title":"Do Language Models Understand Morality? Towards a Robust Detection of Moral Content","date":"2024-06-06","arxiv_id":"2406.04143","repositories_listed":1,"syntology":null},{"url":"/paper/every-answer-matters-evaluating-commonsense","slug":"every-answer-matters-evaluating-commonsense","title":"Every Answer Matters: Evaluating Commonsense with Probabilistic Measures","date":"2024-06-06","arxiv_id":"2406.04145","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-as-evaluators-for","slug":"large-language-models-as-evaluators-for","title":"Large Language Models as Evaluators for Recommendation Explanations","date":"2024-06-05","arxiv_id":"2406.03248","repositories_listed":1,"syntology":null},{"url":"/paper/extended-mind-transformers","slug":"extended-mind-transformers","title":"Extended Mind Transformers","date":"2024-06-04","arxiv_id":"2406.02332","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-accord-closing-the-commonsense","slug":"texttt-accord-closing-the-commonsense","title":"$\\texttt{ACCORD}$: Closing the Commonsense Measurability Gap","date":"2024-06-04","arxiv_id":"2406.02804","repositories_listed":1,"syntology":null},{"url":"/paper/easy-problems-that-llms-get-wrong","slug":"easy-problems-that-llms-get-wrong","title":"Easy Problems That LLMs Get Wrong","date":"2024-05-30","arxiv_id":"2405.19616","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/easy-problems-that-llms-get-wrong#ran","syntology_url":"https://syntology.ai/paper/2405.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19616"}},"official":{"repos":["autogenai/easy-problems-that-llms-get-wrong"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/irel-at-semeval-2024-task-9-improving","slug":"irel-at-semeval-2024-task-9-improving","title":"iREL at SemEval-2024 Task 9: Improving Conventional Prompting Methods for Brain Teasers","date":"2024-05-25","arxiv_id":"2405.16129","repositories_listed":1,"syntology":null},{"url":"/paper/meteor-mamba-based-traversal-of-rationale-for","slug":"meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","arxiv_id":"2405.15574","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meteor-mamba-based-traversal-of-rationale-for#ran","syntology_url":"https://syntology.ai/paper/2405.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15574"}},"official":{"repos":["byungkwanlee/meteor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/openba-v2-reaching-77-3-high-compression","slug":"openba-v2-reaching-77-3-high-compression","title":"OpenBA-V2: Reaching 77.3% High Compression Ratio with Fast Multi-Stage Pruning","date":"2024-05-09","arxiv_id":"2405.05957","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-aigc-video-quality-a-focus-on","slug":"exploring-aigc-video-quality-a-focus-on","title":"Exploring AIGC Video Quality: A Focus on Visual Harmony, Video-Text Consistency and Domain Distribution Gap","date":"2024-04-21","arxiv_id":"2404.13573","repositories_listed":1,"syntology":null},{"url":"/paper/memory-sharing-for-large-language-model-based","slug":"memory-sharing-for-large-language-model-based","title":"Memory Sharing for Large Language Model based Agents","date":"2024-04-15","arxiv_id":"2404.09982","repositories_listed":1,"syntology":null},{"url":"/paper/vllms-provide-better-context-for-emotion","slug":"vllms-provide-better-context-for-emotion","title":"VLLMs Provide Better Context for Emotion Understanding Through Common Sense Reasoning","date":"2024-04-10","arxiv_id":"2404.07078","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-llms-the-evolution-of-latent","slug":"unveiling-llms-the-evolution-of-latent","title":"Unveiling LLMs: The Evolution of Latent Representations in a Dynamic Knowledge Graph","date":"2024-04-04","arxiv_id":"2404.03623","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unveiling-llms-the-evolution-of-latent#ran","syntology_url":"https://syntology.ai/paper/2404.03623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03623"}},"official":{"repos":["Ipazia-AI/latent-explorer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ails-ntua-at-semeval-2024-task-9-cracking","slug":"ails-ntua-at-semeval-2024-task-9-cracking","title":"AILS-NTUA at SemEval-2024 Task 9: Cracking Brain Teasers: Transformer Models for Lateral Thinking Puzzles","date":"2024-04-01","arxiv_id":"2404.01084","repositories_listed":1,"syntology":null},{"url":"/paper/common-sense-enhanced-knowledge-based","slug":"common-sense-enhanced-knowledge-based","title":"Common Sense Enhanced Knowledge-based Recommendation with Large Language Model","date":"2024-03-27","arxiv_id":"2403.18325","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-need-consultants-for","slug":"large-language-models-need-consultants-for","title":"Large Language Models Need Consultants for Reasoning: Becoming an Expert in a Complex Human System Through Behavior Simulation","date":"2024-03-27","arxiv_id":"2403.18230","repositories_listed":1,"syntology":null},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/to-help-or-not-to-help-llm-based-attentive","slug":"to-help-or-not-to-help-llm-based-attentive","title":"To Help or Not to Help: LLM-based Attentive Support for Human-Robot Group Interactions","date":"2024-03-19","arxiv_id":"2403.12533","repositories_listed":1,"syntology":null},{"url":"/paper/phd-a-prompted-visual-hallucination","slug":"phd-a-prompted-visual-hallucination","title":"PhD: A ChatGPT-Prompted Visual hallucination Evaluation Dataset","date":"2024-03-17","arxiv_id":"2403.11116","repositories_listed":1,"syntology":null},{"url":"/paper/branch-train-mix-mixing-expert-llms-into-a","slug":"branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","arxiv_id":"2403.07816","repositories_listed":1,"syntology":null},{"url":"/paper/repeated-padding-as-data-augmentation-for","slug":"repeated-padding-as-data-augmentation-for","title":"Repeated Padding for Sequential Recommendation","date":"2024-03-11","arxiv_id":"2403.06372","repositories_listed":1,"syntology":null},{"url":"/paper/fact-and-reflection-far-improves-confidence","slug":"fact-and-reflection-far-improves-confidence","title":"Fact-and-Reflection (FaR) Improves Confidence Calibration of Large Language Models","date":"2024-02-27","arxiv_id":"2402.17124","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fact-and-reflection-far-improves-confidence#ran","syntology_url":"https://syntology.ai/paper/2402.17124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17124"}},"official":{"repos":["colinzhaoust/fact-and-reflection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-reasoning-based-on-large-language","slug":"hybrid-reasoning-based-on-large-language","title":"Hybrid Reasoning Based on Large Language Models for Autonomous Car Driving","date":"2024-02-21","arxiv_id":"2402.13602","repositories_listed":1,"syntology":null},{"url":"/paper/moelora-contrastive-learning-guided-mixture","slug":"moelora-contrastive-learning-guided-mixture","title":"MoELoRA: Contrastive Learning Guided Mixture of Experts on Parameter-Efficient Fine-Tuning for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12851","repositories_listed":1,"syntology":null},{"url":"/paper/openfmnav-towards-open-set-zero-shot-object","slug":"openfmnav-towards-open-set-zero-shot-object","title":"OpenFMNav: Towards Open-Set Zero-Shot Object Navigation via Vision-Language Foundation Models","date":"2024-02-16","arxiv_id":"2402.10670","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openfmnav-towards-open-set-zero-shot-object#ran","syntology_url":"https://syntology.ai/paper/2402.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10670"}},"official":{"repos":["yxKryptonite/OpenFMNav"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/common-sense-reasoning-for-deep-fake","slug":"common-sense-reasoning-for-deep-fake","title":"Common Sense Reasoning for Deepfake Detection","date":"2024-01-31","arxiv_id":"2402.00126","repositories_listed":1,"syntology":null},{"url":"/paper/hazard-challenge-embodied-decision-making-in","slug":"hazard-challenge-embodied-decision-making-in","title":"HAZARD Challenge: Embodied Decision Making in Dynamically Changing Environments","date":"2024-01-23","arxiv_id":"2401.12975","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hazard-challenge-embodied-decision-making-in#ran","syntology_url":"https://syntology.ai/paper/2401.12975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12975"}},"official":{"repos":["umass-foundation-model/hazard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cbvs-a-large-scale-chinese-image-text","slug":"cbvs-a-large-scale-chinese-image-text","title":"CBVS: A Large-Scale Chinese Image-Text Benchmark for Real-World Short Video Search Scenarios","date":"2024-01-19","arxiv_id":"2401.10475","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-neurosymbolic","slug":"large-language-models-are-neurosymbolic","title":"Large Language Models Are Neurosymbolic Reasoners","date":"2024-01-17","arxiv_id":"2401.09334","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-neurosymbolic#ran","syntology_url":"https://syntology.ai/paper/2401.09334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09334"}},"official":{"repos":["hyintell/llmsymbolic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-content-based-novelty-measure-for-scholarly","slug":"a-content-based-novelty-measure-for-scholarly","title":"A Content-Based Novelty Measure for Scholarly Publications: A Proof of Concept","date":"2024-01-08","arxiv_id":"2401.03642","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-synthesis-of-patient-records","slug":"collaborative-synthesis-of-patient-records","title":"Collaborative Synthesis of Patient Records through Multi-Visit Health State Inference","date":"2023-12-22","arxiv_id":"2312.14646","repositories_listed":1,"syntology":null},{"url":"/paper/a-semantic-space-is-worth-256-language","slug":"a-semantic-space-is-worth-256-language","title":"A Semantic Space is Worth 256 Language Descriptions: Make Stronger Segmentation Models with Descriptive Properties","date":"2023-12-21","arxiv_id":"2312.13764","repositories_listed":1,"syntology":null},{"url":"/paper/corecode-a-common-sense-annotated-dialogue","slug":"corecode-a-common-sense-annotated-dialogue","title":"CORECODE: A Common Sense Annotated Dialogue Dataset with Benchmark Tasks for Chinese Large Language Models","date":"2023-12-20","arxiv_id":"2312.12853","repositories_listed":1,"syntology":null},{"url":"/paper/cldr-contrastive-learning-drug-response","slug":"cldr-contrastive-learning-drug-response","title":"CLDR: Contrastive Learning Drug Response Models from Natural Language Supervision","date":"2023-12-17","arxiv_id":"2312.10707","repositories_listed":1,"syntology":null},{"url":"/paper/holodeck-language-guided-generation-of-3d","slug":"holodeck-language-guided-generation-of-3d","title":"Holodeck: Language Guided Generation of 3D Embodied AI Environments","date":"2023-12-14","arxiv_id":"2312.09067","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/holodeck-language-guided-generation-of-3d#ran","syntology_url":"https://syntology.ai/paper/2312.09067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09067"}},"official":{"repos":["allenai/Holodeck"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-agent-based-modeling-with-actions","slug":"generative-agent-based-modeling-with-actions","title":"Generative agent-based modeling with actions grounded in physical, social, or digital space using Concordia","date":"2023-12-06","arxiv_id":"2312.03664","repositories_listed":1,"syntology":null},{"url":"/paper/speak-like-a-native-prompting-large-language","slug":"speak-like-a-native-prompting-large-language","title":"AlignedCoT: Prompting Large Language Models via Native-Speaking Demonstrations","date":"2023-11-22","arxiv_id":"2311.13538","repositories_listed":1,"syntology":null},{"url":"/paper/a-language-agent-for-autonomous-driving","slug":"a-language-agent-for-autonomous-driving","title":"A Language Agent for Autonomous Driving","date":"2023-11-17","arxiv_id":"2311.10813","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-temporally-grounded","slug":"are-large-language-models-temporally-grounded","title":"Are Large Language Models Temporally Grounded?","date":"2023-11-14","arxiv_id":"2311.08398","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-images-for-intuitively-reasoning","slug":"chain-of-images-for-intuitively-reasoning","title":"Chain of Images for Intuitively Reasoning","date":"2023-11-09","arxiv_id":"2311.09241","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-road-with-gpt-4v-ision-early","slug":"on-the-road-with-gpt-4v-ision-early","title":"On the Road with GPT-4V(ision): Early Explorations of Visual-Language Model on Autonomous Driving","date":"2023-11-09","arxiv_id":"2311.05332","repositories_listed":1,"syntology":null},{"url":"/paper/neusyre-neuro-symbolic-visual-understanding","slug":"neusyre-neuro-symbolic-visual-understanding","title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","date":"2023-11-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/copal-id-indonesian-language-reasoning-with","slug":"copal-id-indonesian-language-reasoning-with","title":"COPAL-ID: Indonesian Language Reasoning with Local Culture and Nuances","date":"2023-11-02","arxiv_id":"2311.01012","repositories_listed":1,"syntology":null},{"url":"/paper/rome-evaluating-pre-trained-vision-language","slug":"rome-evaluating-pre-trained-vision-language","title":"ROME: Evaluating Pre-trained Vision-Language Models on Reasoning beyond Visual Common Sense","date":"2023-10-30","arxiv_id":"2310.19301","repositories_listed":1,"syntology":null},{"url":"/paper/dcqa-document-level-chart-question-answering","slug":"dcqa-document-level-chart-question-answering","title":"DCQA: Document-Level Chart Question Answering towards Complex Reasoning and Common-Sense Understanding","date":"2023-10-29","arxiv_id":"2310.18983","repositories_listed":1,"syntology":null},{"url":"/paper/llm-fp4-4-bit-floating-point-quantized","slug":"llm-fp4-4-bit-floating-point-quantized","title":"LLM-FP4: 4-Bit Floating-Point Quantized Transformers","date":"2023-10-25","arxiv_id":"2310.16836","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/llm-fp4-4-bit-floating-point-quantized#ran","syntology_url":"https://syntology.ai/paper/2310.16836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16836"}},"official":{"repos":["nbasyl/llm-fp4"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-large-language-models-as-a-source","slug":"exploring-large-language-models-as-a-source","title":"Exploring Large Language Models as a Source of Common-Sense Knowledge for Robots","date":"2023-10-19","arxiv_id":"2311.08412","repositories_listed":1,"syntology":null},{"url":"/paper/gesturegpt-zero-shot-interactive-gesture","slug":"gesturegpt-zero-shot-interactive-gesture","title":"GestureGPT: Toward Zero-Shot Free-Form Hand Gesture Understanding with Large Language Model Agents","date":"2023-10-19","arxiv_id":"2310.12821","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-multi-agent-coordination-abilities","slug":"evaluating-multi-agent-coordination-abilities","title":"LLM-Coordination: Evaluating and Analyzing Multi-agent Coordination Abilities in Large Language Models","date":"2023-10-05","arxiv_id":"2310.03903","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-multi-agent-coordination-abilities#ran","syntology_url":"https://syntology.ai/paper/2310.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03903"}},"official":{"repos":["eric-ai-lab/llm_coordination"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rladapter-bridging-large-language-models-to","slug":"rladapter-bridging-large-language-models-to","title":"AdaRefiner: Refining Decisions of Language Models with Adaptive Feedback","date":"2023-09-29","arxiv_id":"2309.17176","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rladapter-bridging-large-language-models-to#ran","syntology_url":"https://syntology.ai/paper/2309.17176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17176"}},"official":{"repos":["pku-rl/adarefiner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/telling-stories-for-common-sense-zero-shot","slug":"telling-stories-for-common-sense-zero-shot","title":"Telling Stories for Common Sense Zero-Shot Action Recognition","date":"2023-09-29","arxiv_id":"2309.17327","repositories_listed":1,"syntology":null},{"url":"/paper/self-refined-large-language-model-as","slug":"self-refined-large-language-model-as","title":"Self-Refined Large Language Model as Automated Reward Function Designer for Deep Reinforcement Learning in Robotics","date":"2023-09-13","arxiv_id":"2309.06687","repositories_listed":1,"syntology":null},{"url":"/paper/trafficgpt-viewing-processing-and-interacting","slug":"trafficgpt-viewing-processing-and-interacting","title":"TrafficGPT: Viewing, Processing and Interacting with Traffic Foundation Models","date":"2023-09-13","arxiv_id":"2309.06719","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trafficgpt-viewing-processing-and-interacting#ran","syntology_url":"https://syntology.ai/paper/2309.06719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06719"}},"official":{"repos":["lijlansg/trafficgpt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2309-06256","slug":"2309-06256","title":"Mitigating the Alignment Tax of RLHF","date":"2023-09-12","arxiv_id":"2309.06256","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2309-06256#ran","syntology_url":"https://syntology.ai/paper/2309.06256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06256"}},"official":{"repos":["avalonstrel/mitigating-the-alignment-tax-of-rlhf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2309-06363","slug":"2309-06363","title":"Learning to Predict Concept Ordering for Common Sense Generation","date":"2023-09-12","arxiv_id":"2309.06363","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2309-06363#ran","syntology_url":"https://syntology.ai/paper/2309.06363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06363"}},"official":{"repos":["tianhuizhang/concept_ordering"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/textbooks-are-all-you-need-ii-phi-1-5","slug":"textbooks-are-all-you-need-ii-phi-1-5","title":"Textbooks Are All You Need II: phi-1.5 technical report","date":"2023-09-11","arxiv_id":"2309.05463","repositories_listed":1,"syntology":null},{"url":"/paper/saynav-grounding-large-language-models-for","slug":"saynav-grounding-large-language-models-for","title":"SayNav: Grounding Large Language Models for Dynamic Planning to Navigation in New Environments","date":"2023-09-08","arxiv_id":"2309.04077","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/saynav-grounding-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2309.04077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04077"}},"official":{"repos":["arajv/SayNav"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/declarative-reasoning-on-explanations-using","slug":"declarative-reasoning-on-explanations-using","title":"Declarative Reasoning on Explanations Using Constraint Logic Programming","date":"2023-09-01","arxiv_id":"2309.00422","repositories_listed":1,"syntology":null},{"url":"/paper/towards-one-shot-learning-for-text","slug":"towards-one-shot-learning-for-text","title":"Towards One-Shot Learning for Text Classification using Inductive Logic Programming","date":"2023-08-30","arxiv_id":"2308.15885","repositories_listed":1,"syntology":null},{"url":"/paper/token-scaled-logit-distillation-for-ternary-1","slug":"token-scaled-logit-distillation-for-ternary-1","title":"Token-Scaled Logit Distillation for Ternary Weight Generative Language Models","date":"2023-08-13","arxiv_id":"2308.06744","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/token-scaled-logit-distillation-for-ternary-1#ran","syntology_url":"https://syntology.ai/paper/2308.06744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06744"}},"official":{"repos":["aiha-lab/tsld"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ketm-a-knowledge-enhanced-text-matching","slug":"ketm-a-knowledge-enhanced-text-matching","title":"KETM:A Knowledge-Enhanced Text Matching method","date":"2023-08-11","arxiv_id":"2308.06235","repositories_listed":1,"syntology":null},{"url":"/paper/do-multilingual-language-models-think-better","slug":"do-multilingual-language-models-think-better","title":"Do Multilingual Language Models Think Better in English?","date":"2023-08-02","arxiv_id":"2308.01223","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/do-multilingual-language-models-think-better#ran","syntology_url":"https://syntology.ai/paper/2308.01223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01223"}},"official":{"repos":["juletx/self-translate"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/drive-like-a-human-rethinking-autonomous","slug":"drive-like-a-human-rethinking-autonomous","title":"Drive Like a Human: Rethinking Autonomous Driving with Large Language Models","date":"2023-07-14","arxiv_id":"2307.07162","repositories_listed":1,"syntology":null},{"url":"/paper/garbage-in-garbage-out-zero-shot-detection-of","slug":"garbage-in-garbage-out-zero-shot-detection-of","title":"Garbage in, garbage out: Zero-shot detection of crime using Large Language Models","date":"2023-07-04","arxiv_id":"2307.06844","repositories_listed":1,"syntology":null},{"url":"/paper/reflect-summarizing-robot-experiences-for","slug":"reflect-summarizing-robot-experiences-for","title":"REFLECT: Summarizing Robot Experiences for Failure Explanation and Correction","date":"2023-06-27","arxiv_id":"2306.15724","repositories_listed":1,"syntology":{"n":19,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/reflect-summarizing-robot-experiences-for#ran","syntology_url":"https://syntology.ai/paper/2306.15724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15724"}},"official":{"repos":["real-stanford/reflect"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/being-right-for-whose-right-reasons","slug":"being-right-for-whose-right-reasons","title":"Being Right for Whose Right Reasons?","date":"2023-06-01","arxiv_id":"2306.00639","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-not-abstract","slug":"large-language-models-are-not-abstract","title":"Large Language Models Are Not Strong Abstract Reasoners","date":"2023-05-31","arxiv_id":"2305.19555","repositories_listed":1,"syntology":null},{"url":"/paper/plasma-making-small-language-models-better","slug":"plasma-making-small-language-models-better","title":"PlaSma: Making Small Language Models Better Procedural Knowledge Models for (Counterfactual) Planning","date":"2023-05-31","arxiv_id":"2305.19472","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plasma-making-small-language-models-better#ran","syntology_url":"https://syntology.ai/paper/2305.19472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19472"}},"official":{"repos":["allenai/plasma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ghost-in-the-minecraft-generally-capable","slug":"ghost-in-the-minecraft-generally-capable","title":"Ghost in the Minecraft: Generally Capable Agents for Open-World Environments via Large Language Models with Text-based Knowledge and Memory","date":"2023-05-25","arxiv_id":"2305.17144","repositories_listed":1,"syntology":null},{"url":"/paper/memex-detecting-explanatory-evidence-for","slug":"memex-detecting-explanatory-evidence-for","title":"MEMEX: Detecting Explanatory Evidence for Memes via Knowledge-Enriched Contextualization","date":"2023-05-25","arxiv_id":"2305.15913","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memex-detecting-explanatory-evidence-for#ran","syntology_url":"https://syntology.ai/paper/2305.15913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15913"}},"official":{"repos":["lcs2-iiitd/memex_meme_evidence"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}}],"record_sha256":"d0a6d6d980f6c03db466d44c1039c1da07bb8c83f691432ac1e875e75367b336","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}