{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/23","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":23,"pages_in_order":109,"rows_per_page":100,"rows":[2201,2300],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/22","next":"/task/question-answering/papers/24","papers":[{"url":"/paper/opseval-a-comprehensive-task-oriented-aiops","slug":"opseval-a-comprehensive-task-oriented-aiops","title":"OpsEval: A Comprehensive IT Operations Benchmark Suite for Large Language Models","date":"2023-10-11","arxiv_id":"2310.07637","repositories_listed":1,"syntology":null},{"url":"/paper/qacheck-a-demonstration-system-for-question","slug":"qacheck-a-demonstration-system-for-question","title":"QACHECK: A Demonstration System for Question-Guided Multi-Hop Fact-Checking","date":"2023-10-11","arxiv_id":"2310.07609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/qacheck-a-demonstration-system-for-question#ran","syntology_url":"https://syntology.ai/paper/2310.07609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07609"}},"official":{"repos":["xinyuanlu00/qacheck"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memsum-dqa-adapting-an-efficient-long","slug":"memsum-dqa-adapting-an-efficient-long","title":"MemSum-DQA: Adapting An Efficient Long Document Extractive Summarizer for Document Question Answering","date":"2023-10-10","arxiv_id":"2310.06436","repositories_listed":1,"syntology":null},{"url":"/paper/seer-a-knapsack-approach-to-exemplar","slug":"seer-a-knapsack-approach-to-exemplar","title":"SEER : A Knapsack approach to Exemplar Selection for In-Context HybridQA","date":"2023-10-10","arxiv_id":"2310.06675","repositories_listed":1,"syntology":{"n":16,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":10,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/seer-a-knapsack-approach-to-exemplar#ran","syntology_url":"https://syntology.ai/paper/2310.06675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06675"}},"official":{"repos":["jtonglet/seer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-data-bias-in-music-avqa-crafting-a","slug":"tackling-data-bias-in-music-avqa-crafting-a","title":"Tackling Data Bias in MUSIC-AVQA: Crafting a Balanced Dataset for Unbiased Question-Answering","date":"2023-10-10","arxiv_id":"2310.06238","repositories_listed":1,"syntology":null},{"url":"/paper/a-bias-variance-covariance-decomposition-of","slug":"a-bias-variance-covariance-decomposition-of","title":"A Bias-Variance-Covariance Decomposition of Kernel Scores for Generative Models","date":"2023-10-09","arxiv_id":"2310.05833","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-bias-variance-covariance-decomposition-of#ran","syntology_url":"https://syntology.ai/paper/2310.05833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05833"}},"official":{"repos":["mlo-lab/bvcd_generative_models"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/enhancing-long-form-text-generation-in-mental","slug":"enhancing-long-form-text-generation-in-mental","title":"Task-Adaptive Tokenization: Enhancing Long-Form Text Generation Efficacy in Mental Health and Beyond","date":"2023-10-09","arxiv_id":"2310.05317","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-long-form-text-generation-in-mental#ran","syntology_url":"https://syntology.ai/paper/2310.05317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05317"}},"official":{"repos":["michigannlp/task-adaptive_tokenization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interrolang-exploring-nlp-models-and-datasets","slug":"interrolang-exploring-nlp-models-and-datasets","title":"InterroLang: Exploring NLP Models and Datasets through Dialogue-based Explanations","date":"2023-10-09","arxiv_id":"2310.05592","repositories_listed":1,"syntology":null},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-investigation-of-llms-inefficacy-in","slug":"an-investigation-of-llms-inefficacy-in","title":"An Investigation of LLMs' Inefficacy in Understanding Converse Relations","date":"2023-10-08","arxiv_id":"2310.05163","repositories_listed":1,"syntology":null},{"url":"/paper/minprompt-graph-based-minimal-prompt-data","slug":"minprompt-graph-based-minimal-prompt-data","title":"MinPrompt: Graph-based Minimal Prompt Data Augmentation for Few-shot Question Answering","date":"2023-10-08","arxiv_id":"2310.05007","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-generation-synergy-augmented-large","slug":"retrieval-generation-synergy-augmented-large","title":"Retrieval-Generation Synergy Augmented Large Language Models","date":"2023-10-08","arxiv_id":"2310.05149","repositories_listed":1,"syntology":null},{"url":"/paper/self-knowledge-guided-retrieval-augmentation","slug":"self-knowledge-guided-retrieval-augmentation","title":"Self-Knowledge Guided Retrieval Augmentation for Large Language Models","date":"2023-10-08","arxiv_id":"2310.05002","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-zero-shot-abilities-of-vision","slug":"analyzing-zero-shot-abilities-of-vision","title":"Analyzing Zero-Shot Abilities of Vision-Language Models on Video Understanding Tasks","date":"2023-10-07","arxiv_id":"2310.04914","repositories_listed":1,"syntology":null},{"url":"/paper/a-long-way-to-go-investigating-length","slug":"a-long-way-to-go-investigating-length","title":"A Long Way to Go: Investigating Length Correlations in RLHF","date":"2023-10-05","arxiv_id":"2310.03716","repositories_listed":1,"syntology":null},{"url":"/paper/biobridge-bridging-biomedical-foundation","slug":"biobridge-bridging-biomedical-foundation","title":"BioBridge: Bridging Biomedical Foundation Models via Knowledge Graphs","date":"2023-10-05","arxiv_id":"2310.03320","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/biobridge-bridging-biomedical-foundation#ran","syntology_url":"https://syntology.ai/paper/2310.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03320"}},"official":{"repos":["ryanwangzf/biobridge"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/decoderlens-layerwise-interpretation-of","slug":"decoderlens-layerwise-interpretation-of","title":"DecoderLens: Layerwise Interpretation of Encoder-Decoder Transformers","date":"2023-10-05","arxiv_id":"2310.03686","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-multi-agent-coordination-abilities","slug":"evaluating-multi-agent-coordination-abilities","title":"LLM-Coordination: Evaluating and Analyzing Multi-agent Coordination Abilities in Large Language Models","date":"2023-10-05","arxiv_id":"2310.03903","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-multi-agent-coordination-abilities#ran","syntology_url":"https://syntology.ai/paper/2310.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03903"}},"official":{"repos":["eric-ai-lab/llm_coordination"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mappergpt-large-language-models-for-linking","slug":"mappergpt-large-language-models-for-linking","title":"MapperGPT: Large Language Models for Linking and Mapping Entities","date":"2023-10-05","arxiv_id":"2310.03666","repositories_listed":1,"syntology":null},{"url":"/paper/procedural-text-mining-with-large-language","slug":"procedural-text-mining-with-large-language","title":"Procedural Text Mining with Large Language Models","date":"2023-10-05","arxiv_id":"2310.03376","repositories_listed":1,"syntology":null},{"url":"/paper/a-umls-augmented-framework-for-improving","slug":"a-umls-augmented-framework-for-improving","title":"Integrating UMLS Knowledge into Large Language Models for Medical Question Answering","date":"2023-10-04","arxiv_id":"2310.02778","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-are-large-language-models-from-agents","slug":"how-far-are-large-language-models-from-agents","title":"How FaR Are Large Language Models From Agents with Theory-of-Mind?","date":"2023-10-04","arxiv_id":"2310.03051","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-question-answering-for-unified","slug":"multimodal-question-answering-for-unified","title":"Multimodal Question Answering for Unified Information Extraction","date":"2023-10-04","arxiv_id":"2310.03017","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-as-knowledge-bases-for-visual","slug":"language-models-as-knowledge-bases-for-visual","title":"Language Models as Knowledge Bases for Visual Word Sense Disambiguation","date":"2023-10-03","arxiv_id":"2310.01960","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-models-as-knowledge-bases-for-visual#ran","syntology_url":"https://syntology.ai/paper/2310.01960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01960"}},"official":{"repos":["anastasiakrith/llm-for-vwsd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/fool-your-vision-and-language-model-with","slug":"fool-your-vision-and-language-model-with","title":"Fool Your (Vision and) Language Model With Embarrassingly Simple Permutations","date":"2023-10-02","arxiv_id":"2310.01651","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fool-your-vision-and-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2310.01651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01651"}},"official":{"repos":["ys-zong/foolyourvllms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/understanding-ai-cognition-a-neural-module","slug":"understanding-ai-cognition-a-neural-module","title":"A Framework for Inference Inspired by Human Memory Mechanisms","date":"2023-10-01","arxiv_id":"2310.09297","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-late-interaction-multi-modal-1","slug":"fine-grained-late-interaction-multi-modal-1","title":"Fine-grained Late-interaction Multi-modal Retrieval for Retrieval Augmented Visual Question Answering","date":"2023-09-29","arxiv_id":"2309.17133","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-late-interaction-multi-modal-1#ran","syntology_url":"https://syntology.ai/paper/2309.17133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17133"}},"official":{"repos":["linweizhedragon/retrieval-augmented-visual-question-answering"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-long-form-legal-question","slug":"interpretable-long-form-legal-question","title":"Interpretable Long-Form Legal Question Answering with Retrieval-Augmented Large Language Models","date":"2023-09-29","arxiv_id":"2309.17050","repositories_listed":1,"syntology":null},{"url":"/paper/promoting-generalized-cross-lingual-question","slug":"promoting-generalized-cross-lingual-question","title":"Promoting Generalized Cross-lingual Question Answering in Few-resource Scenarios via Self-knowledge Distillation","date":"2023-09-29","arxiv_id":"2309.17134","repositories_listed":1,"syntology":null},{"url":"/paper/at-which-training-stage-does-cocde-data-help","slug":"at-which-training-stage-does-cocde-data-help","title":"At Which Training Stage Does Code Data Help LLMs Reasoning?","date":"2023-09-28","arxiv_id":"2309.16298","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/at-which-training-stage-does-cocde-data-help#ran","syntology_url":"https://syntology.ai/paper/2309.16298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16298"}},"official":{"repos":["yingweima2022/codellm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spider4sparql-a-complex-benchmark-for","slug":"spider4sparql-a-complex-benchmark-for","title":"Spider4SPARQL: A Complex Benchmark for Evaluating Knowledge Graph Question Answering Systems","date":"2023-09-28","arxiv_id":"2309.16248","repositories_listed":1,"syntology":null},{"url":"/paper/toloka-visual-question-answering-benchmark","slug":"toloka-visual-question-answering-benchmark","title":"Toloka Visual Question Answering Benchmark","date":"2023-09-28","arxiv_id":"2309.16511","repositories_listed":1,"syntology":null},{"url":"/paper/vdc-versatile-data-cleanser-for-detecting","slug":"vdc-versatile-data-cleanser-for-detecting","title":"VDC: Versatile Data Cleanser based on Visual-Linguistic Inconsistency by Multimodal Large Language Models","date":"2023-09-28","arxiv_id":"2309.16211","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":4,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/vdc-versatile-data-cleanser-for-detecting#ran","syntology_url":"https://syntology.ai/paper/2309.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16211"}},"official":{"repos":["zihao-ai/vdc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-and-few-shot-video-question","slug":"zero-shot-and-few-shot-video-question","title":"Zero-Shot and Few-Shot Video Question Answering with Multi-Modal Prompts","date":"2023-09-27","arxiv_id":"2309.15915","repositories_listed":1,"syntology":null},{"url":"/paper/question-answering-approach-to-evaluate-legal","slug":"question-answering-approach-to-evaluate-legal","title":"Question-Answering Approach to Evaluating Legal Summaries","date":"2023-09-26","arxiv_id":"2309.15016","repositories_listed":1,"syntology":null},{"url":"/paper/physics-of-language-models-part-3-1-knowledge","slug":"physics-of-language-models-part-3-1-knowledge","title":"Physics of Language Models: Part 3.1, Knowledge Storage and Extraction","date":"2023-09-25","arxiv_id":"2309.14316","repositories_listed":1,"syntology":null},{"url":"/paper/qasports-a-question-answering-dataset-about","slug":"qasports-a-question-answering-dataset-about","title":"QASports: A Question Answering Dataset about Sports","date":"2023-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bamboo-a-comprehensive-benchmark-for","slug":"bamboo-a-comprehensive-benchmark-for","title":"BAMBOO: A Comprehensive Benchmark for Evaluating Long Text Modeling Capacities of Large Language Models","date":"2023-09-23","arxiv_id":"2309.13345","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bamboo-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2309.13345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13345"}},"official":{"repos":["rucaibox/bamboo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-student-performance-in-game-based","slug":"modeling-student-performance-in-game-based","title":"Modeling Student Performance in Game-Based Learning Environments","date":"2023-09-23","arxiv_id":"2309.13429","repositories_listed":1,"syntology":null},{"url":"/paper/openai-s-gpt4-as-coding-assistant","slug":"openai-s-gpt4-as-coding-assistant","title":"OpenAi's GPT4 as coding assistant","date":"2023-09-22","arxiv_id":"2309.12732","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-sanitization-of-large-language","slug":"knowledge-sanitization-of-large-language","title":"Knowledge Sanitization of Large Language Models","date":"2023-09-21","arxiv_id":"2309.11852","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/knowledge-sanitization-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2309.11852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11852"}},"official":{"repos":["yoichi1484/knowledge-sanitization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieve-rewrite-answer-a-kg-to-text-enhanced","slug":"retrieve-rewrite-answer-a-kg-to-text-enhanced","title":"Retrieve-Rewrite-Answer: A KG-to-Text Enhanced LLMs Framework for Knowledge Graph Question Answering","date":"2023-09-20","arxiv_id":"2309.11206","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retrieve-rewrite-answer-a-kg-to-text-enhanced#ran","syntology_url":"https://syntology.ai/paper/2309.11206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11206"}},"official":{"repos":["wuyike2000/retrieve-rewrite-answer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/screws-a-modular-framework-for-reasoning-with","slug":"screws-a-modular-framework-for-reasoning-with","title":"SCREWS: A Modular Framework for Reasoning with Revisions","date":"2023-09-20","arxiv_id":"2309.13075","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-to-sequence-spanish-pre-trained","slug":"sequence-to-sequence-spanish-pre-trained","title":"Sequence-to-Sequence Spanish Pre-trained Language Models","date":"2023-09-20","arxiv_id":"2309.11259","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-open-domain-table-question","slug":"enhancing-open-domain-table-question","title":"Enhancing Open-Domain Table Question Answering via Syntax- and Structure-aware Dense Retrieval","date":"2023-09-19","arxiv_id":"2309.10506","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-embedded-programs-for-hybrid","slug":"natural-language-embedded-programs-for-hybrid","title":"Natural Language Embedded Programs for Hybrid Language Symbolic Reasoning","date":"2023-09-19","arxiv_id":"2309.10814","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/natural-language-embedded-programs-for-hybrid#ran","syntology_url":"https://syntology.ai/paper/2309.10814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10814"}},"official":{"repos":["luohongyin/langcode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/adapting-large-language-models-via-reading","slug":"adapting-large-language-models-via-reading","title":"Adapting Large Language Models to Domains via Reading Comprehension","date":"2023-09-18","arxiv_id":"2309.09530","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adapting-large-language-models-via-reading#ran","syntology_url":"https://syntology.ai/paper/2309.09530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09530"}},"official":{"repos":["microsoft/lmops"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fabricator-an-open-source-toolkit-for","slug":"fabricator-an-open-source-toolkit-for","title":"Fabricator: An Open Source Toolkit for Generating Labeled Training Data with Teacher LLMs","date":"2023-09-18","arxiv_id":"2309.09582","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fabricator-an-open-source-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2309.09582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09582"}},"official":{"repos":["flairnlp/fabricator"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/monolingual-or-multilingual-instruction","slug":"monolingual-or-multilingual-instruction","title":"Monolingual or Multilingual Instruction Tuning: Which Makes a Better Alpaca","date":"2023-09-16","arxiv_id":"2309.08958","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-the-evaluation-of-traditional","slug":"advancing-the-evaluation-of-traditional","title":"Advancing the Evaluation of Traditional Chinese Language Models: Towards a Comprehensive Benchmark Suite","date":"2023-09-15","arxiv_id":"2309.08448","repositories_listed":1,"syntology":null},{"url":"/paper/are-multilingual-llms-culturally-diverse","slug":"are-multilingual-llms-culturally-diverse","title":"Are Multilingual LLMs Culturally-Diverse Reasoners? An Investigation into Multicultural Proverbs and Sayings","date":"2023-09-15","arxiv_id":"2309.08591","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/are-multilingual-llms-culturally-diverse#ran","syntology_url":"https://syntology.ai/paper/2309.08591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.08591"}},"official":{"repos":["UKPLab/maps"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/d3-data-diversity-design-for-systematic","slug":"d3-data-diversity-design-for-systematic","title":"D3: Data Diversity Design for Systematic Generalization in Visual Question Answering","date":"2023-09-15","arxiv_id":"2309.08798","repositories_listed":1,"syntology":null},{"url":"/paper/data-distribution-bottlenecks-in-grounding","slug":"data-distribution-bottlenecks-in-grounding","title":"Data Distribution Bottlenecks in Grounding Language Models to Knowledge Bases","date":"2023-09-15","arxiv_id":"2309.08345","repositories_listed":1,"syntology":null},{"url":"/paper/structural-self-supervised-objectives-for","slug":"structural-self-supervised-objectives-for","title":"Structural Self-Supervised Objectives for Transformers","date":"2023-09-15","arxiv_id":"2309.08272","repositories_listed":1,"syntology":null},{"url":"/paper/catfood-counterfactual-augmented-training-for","slug":"catfood-counterfactual-augmented-training-for","title":"CATfOOD: Counterfactual Augmented Training for Improving Out-of-Domain Performance and Calibration","date":"2023-09-14","arxiv_id":"2309.07822","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catfood-counterfactual-augmented-training-for#ran","syntology_url":"https://syntology.ai/paper/2309.07822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07822"}},"official":{"repos":["ukplab/catfood"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textbooks-are-all-you-need-ii-phi-1-5","slug":"textbooks-are-all-you-need-ii-phi-1-5","title":"Textbooks Are All You Need II: phi-1.5 technical report","date":"2023-09-11","arxiv_id":"2309.05463","repositories_listed":1,"syntology":null},{"url":"/paper/agent-a-novel-pipeline-for-automatically","slug":"agent-a-novel-pipeline-for-automatically","title":"AGent: A Novel Pipeline for Automatically Creating Unanswerable Questions","date":"2023-09-10","arxiv_id":"2309.05103","repositories_listed":1,"syntology":null},{"url":"/paper/code-style-in-context-learning-for-knowledge","slug":"code-style-in-context-learning-for-knowledge","title":"Code-Style In-Context Learning for Knowledge-Based Question Answering","date":"2023-09-09","arxiv_id":"2309.04695","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-style-in-context-learning-for-knowledge#ran","syntology_url":"https://syntology.ai/paper/2309.04695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04695"}},"official":{"repos":["arthurizijar/kb-coder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-tuning-large-language-models-with","slug":"knowledge-tuning-large-language-models-with","title":"Knowledge-tuning Large Language Models with Structured Medical Knowledge Bases for Reliable Response Generation in Chinese","date":"2023-09-08","arxiv_id":"2309.04175","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-tuning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2309.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04175"}},"official":null}},{"url":"/paper/aligning-large-language-models-for-clinical","slug":"aligning-large-language-models-for-clinical","title":"Aligning Large Language Models for Clinical Tasks","date":"2023-09-06","arxiv_id":"2309.02884","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-large-language-models-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2309.02884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02884"}},"official":{"repos":["ssm123ssm/medGPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hc3-plus-a-semantic-invariant-human-chatgpt","slug":"hc3-plus-a-semantic-invariant-human-chatgpt","title":"HC3 Plus: A Semantic-Invariant Human ChatGPT Comparison Corpus","date":"2023-09-06","arxiv_id":"2309.02731","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-black-box-llms-with-medical","slug":"augmenting-black-box-llms-with-medical","title":"Augmenting Black-box LLMs with Medical Textbooks for Biomedical Question Answering (Published in Findings of EMNLP 2024)","date":"2023-09-05","arxiv_id":"2309.02233","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/augmenting-black-box-llms-with-medical#ran","syntology_url":"https://syntology.ai/paper/2309.02233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02233"}},"official":{"repos":["TIGER-AI-Lab/LLM-AMT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-i-trust-your-answer-visually-grounded","slug":"can-i-trust-your-answer-visually-grounded","title":"Can I Trust Your Answer? Visually Grounded Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-i-trust-your-answer-visually-grounded#ran","syntology_url":"https://syntology.ai/paper/2309.01327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01327"}},"official":{"repos":["doc-doc/next-gqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cruise-screening-living-literature-reviews","slug":"cruise-screening-living-literature-reviews","title":"CRUISE-Screening: Living Literature Reviews Toolbox","date":"2023-09-04","arxiv_id":"2309.01684","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cruise-screening-living-literature-reviews#ran","syntology_url":"https://syntology.ai/paper/2309.01684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01684"}},"official":{"repos":["projectdossier/cruise-screening"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-data-augmentation-using-llms","slug":"generative-data-augmentation-using-llms","title":"Generative Data Augmentation using LLMs improves Distributional Robustness in Question Answering","date":"2023-09-03","arxiv_id":"2309.06358","repositories_listed":1,"syntology":null},{"url":"/paper/medchatzh-a-better-medical-adviser-learns","slug":"medchatzh-a-better-medical-adviser-learns","title":"MedChatZH: a Better Medical Adviser Learns from Better Instructions","date":"2023-09-03","arxiv_id":"2309.01114","repositories_listed":1,"syntology":null},{"url":"/paper/batchprompt-accomplish-more-with-less","slug":"batchprompt-accomplish-more-with-less","title":"BatchPrompt: Accomplish more with less","date":"2023-09-01","arxiv_id":"2309.00384","repositories_listed":1,"syntology":null},{"url":"/paper/towards-addressing-the-misalignment-of-object","slug":"towards-addressing-the-misalignment-of-object","title":"Towards Addressing the Misalignment of Object Proposal Evaluation for Vision-Language Tasks via Semantic Grounding","date":"2023-09-01","arxiv_id":"2309.00215","repositories_listed":1,"syntology":null},{"url":"/paper/separate-and-locate-rethink-the-text-in-text","slug":"separate-and-locate-rethink-the-text-in-text","title":"Separate and Locate: Rethink the Text in Text-based Visual Question Answering","date":"2023-08-31","arxiv_id":"2308.16383","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-kb-text-gap-leveraging","slug":"bridging-the-kb-text-gap-leveraging","title":"Bridging the KB-Text Gap: Leveraging Structured Knowledge-aware Pre-training for KBQA","date":"2023-08-28","arxiv_id":"2308.14436","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bridging-the-kb-text-gap-leveraging#ran","syntology_url":"https://syntology.ai/paper/2308.14436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14436"}},"official":{"repos":["dongguanting/skp-for-kbqa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unipt-universal-parallel-tuning-for-transfer","slug":"unipt-universal-parallel-tuning-for-transfer","title":"UniPT: Universal Parallel Tuning for Transfer Learning with Efficient Parameter and Memory","date":"2023-08-28","arxiv_id":"2308.14316","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-cross-lingual-abilities-of","slug":"empowering-cross-lingual-abilities-of","title":"Empowering Cross-lingual Abilities of Instruction-tuned Large Language Models by Translation-following demonstrations","date":"2023-08-27","arxiv_id":"2308.14186","repositories_listed":1,"syntology":null},{"url":"/paper/towards-vision-language-mechanistic","slug":"towards-vision-language-mechanistic","title":"Towards Vision-Language Mechanistic Interpretability: A Causal Tracing Tool for BLIP","date":"2023-08-27","arxiv_id":"2308.14179","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-driven-cot-exploring-faithful","slug":"knowledge-driven-cot-exploring-faithful","title":"Knowledge-Driven CoT: Exploring Faithful Reasoning in LLMs for Knowledge-intensive Question Answering","date":"2023-08-25","arxiv_id":"2308.13259","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-driven-cot-exploring-faithful#ran","syntology_url":"https://syntology.ai/paper/2308.13259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13259"}},"official":{"repos":["adelwang/kd-cot"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/calm-a-multi-task-benchmark-for-comprehensive","slug":"calm-a-multi-task-benchmark-for-comprehensive","title":"CALM : A Multi-task Benchmark for Comprehensive Assessment of Language Model Bias","date":"2023-08-24","arxiv_id":"2308.12539","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calm-a-multi-task-benchmark-for-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2308.12539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12539"}},"official":{"repos":["vipulgupta1011/calm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flexkbqa-a-flexible-llm-powered-framework-for","slug":"flexkbqa-a-flexible-llm-powered-framework-for","title":"FlexKBQA: A Flexible LLM-Powered Framework for Few-Shot Knowledge Base Question Answering","date":"2023-08-23","arxiv_id":"2308.12060","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flexkbqa-a-flexible-llm-powered-framework-for#ran","syntology_url":"https://syntology.ai/paper/2308.12060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12060"}},"official":{"repos":["leezythu/flexkbqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-graph-prompting-for-multi-document","slug":"knowledge-graph-prompting-for-multi-document","title":"Knowledge Graph Prompting for Multi-Document Question Answering","date":"2023-08-22","arxiv_id":"2308.11730","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-graph-prompting-for-multi-document#ran","syntology_url":"https://syntology.ai/paper/2308.11730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11730"}},"official":{"repos":["yuwvandy/kg-llm-mdqa"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-on-wikipedia-style","slug":"large-language-models-on-wikipedia-style","title":"Large Language Models on Wikipedia-Style Survey Generation: an Evaluation in NLP Concepts","date":"2023-08-21","arxiv_id":"2308.10410","repositories_listed":1,"syntology":null},{"url":"/paper/ralle-a-framework-for-developing-and","slug":"ralle-a-framework-for-developing-and","title":"RaLLe: A Framework for Developing and Evaluating Retrieval-Augmented Large Language Models","date":"2023-08-21","arxiv_id":"2308.10633","repositories_listed":1,"syntology":{"n":15,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/ralle-a-framework-for-developing-and#ran","syntology_url":"https://syntology.ai/paper/2308.10633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10633"}},"official":{"repos":["yhoshi3/ralle"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-baselines-for-interactive-video","slug":"simple-baselines-for-interactive-video","title":"Simple Baselines for Interactive Video Retrieval with Questions and Answers","date":"2023-08-21","arxiv_id":"2308.10402","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/simple-baselines-for-interactive-video#ran","syntology_url":"https://syntology.ai/paper/2308.10402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10402"}},"official":{"repos":["kevinliang888/ivr-qa-baselines"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vqa-therapy-exploring-answer-differences-by","slug":"vqa-therapy-exploring-answer-differences-by","title":"VQA Therapy: Exploring Answer Differences by Visually Grounding Answers","date":"2023-08-21","arxiv_id":"2308.11662","repositories_listed":1,"syntology":null},{"url":"/paper/librisqa-pioneering-free-form-and-open-ended","slug":"librisqa-pioneering-free-form-and-open-ended","title":"LibriSQA: A Novel Dataset and Framework for Spoken Question Answering with Large Language Models","date":"2023-08-20","arxiv_id":"2308.10390","repositories_listed":1,"syntology":null},{"url":"/paper/vit-lens-towards-omni-modal-representations","slug":"vit-lens-towards-omni-modal-representations","title":"ViT-Lens: Initiating Omni-Modal Exploration through 3D Insights","date":"2023-08-20","arxiv_id":"2308.10185","repositories_listed":1,"syntology":null},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gameeval-evaluating-llms-on-conversational","slug":"gameeval-evaluating-llms-on-conversational","title":"GameEval: Evaluating LLMs on Conversational Games","date":"2023-08-19","arxiv_id":"2308.10032","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gameeval-evaluating-llms-on-conversational#ran","syntology_url":"https://syntology.ai/paper/2308.10032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10032"}},"official":{"repos":["gameeval/gameeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedgpt-open-multimodal-generative-pre","slug":"biomedgpt-open-multimodal-generative-pre","title":"BioMedGPT: Open Multimodal Generative Pre-trained Transformer for BioMedicine","date":"2023-08-18","arxiv_id":"2308.09442","repositories_listed":1,"syntology":null},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoschema-a-diagnostic-benchmark-for-very-1#ran","syntology_url":"https://syntology.ai/paper/2308.09126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09126"}},"official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mindmap-knowledge-graph-prompting-sparks","slug":"mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mindmap-knowledge-graph-prompting-sparks#ran","syntology_url":"https://syntology.ai/paper/2308.09729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09729"}},"official":{"repos":["wyl-willing/MindMap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-the-meanings-of-function-words-from","slug":"learning-the-meanings-of-function-words-from","title":"Learning the meanings of function words from grounded language using a visual question answering model","date":"2023-08-16","arxiv_id":"2308.08628","repositories_listed":1,"syntology":null},{"url":"/paper/tech-text-guided-reconstruction-of-lifelike","slug":"tech-text-guided-reconstruction-of-lifelike","title":"TeCH: Text-guided Reconstruction of Lifelike Clothed Humans","date":"2023-08-16","arxiv_id":"2308.08545","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tech-text-guided-reconstruction-of-lifelike#ran","syntology_url":"https://syntology.ai/paper/2308.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08545"}},"official":{"repos":["huangyangyi/tech"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tem-adapter-adapting-image-text-pretraining","slug":"tem-adapter-adapting-image-text-pretraining","title":"Tem-adapter: Adapting Image-Text Pretraining for Video Question Answer","date":"2023-08-16","arxiv_id":"2308.08414","repositories_listed":1,"syntology":null},{"url":"/paper/question-answering-over-linked-data-with-gpt","slug":"question-answering-over-linked-data-with-gpt","title":"Question Answering over Linked Data with GPT-3","date":"2023-08-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-information","slug":"large-language-models-for-information","title":"Large Language Models for Information Retrieval: A Survey","date":"2023-08-14","arxiv_id":"2308.07107","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-and-preventing-hallucinations-in","slug":"detecting-and-preventing-hallucinations-in","title":"Detecting and Preventing Hallucinations in Large Vision Language Models","date":"2023-08-11","arxiv_id":"2308.06394","repositories_listed":1,"syntology":null},{"url":"/paper/foundation-model-is-efficient-multimodal-1","slug":"foundation-model-is-efficient-multimodal-1","title":"Foundation Model is Efficient Multimodal Multitask Model Selector","date":"2023-08-11","arxiv_id":"2308.06262","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foundation-model-is-efficient-multimodal-1#ran","syntology_url":"https://syntology.ai/paper/2308.06262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06262"}},"official":{"repos":["opengvlab/multitask-model-selector"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/ketm-a-knowledge-enhanced-text-matching","slug":"ketm-a-knowledge-enhanced-text-matching","title":"KETM:A Knowledge-Enhanced Text Matching method","date":"2023-08-11","arxiv_id":"2308.06235","repositories_listed":1,"syntology":null},{"url":"/paper/littlemu-deploying-an-online-virtual-teaching","slug":"littlemu-deploying-an-online-virtual-teaching","title":"LittleMu: Deploying an Online Virtual Teaching Assistant via Heterogeneous Sources Integration and Chain of Teach Prompts","date":"2023-08-11","arxiv_id":"2308.05935","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-spatio-temporal-perception-for","slug":"progressive-spatio-temporal-perception-for","title":"Progressive Spatio-temporal Perception for Audio-Visual Question Answering","date":"2023-08-10","arxiv_id":"2308.05421","repositories_listed":1,"syntology":null}],"record_sha256":"2818eda7b52b0fef0c28053651a15cf5c5863f1ea99119da38b523326e035df4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}