{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/25","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":109,"rows_per_page":100,"rows":[2401,2500],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/24","next":"/task/question-answering/papers/26","papers":[{"url":"/paper/a-critical-evaluation-of-evaluations-for-long","slug":"a-critical-evaluation-of-evaluations-for-long","title":"A Critical Evaluation of Evaluations for Long-form Question Answering","date":"2023-05-29","arxiv_id":"2305.18201","repositories_listed":1,"syntology":null},{"url":"/paper/a-systematic-study-and-comprehensive","slug":"a-systematic-study-and-comprehensive","title":"A Systematic Study and Comprehensive Evaluation of ChatGPT on Benchmark Datasets","date":"2023-05-29","arxiv_id":"2305.18486","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-object-detection-with-multimodal","slug":"contextual-object-detection-with-multimodal","title":"Contextual Object Detection with Multimodal Large Language Models","date":"2023-05-29","arxiv_id":"2305.18279","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-object-detection-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2305.18279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18279"}},"official":{"repos":["yuhangzang/contextdet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-scale-attention-for-audio-question","slug":"multi-scale-attention-for-audio-question","title":"Multi-Scale Attention for Audio Question Answering","date":"2023-05-29","arxiv_id":"2305.17993","repositories_listed":1,"syntology":null},{"url":"/paper/conformal-prediction-with-large-language","slug":"conformal-prediction-with-large-language","title":"Conformal Prediction with Large Language Models for Multi-Choice Question Answering","date":"2023-05-28","arxiv_id":"2305.18404","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/conformal-prediction-with-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.18404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18404"}},"official":{"repos":["bhaweshiitk/conformalllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/havqa-a-dataset-for-visual-question-answering","slug":"havqa-a-dataset-for-visual-question-answering","title":"HaVQA: A Dataset for Visual Question Answering and Multimodal Research in Hausa Language","date":"2023-05-28","arxiv_id":"2305.17690","repositories_listed":1,"syntology":null},{"url":"/paper/plug-and-play-document-modules-for-pre","slug":"plug-and-play-document-modules-for-pre","title":"Plug-and-Play Document Modules for Pre-trained Models","date":"2023-05-28","arxiv_id":"2305.17660","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-positive-scaling-how-negation-impacts","slug":"beyond-positive-scaling-how-negation-impacts","title":"Beyond Positive Scaling: How Negation Impacts Scaling Trends of Language Models","date":"2023-05-27","arxiv_id":"2305.17311","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/modularized-zero-shot-vqa-with-pre-trained","slug":"modularized-zero-shot-vqa-with-pre-trained","title":"Modularized Zero-shot VQA with Pre-trained Models","date":"2023-05-27","arxiv_id":"2305.17369","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-comparison-of-lm-based-question","slug":"an-empirical-comparison-of-lm-based-question","title":"An Empirical Comparison of LM-based Question and Answer Generation Methods","date":"2023-05-26","arxiv_id":"2305.17002","repositories_listed":1,"syntology":null},{"url":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/biomedgpt-a-unified-and-generalist-biomedical#ran","syntology_url":"https://syntology.ai/paper/2305.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17100"}},"official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploiting-abstract-meaning-representation","slug":"exploiting-abstract-meaning-representation","title":"Exploiting Abstract Meaning Representation for Open-Domain Question Answering","date":"2023-05-26","arxiv_id":"2305.17050","repositories_listed":1,"syntology":null},{"url":"/paper/rfid-towards-rational-fusion-in-decoder-for","slug":"rfid-towards-rational-fusion-in-decoder-for","title":"RFiD: Towards Rational Fusion-in-Decoder for Open-Domain Question Answering","date":"2023-05-26","arxiv_id":"2305.17041","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-visual-question-answering-with","slug":"zero-shot-visual-question-answering-with","title":"Zero-shot Visual Question Answering with Language Model Feedback","date":"2023-05-26","arxiv_id":"2305.17006","repositories_listed":1,"syntology":null},{"url":"/paper/self-contradictory-hallucinations-of-large","slug":"self-contradictory-hallucinations-of-large","title":"Self-contradictory Hallucinations of Large Language Models: Evaluation, Detection and Mitigation","date":"2023-05-25","arxiv_id":"2305.15852","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-contradictory-hallucinations-of-large#ran","syntology_url":"https://syntology.ai/paper/2305.15852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15852"}},"official":{"repos":["eth-sri/chatprotect"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ufo-unified-fact-obtaining-for-commonsense","slug":"ufo-unified-fact-obtaining-for-commonsense","title":"UFO: Unified Fact Obtaining for Commonsense Question Answering","date":"2023-05-25","arxiv_id":"2305.16048","repositories_listed":1,"syntology":null},{"url":"/paper/beamsearchqa-large-language-models-are-strong","slug":"beamsearchqa-large-language-models-are-strong","title":"Allies: Prompting Large Language Model with Beam Search","date":"2023-05-24","arxiv_id":"2305.14766","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/beamsearchqa-large-language-models-are-strong#ran","syntology_url":"https://syntology.ai/paper/2305.14766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14766"}},"official":{"repos":["microsoft/simxns"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/car-conceptualization-augmented-reasoner-for","slug":"car-conceptualization-augmented-reasoner-for","title":"CAR: Conceptualization-Augmented Reasoner for Zero-Shot Commonsense Question Answering","date":"2023-05-24","arxiv_id":"2305.14869","repositories_listed":1,"syntology":null},{"url":"/paper/cheap-and-quick-efficient-vision-language","slug":"cheap-and-quick-efficient-vision-language","title":"Cheap and Quick: Efficient Vision-Language Instruction Tuning for Large Language Models","date":"2023-05-24","arxiv_id":"2305.15023","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-humans-and-models-on-a-similar","slug":"comparing-humans-and-models-on-a-similar","title":"Comparing Humans and Models on a Similar Scale: Towards Cognitive Gender Bias Evaluation in Coreference Resolution","date":"2023-05-24","arxiv_id":"2305.15389","repositories_listed":1,"syntology":null},{"url":"/paper/cream-visually-situated-natural-language","slug":"cream-visually-situated-natural-language","title":"Visually-Situated Natural Language Understanding with Contrastive Reading Model and Frozen Large Language Models","date":"2023-05-24","arxiv_id":"2305.15080","repositories_listed":1,"syntology":null},{"url":"/paper/csts-conditional-semantic-textual-similarity","slug":"csts-conditional-semantic-textual-similarity","title":"C-STS: Conditional Semantic Textual Similarity","date":"2023-05-24","arxiv_id":"2305.15093","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/csts-conditional-semantic-textual-similarity#ran","syntology_url":"https://syntology.ai/paper/2305.15093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15093"}},"official":{"repos":["princeton-nlp/c-sts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-retrieval-augmented-large-language","slug":"enhancing-retrieval-augmented-large-language","title":"Enhancing Retrieval-Augmented Large Language Models with Iterative Retrieval-Generation Synergy","date":"2023-05-24","arxiv_id":"2305.15294","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-faithful-and-plausible-visual","slug":"measuring-faithful-and-plausible-visual","title":"Measuring Faithful and Plausible Visual Grounding in VQA","date":"2023-05-24","arxiv_id":"2305.15015","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-online-adaptation-of-language","slug":"meta-learning-online-adaptation-of-language","title":"Meta-Learning Online Adaptation of Language Models","date":"2023-05-24","arxiv_id":"2305.15076","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-learning-online-adaptation-of-language#ran","syntology_url":"https://syntology.ai/paper/2305.15076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15076"}},"official":{"repos":["nathanhu0/CaMeLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-temporal-misalignment-by","slug":"mitigating-temporal-misalignment-by","title":"Mitigating Temporal Misalignment by Discarding Outdated Facts","date":"2023-05-24","arxiv_id":"2305.14824","repositories_listed":1,"syntology":null},{"url":"/paper/nuscenes-qa-a-multi-modal-visual-question","slug":"nuscenes-qa-a-multi-modal-visual-question","title":"NuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario","date":"2023-05-24","arxiv_id":"2305.14836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuscenes-qa-a-multi-modal-visual-question#ran","syntology_url":"https://syntology.ai/paper/2305.14836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14836"}},"official":{"repos":["qiantianwen/nuscenes-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openpi2-0-an-improved-dataset-for-entity","slug":"openpi2-0-an-improved-dataset-for-entity","title":"OpenPI2.0: An Improved Dataset for Entity Tracking in Texts","date":"2023-05-24","arxiv_id":"2305.14603","repositories_listed":1,"syntology":null},{"url":"/paper/peek-across-improving-multi-document-modeling","slug":"peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","arxiv_id":"2305.15387","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/peek-across-improving-multi-document-modeling#ran","syntology_url":"https://syntology.ai/paper/2305.15387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15387"}},"official":{"repos":["aviclu/peekacross"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-sentence-union-generation-as-a","slug":"revisiting-sentence-union-generation-as-a","title":"Revisiting Sentence Union Generation as a Testbed for Text Consolidation","date":"2023-05-24","arxiv_id":"2305.15605","repositories_listed":1,"syntology":null},{"url":"/paper/the-art-of-socratic-questioning-zero-shot","slug":"the-art-of-socratic-questioning-zero-shot","title":"The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14999","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-art-of-socratic-questioning-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2305.14999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14999"}},"official":{"repos":["vt-nlp/socratic-questioning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-role-of-output-vocabulary-in-t2t-lms-for","slug":"the-role-of-output-vocabulary-in-t2t-lms-for","title":"The Role of Output Vocabulary in T2T LMs for SPARQL Semantic Parsing","date":"2023-05-24","arxiv_id":"2305.15108","repositories_listed":1,"syntology":null},{"url":"/paper/tomchallenges-a-principle-guided-dataset-and","slug":"tomchallenges-a-principle-guided-dataset-and","title":"ToMChallenges: A Principle-Guided Dataset and Diverse Evaluation Tasks for Exploring Theory of Mind","date":"2023-05-24","arxiv_id":"2305.15068","repositories_listed":1,"syntology":null},{"url":"/paper/unichart-a-universal-vision-language","slug":"unichart-a-universal-vision-language","title":"UniChart: A Universal Vision-language Pretrained Model for Chart Comprehension and Reasoning","date":"2023-05-24","arxiv_id":"2305.14761","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/unichart-a-universal-vision-language#ran","syntology_url":"https://syntology.ai/paper/2305.14761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14761"}},"official":{"repos":["vis-nlp/unichart"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unlocking-temporal-question-answering-for","slug":"unlocking-temporal-question-answering-for","title":"Unlocking Temporal Question Answering for Large Language Models with Tailor-Made Reasoning Logic","date":"2023-05-24","arxiv_id":"2305.15014","repositories_listed":1,"syntology":null},{"url":"/paper/using-natural-language-explanations-to","slug":"using-natural-language-explanations-to","title":"Using Natural Language Explanations to Rescale Human Judgments","date":"2023-05-24","arxiv_id":"2305.14770","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/using-natural-language-explanations-to#ran","syntology_url":"https://syntology.ai/paper/2305.14770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14770"}},"official":{"repos":["manyawadhwa/explanation_based_rescaling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/asking-clarification-questions-to-handle","slug":"asking-clarification-questions-to-handle","title":"Asking Clarification Questions to Handle Ambiguity in Open-Domain QA","date":"2023-05-23","arxiv_id":"2305.13808","repositories_listed":1,"syntology":null},{"url":"/paper/band-biomedical-alert-news-dataset","slug":"band-biomedical-alert-news-dataset","title":"BAND: Biomedical Alert News Dataset","date":"2023-05-23","arxiv_id":"2305.14480","repositories_listed":1,"syntology":null},{"url":"/paper/complementing-gpt-3-with-few-shot-sequence-to","slug":"complementing-gpt-3-with-few-shot-sequence-to","title":"Fine-tuned LLMs Know More, Hallucinate Less with Few-Shot Sequence-to-Sequence Semantic Parsing over Wikidata","date":"2023-05-23","arxiv_id":"2305.14202","repositories_listed":1,"syntology":null},{"url":"/paper/continual-dialogue-state-tracking-via-example","slug":"continual-dialogue-state-tracking-via-example","title":"Continual Dialogue State Tracking via Example-Guided Question Answering","date":"2023-05-23","arxiv_id":"2305.13721","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-contrast-consistency-of-open-domain","slug":"exploring-contrast-consistency-of-open-domain","title":"Exploring Contrast Consistency of Open-Domain Question Answering Systems on Minimally Edited Questions","date":"2023-05-23","arxiv_id":"2305.14441","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-of-knowledge-exploring-known","slug":"knowledge-of-knowledge-exploring-known","title":"Knowledge of Knowledge: Exploring Known-Unknowns Uncertainty with Large Language Models","date":"2023-05-23","arxiv_id":"2305.13712","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowledge-of-knowledge-exploring-known#ran","syntology_url":"https://syntology.ai/paper/2305.13712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13712"}},"official":{"repos":["amayuelas/knowledge-of-knowledge"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/memecap-a-dataset-for-captioning-and","slug":"memecap-a-dataset-for-captioning-and","title":"MemeCap: A Dataset for Captioning and Interpreting Memes","date":"2023-05-23","arxiv_id":"2305.13703","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-risk-of-misinformation-pollution-with","slug":"on-the-risk-of-misinformation-pollution-with","title":"On the Risk of Misinformation Pollution with Large Language Models","date":"2023-05-23","arxiv_id":"2305.13661","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-risk-of-misinformation-pollution-with#ran","syntology_url":"https://syntology.ai/paper/2305.13661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13661"}},"official":{"repos":["MexicanLemonade/LLM-Misinfo-QA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/question-answering-as-programming-for-solving","slug":"question-answering-as-programming-for-solving","title":"Question Answering as Programming for Solving Time-Sensitive Questions","date":"2023-05-23","arxiv_id":"2305.14221","repositories_listed":1,"syntology":null},{"url":"/paper/ret-llm-towards-a-general-read-write-memory","slug":"ret-llm-towards-a-general-read-write-memory","title":"RET-LLM: Towards a General Read-Write Memory for Large Language Models","date":"2023-05-23","arxiv_id":"2305.14322","repositories_listed":1,"syntology":null},{"url":"/paper/sources-of-hallucination-by-large-language","slug":"sources-of-hallucination-by-large-language","title":"Sources of Hallucination by Large Language Models on Inference Tasks","date":"2023-05-23","arxiv_id":"2305.14552","repositories_listed":1,"syntology":null},{"url":"/paper/beneath-surface-similarity-large-language","slug":"beneath-surface-similarity-large-language","title":"Beneath Surface Similarity: Large Language Models Make Reasonable Scientific Analogies after Structure Abduction","date":"2023-05-22","arxiv_id":"2305.12660","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-prompt-based-question-answering","slug":"evaluating-prompt-based-question-answering","title":"Evaluating Prompt-based Question Answering for Object Prediction in the Open Research Knowledge Graph","date":"2023-05-22","arxiv_id":"2305.12900","repositories_listed":1,"syntology":null},{"url":"/paper/how-language-model-hallucinations-can","slug":"how-language-model-hallucinations-can","title":"How Language Model Hallucinations Can Snowball","date":"2023-05-22","arxiv_id":"2305.13534","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-retrieval-task-oriented-dialog","slug":"knowledge-retrieval-task-oriented-dialog","title":"Knowledge-Retrieval Task-Oriented Dialog Systems with Semi-Supervision","date":"2023-05-22","arxiv_id":"2305.13199","repositories_listed":1,"syntology":null},{"url":"/paper/llms-for-knowledge-graph-construction-and","slug":"llms-for-knowledge-graph-construction-and","title":"LLMs for Knowledge Graph Construction and Reasoning: Recent Capabilities and Future Opportunities","date":"2023-05-22","arxiv_id":"2305.13168","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/llms-for-knowledge-graph-construction-and#ran","syntology_url":"https://syntology.ai/paper/2305.13168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13168"}},"official":{"repos":["zjunlp/autokg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/multitabqa-generating-tabular-answers-for","slug":"multitabqa-generating-tabular-answers-for","title":"MultiTabQA: Generating Tabular Answers for Multi-Table Question Answering","date":"2023-05-22","arxiv_id":"2305.12820","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multitabqa-generating-tabular-answers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12820"}},"official":{"repos":["kolk/multitabqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/refind-relation-extraction-financial-dataset","slug":"refind-relation-extraction-financial-dataset","title":"REFinD: Relation Extraction Financial Dataset","date":"2023-05-22","arxiv_id":"2305.18322","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-probabilistic-logical-reasoning-to","slug":"teaching-probabilistic-logical-reasoning-to","title":"Teaching Probabilistic Logical Reasoning to Transformers","date":"2023-05-22","arxiv_id":"2305.13179","repositories_listed":1,"syntology":null},{"url":"/paper/continually-improving-extractive-qa-via-human","slug":"continually-improving-extractive-qa-via-human","title":"Continually Improving Extractive QA via Human Feedback","date":"2023-05-21","arxiv_id":"2305.12473","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-open-qa-evaluation-1","slug":"evaluating-open-qa-evaluation-1","title":"Evaluating Open-QA Evaluation","date":"2023-05-21","arxiv_id":"2305.12421","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/evaluating-open-qa-evaluation-1#ran","syntology_url":"https://syntology.ai/paper/2305.12421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12421"}},"official":{"repos":["wangcunxiang/qa-eval"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/model-analysis-evaluation-for-ambiguous","slug":"model-analysis-evaluation-for-ambiguous","title":"Model Analysis & Evaluation for Ambiguous Question Answering","date":"2023-05-21","arxiv_id":"2305.12483","repositories_listed":1,"syntology":null},{"url":"/paper/pruning-pre-trained-language-models-with","slug":"pruning-pre-trained-language-models-with","title":"Pruning Pre-trained Language Models with Principled Importance and Self-regularization","date":"2023-05-21","arxiv_id":"2305.12394","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pruning-pre-trained-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2305.12394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12394"}},"official":{"repos":["drsy/pins"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/target-aware-spatio-temporal-reasoning-via","slug":"target-aware-spatio-temporal-reasoning-via","title":"Target-Aware Spatio-Temporal Reasoning via Answering Questions in Dynamics Audio-Visual Scenarios","date":"2023-05-21","arxiv_id":"2305.12397","repositories_listed":1,"syntology":null},{"url":"/paper/theoremqa-a-theorem-driven-question-answering","slug":"theoremqa-a-theorem-driven-question-answering","title":"TheoremQA: A Theorem-driven Question Answering dataset","date":"2023-05-21","arxiv_id":"2305.12524","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/theoremqa-a-theorem-driven-question-answering#ran","syntology_url":"https://syntology.ai/paper/2305.12524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12524"}},"official":{"repos":["wenhuchen/theoremqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/vnhsge-vietnamese-high-school-graduation","slug":"vnhsge-vietnamese-high-school-graduation","title":"VNHSGE: VietNamese High School Graduation Examination Dataset for Large Language Models","date":"2023-05-20","arxiv_id":"2305.12199","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-makes-for-good-visual-tokenizers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12223"}},"official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critic-large-language-models-can-self-correct","slug":"critic-large-language-models-can-self-correct","title":"CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing","date":"2023-05-19","arxiv_id":"2305.11738","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/critic-large-language-models-can-self-correct#ran","syntology_url":"https://syntology.ai/paper/2305.11738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11738"}},"official":{"repos":["microsoft/ProphetNet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/empower-large-language-model-to-perform","slug":"empower-large-language-model-to-perform","title":"Empower Large Language Model to Perform Better on Industrial Domain-Specific Question Answering","date":"2023-05-19","arxiv_id":"2305.11541","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-vision-language-pre-training-with","slug":"enhancing-vision-language-pre-training-with","title":"Enhancing Vision-Language Pre-Training with Jointly Learned Questioner and Dense Captioner","date":"2023-05-19","arxiv_id":"2305.11769","repositories_listed":1,"syntology":null},{"url":"/paper/llm-itself-can-read-and-generate-cxr-images","slug":"llm-itself-can-read-and-generate-cxr-images","title":"LLM-CXR: Instruction-Finetuned LLM for CXR Image Understanding and Generation","date":"2023-05-19","arxiv_id":"2305.11490","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/llm-itself-can-read-and-generate-cxr-images#ran","syntology_url":"https://syntology.ai/paper/2305.11490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11490"}},"official":{"repos":["hyn2028/llm-cxr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pengi-an-audio-language-model-for-audio-tasks-1","slug":"pengi-an-audio-language-model-for-audio-tasks-1","title":"Pengi: An Audio Language Model for Audio Tasks","date":"2023-05-19","arxiv_id":"2305.11834","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pengi-an-audio-language-model-for-audio-tasks-1#ran","syntology_url":"https://syntology.ai/paper/2305.11834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11834"}},"official":{"repos":["microsoft/pengi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/s-3-hqa-a-three-stage-approach-for-multi-hop","slug":"s-3-hqa-a-three-stage-approach-for-multi-hop","title":"S$^3$HQA: A Three-Stage Approach for Multi-hop Text-Table Hybrid Question Answering","date":"2023-05-19","arxiv_id":"2305.11725","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/s-3-hqa-a-three-stage-approach-for-multi-hop#ran","syntology_url":"https://syntology.ai/paper/2305.11725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11725"}},"official":{"repos":["lfy79001/s3hqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/self-qa-unsupervised-knowledge-guided","slug":"self-qa-unsupervised-knowledge-guided","title":"Self-QA: Unsupervised Knowledge Guided Language Model Alignment","date":"2023-05-19","arxiv_id":"2305.11952","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-instruction-tasks-unlocks-large","slug":"aligning-instruction-tasks-unlocks-large","title":"Aligning Instruction Tasks Unlocks Large Language Models as Zero-Shot Relation Extractors","date":"2023-05-18","arxiv_id":"2305.11159","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-can-be-guided-to-evade","slug":"large-language-models-can-be-guided-to-evade","title":"Large Language Models can be Guided to Evade AI-Generated Text Detection","date":"2023-05-18","arxiv_id":"2305.10847","repositories_listed":1,"syntology":null},{"url":"/paper/medblip-bootstrapping-language-image-pre","slug":"medblip-bootstrapping-language-image-pre","title":"MedBLIP: Bootstrapping Language-Image Pre-training from 3D Medical Images and Texts","date":"2023-05-18","arxiv_id":"2305.10799","repositories_listed":1,"syntology":null},{"url":"/paper/mlongt5-a-multilingual-and-efficient-text-to","slug":"mlongt5-a-multilingual-and-efficient-text-to","title":"mLongT5: A Multilingual and Efficient Text-To-Text Transformer for Longer Sequences","date":"2023-05-18","arxiv_id":"2305.11129","repositories_listed":1,"syntology":null},{"url":"/paper/imad-image-augmented-multi-modal-dialogue","slug":"imad-image-augmented-multi-modal-dialogue","title":"IMAD: IMage-Augmented multi-modal Dialogue","date":"2023-05-17","arxiv_id":"2305.10512","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imad-image-augmented-multi-modal-dialogue#ran","syntology_url":"https://syntology.ai/paper/2305.10512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10512"}},"official":{"repos":["vityavitalich/imad"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/palm-2-technical-report-1","slug":"palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","arxiv_id":"2305.10403","repositories_listed":1,"syntology":null},{"url":"/paper/what-you-see-is-what-you-read-improving-text-1","slug":"what-you-see-is-what-you-read-improving-text-1","title":"What You See is What You Read? Improving Text-Image Alignment Evaluation","date":"2023-05-17","arxiv_id":"2305.10400","repositories_listed":1,"syntology":null},{"url":"/paper/a-video-is-worth-4096-tokens-verbalize-story","slug":"a-video-is-worth-4096-tokens-verbalize-story","title":"A Video Is Worth 4096 Tokens: Verbalize Videos To Understand Them In Zero Shot","date":"2023-05-16","arxiv_id":"2305.09758","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-built-in","slug":"large-language-models-are-built-in","title":"Large Language Models are Built-in Autoregressive Search Engines","date":"2023-05-16","arxiv_id":"2305.09612","repositories_listed":1,"syntology":null},{"url":"/paper/structgpt-a-general-framework-for-large","slug":"structgpt-a-general-framework-for-large","title":"StructGPT: A General Framework for Large Language Model to Reason over Structured Data","date":"2023-05-16","arxiv_id":"2305.09645","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":2,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":3,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 3 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/structgpt-a-general-framework-for-large#ran","syntology_url":"https://syntology.ai/paper/2305.09645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09645"}},"official":{"repos":["rucaibox/structgpt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-interpreter-understands-your-meaning-end","slug":"the-interpreter-understands-your-meaning-end","title":"The Interpreter Understands Your Meaning: End-to-end Spoken Language Understanding Aided by Speech Translation","date":"2023-05-16","arxiv_id":"2305.09652","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-interpreter-understands-your-meaning-end#ran","syntology_url":"https://syntology.ai/paper/2305.09652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09652"}},"official":{"repos":["idiap/translation-aided-slu"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-expert-level-medical-question","slug":"towards-expert-level-medical-question","title":"Towards Expert-Level Medical Question Answering with Large Language Models","date":"2023-05-16","arxiv_id":"2305.09617","repositories_listed":1,"syntology":null},{"url":"/paper/xpqa-cross-lingual-product-question-answering","slug":"xpqa-cross-lingual-product-question-answering","title":"xPQA: Cross-Lingual Product Question Answering across 12 Languages","date":"2023-05-16","arxiv_id":"2305.09249","repositories_listed":1,"syntology":null},{"url":"/paper/capturing-humans-mental-models-of-ai-an-item","slug":"capturing-humans-mental-models-of-ai-an-item","title":"Capturing Humans' Mental Models of AI: An Item Response Theory Approach","date":"2023-05-15","arxiv_id":"2305.09064","repositories_listed":1,"syntology":null},{"url":"/paper/kepr-knowledge-enhancement-and-plausibility","slug":"kepr-knowledge-enhancement-and-plausibility","title":"KEPR: Knowledge Enhancement and Plausibility Ranking for Generative Commonsense Question Answering","date":"2023-05-15","arxiv_id":"2305.08347","repositories_listed":1,"syntology":null},{"url":"/paper/meeqa-natural-questions-in-meeting","slug":"meeqa-natural-questions-in-meeting","title":"MeeQA: Natural Questions in Meeting Transcripts","date":"2023-05-15","arxiv_id":"2305.08502","repositories_listed":1,"syntology":null},{"url":"/paper/motion-question-answering-via-modular-motion","slug":"motion-question-answering-via-modular-motion","title":"Motion Question Answering via Modular Motion Programs","date":"2023-05-15","arxiv_id":"2305.08953","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/motion-question-answering-via-modular-motion#ran","syntology_url":"https://syntology.ai/paper/2305.08953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.08953"}},"official":{"repos":["markendo/humanmotionqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/parameter-efficient-fine-tuning-with-layer","slug":"parameter-efficient-fine-tuning-with-layer","title":"Parameter-Efficient Fine-Tuning with Layer Pruning on Free-Text Sequence-to-Sequence Modeling","date":"2023-05-15","arxiv_id":"2305.08285","repositories_listed":1,"syntology":null},{"url":"/paper/question-answering-system-extracts","slug":"question-answering-system-extracts","title":"Question-Answering System Extracts Information on Injection Drug Use from Clinical Notes","date":"2023-05-15","arxiv_id":"2305.08777","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generalize-for-cross-domain-qa","slug":"learning-to-generalize-for-cross-domain-qa","title":"Learning to Generalize for Cross-domain QA","date":"2023-05-14","arxiv_id":"2305.08208","repositories_listed":1,"syntology":null},{"url":"/paper/matsci-nlp-evaluating-scientific-language","slug":"matsci-nlp-evaluating-scientific-language","title":"MatSci-NLP: Evaluating Scientific Language Models on Materials Science Language Tasks Using Text-to-Schema Modeling","date":"2023-05-14","arxiv_id":"2305.08264","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/matsci-nlp-evaluating-scientific-language#ran","syntology_url":"https://syntology.ai/paper/2305.08264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.08264"}},"official":{"repos":["banglab-udem-mila/nlp4matsci-acl23"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-hidden-mystery-of-ocr-in-large","slug":"on-the-hidden-mystery-of-ocr-in-large","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","date":"2023-05-13","arxiv_id":"2305.07895","repositories_listed":1,"syntology":null},{"url":"/paper/scene-self-labeled-counterfactuals-for","slug":"scene-self-labeled-counterfactuals-for","title":"SCENE: Self-Labeled Counterfactuals for Extrapolating to Negative Examples","date":"2023-05-13","arxiv_id":"2305.07984","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scene-self-labeled-counterfactuals-for#ran","syntology_url":"https://syntology.ai/paper/2305.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07984"}},"official":{"repos":["deqingfu/scene"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/open-wikitable-dataset-for-open-domain","slug":"open-wikitable-dataset-for-open-domain","title":"Open-WikiTable: Dataset for Open Domain Question Answering with Complex Reasoning over Table","date":"2023-05-12","arxiv_id":"2305.07288","repositories_listed":1,"syntology":null},{"url":"/paper/afriqa-cross-lingual-open-retrieval-question","slug":"afriqa-cross-lingual-open-retrieval-question","title":"AfriQA: Cross-lingual Open-Retrieval Question Answering for African Languages","date":"2023-05-11","arxiv_id":"2305.06897","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-open-domain-question-answering-in","slug":"evaluating-open-domain-question-answering-in","title":"Evaluating Open-Domain Question Answering in the Era of Large Language Models","date":"2023-05-11","arxiv_id":"2305.06984","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-open-domain-question-answering-in#ran","syntology_url":"https://syntology.ai/paper/2305.06984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06984"}},"official":{"repos":["ehsk/openqa-eval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/long-tailed-question-answering-in-an-open","slug":"long-tailed-question-answering-in-an-open","title":"Long-Tailed Question Answering in an Open World","date":"2023-05-11","arxiv_id":"2305.06557","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-languages-are-created-equal-in-llms","slug":"not-all-languages-are-created-equal-in-llms","title":"Not All Languages Are Created Equal in LLMs: Improving Multilingual Capability by Cross-Lingual-Thought Prompting","date":"2023-05-11","arxiv_id":"2305.07004","repositories_listed":1,"syntology":null},{"url":"/paper/self-chained-image-language-model-for-video-1","slug":"self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","arxiv_id":"2305.06988","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-chained-image-language-model-for-video-1#ran","syntology_url":"https://syntology.ai/paper/2305.06988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06988"}},"official":{"repos":["yui010206/sevila"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"1c8f52e18c821701e0e377f7c46bece8015da654b0f637a7d6559589abff1cd5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}