{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/22","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":22,"pages_in_order":109,"rows_per_page":100,"rows":[2101,2200],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/21","next":"/task/question-answering/papers/23","papers":[{"url":"/paper/xplainllm-a-qa-explanation-dataset-for","slug":"xplainllm-a-qa-explanation-dataset-for","title":"XplainLLM: A Knowledge-Augmented Dataset for Reliable Grounded Explanations in LLMs","date":"2023-11-15","arxiv_id":"2311.08614","repositories_listed":1,"syntology":null},{"url":"/paper/carpe-diem-on-the-evaluation-of-world","slug":"carpe-diem-on-the-evaluation-of-world","title":"Carpe Diem: On the Evaluation of World Knowledge in Lifelong Language Models","date":"2023-11-14","arxiv_id":"2311.08106","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/carpe-diem-on-the-evaluation-of-world#ran","syntology_url":"https://syntology.ai/paper/2311.08106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08106"}},"official":{"repos":["kimyuji/evolvingqa_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-well-do-large-language-models-understand","slug":"how-well-do-large-language-models-understand","title":"How Well Do Large Language Models Understand Syntax? An Evaluation by Asking Natural Language Questions","date":"2023-11-14","arxiv_id":"2311.08287","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-gpt-4v-on","slug":"a-comprehensive-evaluation-of-gpt-4v-on","title":"A Comprehensive Evaluation of GPT-4V on Knowledge-Intensive Visual Question Answering","date":"2023-11-13","arxiv_id":"2311.07536","repositories_listed":1,"syntology":null},{"url":"/paper/a-step-closer-to-comprehensive-answers","slug":"a-step-closer-to-comprehensive-answers","title":"A Step Closer to Comprehensive Answers: Constrained Multi-Stage Question Decomposition with Large Language Models","date":"2023-11-13","arxiv_id":"2311.07491","repositories_listed":1,"syntology":null},{"url":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-robustness-of-question-rewriting-1","slug":"on-the-robustness-of-question-rewriting-1","title":"On the Robustness of Question Rewriting Systems to Questions of Varying Hardness","date":"2023-11-12","arxiv_id":"2311.06807","repositories_listed":1,"syntology":null},{"url":"/paper/knowledgeable-preference-alignment-for-llms","slug":"knowledgeable-preference-alignment-for-llms","title":"Knowledgeable Preference Alignment for LLMs in Domain-specific Question Answering","date":"2023-11-11","arxiv_id":"2311.06503","repositories_listed":1,"syntology":null},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chimed-gpt-a-chinese-medical-large-language","slug":"chimed-gpt-a-chinese-medical-large-language","title":"ChiMed-GPT: A Chinese Medical Large Language Model with Full Training Regime and Better Alignment to Human Preferences","date":"2023-11-10","arxiv_id":"2311.06025","repositories_listed":1,"syntology":null},{"url":"/paper/tencentllmeval-a-hierarchical-evaluation-of","slug":"tencentllmeval-a-hierarchical-evaluation-of","title":"TencentLLMEval: A Hierarchical Evaluation of Real-World Capabilities for Human-Aligned LLMs","date":"2023-11-09","arxiv_id":"2311.05374","repositories_listed":1,"syntology":null},{"url":"/paper/genome-generative-neuro-symbolic-visual","slug":"genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","arxiv_id":"2311.04901","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/genome-generative-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2311.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04901"}},"official":null}},{"url":"/paper/loogle-can-long-context-language-models","slug":"loogle-can-long-context-language-models","title":"LooGLE: Can Long-Context Language Models Understand Long Contexts?","date":"2023-11-08","arxiv_id":"2311.04939","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/loogle-can-long-context-language-models#ran","syntology_url":"https://syntology.ai/paper/2311.04939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04939"}},"official":{"repos":["bigai-nlco/loogle"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/massive-editing-for-large-language-models-via","slug":"massive-editing-for-large-language-models-via","title":"Massive Editing for Large Language Models via Meta Learning","date":"2023-11-08","arxiv_id":"2311.04661","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/massive-editing-for-large-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2311.04661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04661"}},"official":{"repos":["chenmientan/malmen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nlqxform-a-language-model-based-question-to","slug":"nlqxform-a-language-model-based-question-to","title":"NLQxform: A Language Model-based Question to SPARQL Transformer","date":"2023-11-08","arxiv_id":"2311.07588","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-translation-of-attention-patterns","slug":"zero-shot-translation-of-attention-patterns","title":"Zero-shot Translation of Attention Patterns in VQA Models to Natural Language","date":"2023-11-08","arxiv_id":"2311.05043","repositories_listed":1,"syntology":null},{"url":"/paper/jpave-a-generation-and-classification-based","slug":"jpave-a-generation-and-classification-based","title":"JPAVE: A Generation and Classification-based Model for Joint Product Attribute Prediction and Value Extraction","date":"2023-11-07","arxiv_id":"2311.04196","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-structured-information-for","slug":"leveraging-structured-information-for","title":"Leveraging Structured Information for Explainable Multi-hop Question Answering and Reasoning","date":"2023-11-07","arxiv_id":"2311.03734","repositories_listed":1,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":7,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":13,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/leveraging-structured-information-for#ran","syntology_url":"https://syntology.ai/paper/2311.03734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03734"}},"official":{"repos":["bcdnlp/structure-qa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-cache-modular-attention-reuse-for-low","slug":"prompt-cache-modular-attention-reuse-for-low","title":"Prompt Cache: Modular Attention Reuse for Low-Latency Inference","date":"2023-11-07","arxiv_id":"2311.04934","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompt-cache-modular-attention-reuse-for-low#ran","syntology_url":"https://syntology.ai/paper/2311.04934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04934"}},"official":{"repos":["yale-sys/prompt-cache"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-geospatial-question-answering","slug":"benchmarking-geospatial-question-answering","title":"Benchmarking Geospatial Question Answering Engines using the Dataset GeoQuestions1089","date":"2023-11-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tailoring-self-rationalizers-with-multi","slug":"tailoring-self-rationalizers-with-multi","title":"Tailoring Self-Rationalizers with Multi-Reward Distillation","date":"2023-11-06","arxiv_id":"2311.02805","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tailoring-self-rationalizers-with-multi#ran","syntology_url":"https://syntology.ai/paper/2311.02805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02805"}},"official":{"repos":["ink-usc/rationalemultirewarddistillation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-question-answering-with-reinforcement","slug":"causal-question-answering-with-reinforcement","title":"Causal Question Answering with Reinforcement Learning","date":"2023-11-05","arxiv_id":"2311.02760","repositories_listed":1,"syntology":null},{"url":"/paper/chata-towards-an-intelligent-question-answer","slug":"chata-towards-an-intelligent-question-answer","title":"AI-TA: Towards an Intelligent Question-Answer Teaching Assistant using Open-Source LLMs","date":"2023-11-05","arxiv_id":"2311.02775","repositories_listed":1,"syntology":null},{"url":"/paper/chef-a-comprehensive-evaluation-framework-for","slug":"chef-a-comprehensive-evaluation-framework-for","title":"ChEF: A Comprehensive Evaluation Framework for Standardized Assessment of Multimodal Large Language Models","date":"2023-11-05","arxiv_id":"2311.02692","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-grounding-potential-of-vqa-oriented","slug":"exploring-grounding-potential-of-vqa-oriented","title":"GPT-4V-AD: Exploring Grounding Potential of VQA-oriented GPT-4V for Zero-shot Anomaly Detection","date":"2023-11-05","arxiv_id":"2311.02612","repositories_listed":1,"syntology":null},{"url":"/paper/sac-3-reliable-hallucination-detection-in","slug":"sac-3-reliable-hallucination-detection-in","title":"SAC3: Reliable Hallucination Detection in Black-Box Language Models via Semantic-aware Cross-check Consistency","date":"2023-11-03","arxiv_id":"2311.01740","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/sac-3-reliable-hallucination-detection-in#ran","syntology_url":"https://syntology.ai/paper/2311.01740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01740"}},"official":{"repos":["intuit/sac3"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/acquired-a-dataset-for-answering","slug":"acquired-a-dataset-for-answering","title":"ACQUIRED: A Dataset for Answering Counterfactual Questions In Real-Life Videos","date":"2023-11-02","arxiv_id":"2311.01620","repositories_listed":1,"syntology":null},{"url":"/paper/effective-human-ai-teams-via-learned-natural-1","slug":"effective-human-ai-teams-via-learned-natural-1","title":"Effective Human-AI Teams via Learned Natural Language Rules and Onboarding","date":"2023-11-02","arxiv_id":"2311.01007","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/effective-human-ai-teams-via-learned-natural-1#ran","syntology_url":"https://syntology.ai/paper/2311.01007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01007"}},"official":{"repos":["clinicalml/onboarding_human_ai"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/long-story-short-a-summarize-then-search","slug":"long-story-short-a-summarize-then-search","title":"Long Story Short: a Summarize-then-Search Method for Long Video Question Answering","date":"2023-11-02","arxiv_id":"2311.01233","repositories_listed":1,"syntology":null},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-lightweight-method-to-generate-unanswerable","slug":"a-lightweight-method-to-generate-unanswerable","title":"A Lightweight Method to Generate Unanswerable Questions in English","date":"2023-10-30","arxiv_id":"2310.19403","repositories_listed":1,"syntology":null},{"url":"/paper/generating-context-aware-natural-answers-for","slug":"generating-context-aware-natural-answers-for","title":"Generating Context-Aware Natural Answers for Questions in 3D Scenes","date":"2023-10-30","arxiv_id":"2310.19516","repositories_listed":1,"syntology":null},{"url":"/paper/harvest-video-foundation-models-via-efficient","slug":"harvest-video-foundation-models-via-efficient","title":"Harvest Video Foundation Models via Efficient Post-Pretraining","date":"2023-10-30","arxiv_id":"2310.19554","repositories_listed":1,"syntology":null},{"url":"/paper/split-ner-named-entity-recognition-via-two","slug":"split-ner-named-entity-recognition-via-two","title":"Split-NER: Named Entity Recognition via Two Question-Answering-based Classifications","date":"2023-10-30","arxiv_id":"2310.19942","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/split-ner-named-entity-recognition-via-two#ran","syntology_url":"https://syntology.ai/paper/2310.19942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19942"}},"official":{"repos":["c3sr/split-ner"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/dcqa-document-level-chart-question-answering","slug":"dcqa-document-level-chart-question-answering","title":"DCQA: Document-Level Chart Question Answering towards Complex Reasoning and Common-Sense Understanding","date":"2023-10-29","arxiv_id":"2310.18983","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-task-and-weight-prioritization","slug":"dynamic-task-and-weight-prioritization","title":"Dynamic Task and Weight Prioritization Curriculum Learning for Multimodal Imagery","date":"2023-10-29","arxiv_id":"2310.19109","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-follow-object-centric-image","slug":"learning-to-follow-object-centric-image","title":"Learning to Follow Object-Centric Image Editing Instructions Faithfully","date":"2023-10-29","arxiv_id":"2310.19145","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-chatgpt-for-medical-applications","slug":"multimodal-chatgpt-for-medical-applications","title":"Multimodal ChatGPT for Medical Applications: an Experimental Study of GPT-4V","date":"2023-10-29","arxiv_id":"2310.19061","repositories_listed":1,"syntology":null},{"url":"/paper/detrimental-contexts-in-open-domain-question","slug":"detrimental-contexts-in-open-domain-question","title":"Detrimental Contexts in Open-Domain Question Answering","date":"2023-10-27","arxiv_id":"2310.18077","repositories_listed":1,"syntology":null},{"url":"/paper/from-values-to-opinions-predicting-human","slug":"from-values-to-opinions-predicting-human","title":"From Values to Opinions: Predicting Human Behaviors and Stances Using Value-Injected Large Language Models","date":"2023-10-27","arxiv_id":"2310.17857","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/from-values-to-opinions-predicting-human#ran","syntology_url":"https://syntology.ai/paper/2310.17857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17857"}},"official":{"repos":["dongjunkang/vim"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/knowledge-corpus-error-in-question-answering","slug":"knowledge-corpus-error-in-question-answering","title":"Knowledge Corpus Error in Question Answering","date":"2023-10-27","arxiv_id":"2310.18076","repositories_listed":1,"syntology":null},{"url":"/paper/viclevr-a-visual-reasoning-dataset-and-hybrid","slug":"viclevr-a-visual-reasoning-dataset-and-hybrid","title":"ViCLEVR: A Visual Reasoning Dataset and Hybrid Multimodal Fusion Model for Visual Question Answering in Vietnamese","date":"2023-10-27","arxiv_id":"2310.18046","repositories_listed":1,"syntology":null},{"url":"/paper/antifakeprompt-prompt-tuned-vision-language","slug":"antifakeprompt-prompt-tuned-vision-language","title":"AntifakePrompt: Prompt-Tuned Vision-Language Models are Fake Image Detectors","date":"2023-10-26","arxiv_id":"2310.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/antifakeprompt-prompt-tuned-vision-language#ran","syntology_url":"https://syntology.ai/paper/2310.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17419"}},"official":{"repos":["nctu-eva-lab/antifakeprompt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/incorporating-probing-signals-into-multimodal","slug":"incorporating-probing-signals-into-multimodal","title":"Incorporating Probing Signals into Multimodal Machine Translation via Visual Question-Answering Pairs","date":"2023-10-26","arxiv_id":"2310.17133","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-enhanced-narrative-question","slug":"diversity-enhanced-narrative-question","title":"Diversity Enhanced Narrative Question Generation for Storybooks","date":"2023-10-25","arxiv_id":"2310.16446","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diversity-enhanced-narrative-question#ran","syntology_url":"https://syntology.ai/paper/2310.16446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16446"}},"official":{"repos":["hkyoon95/mqg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quality-quantity-synthetic-corpora-from","slug":"quality-quantity-synthetic-corpora-from","title":"TOP-Training: Target-Oriented Pretraining for Medical Extractive Question Answering","date":"2023-10-25","arxiv_id":"2310.16995","repositories_listed":1,"syntology":null},{"url":"/paper/background-summarization-of-event-timelines","slug":"background-summarization-of-event-timelines","title":"Background Summarization of Event Timelines","date":"2023-10-24","arxiv_id":"2310.16197","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-temporal-and-causal","slug":"large-language-models-are-temporal-and-causal","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","date":"2023-10-24","arxiv_id":"2310.15747","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/large-language-models-are-temporal-and-causal#ran","syntology_url":"https://syntology.ai/paper/2310.15747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15747"}},"official":{"repos":["mlvlab/Flipped-VQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/nutrea-neural-tree-search-for-context-guided-1","slug":"nutrea-neural-tree-search-for-context-guided-1","title":"NuTrea: Neural Tree Search for Context-guided Multi-hop KGQA","date":"2023-10-24","arxiv_id":"2310.15484","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nutrea-neural-tree-search-for-context-guided-1#ran","syntology_url":"https://syntology.ai/paper/2310.15484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15484"}},"official":{"repos":["mlvlab/nutrea"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tage-enabling-an-embodied-agent-to-understand","slug":"tage-enabling-an-embodied-agent-to-understand","title":"tagE: Enabling an Embodied Agent to Understand Human Instructions","date":"2023-10-24","arxiv_id":"2310.15605","repositories_listed":1,"syntology":null},{"url":"/paper/visual-cropping-improves-zero-shot-question","slug":"visual-cropping-improves-zero-shot-question","title":"Towards Perceiving Small Visual Details in Zero-shot Visual Question Answering with Multimodal LLMs","date":"2023-10-24","arxiv_id":"2310.16033","repositories_listed":1,"syntology":null},{"url":"/paper/disc-finllm-a-chinese-financial-large","slug":"disc-finllm-a-chinese-financial-large","title":"DISC-FinLLM: A Chinese Financial Large Language Model based on Multiple Experts Fine-tuning","date":"2023-10-23","arxiv_id":"2310.15205","repositories_listed":1,"syntology":null},{"url":"/paper/diversify-question-generation-with-retrieval","slug":"diversify-question-generation-with-retrieval","title":"Diversify Question Generation with Retrieval-Augmented Style Transfer","date":"2023-10-23","arxiv_id":"2310.14503","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diversify-question-generation-with-retrieval#ran","syntology_url":"https://syntology.ai/paper/2310.14503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14503"}},"official":{"repos":["gouqi666/rast"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/epik-eval-evaluation-for-language-models-as","slug":"epik-eval-evaluation-for-language-models-as","title":"EpiK-Eval: Evaluation for Language Models as Epistemic Models","date":"2023-10-23","arxiv_id":"2310.15372","repositories_listed":1,"syntology":null},{"url":"/paper/once-upon-a-textit-time-in-textit-graph","slug":"once-upon-a-textit-time-in-textit-graph","title":"Once Upon a $\\textit{Time}$ in $\\textit{Graph}$: Relative-Time Pretraining for Complex Temporal Reasoning","date":"2023-10-23","arxiv_id":"2310.14709","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/once-upon-a-textit-time-in-textit-graph#ran","syntology_url":"https://syntology.ai/paper/2310.14709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14709"}},"official":{"repos":["damo-nlp-sg/rememo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reference-free-domain-adaptation-for","slug":"reference-free-domain-adaptation-for","title":"Reference Free Domain Adaptation for Translation of Noisy Questions with Question Specific Rewards","date":"2023-10-23","arxiv_id":"2310.15259","repositories_listed":1,"syntology":null},{"url":"/paper/text-fact-transfer","slug":"text-fact-transfer","title":"Text Fact Transfer","date":"2023-10-23","arxiv_id":"2310.14486","repositories_listed":1,"syntology":null},{"url":"/paper/tree-of-clarifications-answering-ambiguous","slug":"tree-of-clarifications-answering-ambiguous","title":"Tree of Clarifications: Answering Ambiguous Questions with Retrieval-Augmented Large Language Models","date":"2023-10-23","arxiv_id":"2310.14696","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tree-of-clarifications-answering-ambiguous#ran","syntology_url":"https://syntology.ai/paper/2310.14696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14696"}},"official":{"repos":["gankim/tree-of-clarifications"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cxr-llava-multimodal-large-language-model-for","slug":"cxr-llava-multimodal-large-language-model-for","title":"CXR-LLAVA: a multimodal large language model for interpreting chest X-ray images","date":"2023-10-22","arxiv_id":"2310.18341","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cxr-llava-multimodal-large-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2310.18341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18341"}},"official":{"repos":["ecofri/cxr_llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/merging-generated-and-retrieved-knowledge-for","slug":"merging-generated-and-retrieved-knowledge-for","title":"Merging Generated and Retrieved Knowledge for Open-Domain QA","date":"2023-10-22","arxiv_id":"2310.14393","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/merging-generated-and-retrieved-knowledge-for#ran","syntology_url":"https://syntology.ai/paper/2310.14393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14393"}},"official":{"repos":["yunx-z/combo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qa-natver-question-answering-for-natural","slug":"qa-natver-question-answering-for-natural","title":"QA-NatVer: Question Answering for Natural Logic-based Fact Verification","date":"2023-10-22","arxiv_id":"2310.14198","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/qa-natver-question-answering-for-natural#ran","syntology_url":"https://syntology.ai/paper/2310.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14198"}},"official":{"repos":["raldir/qa-natver"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/large-language-models-and-multimodal","slug":"large-language-models-and-multimodal","title":"Large Language Models and Multimodal Retrieval for Visual Word Sense Disambiguation","date":"2023-10-21","arxiv_id":"2310.14025","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-and-multimodal#ran","syntology_url":"https://syntology.ai/paper/2310.14025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14025"}},"official":{"repos":["anastasiakrith/multimodal-retrieval-for-vwsd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/moqagpt-zero-shot-multi-modal-open-domain","slug":"moqagpt-zero-shot-multi-modal-open-domain","title":"MoqaGPT : Zero-Shot Multi-modal Open-domain Question Answering with Large Language Model","date":"2023-10-20","arxiv_id":"2310.13265","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-retrieval-augmented-reader-models","slug":"optimizing-retrieval-augmented-reader-models","title":"Optimizing Retrieval-augmented Reader Models via Token Elimination","date":"2023-10-20","arxiv_id":"2310.13682","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimizing-retrieval-augmented-reader-models#ran","syntology_url":"https://syntology.ai/paper/2310.13682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13682"}},"official":{"repos":["mosheber/token_elimination"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/posqa-probe-the-world-models-of-llms-with","slug":"posqa-probe-the-world-models-of-llms-with","title":"POSQA: Probe the World Models of LLMs with Size Comparisons","date":"2023-10-20","arxiv_id":"2310.13394","repositories_listed":1,"syntology":null},{"url":"/paper/primacy-effect-of-chatgpt","slug":"primacy-effect-of-chatgpt","title":"Primacy Effect of ChatGPT","date":"2023-10-20","arxiv_id":"2310.13206","repositories_listed":1,"syntology":null},{"url":"/paper/salmonn-towards-generic-hearing-abilities-for","slug":"salmonn-towards-generic-hearing-abilities-for","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","date":"2023-10-20","arxiv_id":"2310.13289","repositories_listed":1,"syntology":null},{"url":"/paper/self-consistency-of-large-language-models","slug":"self-consistency-of-large-language-models","title":"Self-Consistency of Large Language Models under Ambiguity","date":"2023-10-20","arxiv_id":"2310.13439","repositories_listed":1,"syntology":null},{"url":"/paper/self-prompted-chain-of-thought-on-large","slug":"self-prompted-chain-of-thought-on-large","title":"Self-prompted Chain-of-Thought on Large Language Models for Open-domain Multi-hop Reasoning","date":"2023-10-20","arxiv_id":"2310.13552","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-self-adaptive-small-language-models","slug":"test-time-self-adaptive-small-language-models","title":"Test-Time Self-Adaptive Small Language Models for Question Answering","date":"2023-10-20","arxiv_id":"2310.13307","repositories_listed":1,"syntology":null},{"url":"/paper/clift-analysing-natural-distribution-shift-on","slug":"clift-analysing-natural-distribution-shift-on","title":"CLIFT: Analysing Natural Distribution Shift on Question Answering Models in Clinical Domain","date":"2023-10-19","arxiv_id":"2310.13146","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-augmented-language-model","slug":"knowledge-augmented-language-model","title":"Knowledge-Augmented Language Model Verification","date":"2023-10-19","arxiv_id":"2310.12836","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":3,"n_honours":5,"n_violates":0,"n_no_contract":4,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 5 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-augmented-language-model#ran","syntology_url":"https://syntology.ai/paper/2310.12836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12836"}},"official":{"repos":["jinheonbaek/kalmv"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/psychic-a-neuro-symbolic-framework-for","slug":"psychic-a-neuro-symbolic-framework-for","title":"PSYCHIC: A Neuro-Symbolic Framework for Knowledge Graph Question-Answering Grounding","date":"2023-10-19","arxiv_id":"2310.12638","repositories_listed":1,"syntology":null},{"url":"/paper/reliable-academic-conference-question","slug":"reliable-academic-conference-question","title":"Reliable Academic Conference Question Answering: A Study Based on Large Language Model","date":"2023-10-19","arxiv_id":"2310.13028","repositories_listed":1,"syntology":null},{"url":"/paper/rsadapter-adapting-multimodal-models-for","slug":"rsadapter-adapting-multimodal-models-for","title":"RSAdapter: Adapting Multimodal Models for Remote Sensing Visual Question Answering","date":"2023-10-19","arxiv_id":"2310.13120","repositories_listed":1,"syntology":null},{"url":"/paper/time-aware-representation-learning-for-time","slug":"time-aware-representation-learning-for-time","title":"Time-Aware Representation Learning for Time-Sensitive Question Answering","date":"2023-10-19","arxiv_id":"2310.12585","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/time-aware-representation-learning-for-time#ran","syntology_url":"https://syntology.ai/paper/2310.12585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12585"}},"official":{"repos":["sonjbin/tcqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gold-a-global-and-local-aware-denoising","slug":"gold-a-global-and-local-aware-denoising","title":"Gold: A Global and Local-aware Denoising Framework for Commonsense Knowledge Graph Noise Detection","date":"2023-10-18","arxiv_id":"2310.12011","repositories_listed":1,"syntology":null},{"url":"/paper/qadynamics-training-dynamics-driven-synthetic","slug":"qadynamics-training-dynamics-driven-synthetic","title":"QADYNAMICS: Training Dynamics-Driven Synthetic QA Diagnostic for Zero-Shot Commonsense Question Answering","date":"2023-10-17","arxiv_id":"2310.11303","repositories_listed":1,"syntology":null},{"url":"/paper/unanswerable-visual-question-answering","slug":"unanswerable-visual-question-answering","title":"UNK-VQA: A Dataset and a Probe into the Abstention Ability of Multi-modal Large Models","date":"2023-10-17","arxiv_id":"2310.10942","repositories_listed":1,"syntology":null},{"url":"/paper/bioplanner-automatic-evaluation-of-llms-on","slug":"bioplanner-automatic-evaluation-of-llms-on","title":"BioPlanner: Automatic Evaluation of LLMs on Protocol Planning in Biology","date":"2023-10-16","arxiv_id":"2310.10632","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bioplanner-automatic-evaluation-of-llms-on#ran","syntology_url":"https://syntology.ai/paper/2310.10632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10632"}},"official":{"repos":["bioplanner/bioplanner"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficacy-of-dual-encoders-for-extreme-multi","slug":"efficacy-of-dual-encoders-for-extreme-multi","title":"Dual-Encoders for Extreme Multi-Label Classification","date":"2023-10-16","arxiv_id":"2310.10636","repositories_listed":1,"syntology":null},{"url":"/paper/emerging-challenges-in-personalized-medicine","slug":"emerging-challenges-in-personalized-medicine","title":"Emerging Challenges in Personalized Medicine: Assessing Demographic Effects on Biomedical Question Answering Systems","date":"2023-10-16","arxiv_id":"2310.10571","repositories_listed":1,"syntology":null},{"url":"/paper/empirical-study-of-zero-shot-ner-with-chatgpt","slug":"empirical-study-of-zero-shot-ner-with-chatgpt","title":"Empirical Study of Zero-Shot NER with ChatGPT","date":"2023-10-16","arxiv_id":"2310.10035","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/empirical-study-of-zero-shot-ner-with-chatgpt#ran","syntology_url":"https://syntology.ai/paper/2310.10035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10035"}},"official":{"repos":["emma1066/zero-shot-ner-with-chatgpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/if-the-sources-could-talk-evaluating-large","slug":"if-the-sources-could-talk-evaluating-large","title":"If the Sources Could Talk: Evaluating Large Language Models for Research Assistance in History","date":"2023-10-16","arxiv_id":"2310.10808","repositories_listed":1,"syntology":null},{"url":"/paper/jmedlora-medical-domain-adaptation-on","slug":"jmedlora-medical-domain-adaptation-on","title":"JMedLoRA:Medical Domain Adaptation on Japanese Large Language Models using Instruction-tuning","date":"2023-10-16","arxiv_id":"2310.10083","repositories_listed":1,"syntology":null},{"url":"/paper/on-position-bias-in-summarization-with-large","slug":"on-position-bias-in-summarization-with-large","title":"On Context Utilization in Summarization with Large Language Models","date":"2023-10-16","arxiv_id":"2310.10570","repositories_listed":1,"syntology":null},{"url":"/paper/untying-the-reversal-curse-via-bidirectional","slug":"untying-the-reversal-curse-via-bidirectional","title":"Untying the Reversal Curse via Bidirectional Language Model Editing","date":"2023-10-16","arxiv_id":"2310.10322","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/untying-the-reversal-curse-via-bidirectional#ran","syntology_url":"https://syntology.ai/paper/2310.10322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10322"}},"official":{"repos":["mjy1111/bake"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-learning-with-iterative","slug":"in-context-learning-with-iterative","title":"In-Context Learning with Iterative Demonstration Selection","date":"2023-10-15","arxiv_id":"2310.09881","repositories_listed":1,"syntology":null},{"url":"/paper/chatkbqa-a-generate-then-retrieve-framework","slug":"chatkbqa-a-generate-then-retrieve-framework","title":"ChatKBQA: A Generate-then-Retrieve Framework for Knowledge Base Question Answering with Fine-tuned Large Language Models","date":"2023-10-13","arxiv_id":"2310.08975","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/chatkbqa-a-generate-then-retrieve-framework#ran","syntology_url":"https://syntology.ai/paper/2310.08975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08975"}},"official":{"repos":["lhrlab/chatkbqa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/from-clip-to-dino-visual-encoders-shout-in","slug":"from-clip-to-dino-visual-encoders-shout-in","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","date":"2023-10-13","arxiv_id":"2310.08825","repositories_listed":1,"syntology":null},{"url":"/paper/qilin-med-multi-stage-knowledge-injection","slug":"qilin-med-multi-stage-knowledge-injection","title":"Qilin-Med: Multi-stage Knowledge Injection Advanced Medical Large Language Model","date":"2023-10-13","arxiv_id":"2310.09089","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/qilin-med-multi-stage-knowledge-injection#ran","syntology_url":"https://syntology.ai/paper/2310.09089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09089"}},"official":{"repos":["williamliujl/Qilin-Med"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/expanding-the-vocabulary-of-bert-for","slug":"expanding-the-vocabulary-of-bert-for","title":"Expanding the Vocabulary of BERT for Knowledge Base Construction","date":"2023-10-12","arxiv_id":"2310.08291","repositories_listed":1,"syntology":null},{"url":"/paper/graphextqa-a-benchmark-for-evaluating-graph","slug":"graphextqa-a-benchmark-for-evaluating-graph","title":"GraphextQA: A Benchmark for Evaluating Graph-Enhanced Large Language Models","date":"2023-10-12","arxiv_id":"2310.08487","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-scientific","slug":"large-language-models-for-scientific","title":"Large Language Models for Scientific Synthesis, Inference and Explanation","date":"2023-10-12","arxiv_id":"2310.07984","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-for-scientific#ran","syntology_url":"https://syntology.ai/paper/2310.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07984"}},"official":{"repos":["zyzisastudyreallyhardguy/llm4sd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/loftq-lora-fine-tuning-aware-quantization-for","slug":"loftq-lora-fine-tuning-aware-quantization-for","title":"LoftQ: LoRA-Fine-Tuning-Aware Quantization for Large Language Models","date":"2023-10-12","arxiv_id":"2310.08659","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/loftq-lora-fine-tuning-aware-quantization-for#ran","syntology_url":"https://syntology.ai/paper/2310.08659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08659"}},"official":{"repos":["yxli2123/loftq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-set-knowledge-based-visual-question","slug":"open-set-knowledge-based-visual-question","title":"Open-Set Knowledge-Based Visual Question Answering with Inference Paths","date":"2023-10-12","arxiv_id":"2310.08148","repositories_listed":1,"syntology":null},{"url":"/paper/qasina-religious-domain-question-answering","slug":"qasina-religious-domain-question-answering","title":"QASiNa: Religious Domain Question Answering using Sirah Nabawiyah","date":"2023-10-12","arxiv_id":"2310.08102","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-factuality-a-comprehensive-evaluation","slug":"beyond-factuality-a-comprehensive-evaluation","title":"Beyond Factuality: A Comprehensive Evaluation of Large Language Models as Knowledge Generators","date":"2023-10-11","arxiv_id":"2310.07289","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-factuality-a-comprehensive-evaluation#ran","syntology_url":"https://syntology.ai/paper/2310.07289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07289"}},"official":{"repos":["chanliang/conner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/instructretro-instruction-tuning-post","slug":"instructretro-instruction-tuning-post","title":"InstructRetro: Instruction Tuning post Retrieval-Augmented Pretraining","date":"2023-10-11","arxiv_id":"2310.07713","repositories_listed":1,"syntology":null},{"url":"/paper/mini-dalle3-interactive-text-to-image-by","slug":"mini-dalle3-interactive-text-to-image-by","title":"Mini-DALLE3: Interactive Text to Image by Prompting Large Language Models","date":"2023-10-11","arxiv_id":"2310.07653","repositories_listed":1,"syntology":null}],"record_sha256":"5821eff912d974b56d63e3b4cfc15ed2f037200f84cd6a96c9702bd4f8953fcb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}