{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/18","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":109,"rows_per_page":100,"rows":[1701,1800],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/17","next":"/task/question-answering/papers/19","papers":[{"url":"/paper/overview-of-the-ehrsql-2024-shared-task-on","slug":"overview-of-the-ehrsql-2024-shared-task-on","title":"Overview of the EHRSQL 2024 Shared Task on Reliable Text-to-SQL Modeling on Electronic Health Records","date":"2024-05-04","arxiv_id":"2405.06673","repositories_listed":1,"syntology":null},{"url":"/paper/semi-parametric-retrieval-via-binary-token","slug":"semi-parametric-retrieval-via-binary-token","title":"Semi-Parametric Retrieval via Binary Bag-of-Tokens Index","date":"2024-05-03","arxiv_id":"2405.01924","repositories_listed":1,"syntology":null},{"url":"/paper/single-and-multi-hop-question-answering","slug":"single-and-multi-hop-question-answering","title":"Single and Multi-Hop Question-Answering Datasets for Reticular Chemistry with GPT-4-Turbo","date":"2024-05-03","arxiv_id":"2405.02128","repositories_listed":1,"syntology":null},{"url":"/paper/omnidrive-a-holistic-llm-agent-framework-for","slug":"omnidrive-a-holistic-llm-agent-framework-for","title":"OmniDrive: A Holistic Vision-Language Dataset for Autonomous Driving with Counterfactual Reasoning","date":"2024-05-02","arxiv_id":"2405.01533","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/omnidrive-a-holistic-llm-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.01533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01533"}},"official":{"repos":["nvlabs/omnidrive"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-flute-visual-figurative-language","slug":"v-flute-visual-figurative-language","title":"Understanding Figurative Meaning through Explainable Visual Entailment","date":"2024-05-02","arxiv_id":"2405.01474","repositories_listed":1,"syntology":null},{"url":"/paper/biomedrag-a-retrieval-augmented-large","slug":"biomedrag-a-retrieval-augmented-large","title":"BiomedRAG: A Retrieval Augmented Large Language Model for Biomedicine","date":"2024-05-01","arxiv_id":"2405.00465","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/biomedrag-a-retrieval-augmented-large#ran","syntology_url":"https://syntology.ai/paper/2405.00465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00465"}},"official":{"repos":["toneli/petailor-for-bio-triple-extraction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-and-retrieval-augmented","slug":"fine-tuning-and-retrieval-augmented","title":"Fine-Tuning and Retrieval Augmented Generation for Question Answering Using Affordable Large Language Models","date":"2024-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lito-learnable-intervention-for-truthfulness","slug":"lito-learnable-intervention-for-truthfulness","title":"Enhanced Language Model Truthfulness with Learnable Intervention and Uncertainty Expression","date":"2024-05-01","arxiv_id":"2405.00301","repositories_listed":1,"syntology":null},{"url":"/paper/tablevqa-bench-a-visual-question-answering","slug":"tablevqa-bench-a-visual-question-answering","title":"TableVQA-Bench: A Visual Question Answering Benchmark on Multiple Table Domains","date":"2024-04-30","arxiv_id":"2404.19205","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tablevqa-bench-a-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2404.19205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.19205"}},"official":{"repos":["naver-ai/tablevqabench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-search-engine-for-machines-unified","slug":"towards-a-search-engine-for-machines-unified","title":"Towards a Search Engine for Machines: Unified Ranking for Multiple Retrieval-Augmented Large Language Models","date":"2024-04-30","arxiv_id":"2405.00175","repositories_listed":1,"syntology":null},{"url":"/paper/when-to-retrieve-teaching-llms-to-utilize","slug":"when-to-retrieve-teaching-llms-to-utilize","title":"When to Retrieve: Teaching LLMs to Utilize Information Retrieval Effectively","date":"2024-04-30","arxiv_id":"2404.19705","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/when-to-retrieve-teaching-llms-to-utilize#ran","syntology_url":"https://syntology.ai/paper/2404.19705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.19705"}},"official":{"repos":["tLabruna/Adapt-LLM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-page-document-visual-question-answering","slug":"multi-page-document-visual-question-answering","title":"Multi-Page Document Visual Question Answering using Self-Attention Scoring Mechanism","date":"2024-04-29","arxiv_id":"2404.19024","repositories_listed":1,"syntology":null},{"url":"/paper/medifact-at-mediqa-corr-2024-why-ai-needs-a","slug":"medifact-at-mediqa-corr-2024-why-ai-needs-a","title":"MediFact at MEDIQA-CORR 2024: Why AI Needs a Human Touch","date":"2024-04-27","arxiv_id":"2404.17999","repositories_listed":1,"syntology":null},{"url":"/paper/medifact-at-mediqa-m3g-2024-medical-question","slug":"medifact-at-mediqa-m3g-2024-medical-question","title":"MediFact at MEDIQA-M3G 2024: Medical Question Answering in Dermatology with Multimodal Learning","date":"2024-04-27","arxiv_id":"2405.01583","repositories_listed":1,"syntology":null},{"url":"/paper/can-a-multichoice-dataset-be-repurposed-for","slug":"can-a-multichoice-dataset-be-repurposed-for","title":"From Multiple-Choice to Extractive QA: A Case Study for English and Arabic","date":"2024-04-26","arxiv_id":"2404.17342","repositories_listed":1,"syntology":null},{"url":"/paper/moviechat-question-aware-sparse-memory-for","slug":"moviechat-question-aware-sparse-memory-for","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","date":"2024-04-26","arxiv_id":"2404.17176","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/moviechat-question-aware-sparse-memory-for#ran","syntology_url":"https://syntology.ai/paper/2404.17176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17176"}},"official":{"repos":["rese1f/MovieChat"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/asking-and-answering-questions-to-extract","slug":"asking-and-answering-questions-to-extract","title":"Asking and Answering Questions to Extract Event-Argument Structures","date":"2024-04-25","arxiv_id":"2404.16413","repositories_listed":1,"syntology":null},{"url":"/paper/indicgenbench-a-multilingual-benchmark-to","slug":"indicgenbench-a-multilingual-benchmark-to","title":"IndicGenBench: A Multilingual Benchmark to Evaluate Generation Capabilities of LLMs on Indic Languages","date":"2024-04-25","arxiv_id":"2404.16816","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-in-healthcare-a","slug":"large-language-models-in-healthcare-a","title":"Large Language Models in the Clinic: A Comprehensive Benchmark","date":"2024-04-25","arxiv_id":"2405.00716","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-graph-completion-using-structural","slug":"knowledge-graph-completion-using-structural","title":"Knowledge Graph Completion using Structural and Textual Embeddings","date":"2024-04-24","arxiv_id":"2404.16206","repositories_listed":1,"syntology":null},{"url":"/paper/bias-patterns-in-the-application-of-llms-for","slug":"bias-patterns-in-the-application-of-llms-for","title":"Bias patterns in the application of LLMs for clinical decision support: A comprehensive study","date":"2024-04-23","arxiv_id":"2404.15149","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bias-patterns-in-the-application-of-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15149"}},"official":{"repos":["healthylaife/faircdsllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/from-matching-to-generation-a-survey-on","slug":"from-matching-to-generation-a-survey-on","title":"From Matching to Generation: A Survey on Generative Information Retrieval","date":"2024-04-23","arxiv_id":"2404.14851","repositories_listed":1,"syntology":null},{"url":"/paper/generate-on-graph-treat-llm-as-both-agent-and","slug":"generate-on-graph-treat-llm-as-both-agent-and","title":"Generate-on-Graph: Treat LLM as both Agent and KG in Incomplete Knowledge Graph Question Answering","date":"2024-04-23","arxiv_id":"2404.14741","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generate-on-graph-treat-llm-as-both-agent-and#ran","syntology_url":"https://syntology.ai/paper/2404.14741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14741"}},"official":{"repos":["yaooxu/gog"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/insights-into-alignment-evaluating-dpo-and","slug":"insights-into-alignment-evaluating-dpo-and","title":"Insights into Alignment: Evaluating DPO and its Variants Across Multiple Tasks","date":"2024-04-23","arxiv_id":"2404.14723","repositories_listed":1,"syntology":null},{"url":"/paper/meddr-diagnosis-guided-bootstrapping-for","slug":"meddr-diagnosis-guided-bootstrapping-for","title":"GSCo: Towards Generalizable AI in Medicine via Generalist-Specialist Collaboration","date":"2024-04-23","arxiv_id":"2404.15127","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meddr-diagnosis-guided-bootstrapping-for#ran","syntology_url":"https://syntology.ai/paper/2404.15127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15127"}},"official":{"repos":["sunanhe/meddr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/simulating-task-oriented-dialogues-with-state","slug":"simulating-task-oriented-dialogues-with-state","title":"Simulating Task-Oriented Dialogues with State Transition Graphs and Large Language Models","date":"2024-04-23","arxiv_id":"2404.14772","repositories_listed":1,"syntology":null},{"url":"/paper/towards-systematic-evaluation-of-logical","slug":"towards-systematic-evaluation-of-logical","title":"LogicBench: Towards Systematic Evaluation of Logical Reasoning Ability of Large Language Models","date":"2024-04-23","arxiv_id":"2404.15522","repositories_listed":1,"syntology":null},{"url":"/paper/boter-bootstrapping-knowledge-selection-and","slug":"boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","arxiv_id":"2404.13947","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/boter-bootstrapping-knowledge-selection-and#ran","syntology_url":"https://syntology.ai/paper/2404.13947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13947"}},"official":{"repos":["haodongze/self-ksel-qans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/listen-then-see-video-alignment-with-speaker","slug":"listen-then-see-video-alignment-with-speaker","title":"Listen Then See: Video Alignment with Speaker Attention","date":"2024-04-21","arxiv_id":"2404.13530","repositories_listed":1,"syntology":null},{"url":"/paper/lost-in-space-probing-fine-grained-spatial","slug":"lost-in-space-probing-fine-grained-spatial","title":"Lost in Space: Probing Fine-grained Spatial Understanding in Vision and Language Resamplers","date":"2024-04-21","arxiv_id":"2404.13594","repositories_listed":1,"syntology":null},{"url":"/paper/fakebench-uncover-the-achilles-heels-of-fake","slug":"fakebench-uncover-the-achilles-heels-of-fake","title":"FakeBench: Probing Explainable Fake Image Detection via Large Multimodal Models","date":"2024-04-20","arxiv_id":"2404.13306","repositories_listed":1,"syntology":null},{"url":"/paper/isqa-informative-factuality-feedback-for","slug":"isqa-informative-factuality-feedback-for","title":"ISQA: Informative Factuality Feedback for Scientific Summarization","date":"2024-04-20","arxiv_id":"2404.13246","repositories_listed":1,"syntology":null},{"url":"/paper/mahasquad-bridging-linguistic-divides-in","slug":"mahasquad-bridging-linguistic-divides-in","title":"MahaSQuAD: Bridging Linguistic Divides in Marathi Question-Answering","date":"2024-04-20","arxiv_id":"2404.13364","repositories_listed":1,"syntology":null},{"url":"/paper/lapa-latent-prompt-assist-model-for-medical","slug":"lapa-latent-prompt-assist-model-for-medical","title":"LaPA: Latent Prompt Assist Model For Medical Visual Question Answering","date":"2024-04-19","arxiv_id":"2404.13039","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lapa-latent-prompt-assist-model-for-medical#ran","syntology_url":"https://syntology.ai/paper/2404.13039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13039"}},"official":{"repos":["garygutc/lapa_model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/advisorqa-towards-helpful-and-harmless-advice","slug":"advisorqa-towards-helpful-and-harmless-advice","title":"AdvisorQA: Towards Helpful and Harmless Advice-seeking Question Answering with Collective Intelligence","date":"2024-04-18","arxiv_id":"2404.11826","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/advisorqa-towards-helpful-and-harmless-advice#ran","syntology_url":"https://syntology.ai/paper/2404.11826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11826"}},"official":{"repos":["minbeomkim/advisorqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-language-models-to-explicitly-handle","slug":"aligning-language-models-to-explicitly-handle","title":"Aligning Language Models to Explicitly Handle Ambiguity","date":"2024-04-18","arxiv_id":"2404.11972","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-language-models-to-explicitly-handle#ran","syntology_url":"https://syntology.ai/paper/2404.11972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11972"}},"official":{"repos":["heyjoonkim/apa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eusquad-automatically-translated-and-aligned","slug":"eusquad-automatically-translated-and-aligned","title":"EuSQuAD: Automatically Translated and Aligned SQuAD2.0 for Basque","date":"2024-04-18","arxiv_id":"2404.12177","repositories_listed":1,"syntology":null},{"url":"/paper/look-listen-and-answer-overcoming-biases-for","slug":"look-listen-and-answer-overcoming-biases-for","title":"Look, Listen, and Answer: Overcoming Biases for Audio-Visual Question Answering","date":"2024-04-18","arxiv_id":"2404.12020","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/look-listen-and-answer-overcoming-biases-for#ran","syntology_url":"https://syntology.ai/paper/2404.12020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12020"}},"official":{"repos":["reml-group/music-avqa-r"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/consistency-training-by-synthetic-question","slug":"consistency-training-by-synthetic-question","title":"Consistency Training by Synthetic Question Generation for Conversational Question Answering","date":"2024-04-17","arxiv_id":"2404.11109","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/consistency-training-by-synthetic-question#ran","syntology_url":"https://syntology.ai/paper/2404.11109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11109"}},"official":{"repos":["hamedhematian/syncqg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-faithful-are-rag-models-quantifying-the","slug":"how-faithful-are-rag-models-quantifying-the","title":"ClashEval: Quantifying the tug-of-war between an LLM's internal prior and external evidence","date":"2024-04-16","arxiv_id":"2404.10198","repositories_listed":1,"syntology":null},{"url":"/paper/spiral-of-silences-how-is-large-language","slug":"spiral-of-silences-how-is-large-language","title":"Spiral of Silence: How is Large Language Model Killing Information Retrieval? -- A Case Study on Open Domain Question Answering","date":"2024-04-16","arxiv_id":"2404.10496","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-vision-and-language-spaces-with","slug":"bridging-vision-and-language-spaces-with","title":"Bridging Vision and Language Spaces with Assignment Prediction","date":"2024-04-15","arxiv_id":"2404.09632","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-vision-and-language-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2404.09632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09632"}},"official":{"repos":["park-jungin/vlap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/constructing-benchmarks-and-interventions-for","slug":"constructing-benchmarks-and-interventions-for","title":"Constructing Benchmarks and Interventions for Combating Hallucinations in LLMs","date":"2024-04-15","arxiv_id":"2404.09971","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constructing-benchmarks-and-interventions-for#ran","syntology_url":"https://syntology.ai/paper/2404.09971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09971"}},"official":{"repos":["technion-cs-nlp/hallucination-mitigation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textcot-zoom-in-for-enhanced-multimodal-text","slug":"textcot-zoom-in-for-enhanced-multimodal-text","title":"TextCoT: Zoom In for Enhanced Multimodal Text-Rich Image Understanding","date":"2024-04-15","arxiv_id":"2404.09797","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textcot-zoom-in-for-enhanced-multimodal-text#ran","syntology_url":"https://syntology.ai/paper/2404.09797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09797"}},"official":{"repos":["bzluan/textcot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/curiousllm-elevating-multi-document-qa-with","slug":"curiousllm-elevating-multi-document-qa-with","title":"CuriousLLM: Elevating Multi-Document QA with Reasoning-Infused Knowledge Graph Prompting","date":"2024-04-13","arxiv_id":"2404.09077","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-health-question-answering-with","slug":"improving-health-question-answering-with","title":"Improving Health Question Answering with Reliable and Time-Aware Evidence Retrieval","date":"2024-04-12","arxiv_id":"2404.08359","repositories_listed":1,"syntology":null},{"url":"/paper/small-models-are-still-effective-cross-domain","slug":"small-models-are-still-effective-cross-domain","title":"Small Models Are (Still) Effective Cross-Domain Argument Extractors","date":"2024-04-12","arxiv_id":"2404.08579","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-localize-objects-improves-spatial","slug":"learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","arxiv_id":"2404.07449","repositories_listed":1,"syntology":null},{"url":"/paper/lloco-learning-long-contexts-offline","slug":"lloco-learning-long-contexts-offline","title":"LLoCO: Learning Long Contexts Offline","date":"2024-04-11","arxiv_id":"2404.07979","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/lloco-learning-long-contexts-offline#ran","syntology_url":"https://syntology.ai/paper/2404.07979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07979"}},"official":{"repos":["jeffreysijuntan/lloco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-image-visual-question-answering-for","slug":"multi-image-visual-question-answering-for","title":"Language Models Meet Anomaly Detection for Better Interpretability and Generalizability","date":"2024-04-11","arxiv_id":"2404.07622","repositories_listed":1,"syntology":null},{"url":"/paper/openbias-open-set-bias-detection-in-text-to","slug":"openbias-open-set-bias-detection-in-text-to","title":"OpenBias: Open-set Bias Detection in Text-to-Image Generative Models","date":"2024-04-11","arxiv_id":"2404.07990","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openbias-open-set-bias-detection-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2404.07990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07990"}},"official":{"repos":["picsart-ai-research/openbias"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/superposition-prompting-improving-and","slug":"superposition-prompting-improving-and","title":"Superposition Prompting: Improving and Accelerating Retrieval-Augmented Generation","date":"2024-04-10","arxiv_id":"2404.06910","repositories_listed":1,"syntology":null},{"url":"/paper/transferable-and-efficient-non-factual","slug":"transferable-and-efficient-non-factual","title":"Transferable and Efficient Non-Factual Content Detection via Probe Training with Offline Consistency Checking","date":"2024-04-10","arxiv_id":"2404.06742","repositories_listed":1,"syntology":null},{"url":"/paper/ada-leval-evaluating-long-context-llms-with","slug":"ada-leval-evaluating-long-context-llms-with","title":"Ada-LEval: Evaluating long-context LLMs with length-adaptable benchmarks","date":"2024-04-09","arxiv_id":"2404.06480","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ada-leval-evaluating-long-context-llms-with#ran","syntology_url":"https://syntology.ai/paper/2404.06480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06480"}},"official":{"repos":["open-compass/ada-leval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-based-reasoning-about-vector-graphics","slug":"text-based-reasoning-about-vector-graphics","title":"Visually Descriptive Language Model for Vector Graphics Reasoning","date":"2024-04-09","arxiv_id":"2404.06479","repositories_listed":1,"syntology":null},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-visual-and-text-prompting-for-improved","slug":"joint-visual-and-text-prompting-for-improved","title":"Joint Visual and Text Prompting for Improved Object-Centric Perception with Multimodal Large Language Models","date":"2024-04-06","arxiv_id":"2404.04514","repositories_listed":1,"syntology":null},{"url":"/paper/kazqad-kazakh-open-domain-question-answering","slug":"kazqad-kazakh-open-domain-question-answering","title":"KazQAD: Kazakh Open-Domain Question Answering Dataset","date":"2024-04-06","arxiv_id":"2404.04487","repositories_listed":1,"syntology":null},{"url":"/paper/soft-prompting-with-graph-of-thought-for","slug":"soft-prompting-with-graph-of-thought-for","title":"Soft-Prompting with Graph-of-Thought for Multi-modal Representation Learning","date":"2024-04-06","arxiv_id":"2404.04538","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soft-prompting-with-graph-of-thought-for#ran","syntology_url":"https://syntology.ai/paper/2404.04538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04538"}},"official":{"repos":["shishicode/agot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cbr-rag-case-based-reasoning-for-retrieval","slug":"cbr-rag-case-based-reasoning-for-retrieval","title":"CBR-RAG: Case-Based Reasoning for Retrieval Augmented Generation in LLMs for Legal Question Answering","date":"2024-04-04","arxiv_id":"2404.04302","repositories_listed":1,"syntology":null},{"url":"/paper/longvlm-efficient-long-video-understanding","slug":"longvlm-efficient-long-video-understanding","title":"LongVLM: Efficient Long Video Understanding via Large Language Models","date":"2024-04-04","arxiv_id":"2404.03384","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"0 ran · 4 unverified","sample_list":"/paper/longvlm-efficient-long-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2404.03384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03384"}},"official":{"repos":["ziplab/longvlm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/untangle-the-knot-interweaving-conflicting","slug":"untangle-the-knot-interweaving-conflicting","title":"Untangle the KNOT: Interweaving Conflicting Knowledge and Reasoning Skills in Large Language Models","date":"2024-04-04","arxiv_id":"2404.03577","repositories_listed":1,"syntology":null},{"url":"/paper/multi-granularity-guided-fusion-in-decoder","slug":"multi-granularity-guided-fusion-in-decoder","title":"Multi-Granularity Guided Fusion-in-Decoder","date":"2024-04-03","arxiv_id":"2404.02581","repositories_listed":1,"syntology":null},{"url":"/paper/clapnq-cohesive-long-form-answers-from","slug":"clapnq-cohesive-long-form-answers-from","title":"CLAPNQ: Cohesive Long-form Answers from Passages in Natural Questions for RAG systems","date":"2024-04-02","arxiv_id":"2404.02103","repositories_listed":1,"syntology":null},{"url":"/paper/helmsman-of-the-masses-evaluate-the-opinion","slug":"helmsman-of-the-masses-evaluate-the-opinion","title":"Helmsman of the Masses? Evaluate the Opinion Leadership of Large Language Models in the Werewolf Game","date":"2024-04-02","arxiv_id":"2404.01602","repositories_listed":1,"syntology":null},{"url":"/paper/improving-retrieval-augmented-open-domain","slug":"improving-retrieval-augmented-open-domain","title":"Improving Retrieval Augmented Open-Domain Question-Answering with Vectorized Contexts","date":"2024-04-02","arxiv_id":"2404.02022","repositories_listed":1,"syntology":null},{"url":"/paper/rematch-robust-and-efficient-matching-of","slug":"rematch-robust-and-efficient-matching-of","title":"Rematch: Robust and Efficient Matching of Local Knowledge Graphs to Improve Structural and Semantic Similarity","date":"2024-04-02","arxiv_id":"2404.02126","repositories_listed":1,"syntology":null},{"url":"/paper/causalchaos-dataset-for-comprehensive-causal","slug":"causalchaos-dataset-for-comprehensive-causal","title":"CausalChaos! Dataset for Comprehensive Causal Action Question Answering Over Longer Causal Chains Grounded in Dynamic Visual Scenes","date":"2024-04-01","arxiv_id":"2404.01299","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalchaos-dataset-for-comprehensive-causal#ran","syntology_url":"https://syntology.ai/paper/2404.01299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01299"}},"official":{"repos":["lunaproject22/causalchaos"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-optimization-of-video-large","slug":"direct-preference-optimization-of-video-large","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","date":"2024-04-01","arxiv_id":"2404.01258","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-optimization-of-video-large#ran","syntology_url":"https://syntology.ai/paper/2404.01258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01258"}},"official":{"repos":["riflezhang/llava-hound-dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/traveler-a-multi-lmm-agent-framework-for","slug":"traveler-a-multi-lmm-agent-framework-for","title":"TraveLER: A Modular Multi-LMM Agent Framework for Video Question-Answering","date":"2024-04-01","arxiv_id":"2404.01476","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-divergent-inductive-biases-of-llms","slug":"unveiling-divergent-inductive-biases-of-llms","title":"Unveiling Divergent Inductive Biases of LLMs on Temporal Data","date":"2024-04-01","arxiv_id":"2404.01453","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-multi-hop-question-generation-an","slug":"explainable-multi-hop-question-generation-an","title":"Explainable Multi-hop Question Generation: An End-to-End Approach without Intermediate Question Labeling","date":"2024-03-31","arxiv_id":"2404.00571","repositories_listed":1,"syntology":null},{"url":"/paper/how-much-are-llms-contaminated-a","slug":"how-much-are-llms-contaminated-a","title":"How Much are Large Language Models Contaminated? A Comprehensive Survey and the LLMSanitize Library","date":"2024-03-31","arxiv_id":"2404.00699","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-much-are-llms-contaminated-a#ran","syntology_url":"https://syntology.ai/paper/2404.00699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00699"}},"official":{"repos":["ntunlp/llmsanitize"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-robust-are-the-tabular-qa-models-for","slug":"how-robust-are-the-tabular-qa-models-for","title":"How Robust are the Tabular QA Models for Scientific Tables? A Study using Customized Dataset","date":"2024-03-30","arxiv_id":"2404.00401","repositories_listed":1,"syntology":null},{"url":"/paper/linguistic-calibration-of-language-models","slug":"linguistic-calibration-of-language-models","title":"Linguistic Calibration of Long-Form Generations","date":"2024-03-30","arxiv_id":"2404.00474","repositories_listed":1,"syntology":null},{"url":"/paper/draw-and-understand-leveraging-visual-prompts","slug":"draw-and-understand-leveraging-visual-prompts","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","date":"2024-03-29","arxiv_id":"2403.20271","repositories_listed":1,"syntology":null},{"url":"/paper/mango-a-benchmark-for-evaluating-mapping-and","slug":"mango-a-benchmark-for-evaluating-mapping-and","title":"MANGO: A Benchmark for Evaluating Mapping and Navigation Abilities of Large Language Models","date":"2024-03-29","arxiv_id":"2403.19913","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mango-a-benchmark-for-evaluating-mapping-and#ran","syntology_url":"https://syntology.ai/paper/2403.19913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19913"}},"official":{"repos":["oaklight/mango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsolvable-problem-detection-evaluating","slug":"unsolvable-problem-detection-evaluating","title":"Unsolvable Problem Detection: Evaluating Trustworthiness of Vision Language Models","date":"2024-03-29","arxiv_id":"2403.20331","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsolvable-problem-detection-evaluating#ran","syntology_url":"https://syntology.ai/paper/2403.20331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20331"}},"official":{"repos":["atsumiyai/upd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-good-at-utility","slug":"are-large-language-models-good-at-utility","title":"Are Large Language Models Good at Utility Judgments?","date":"2024-03-28","arxiv_id":"2403.19216","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/are-large-language-models-good-at-utility#ran","syntology_url":"https://syntology.ai/paper/2403.19216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19216"}},"official":{"repos":["ict-bigdatalab/utility_judgments"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/jdocqa-japanese-document-question-answering","slug":"jdocqa-japanese-document-question-answering","title":"JDocQA: Japanese Document Question Answering Dataset for Generative Language Models","date":"2024-03-28","arxiv_id":"2403.19454","repositories_listed":1,"syntology":null},{"url":"/paper/multi-frame-lightweight-efficient-vision","slug":"multi-frame-lightweight-efficient-vision","title":"Multi-Frame, Lightweight & Efficient Vision-Language Models for Question Answering in Autonomous Driving","date":"2024-03-28","arxiv_id":"2403.19838","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-enhanced-knowledge-editing-for","slug":"retrieval-enhanced-knowledge-editing-for","title":"Retrieval-enhanced Knowledge Editing in Language Models for Multi-Hop Question Answering","date":"2024-03-28","arxiv_id":"2403.19631","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retrieval-enhanced-knowledge-editing-for#ran","syntology_url":"https://syntology.ai/paper/2403.19631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19631"}},"official":{"repos":["sycny/rae"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","slug":"an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","arxiv_id":"2403.18406","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-image-grid-can-be-worth-a-video-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.18406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18406"}},"official":{"repos":["imagegridworth/IG-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedlm-a-2-7b-parameter-language-model","slug":"biomedlm-a-2-7b-parameter-language-model","title":"BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text","date":"2024-03-27","arxiv_id":"2403.18421","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/biomedlm-a-2-7b-parameter-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.18421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18421"}},"official":{"repos":["stanford-crfm/biomedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-language-beat-numerical-regression","slug":"can-language-beat-numerical-regression","title":"Can Language Beat Numerical Regression? Language-Based Multimodal Trajectory Prediction","date":"2024-03-27","arxiv_id":"2403.18447","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-language-beat-numerical-regression#ran","syntology_url":"https://syntology.ai/paper/2403.18447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18447"}},"official":{"repos":["inhwanbae/lmtrajectory"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/nl-iti-optimizing-probing-and-intervention","slug":"nl-iti-optimizing-probing-and-intervention","title":"Non-Linear Inference Time Intervention: Improving LLM Truthfulness","date":"2024-03-27","arxiv_id":"2403.18680","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-and-mitigating-unimodal-biases-in","slug":"quantifying-and-mitigating-unimodal-biases-in","title":"Quantifying and Mitigating Unimodal Biases in Multimodal Large Language Models: A Causal Perspective","date":"2024-03-27","arxiv_id":"2403.18346","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-and-mitigating-unimodal-biases-in#ran","syntology_url":"https://syntology.ai/paper/2403.18346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18346"}},"official":{"repos":["opencausalab/more"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reshaping-free-text-radiology-notes-into","slug":"reshaping-free-text-radiology-notes-into","title":"Reshaping Free-Text Radiology Notes Into Structured Reports With Generative Transformers","date":"2024-03-27","arxiv_id":"2403.18938","repositories_listed":1,"syntology":null},{"url":"/paper/triviahg-a-dataset-for-automatic-hint","slug":"triviahg-a-dataset-for-automatic-hint","title":"TriviaHG: A Dataset for Automatic Hint Generation from Factoid Questions","date":"2024-03-27","arxiv_id":"2403.18426","repositories_listed":1,"syntology":null},{"url":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic","slug":"arabicaqa-a-comprehensive-dataset-for-arabic","title":"ArabicaQA: A Comprehensive Dataset for Arabic Question Answering","date":"2024-03-26","arxiv_id":"2403.17848","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic#ran","syntology_url":"https://syntology.ai/paper/2403.17848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17848"}},"official":{"repos":["datascienceuibk/arabicaqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-multiple-choice-questions-really-be","slug":"can-multiple-choice-questions-really-be","title":"Can multiple-choice questions really be useful in detecting the abilities of LLMs?","date":"2024-03-26","arxiv_id":"2403.17752","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-action-faithful-and-multimodal","slug":"chain-of-action-faithful-and-multimodal","title":"Chain-of-Action: Faithful and Multimodal Question Answering through Large Language Models","date":"2024-03-26","arxiv_id":"2403.17359","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-action-faithful-and-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.17359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17359"}},"official":{"repos":["MAGICS-LAB/Chain-of-Actions"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chroniclingamericaqa-a-large-scale-question","slug":"chroniclingamericaqa-a-large-scale-question","title":"ChroniclingAmericaQA: A Large-scale Question Answering Dataset based on Historical American Newspaper Pages","date":"2024-03-26","arxiv_id":"2403.17859","repositories_listed":1,"syntology":null},{"url":"/paper/denoising-table-text-retrieval-for-open","slug":"denoising-table-text-retrieval-for-open","title":"Denoising Table-Text Retrieval for Open-Domain Question Answering","date":"2024-03-26","arxiv_id":"2403.17611","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-subgraph-generation-for","slug":"intrinsic-subgraph-generation-for","title":"Intrinsic Subgraph Generation for Interpretable Graph based Visual Question Answering","date":"2024-03-26","arxiv_id":"2403.17647","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/intrinsic-subgraph-generation-for#ran","syntology_url":"https://syntology.ai/paper/2403.17647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17647"}},"official":{"repos":["digitalphonetics/intrinsic-subgraph-generation-for-vqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omnivid-a-generative-framework-for-universal#ran","syntology_url":"https://syntology.ai/paper/2403.17935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17935"}},"official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pctoolkit-a-unified-plug-and-play-prompt","slug":"pctoolkit-a-unified-plug-and-play-prompt","title":"PCToolkit: A Unified Plug-and-Play Prompt Compression Toolkit of Large Language Models","date":"2024-03-26","arxiv_id":"2403.17411","repositories_listed":1,"syntology":null},{"url":"/paper/attribute-first-then-generate-locally","slug":"attribute-first-then-generate-locally","title":"Attribute First, then Generate: Locally-attributable Grounded Text Generation","date":"2024-03-25","arxiv_id":"2403.17104","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attribute-first-then-generate-locally#ran","syntology_url":"https://syntology.ai/paper/2403.17104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17104"}},"official":{"repos":["lovodkin93/attribute-first-then-generate"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"745f6023193d3e58ca9673abbc399ab08b990c80fa6666d961d9ff054da477ad","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}