{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/13","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":109,"rows_per_page":100,"rows":[1201,1300],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/12","next":"/task/question-answering/papers/14","papers":[{"url":"/paper/a-survey-of-medical-vision-and-language","slug":"a-survey-of-medical-vision-and-language","title":"A Survey of Medical Vision-and-Language Applications and Their Techniques","date":"2024-11-19","arxiv_id":"2411.12195","repositories_listed":1,"syntology":null},{"url":"/paper/mc-llava-multi-concept-personalized-vision","slug":"mc-llava-multi-concept-personalized-vision","title":"MC-LLaVA: Multi-Concept Personalized Vision-Language Model","date":"2024-11-18","arxiv_id":"2411.11706","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mc-llava-multi-concept-personalized-vision#ran","syntology_url":"https://syntology.ai/paper/2411.11706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11706"}},"official":{"repos":["arctanxarc/mc-llava"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/quantifying-preferences-of-vision-language","slug":"quantifying-preferences-of-vision-language","title":"Value-Spectrum: Quantifying Preferences of Vision-Language Models via Value Decomposition in Social Media Contexts","date":"2024-11-18","arxiv_id":"2411.11479","repositories_listed":1,"syntology":null},{"url":"/paper/backdoormbti-a-backdoor-learning-multimodal","slug":"backdoormbti-a-backdoor-learning-multimodal","title":"BackdoorMBTI: A Backdoor Learning Multimodal Benchmark Tool Kit for Backdoor Defense Evaluation","date":"2024-11-17","arxiv_id":"2411.11006","repositories_listed":1,"syntology":null},{"url":"/paper/forpkg-1-0-a-framework-for-constructing","slug":"forpkg-1-0-a-framework-for-constructing","title":"ForPKG: A Framework for Constructing Forestry Policy Knowledge Graph and Application Analysis","date":"2024-11-17","arxiv_id":"2411.11090","repositories_listed":1,"syntology":null},{"url":"/paper/the-limited-impact-of-medical-adaptation-of","slug":"the-limited-impact-of-medical-adaptation-of","title":"The Limited Impact of Medical Adaptation of Large Language and Vision-Language Models","date":"2024-11-13","arxiv_id":"2411.08870","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-limited-impact-of-medical-adaptation-of#ran","syntology_url":"https://syntology.ai/paper/2411.08870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.08870"}},"official":{"repos":["taekb/eval-medical-dapt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deceiving-question-answering-models-a-hybrid","slug":"deceiving-question-answering-models-a-hybrid","title":"Deceiving Question-Answering Models: A Hybrid Word-Level Adversarial Approach","date":"2024-11-12","arxiv_id":"2411.08248","repositories_listed":1,"syntology":null},{"url":"/paper/likelihood-as-a-performance-gauge-for","slug":"likelihood-as-a-performance-gauge-for","title":"Likelihood as a Performance Gauge for Retrieval-Augmented Generation","date":"2024-11-12","arxiv_id":"2411.07773","repositories_listed":1,"syntology":null},{"url":"/paper/sparrowvqe-visual-question-explanation-for","slug":"sparrowvqe-visual-question-explanation-for","title":"SparrowVQE: Visual Question Explanation for Course Content Understanding","date":"2024-11-12","arxiv_id":"2411.07516","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-context-sensitivity-and-the-knob","slug":"controllable-context-sensitivity-and-the-knob","title":"Controllable Context Sensitivity and the Knob Behind It","date":"2024-11-11","arxiv_id":"2411.07404","repositories_listed":1,"syntology":{"n":24,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":24,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/controllable-context-sensitivity-and-the-knob#ran","syntology_url":"https://syntology.ai/paper/2411.07404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07404"}},"official":{"repos":["kdu4108/context-vs-prior-finetuning"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-optimal-search-and-retrieval-for-rag","slug":"toward-optimal-search-and-retrieval-for-rag","title":"Toward Optimal Search and Retrieval for RAG","date":"2024-11-11","arxiv_id":"2411.07396","repositories_listed":1,"syntology":null},{"url":"/paper/self-training-meets-consistency-improving","slug":"self-training-meets-consistency-improving","title":"Self-Training Meets Consistency: Improving LLMs' Reasoning With Consistency-Driven Rationale Evaluation","date":"2024-11-10","arxiv_id":"2411.06387","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-training-meets-consistency-improving#ran","syntology_url":"https://syntology.ai/paper/2411.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06387"}},"official":{"repos":["jaehyeoklee-119/crest"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-navigation-with-vision-language","slug":"end-to-end-navigation-with-vision-language","title":"End-to-End Navigation with Vision Language Models: Transforming Spatial Reasoning into Question-Answering","date":"2024-11-08","arxiv_id":"2411.05755","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-navigation-with-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.05755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05755"}},"official":{"repos":["Jirl-upenn/VLMnav"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guideq-framework-for-guided-questioning-for","slug":"guideq-framework-for-guided-questioning-for","title":"GUIDEQ: Framework for Guided Questioning for progressive informational collection and classification","date":"2024-11-08","arxiv_id":"2411.05991","repositories_listed":1,"syntology":null},{"url":"/paper/scidqa-a-deep-reading-comprehension-dataset","slug":"scidqa-a-deep-reading-comprehension-dataset","title":"SciDQA: A Deep Reading Comprehension Dataset over Scientific Papers","date":"2024-11-08","arxiv_id":"2411.05338","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scidqa-a-deep-reading-comprehension-dataset#ran","syntology_url":"https://syntology.ai/paper/2411.05338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05338"}},"official":{"repos":["yale-nlp/scidqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/unmasking-the-limits-of-large-language-models","slug":"unmasking-the-limits-of-large-language-models","title":"Unmasking the Limits of Large Language Models: A Systematic Evaluation of Masked Text Processing Ability through MskQA and MskCal","date":"2024-11-08","arxiv_id":"2411.05665","repositories_listed":1,"syntology":null},{"url":"/paper/delift-data-efficient-language-model","slug":"delift-data-efficient-language-model","title":"DELIFT: Data Efficient Language model Instruction Fine Tuning","date":"2024-11-07","arxiv_id":"2411.04425","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/delift-data-efficient-language-model#ran","syntology_url":"https://syntology.ai/paper/2411.04425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04425"}},"official":{"repos":["agarwalishika/delift"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lexicalization-is-all-you-need-examining-the","slug":"lexicalization-is-all-you-need-examining-the","title":"Lexicalization Is All You Need: Examining the Impact of Lexical Knowledge in a Compositional QALD System","date":"2024-11-06","arxiv_id":"2411.03906","repositories_listed":1,"syntology":null},{"url":"/paper/medical-adaptation-of-large-language-and","slug":"medical-adaptation-of-large-language-and","title":"Medical Adaptation of Large Language and Vision-Language Models: Are We Making Progress?","date":"2024-11-06","arxiv_id":"2411.04118","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medical-adaptation-of-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2411.04118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04118"}},"official":{"repos":["taekb/eval-medical-dapt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meg-medical-knowledge-augmented-large","slug":"meg-medical-knowledge-augmented-large","title":"MEG: Medical Knowledge-Augmented Large Language Models for Question Answering","date":"2024-11-06","arxiv_id":"2411.03883","repositories_listed":1,"syntology":null},{"url":"/paper/vqa-2-visual-question-answering-for-video","slug":"vqa-2-visual-question-answering-for-video","title":"VQA$^2$: Visual Question Answering for Video Quality Assessment","date":"2024-11-06","arxiv_id":"2411.03795","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-large-language-models-in-code","slug":"leveraging-large-language-models-in-code","title":"Leveraging Large Language Models in Code Question Answering: Baselines and Issues","date":"2024-11-05","arxiv_id":"2411.03012","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-survey-of-small-language","slug":"a-comprehensive-survey-of-small-language","title":"A Comprehensive Survey of Small Language Models in the Era of Large Language Models: Techniques, Enhancements, Applications, Collaboration with LLMs, and Trustworthiness","date":"2024-11-04","arxiv_id":"2411.03350","repositories_listed":1,"syntology":null},{"url":"/paper/milu-a-multi-task-indic-language","slug":"milu-a-multi-task-indic-language","title":"MILU: A Multi-task Indic Language Understanding Benchmark","date":"2024-11-04","arxiv_id":"2411.02538","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-medical-datasets-with-training","slug":"diagnosing-medical-datasets-with-training","title":"Diagnosing Medical Datasets with Training Dynamics","date":"2024-11-03","arxiv_id":"2411.01653","repositories_listed":1,"syntology":null},{"url":"/paper/birdie-advancing-state-space-models-with","slug":"birdie-advancing-state-space-models-with","title":"Birdie: Advancing State Space Models with Reward-Driven Objectives and Curricula","date":"2024-11-01","arxiv_id":"2411.01030","repositories_listed":1,"syntology":null},{"url":"/paper/latent-paraphrasing-perturbation-on-layers","slug":"latent-paraphrasing-perturbation-on-layers","title":"Latent Paraphrasing: Perturbation on Layers Improves Knowledge Injection in Language Models","date":"2024-11-01","arxiv_id":"2411.00686","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-paraphrasing-perturbation-on-layers#ran","syntology_url":"https://syntology.ai/paper/2411.00686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00686"}},"official":{"repos":["krafton-ai/LaPael"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rationale-guided-retrieval-augmented","slug":"rationale-guided-retrieval-augmented","title":"Rationale-Guided Retrieval Augmented Generation for Medical Question Answering","date":"2024-11-01","arxiv_id":"2411.00300","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rationale-guided-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.00300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00300"}},"official":{"repos":["dmis-lab/rag2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/right-this-way-can-vlms-guide-us-to-see-more","slug":"right-this-way-can-vlms-guide-us-to-see-more","title":"Right this way: Can VLMs Guide Us to See More to Answer Questions?","date":"2024-11-01","arxiv_id":"2411.00394","repositories_listed":1,"syntology":{"n":17,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":13,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/right-this-way-can-vlms-guide-us-to-see-more#ran","syntology_url":"https://syntology.ai/paper/2411.00394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00394"}},"official":{"repos":["LeoLee7/Directional_guidance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":13,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/nearest-neighbor-normalization-improves","slug":"nearest-neighbor-normalization-improves","title":"Nearest Neighbor Normalization Improves Multimodal Retrieval","date":"2024-10-31","arxiv_id":"2410.24114","repositories_listed":1,"syntology":null},{"url":"/paper/show-me-what-and-where-has-changed-question","slug":"show-me-what-and-where-has-changed-question","title":"Show Me What and Where has Changed? Question Answering and Grounding for Remote Sensing Change Detection","date":"2024-10-31","arxiv_id":"2410.23828","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/show-me-what-and-where-has-changed-question#ran","syntology_url":"https://syntology.ai/paper/2410.23828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23828"}},"official":{"repos":["like413/vista"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/buzz-beehive-structured-sparse-kv-cache-with","slug":"buzz-beehive-structured-sparse-kv-cache-with","title":"BUZZ: Beehive-structured Sparse KV Cache with Segmented Heavy Hitters for Efficient LLM Inference","date":"2024-10-30","arxiv_id":"2410.23079","repositories_listed":1,"syntology":null},{"url":"/paper/mdcure-a-scalable-pipeline-for-multi-document","slug":"mdcure-a-scalable-pipeline-for-multi-document","title":"MDCure: A Scalable Pipeline for Multi-Document Instruction-Following","date":"2024-10-30","arxiv_id":"2410.23463","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-large-language-models-for","slug":"multi-agent-large-language-models-for","title":"Multi-Agent Large Language Models for Conversational Task-Solving","date":"2024-10-30","arxiv_id":"2410.22932","repositories_listed":1,"syntology":null},{"url":"/paper/are-vlms-really-blind","slug":"are-vlms-really-blind","title":"Are VLMs Really Blind","date":"2024-10-29","arxiv_id":"2410.22029","repositories_listed":1,"syntology":null},{"url":"/paper/distinguishing-ignorance-from-error-in-llm","slug":"distinguishing-ignorance-from-error-in-llm","title":"Distinguishing Ignorance from Error in LLM Hallucinations","date":"2024-10-29","arxiv_id":"2410.22071","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-guided-prompt-learning-for-request","slug":"knowledge-guided-prompt-learning-for-request","title":"Knowledge-Guided Prompt Learning for Request Quality Assurance in Public Code Review","date":"2024-10-29","arxiv_id":"2410.21673","repositories_listed":1,"syntology":null},{"url":"/paper/promqa-question-answering-dataset-for","slug":"promqa-question-answering-dataset-for","title":"ProMQA: Question Answering Dataset for Multimodal Procedural Activity Understanding","date":"2024-10-29","arxiv_id":"2410.22211","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-data-generation-with-large-language-1","slug":"synthetic-data-generation-with-large-language-1","title":"Synthetic Data Generation with Large Language Models for Personalized Community Question Answering","date":"2024-10-29","arxiv_id":"2410.22182","repositories_listed":1,"syntology":null},{"url":"/paper/autobench-v-can-large-vision-language-models","slug":"autobench-v-can-large-vision-language-models","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","date":"2024-10-28","arxiv_id":"2410.21259","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobench-v-can-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21259"}},"official":{"repos":["wad3birch/AutoBench-V"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-multimodal-explanation-for-visual","slug":"few-shot-multimodal-explanation-for-visual","title":"Few-Shot Multimodal Explanation for Visual Question Answering","date":"2024-10-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/llm-robustness-against-misinformation-in","slug":"llm-robustness-against-misinformation-in","title":"LLM Robustness Against Misinformation in Biomedical Question Answering","date":"2024-10-27","arxiv_id":"2410.21330","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-heads-matter-a-head-level-kv-cache","slug":"not-all-heads-matter-a-head-level-kv-cache","title":"Not All Heads Matter: A Head-Level KV Cache Compression Method with Integrated Retrieval and Reasoning","date":"2024-10-25","arxiv_id":"2410.19258","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/not-all-heads-matter-a-head-level-kv-cache#ran","syntology_url":"https://syntology.ai/paper/2410.19258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19258"}},"official":{"repos":["fyyfu/headkv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/decore-decoding-by-contrasting-retrieval","slug":"decore-decoding-by-contrasting-retrieval","title":"DeCoRe: Decoding by Contrasting Retrieval Heads to Mitigate Hallucinations","date":"2024-10-24","arxiv_id":"2410.18860","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-reflect-the-ideology-of","slug":"large-language-models-reflect-the-ideology-of","title":"Large Language Models Reflect the Ideology of their Creators","date":"2024-10-24","arxiv_id":"2410.18417","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-reflect-the-ideology-of#ran","syntology_url":"https://syntology.ai/paper/2410.18417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18417"}},"official":{"repos":["aida-ugent/llm-ideology-analysis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-text-matters-improving-text-kvqa-with","slug":"visual-text-matters-improving-text-kvqa-with","title":"Visual Text Matters: Improving Text-KVQA with Visual Text Entity Knowledge-aware Large Multimodal Assistant","date":"2024-10-24","arxiv_id":"2410.19144","repositories_listed":1,"syntology":null},{"url":"/paper/adem-vl-adaptive-and-embedded-fusion-for","slug":"adem-vl-adaptive-and-embedded-fusion-for","title":"ADEM-VL: Adaptive and Embedded Fusion for Efficient Vision-Language Tuning","date":"2024-10-23","arxiv_id":"2410.17779","repositories_listed":1,"syntology":null},{"url":"/paper/an-adaptive-framework-for-generating","slug":"an-adaptive-framework-for-generating","title":"An Adaptive Framework for Generating Systematic Explanatory Answer in Online Q&A Platforms","date":"2024-10-23","arxiv_id":"2410.17694","repositories_listed":1,"syntology":null},{"url":"/paper/an-ontology-enabled-approach-for-user","slug":"an-ontology-enabled-approach-for-user","title":"An Ontology-Enabled Approach For User-Centered and Knowledge-Enabled Explanations of AI Systems","date":"2024-10-23","arxiv_id":"2410.17504","repositories_listed":1,"syntology":null},{"url":"/paper/graphusion-a-rag-framework-for-knowledge","slug":"graphusion-a-rag-framework-for-knowledge","title":"Graphusion: A RAG Framework for Knowledge Graph Construction with a Global Perspective","date":"2024-10-23","arxiv_id":"2410.17600","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/graphusion-a-rag-framework-for-knowledge#ran","syntology_url":"https://syntology.ai/paper/2410.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17600"}},"official":{"repos":["irenezihuili/graphusion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longrag-a-dual-perspective-retrieval","slug":"longrag-a-dual-perspective-retrieval","title":"LongRAG: A Dual-Perspective Retrieval-Augmented Generation Paradigm for Long-Context Question Answering","date":"2024-10-23","arxiv_id":"2410.18050","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longrag-a-dual-perspective-retrieval#ran","syntology_url":"https://syntology.ai/paper/2410.18050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18050"}},"official":{"repos":["qingfei1/longrag"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voicetextblender-augmenting-large-language","slug":"voicetextblender-augmenting-large-language","title":"VoiceTextBlender: Augmenting Large Language Models with Speech Capabilities via Single-Stage Joint Speech-Text Supervised Fine-Tuning","date":"2024-10-23","arxiv_id":"2410.17485","repositories_listed":1,"syntology":null},{"url":"/paper/correct-after-answer-enhancing-multi-span","slug":"correct-after-answer-enhancing-multi-span","title":"Correct after Answer: Enhancing Multi-Span Question Answering with Post-Processing Method","date":"2024-10-22","arxiv_id":"2410.16788","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-compositionality-in-text-to-image","slug":"progressive-compositionality-in-text-to-image","title":"Progressive Compositionality In Text-to-Image Generative Models","date":"2024-10-22","arxiv_id":"2410.16719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/progressive-compositionality-in-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2410.16719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16719"}},"official":{"repos":["evansh666/evogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/griffon-g-bridging-vision-language-and-vision","slug":"griffon-g-bridging-vision-language-and-vision","title":"Griffon-G: Bridging Vision-Language and Vision-Centric Tasks via Large Multimodal Models","date":"2024-10-21","arxiv_id":"2410.16163","repositories_listed":1,"syntology":null},{"url":"/paper/steering-knowledge-selection-behaviours-in","slug":"steering-knowledge-selection-behaviours-in","title":"Steering Knowledge Selection Behaviours in LLMs via SAE-Based Representation Engineering","date":"2024-10-21","arxiv_id":"2410.15999","repositories_listed":1,"syntology":null},{"url":"/paper/brief-bridging-retrieval-and-inference-for","slug":"brief-bridging-retrieval-and-inference-for","title":"BRIEF: Bridging Retrieval and Inference for Multi-hop Reasoning via Compression","date":"2024-10-20","arxiv_id":"2410.15277","repositories_listed":1,"syntology":null},{"url":"/paper/crope-evaluating-in-context-adaptation-of","slug":"crope-evaluating-in-context-adaptation-of","title":"CROPE: Evaluating In-Context Adaptation of Vision and Language Models to Culture-Specific Concepts","date":"2024-10-20","arxiv_id":"2410.15453","repositories_listed":1,"syntology":null},{"url":"/paper/medlogic-aqa-enhancing-medical-question","slug":"medlogic-aqa-enhancing-medical-question","title":"MedLogic-AQA: Enhancing Medical Question Answering with Abstractive Models Focusing on Logical Structures","date":"2024-10-20","arxiv_id":"2410.15463","repositories_listed":1,"syntology":null},{"url":"/paper/reverse-question-answering-can-an-llm-write-a","slug":"reverse-question-answering-can-an-llm-write-a","title":"Reverse Question Answering: Can an LLM Write a Question so Hard (or Bad) that it Can't Answer?","date":"2024-10-20","arxiv_id":"2410.15512","repositories_listed":1,"syntology":null},{"url":"/paper/make-llms-better-zero-shot-reasoners","slug":"make-llms-better-zero-shot-reasoners","title":"Make LLMs better zero-shot reasoners: Structure-orientated autonomous reasoning","date":"2024-10-18","arxiv_id":"2410.19000","repositories_listed":1,"syntology":null},{"url":"/paper/multichartqa-benchmarking-vision-language","slug":"multichartqa-benchmarking-vision-language","title":"MultiChartQA: Benchmarking Vision-Language Models on Multi-Chart Problems","date":"2024-10-18","arxiv_id":"2410.14179","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multichartqa-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.14179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14179"}},"official":{"repos":["zivenzhu/multi-chart-qa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/paths-over-graph-knowledge-graph-enpowered","slug":"paths-over-graph-knowledge-graph-enpowered","title":"Paths-over-Graph: Knowledge Graph Empowered Large Language Model Reasoning","date":"2024-10-18","arxiv_id":"2410.14211","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/paths-over-graph-knowledge-graph-enpowered#ran","syntology_url":"https://syntology.ai/paper/2410.14211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14211"}},"official":null}},{"url":"/paper/rag-confusionqa-a-benchmark-for-evaluating","slug":"rag-confusionqa-a-benchmark-for-evaluating","title":"ELOQ: Resources for Enhancing LLM Detection of Out-of-Scope Questions","date":"2024-10-18","arxiv_id":"2410.14567","repositories_listed":1,"syntology":null},{"url":"/paper/viconsformer-constituting-meaningful-phrases","slug":"viconsformer-constituting-meaningful-phrases","title":"ViConsFormer: Constituting Meaningful Phrases of Scene Texts using Transformer-based Method in Vietnamese Text-based Visual Question Answering","date":"2024-10-18","arxiv_id":"2410.14132","repositories_listed":1,"syntology":null},{"url":"/paper/a-little-human-data-goes-a-long-way","slug":"a-little-human-data-goes-a-long-way","title":"A Little Human Data Goes A Long Way","date":"2024-10-17","arxiv_id":"2410.13098","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-little-human-data-goes-a-long-way#ran","syntology_url":"https://syntology.ai/paper/2410.13098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13098"}},"official":{"repos":["dhananjayashok/littlehumandata"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/help-me-identify-is-an-llm-vqa-system-all-we","slug":"help-me-identify-is-an-llm-vqa-system-all-we","title":"Help Me Identify: Is an LLM+VQA System All We Need to Identify Visual Concepts?","date":"2024-10-17","arxiv_id":"2410.13651","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-free-form-decision-making","slug":"measuring-free-form-decision-making","title":"Measuring Free-Form Decision-Making Inconsistency of Language Models in Military Crisis Simulations","date":"2024-10-17","arxiv_id":"2410.13204","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-free-form-decision-making#ran","syntology_url":"https://syntology.ai/paper/2410.13204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13204"}},"official":{"repos":["aashrivastava/llmwargaminginconsistency"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/remember-retrieve-and-generate-understanding","slug":"remember-retrieve-and-generate-understanding","title":"RAP: Retrieval-Augmented Personalization for Multimodal Large Language Models","date":"2024-10-17","arxiv_id":"2410.13360","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/remember-retrieve-and-generate-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.13360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13360"}},"official":{"repos":["hoar012/rap-mllm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-claim-decomposition-benchmark-for-long-form","slug":"a-claim-decomposition-benchmark-for-long-form","title":"A Claim Decomposition Benchmark for Long-form Answer Verification","date":"2024-10-16","arxiv_id":"2410.12558","repositories_listed":1,"syntology":null},{"url":"/paper/legal-uqa-a-low-resource-urdu-english-dataset","slug":"legal-uqa-a-low-resource-urdu-english-dataset","title":"LEGAL-UQA: A Low-Resource Urdu-English Dataset for Legal Question Answering","date":"2024-10-16","arxiv_id":"2410.13013","repositories_listed":1,"syntology":null},{"url":"/paper/meta-chunking-learning-efficient-text","slug":"meta-chunking-learning-efficient-text","title":"Meta-Chunking: Learning Text Segmentation and Semantic Completion via Logical Perception","date":"2024-10-16","arxiv_id":"2410.12788","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-chunking-learning-efficient-text#ran","syntology_url":"https://syntology.ai/paper/2410.12788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12788"}},"official":{"repos":["IAAR-Shanghai/Meta-Chunking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vividmed-vision-language-model-with-versatile","slug":"vividmed-vision-language-model-with-versatile","title":"VividMed: Vision Language Model with Versatile Visual Grounding for Medicine","date":"2024-10-16","arxiv_id":"2410.12694","repositories_listed":1,"syntology":null},{"url":"/paper/worldcuisines-a-massive-scale-benchmark-for","slug":"worldcuisines-a-massive-scale-benchmark-for","title":"WorldCuisines: A Massive-Scale Benchmark for Multilingual and Multicultural Visual Question Answering on Global Cuisines","date":"2024-10-16","arxiv_id":"2410.12705","repositories_listed":1,"syntology":{"n":17,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/worldcuisines-a-massive-scale-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.12705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12705"}},"official":{"repos":["worldcuisines/worldcuisines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/difficult-task-yes-but-simple-task-no","slug":"difficult-task-yes-but-simple-task-no","title":"Difficult Task Yes but Simple Task No: Unveiling the Laziness in Multimodal LLMs","date":"2024-10-15","arxiv_id":"2410.11437","repositories_listed":1,"syntology":null},{"url":"/paper/rulerag-rule-guided-retrieval-augmented","slug":"rulerag-rule-guided-retrieval-augmented","title":"RuleRAG: Rule-guided retrieval-augmented generation with language models for question answering","date":"2024-10-15","arxiv_id":"2410.22353","repositories_listed":1,"syntology":null},{"url":"/paper/videgothink-assessing-egocentric-video","slug":"videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","arxiv_id":"2410.11623","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videgothink-assessing-egocentric-video#ran","syntology_url":"https://syntology.ai/paper/2410.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11623"}},"official":null}},{"url":"/paper/easyrag-efficient-retrieval-augmented","slug":"easyrag-efficient-retrieval-augmented","title":"EasyRAG: Efficient Retrieval-Augmented Generation Framework for Automated Network Operations","date":"2024-10-14","arxiv_id":"2410.10315","repositories_listed":1,"syntology":null},{"url":"/paper/free-video-llm-prompt-guided-visual","slug":"free-video-llm-prompt-guided-visual","title":"Free Video-LLM: Prompt-guided Visual Perception for Efficient Training-free Video LLMs","date":"2024-10-14","arxiv_id":"2410.10441","repositories_listed":1,"syntology":null},{"url":"/paper/kblam-knowledge-base-augmented-language-model","slug":"kblam-knowledge-base-augmented-language-model","title":"KBLaM: Knowledge Base augmented Language Model","date":"2024-10-14","arxiv_id":"2410.10450","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/kblam-knowledge-base-augmented-language-model#ran","syntology_url":"https://syntology.ai/paper/2410.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10450"}},"official":{"repos":["microsoft/KBLaM"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/longmemeval-benchmarking-chat-assistants-on","slug":"longmemeval-benchmarking-chat-assistants-on","title":"LongMemEval: Benchmarking Chat Assistants on Long-Term Interactive Memory","date":"2024-10-14","arxiv_id":"2410.10813","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/longmemeval-benchmarking-chat-assistants-on#ran","syntology_url":"https://syntology.ai/paper/2410.10813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10813"}},"official":{"repos":["xiaowu0162/longmemeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/quite-quantifying-uncertainty-in-natural","slug":"quite-quantifying-uncertainty-in-natural","title":"QUITE: Quantifying Uncertainty in Natural Language Text in Bayesian Reasoning Scenarios","date":"2024-10-14","arxiv_id":"2410.10449","repositories_listed":1,"syntology":null},{"url":"/paper/sensorllm-aligning-large-language-models-with","slug":"sensorllm-aligning-large-language-models-with","title":"SensorLLM: Human-Intuitive Alignment of Multivariate Sensor Data with LLMs for Activity Recognition","date":"2024-10-14","arxiv_id":"2410.10624","repositories_listed":1,"syntology":null},{"url":"/paper/temporalbench-benchmarking-fine-grained","slug":"temporalbench-benchmarking-fine-grained","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","date":"2024-10-14","arxiv_id":"2410.10818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporalbench-benchmarking-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.10818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10818"}},"official":{"repos":["mu-cai/TemporalBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/will-llms-replace-the-encoder-only-models-in","slug":"will-llms-replace-the-encoder-only-models-in","title":"Will LLMs Replace the Encoder-Only Models in Temporal Relation Classification?","date":"2024-10-14","arxiv_id":"2410.10476","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/will-llms-replace-the-encoder-only-models-in#ran","syntology_url":"https://syntology.ai/paper/2410.10476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10476"}},"official":{"repos":["brownfortress/llms-trc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/declarative-knowledge-distillation-from-large","slug":"declarative-knowledge-distillation-from-large","title":"Declarative Knowledge Distillation from Large Language Models for Visual Question Answering Datasets","date":"2024-10-12","arxiv_id":"2410.09428","repositories_listed":1,"syntology":null},{"url":"/paper/multi-granularity-contrastive-cross-modal","slug":"multi-granularity-contrastive-cross-modal","title":"Multi-granularity Contrastive Cross-modal Collaborative Generation for End-to-End Long-term Video Question Answering","date":"2024-10-12","arxiv_id":"2410.09379","repositories_listed":1,"syntology":null},{"url":"/paper/quebec-automobile-insurance-question","slug":"quebec-automobile-insurance-question","title":"Quebec Automobile Insurance Question-Answering With Retrieval-Augmented Generation","date":"2024-10-12","arxiv_id":"2410.09623","repositories_listed":1,"syntology":null},{"url":"/paper/skipping-computations-in-multimodal-llms","slug":"skipping-computations-in-multimodal-llms","title":"Skipping Computations in Multimodal LLMs","date":"2024-10-12","arxiv_id":"2410.09454","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-knowledge-ingestion-towards","slug":"synthetic-knowledge-ingestion-towards","title":"Synthetic Knowledge Ingestion: Towards Knowledge Refinement and Injection for Enhancing Large Language Models","date":"2024-10-12","arxiv_id":"2410.09629","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/synthetic-knowledge-ingestion-towards#ran","syntology_url":"https://syntology.ai/paper/2410.09629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09629"}},"official":{"repos":["intuit-ai-research/knowledge-infused-ai"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-commonsense-reasoning-over-machine","slug":"zero-shot-commonsense-reasoning-over-machine","title":"Zero-shot Commonsense Reasoning over Machine Imagination","date":"2024-10-12","arxiv_id":"2410.09329","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-multimodal-evaluation-with-flexible","slug":"dynamic-multimodal-evaluation-with-flexible","title":"Dynamic Multimodal Evaluation with Flexible Complexity by Vision-Language Bootstrapping","date":"2024-10-11","arxiv_id":"2410.08695","repositories_listed":1,"syntology":null},{"url":"/paper/generation-with-dynamic-vocabulary","slug":"generation-with-dynamic-vocabulary","title":"Generation with Dynamic Vocabulary","date":"2024-10-11","arxiv_id":"2410.08481","repositories_listed":1,"syntology":null},{"url":"/paper/medmobile-a-mobile-sized-language-model-with","slug":"medmobile-a-mobile-sized-language-model-with","title":"MedMobile: A mobile-sized language model with expert-level clinical capabilities","date":"2024-10-11","arxiv_id":"2410.09019","repositories_listed":1,"syntology":null},{"url":"/paper/retriever-and-memory-towards-adaptive-note","slug":"retriever-and-memory-towards-adaptive-note","title":"Retriever-and-Memory: Towards Adaptive Note-Enhanced Retrieval-Augmented Generation","date":"2024-10-11","arxiv_id":"2410.08821","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retriever-and-memory-towards-adaptive-note#ran","syntology_url":"https://syntology.ai/paper/2410.08821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08821"}},"official":{"repos":["thunlp/adaptive-note"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sportu-a-comprehensive-sports-understanding","slug":"sportu-a-comprehensive-sports-understanding","title":"SPORTU: A Comprehensive Sports Understanding Benchmark for Multimodal Large Language Models","date":"2024-10-11","arxiv_id":"2410.08474","repositories_listed":1,"syntology":null},{"url":"/paper/accept-adaptive-codebook-for-composite-and","slug":"accept-adaptive-codebook-for-composite-and","title":"ACCEPT: Adaptive Codebook for Composite and Efficient Prompt Tuning","date":"2024-10-10","arxiv_id":"2410.12847","repositories_listed":1,"syntology":null},{"url":"/paper/accurate-and-regret-aware-numerical-problem","slug":"accurate-and-regret-aware-numerical-problem","title":"Accurate and Regret-aware Numerical Problem Solver for Tabular Question Answering","date":"2024-10-10","arxiv_id":"2410.12846","repositories_listed":1,"syntology":null},{"url":"/paper/stableprompt-automatic-prompt-tuning-using","slug":"stableprompt-automatic-prompt-tuning-using","title":"StablePrompt: Automatic Prompt Tuning using Reinforcement Learning for Large Language Models","date":"2024-10-10","arxiv_id":"2410.07652","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/stableprompt-automatic-prompt-tuning-using#ran","syntology_url":"https://syntology.ai/paper/2410.07652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07652"}},"official":{"repos":["kmc0207/Stableprompt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"6dd86a6d4fea010ffe52e6e66e80aa1f1d986421e16fa023d859d536ee9416d2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}