{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/15","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":109,"rows_per_page":100,"rows":[1401,1500],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/14","next":"/task/question-answering/papers/16","papers":[{"url":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset","slug":"kvasir-vqa-a-text-image-pair-gi-tract-dataset","title":"Kvasir-VQA: A Text-Image Pair GI Tract Dataset","date":"2024-09-02","arxiv_id":"2409.01437","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.01437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01437"}},"official":{"repos":["simula/Kvasir-VQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/pairing-analogy-augmented-generation-with","slug":"pairing-analogy-augmented-generation-with","title":"Pairing Analogy-Augmented Generation with Procedural Memory for Procedural Q&A","date":"2024-09-02","arxiv_id":"2409.01344","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-the-power-of-semi-structured","slug":"harnessing-the-power-of-semi-structured","title":"Harnessing the Power of Semi-Structured Knowledge and LLMs with Triplet-Based Prefiltering for Question Answering","date":"2024-09-01","arxiv_id":"2409.00861","repositories_listed":1,"syntology":null},{"url":"/paper/wikicausal-corpus-and-evaluation-framework","slug":"wikicausal-corpus-and-evaluation-framework","title":"WikiCausal: Corpus and Evaluation Framework for Causal Knowledge Graph Construction","date":"2024-08-31","arxiv_id":"2409.00331","repositories_listed":1,"syntology":null},{"url":"/paper/gradbias-unveiling-word-influence-on-bias-in","slug":"gradbias-unveiling-word-influence-on-bias-in","title":"GradBias: Unveiling Word Influence on Bias in Text-to-Image Generative Models","date":"2024-08-29","arxiv_id":"2408.16700","repositories_listed":1,"syntology":null},{"url":"/paper/grounded-multi-hop-videoqa-in-long-form","slug":"grounded-multi-hop-videoqa-in-long-form","title":"Grounded Multi-Hop VideoQA in Long-Form Egocentric Videos","date":"2024-08-26","arxiv_id":"2408.14469","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-multi-hop-videoqa-in-long-form#ran","syntology_url":"https://syntology.ai/paper/2408.14469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14469"}},"official":null}},{"url":"/paper/question-answering-system-of-bridge-design","slug":"question-answering-system-of-bridge-design","title":"Question answering system of bridge design specification based on large language model","date":"2024-08-26","arxiv_id":"2408.13282","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-attribute-comprehension-in-large","slug":"evaluating-attribute-comprehension-in-large","title":"Evaluating Attribute Comprehension in Large Vision-Language Models","date":"2024-08-25","arxiv_id":"2408.13898","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-multivariate-time-series-anomaly","slug":"efficient-multivariate-time-series-anomaly","title":"Enhanced Fine-Tuning of Lightweight Domain-Specific Q&A Model Based on Large Language Models","date":"2024-08-22","arxiv_id":"2408.12247","repositories_listed":1,"syntology":null},{"url":"/paper/roundtable-leveraging-dynamic-schema-and","slug":"roundtable-leveraging-dynamic-schema-and","title":"RoundTable: Leveraging Dynamic Schema and Contextual Autocomplete for Enhanced Query Precision in Tabular Question Answering","date":"2024-08-22","arxiv_id":"2408.12369","repositories_listed":1,"syntology":null},{"url":"/paper/show-o-one-single-transformer-to-unify","slug":"show-o-one-single-transformer-to-unify","title":"Show-o: One Single Transformer to Unify Multimodal Understanding and Generation","date":"2024-08-22","arxiv_id":"2408.12528","repositories_listed":1,"syntology":null},{"url":"/paper/towards-evaluating-and-building-versatile","slug":"towards-evaluating-and-building-versatile","title":"Towards Evaluating and Building Versatile Large Language Models for Medicine","date":"2024-08-22","arxiv_id":"2408.12547","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-evaluating-and-building-versatile#ran","syntology_url":"https://syntology.ai/paper/2408.12547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12547"}},"official":{"repos":["magic-ai4med/meds-ins"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ancient-wisdom-modern-tools-exploring","slug":"ancient-wisdom-modern-tools-exploring","title":"Ancient Wisdom, Modern Tools: Exploring Retrieval-Augmented LLMs for Ancient Indian Philosophy","date":"2024-08-21","arxiv_id":"2408.11903","repositories_listed":1,"syntology":null},{"url":"/paper/clumo-cluster-based-modality-fusion-prompt","slug":"clumo-cluster-based-modality-fusion-prompt","title":"CluMo: Cluster-based Modality Fusion Prompt for Continual Learning in Visual Question Answering","date":"2024-08-21","arxiv_id":"2408.11742","repositories_listed":1,"syntology":null},{"url":"/paper/differentiating-choices-via-commonality-for","slug":"differentiating-choices-via-commonality-for","title":"Differentiating Choices via Commonality for Multiple-Choice Question Answering","date":"2024-08-21","arxiv_id":"2408.11554","repositories_listed":1,"syntology":null},{"url":"/paper/doctabqa-answering-questions-from-long","slug":"doctabqa-answering-questions-from-long","title":"DocTabQA: Answering Questions from Long Documents Using Tables","date":"2024-08-21","arxiv_id":"2408.11490","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-fine-tuned-retrieval-augmented","slug":"leveraging-fine-tuned-retrieval-augmented","title":"Leveraging Fine-Tuned Retrieval-Augmented Generation with Long-Context Support: For 3GPP Standards","date":"2024-08-21","arxiv_id":"2408.11775","repositories_listed":1,"syntology":null},{"url":"/paper/rcone-rough-cone-embedding-for-multi-hop","slug":"rcone-rough-cone-embedding-for-multi-hop","title":"RConE: Rough Cone Embedding for Multi-Hop Logical Query Answering on Multi-Modal Knowledge Graphs","date":"2024-08-21","arxiv_id":"2408.11526","repositories_listed":1,"syntology":null},{"url":"/paper/colbert-retrieval-and-ensemble-response","slug":"colbert-retrieval-and-ensemble-response","title":"ColBERT Retrieval and Ensemble Response Scoring for Language Model Question Answering","date":"2024-08-20","arxiv_id":"2408.10808","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-retrieval-augmented-generation","slug":"hierarchical-retrieval-augmented-generation","title":"Hierarchical Retrieval-Augmented Generation Model with Rethink for Multi-hop Question Answering","date":"2024-08-20","arxiv_id":"2408.11875","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hierarchical-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2408.11875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11875"}},"official":{"repos":["2282588541a/hirag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multilingual-non-factoid-question-answering","slug":"multilingual-non-factoid-question-answering","title":"Multilingual Non-Factoid Question Answering with Answer Paragraph Selection","date":"2024-08-20","arxiv_id":"2408.10604","repositories_listed":1,"syntology":null},{"url":"/paper/putting-people-in-llms-shoes-generating","slug":"putting-people-in-llms-shoes-generating","title":"Putting People in LLMs' Shoes: Generating Better Answers via Question Rewriter","date":"2024-08-20","arxiv_id":"2408.10573","repositories_listed":1,"syntology":null},{"url":"/paper/v-roast-a-new-dataset-for-visual-road","slug":"v-roast-a-new-dataset-for-visual-road","title":"V-RoAst: Visual Road Assessment. Can VLM be a Road Safety Assessor Using the iRAP Standard?","date":"2024-08-20","arxiv_id":"2408.10872","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-needle-in-a-haystack","slug":"multilingual-needle-in-a-haystack","title":"Multilingual Needle in a Haystack: Investigating Long-Context Behavior of Multilingual Large Language Models","date":"2024-08-19","arxiv_id":"2408.10151","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multilingual-needle-in-a-haystack#ran","syntology_url":"https://syntology.ai/paper/2408.10151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10151"}},"official":{"repos":["AmeyHengle/multilingual-needle-in-a-haystack"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ranking-generated-answers-on-the-agreement-of","slug":"ranking-generated-answers-on-the-agreement-of","title":"Ranking Generated Answers: On the Agreement of Retrieval Models with Humans on Consumer Health Questions","date":"2024-08-19","arxiv_id":"2408.09831","repositories_listed":1,"syntology":null},{"url":"/paper/pa-llava-a-large-language-vision-assistant","slug":"pa-llava-a-large-language-vision-assistant","title":"PA-LLaVA: A Large Language-Vision Assistant for Human Pathology Image Understanding","date":"2024-08-18","arxiv_id":"2408.09530","repositories_listed":1,"syntology":null},{"url":"/paper/fedmeki-a-benchmark-for-scaling-medical","slug":"fedmeki-a-benchmark-for-scaling-medical","title":"FEDMEKI: A Benchmark for Scaling Medical Foundation Models via Federated Knowledge Injection","date":"2024-08-17","arxiv_id":"2408.09227","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedmeki-a-benchmark-for-scaling-medical#ran","syntology_url":"https://syntology.ai/paper/2408.09227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09227"}},"official":{"repos":["psudslab/FEDMEKI"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/a-survey-on-benchmarks-of-multimodal-large","slug":"a-survey-on-benchmarks-of-multimodal-large","title":"A Survey on Benchmarks of Multimodal Large Language Models","date":"2024-08-16","arxiv_id":"2408.08632","repositories_listed":1,"syntology":null},{"url":"/paper/med-pmc-medical-personalized-multi-modal","slug":"med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","arxiv_id":"2408.08693","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/med-pmc-medical-personalized-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2408.08693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08693"}},"official":{"repos":["liuhc0428/med-pmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/realmedqa-a-pilot-biomedical-question","slug":"realmedqa-a-pilot-biomedical-question","title":"RealMedQA: A pilot biomedical question answering dataset containing realistic clinical questions","date":"2024-08-16","arxiv_id":"2408.08624","repositories_listed":1,"syntology":null},{"url":"/paper/visual-agents-as-fast-and-slow-thinkers","slug":"visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","arxiv_id":"2408.08862","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-agents-as-fast-and-slow-thinkers#ran","syntology_url":"https://syntology.ai/paper/2408.08862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08862"}},"official":{"repos":["guangyans/sys2-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iiu-independent-inference-units-for-knowledge","slug":"iiu-independent-inference-units-for-knowledge","title":"IIU: Independent Inference Units for Knowledge-based Visual Question Answering","date":"2024-08-15","arxiv_id":"2408.07989","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-and-understanding-bridging-vision-with","slug":"seeing-and-understanding-bridging-vision-with","title":"ChemVLM: Exploring the Power of Multimodal Large Language Models in Chemistry Area","date":"2024-08-14","arxiv_id":"2408.07246","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seeing-and-understanding-bridging-vision-with#ran","syntology_url":"https://syntology.ai/paper/2408.07246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07246"}},"official":{"repos":["AI4Chem/ChemVlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/maqa-evaluating-uncertainty-quantification-in","slug":"maqa-evaluating-uncertainty-quantification-in","title":"MAQA: Evaluating Uncertainty Quantification in LLMs Regarding Data Uncertainty","date":"2024-08-13","arxiv_id":"2408.06816","repositories_listed":1,"syntology":null},{"url":"/paper/fastfid-improve-inference-efficiency-of-open","slug":"fastfid-improve-inference-efficiency-of-open","title":"FastFiD: Improve Inference Efficiency of Open Domain Question Answering via Sentence Selection","date":"2024-08-12","arxiv_id":"2408.06333","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/fastfid-improve-inference-efficiency-of-open#ran","syntology_url":"https://syntology.ai/paper/2408.06333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06333"}},"official":{"repos":["thunlp/fastfid"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/context-driven-index-trimming-a-data-quality","slug":"context-driven-index-trimming-a-data-quality","title":"Context-Driven Index Trimming: A Data Quality Perspective to Enhancing Precision of RALMs","date":"2024-08-10","arxiv_id":"2408.05524","repositories_listed":1,"syntology":null},{"url":"/paper/msg-chart-multimodal-scene-graph-for-chartqa","slug":"msg-chart-multimodal-scene-graph-for-chartqa","title":"MSG-Chart: Multimodal Scene Graph for ChartQA","date":"2024-08-09","arxiv_id":"2408.04852","repositories_listed":1,"syntology":null},{"url":"/paper/surgical-vqla-adversarial-contrastive","slug":"surgical-vqla-adversarial-contrastive","title":"Surgical-VQLA++: Adversarial Contrastive Learning for Calibrated Robust Visual Question-Localized Answering in Robotic Surgery","date":"2024-08-09","arxiv_id":"2408.04958","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgical-vqla-adversarial-contrastive#ran","syntology_url":"https://syntology.ai/paper/2408.04958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04958"}},"official":{"repos":["longbai1006/surgical-vqlaplus"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficientrag-efficient-retriever-for-multi","slug":"efficientrag-efficient-retriever-for-multi","title":"EfficientRAG: Efficient Retriever for Multi-Hop Question Answering","date":"2024-08-08","arxiv_id":"2408.04259","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficientrag-efficient-retriever-for-multi#ran","syntology_url":"https://syntology.ai/paper/2408.04259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04259"}},"official":{"repos":["nil-zhuang/efficientrag-official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unlocking-the-non-native-language-context","slug":"unlocking-the-non-native-language-context","title":"NatLan: Native Language Prompting Facilitates Knowledge Elicitation Through Language Trigger Provision and Domain Trigger Retention","date":"2024-08-07","arxiv_id":"2408.03544","repositories_listed":1,"syntology":null},{"url":"/paper/2408-03043","slug":"2408-03043","title":"Targeted Visual Prompting for Medical Visual Question Answering","date":"2024-08-06","arxiv_id":"2408.03043","repositories_listed":1,"syntology":null},{"url":"/paper/2408-03094","slug":"2408-03094","title":"500xCompressor: Generalized Prompt Compression for Large Language Models","date":"2024-08-06","arxiv_id":"2408.03094","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-03094#ran","syntology_url":"https://syntology.ai/paper/2408.03094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03094"}},"official":{"repos":["ZongqianLi/500xCompressor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citekit-a-modular-toolkit-for-large-language","slug":"citekit-a-modular-toolkit-for-large-language","title":"Citekit: A Modular Toolkit for Large Language Model Citation Generation","date":"2024-08-06","arxiv_id":"2408.04662","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/citekit-a-modular-toolkit-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2408.04662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04662"}},"official":{"repos":["sjj1017/citekit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gmai-mmbench-a-comprehensive-multimodal","slug":"gmai-mmbench-a-comprehensive-multimodal","title":"GMAI-MMBench: A Comprehensive Multimodal Evaluation Benchmark Towards General Medical AI","date":"2024-08-06","arxiv_id":"2408.03361","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02337","slug":"2408-02337","title":"Developing PUGG for Polish: A Modern Approach to KBQA, MRC, and IR Dataset Construction","date":"2024-08-05","arxiv_id":"2408.02337","repositories_listed":1,"syntology":null},{"url":"/paper/xmainframe-a-large-language-model-for","slug":"xmainframe-a-large-language-model-for","title":"XMainframe: A Large Language Model for Mainframe Modernization","date":"2024-08-05","arxiv_id":"2408.04660","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01933","slug":"2408-01933","title":"DiReCT: Diagnostic Reasoning for Clinical Notes via Large Language Models","date":"2024-08-04","arxiv_id":"2408.01933","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-01933#ran","syntology_url":"https://syntology.ai/paper/2408.01933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01933"}},"official":{"repos":["wbw520/direct"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01262","slug":"2408-01262","title":"RAGEval: Scenario Specific RAG Evaluation Dataset Generation Framework","date":"2024-08-02","arxiv_id":"2408.01262","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01419","slug":"2408-01419","title":"DebateQA: Evaluating Question Answering on Debatable Knowledge","date":"2024-08-02","arxiv_id":"2408.01419","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-01419#ran","syntology_url":"https://syntology.ai/paper/2408.01419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01419"}},"official":{"repos":["pillowsofwind/debateqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00300","slug":"2408-00300","title":"Towards Flexible Evaluation for Generative Visual Question Answering","date":"2024-08-01","arxiv_id":"2408.00300","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00727","slug":"2408-00727","title":"Improving Retrieval-Augmented Generation in Medicine with Iterative Follow-up Questions","date":"2024-08-01","arxiv_id":"2408.00727","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tree-of-traversals-a-zero-shot-reasoning","slug":"tree-of-traversals-a-zero-shot-reasoning","title":"Tree-of-Traversals: A Zero-Shot Reasoning Algorithm for Augmenting Black-box Language Models with Knowledge Graphs","date":"2024-07-31","arxiv_id":"2407.21358","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21170","slug":"2407-21170","title":"Decomposed Prompting to Answer Questions on a Course Discussion Board","date":"2024-07-30","arxiv_id":"2407.21170","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-audio-visual-question-answering-via","slug":"boosting-audio-visual-question-answering-via","title":"Boosting Audio Visual Question Answering via Key Semantic-Aware Cues","date":"2024-07-30","arxiv_id":"2407.20693","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/boosting-audio-visual-question-answering-via#ran","syntology_url":"https://syntology.ai/paper/2407.20693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20693"}},"official":{"repos":["gewu-lab/tspm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dygkt-dynamic-graph-learning-for-knowledge","slug":"dygkt-dynamic-graph-learning-for-knowledge","title":"DyGKT: Dynamic Graph Learning for Knowledge Tracing","date":"2024-07-30","arxiv_id":"2407.20824","repositories_listed":1,"syntology":null},{"url":"/paper/from-feature-importance-to-natural-language","slug":"from-feature-importance-to-natural-language","title":"From Feature Importance to Natural Language Explanations Using LLMs with RAG","date":"2024-07-30","arxiv_id":"2407.20990","repositories_listed":1,"syntology":null},{"url":"/paper/synthvlm-high-efficiency-and-high-quality","slug":"synthvlm-high-efficiency-and-high-quality","title":"SynthVLM: High-Efficiency and High-Quality Synthetic Data for Vision Language Models","date":"2024-07-30","arxiv_id":"2407.20756","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-multimodal-large-language-models-in","slug":"advancing-multimodal-large-language-models-in","title":"Advancing Multimodal Large Language Models in Chart Question Answering with Visualization-Referenced Instruction Tuning","date":"2024-07-29","arxiv_id":"2407.20174","repositories_listed":1,"syntology":null},{"url":"/paper/prometheus-chatbot-knowledge-graph","slug":"prometheus-chatbot-knowledge-graph","title":"Prometheus Chatbot: Knowledge Graph Collaborative Large Language Model for Computer Components Recommendation","date":"2024-07-29","arxiv_id":"2407.19643","repositories_listed":1,"syntology":null},{"url":"/paper/a-role-specific-guided-large-language-model","slug":"a-role-specific-guided-large-language-model","title":"A Role-specific Guided Large Language Model for Ophthalmic Consultation Based on Stylistic Differentiation","date":"2024-07-26","arxiv_id":"2407.18483","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-generalizable-pathology-foundation","slug":"towards-a-generalizable-pathology-foundation","title":"Towards A Generalizable Pathology Foundation Model via Unified Knowledge Distillation","date":"2024-07-26","arxiv_id":"2407.18449","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-a-generalizable-pathology-foundation#ran","syntology_url":"https://syntology.ai/paper/2407.18449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18449"}},"official":null}},{"url":"/paper/i-could-ve-asked-that-reformulating","slug":"i-could-ve-asked-that-reformulating","title":"I Could've Asked That: Reformulating Unanswerable Questions","date":"2024-07-24","arxiv_id":"2407.17469","repositories_listed":1,"syntology":null},{"url":"/paper/educating-llms-like-human-students-structure","slug":"educating-llms-like-human-students-structure","title":"Structure-aware Domain Knowledge Injection for Large Language Models","date":"2024-07-23","arxiv_id":"2407.16724","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-llm-s-cognition-via-structurization","slug":"enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","arxiv_id":"2407.16434","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-llm-s-cognition-via-structurization#ran","syntology_url":"https://syntology.ai/paper/2407.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16434"}},"official":{"repos":["alibaba/struxgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-trimodal-relation-for-avqa-with","slug":"learning-trimodal-relation-for-avqa-with","title":"Learning Trimodal Relation for AVQA with Missing Modality","date":"2024-07-23","arxiv_id":"2407.16171","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":14,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":4,"n_violates":0,"n_no_contract":14,"n_pointer_only":3,"phrase":"20 ran (of which 14 constructed an object rather than computing a result; 18 with no instrument failure: 4 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-trimodal-relation-for-avqa-with#ran","syntology_url":"https://syntology.ai/paper/2407.16171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16171"}},"official":{"repos":["visualaikhu/missing-avqa"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/efficient-retrieval-with-learned-similarities","slug":"efficient-retrieval-with-learned-similarities","title":"Retrieval with Learned Similarities","date":"2024-07-22","arxiv_id":"2407.15462","repositories_listed":1,"syntology":null},{"url":"/paper/haloquest-a-visual-hallucination-dataset-for","slug":"haloquest-a-visual-hallucination-dataset-for","title":"HaloQuest: A Visual Hallucination Dataset for Advancing Multimodal Reasoning","date":"2024-07-22","arxiv_id":"2407.15680","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-acquisition-disentanglement-for","slug":"knowledge-acquisition-disentanglement-for","title":"Knowledge Acquisition Disentanglement for Knowledge-based Visual Question Answering with Large Language Models","date":"2024-07-22","arxiv_id":"2407.15346","repositories_listed":1,"syntology":null},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mminstruct-a-high-quality-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2407.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15838"}},"official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/odyssey-empowering-agents-with-open-world","slug":"odyssey-empowering-agents-with-open-world","title":"Odyssey: Empowering Minecraft Agents with Open-World Skills","date":"2024-07-22","arxiv_id":"2407.15325","repositories_listed":1,"syntology":null},{"url":"/paper/radiorag-factual-large-language-models-for","slug":"radiorag-factual-large-language-models-for","title":"RadioRAG: Factual large language models for enhanced diagnostics in radiology using online retrieval augmented generation","date":"2024-07-22","arxiv_id":"2407.15621","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-language-models-as-risk-scores","slug":"evaluating-language-models-as-risk-scores","title":"Evaluating language models as risk scores","date":"2024-07-19","arxiv_id":"2407.14614","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/evaluating-language-models-as-risk-scores#ran","syntology_url":"https://syntology.ai/paper/2407.14614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14614"}},"official":{"repos":["socialfoundations/folktexts"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/rag-qa-arena-evaluating-domain-robustness-for","slug":"rag-qa-arena-evaluating-domain-robustness-for","title":"RAG-QA Arena: Evaluating Domain Robustness for Long-form Retrieval Augmented Question Answering","date":"2024-07-19","arxiv_id":"2407.13998","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rag-qa-arena-evaluating-domain-robustness-for#ran","syntology_url":"https://syntology.ai/paper/2407.13998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13998"}},"official":{"repos":["awslabs/rag-qa-arena"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quiil-at-t3-challenge-towards-automation-in","slug":"quiil-at-t3-challenge-towards-automation-in","title":"QuIIL at T3 challenge: Towards Automation in Life-Saving Intervention Procedures from First-Person View","date":"2024-07-18","arxiv_id":"2407.13216","repositories_listed":1,"syntology":null},{"url":"/paper/visual-haystacks-answering-harder-questions","slug":"visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","date":"2024-07-18","arxiv_id":"2407.13766","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/visual-haystacks-answering-harder-questions#ran","syntology_url":"https://syntology.ai/paper/2407.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13766"}},"official":{"repos":["visual-haystacks/vhs_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/proctag-process-tagging-for-assessing-the","slug":"proctag-process-tagging-for-assessing-the","title":"ProcTag: Process Tagging for Assessing the Efficacy of Document Instruction Data","date":"2024-07-17","arxiv_id":"2407.12358","repositories_listed":1,"syntology":null},{"url":"/paper/search-engines-llms-or-both-evaluating","slug":"search-engines-llms-or-both-evaluating","title":"Evaluating Search Engines and Large Language Models for Answering Health Questions","date":"2024-07-17","arxiv_id":"2407.12468","repositories_listed":1,"syntology":null},{"url":"/paper/turkishmmlu-measuring-massive-multitask","slug":"turkishmmlu-measuring-massive-multitask","title":"TurkishMMLU: Measuring Massive Multitask Language Understanding in Turkish","date":"2024-07-17","arxiv_id":"2407.12402","repositories_listed":1,"syntology":null},{"url":"/paper/better-rag-using-relevant-information-gain","slug":"better-rag-using-relevant-information-gain","title":"Better RAG using Relevant Information Gain","date":"2024-07-16","arxiv_id":"2407.12101","repositories_listed":1,"syntology":null},{"url":"/paper/bright-a-realistic-and-challenging-benchmark","slug":"bright-a-realistic-and-challenging-benchmark","title":"BRIGHT: A Realistic and Challenging Benchmark for Reasoning-Intensive Retrieval","date":"2024-07-16","arxiv_id":"2407.12883","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-hallucination-detection-and-1","slug":"fine-grained-hallucination-detection-and-1","title":"Localizing and Mitigating Errors in Long-form Question Answering","date":"2024-07-16","arxiv_id":"2407.11930","repositories_listed":1,"syntology":null},{"url":"/paper/scientific-qa-system-with-verifiable-answers","slug":"scientific-qa-system-with-verifiable-answers","title":"Scientific QA System with Verifiable Answers","date":"2024-07-16","arxiv_id":"2407.11485","repositories_listed":1,"syntology":null},{"url":"/paper/video-language-alignment-pre-training-via","slug":"video-language-alignment-pre-training-via","title":"Video-Language Alignment via Spatio-Temporal Graph Transformer","date":"2024-07-16","arxiv_id":"2407.11677","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-large-language-models-with-fmeval","slug":"evaluating-large-language-models-with-fmeval","title":"Evaluating Large Language Models with fmeval","date":"2024-07-15","arxiv_id":"2407.12872","repositories_listed":1,"syntology":null},{"url":"/paper/graphusion-leveraging-large-language-models","slug":"graphusion-leveraging-large-language-models","title":"Graphusion: Leveraging Large Language Models for Scientific Knowledge Graph Fusion and Construction in NLP Education","date":"2024-07-15","arxiv_id":"2407.10794","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-mixgr-enhancing-retriever","slug":"texttt-mixgr-enhancing-retriever","title":"$\\texttt{MixGR}$: Enhancing Retriever Generalization for Scientific Domain through Complementary Granularity","date":"2024-07-15","arxiv_id":"2407.10691","repositories_listed":1,"syntology":null},{"url":"/paper/lost-and-found-overcoming-detector-failures","slug":"lost-and-found-overcoming-detector-failures","title":"Lost and Found: Overcoming Detector Failures in Online Multi-Object Tracking","date":"2024-07-14","arxiv_id":"2407.10151","repositories_listed":1,"syntology":null},{"url":"/paper/iot-lm-large-multisensory-language-models-for","slug":"iot-lm-large-multisensory-language-models-for","title":"IoT-LM: Large Multisensory Language Models for the Internet of Things","date":"2024-07-13","arxiv_id":"2407.09801","repositories_listed":1,"syntology":null},{"url":"/paper/a-mathematical-framework-a-taxonomy-of","slug":"a-mathematical-framework-a-taxonomy-of","title":"A Mathematical Framework, a Taxonomy of Modeling Paradigms, and a Suite of Learning Techniques for Neural-Symbolic Systems","date":"2024-07-12","arxiv_id":"2407.09693","repositories_listed":1,"syntology":null},{"url":"/paper/compact-compressing-retrieved-documents","slug":"compact-compressing-retrieved-documents","title":"CompAct: Compressing Retrieved Documents Actively for Question Answering","date":"2024-07-12","arxiv_id":"2407.09014","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compact-compressing-retrieved-documents#ran","syntology_url":"https://syntology.ai/paper/2407.09014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09014"}},"official":{"repos":["dmis-lab/compact"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gofa-a-generative-one-for-all-model-for-joint","slug":"gofa-a-generative-one-for-all-model-for-joint","title":"GOFA: A Generative One-For-All Model for Joint Graph Language Modeling","date":"2024-07-12","arxiv_id":"2407.09709","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gofa-a-generative-one-for-all-model-for-joint#ran","syntology_url":"https://syntology.ai/paper/2407.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09709"}},"official":{"repos":["jiaruifeng/gofa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/one-stone-four-birds-a-comprehensive-solution","slug":"one-stone-four-birds-a-comprehensive-solution","title":"One Stone, Four Birds: A Comprehensive Solution for QA System Using Supervised Contrastive Learning","date":"2024-07-12","arxiv_id":"2407.09011","repositories_listed":1,"syntology":null},{"url":"/paper/personarag-enhancing-retrieval-augmented","slug":"personarag-enhancing-retrieval-augmented","title":"PersonaRAG: Enhancing Retrieval-Augmented Generation Systems with User-Centric Agents","date":"2024-07-12","arxiv_id":"2407.09394","repositories_listed":1,"syntology":null},{"url":"/paper/spiqa-a-dataset-for-multimodal-question","slug":"spiqa-a-dataset-for-multimodal-question","title":"SPIQA: A Dataset for Multimodal Question Answering on Scientific Papers","date":"2024-07-12","arxiv_id":"2407.09413","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiqa-a-dataset-for-multimodal-question#ran","syntology_url":"https://syntology.ai/paper/2407.09413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09413"}},"official":{"repos":["google/spiqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autobencher-creating-salient-novel-difficult","slug":"autobencher-creating-salient-novel-difficult","title":"AutoBencher: Creating Salient, Novel, Difficult Datasets for Language Models","date":"2024-07-11","arxiv_id":"2407.08351","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobencher-creating-salient-novel-difficult#ran","syntology_url":"https://syntology.ai/paper/2407.08351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08351"}},"official":{"repos":["XiangLi1999/AutoBencher"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-surgery-modulating-llm-s-behavior-via","slug":"model-surgery-modulating-llm-s-behavior-via","title":"Model Surgery: Modulating LLM's Behavior Via Simple Parameter Editing","date":"2024-07-11","arxiv_id":"2407.08770","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-surgery-modulating-llm-s-behavior-via#ran","syntology_url":"https://syntology.ai/paper/2407.08770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08770"}},"official":{"repos":["lucywang720/model-surgery"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-extraction-of-disease-risk-factors","slug":"automatic-extraction-of-disease-risk-factors","title":"Automatic Extraction of Disease Risk Factors from Medical Publications","date":"2024-07-10","arxiv_id":"2407.07373","repositories_listed":1,"syntology":null},{"url":"/paper/ida-vlm-towards-movie-understanding-via-id","slug":"ida-vlm-towards-movie-understanding-via-id","title":"IDA-VLM: Towards Movie Understanding via ID-Aware Large Vision-Language Model","date":"2024-07-10","arxiv_id":"2407.07577","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ida-vlm-towards-movie-understanding-via-id#ran","syntology_url":"https://syntology.ai/paper/2407.07577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07577"}},"official":{"repos":["jiyt17/ida-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"4f459a390e7d02fbe2a09a41bd4d0ec9f4f99f5c668a1e2896eec74be35d24a5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}