{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/17","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":109,"rows_per_page":100,"rows":[1601,1700],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/16","next":"/task/question-answering/papers/18","papers":[{"url":"/paper/mmscan-a-multi-modal-3d-scene-dataset-with","slug":"mmscan-a-multi-modal-3d-scene-dataset-with","title":"MMScan: A Multi-Modal 3D Scene Dataset with Hierarchical Grounded Language Annotations","date":"2024-06-13","arxiv_id":"2406.09401","repositories_listed":1,"syntology":null},{"url":"/paper/needle-in-a-video-haystack-a-scalable","slug":"needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","arxiv_id":"2406.09367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/needle-in-a-video-haystack-a-scalable#ran","syntology_url":"https://syntology.ai/paper/2406.09367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09367"}},"official":{"repos":["joez17/videoniah"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/no-perspective-no-perception-perspective","slug":"no-perspective-no-perception-perspective","title":"No perspective, no perception!! Perspective-aware Healthcare Answer Summarization","date":"2024-06-13","arxiv_id":"2406.08881","repositories_listed":1,"syntology":null},{"url":"/paper/too-many-frames-not-all-useful-efficient","slug":"too-many-frames-not-all-useful-efficient","title":"Too Many Frames, Not All Useful: Efficient Strategies for Long-Form Video QA","date":"2024-06-13","arxiv_id":"2406.09396","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/too-many-frames-not-all-useful-efficient#ran","syntology_url":"https://syntology.ai/paper/2406.09396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09396"}},"official":{"repos":["jongwoopark7978/LVNet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-multilingual-audio-visual-question","slug":"towards-multilingual-audio-visual-question","title":"Towards Multilingual Audio-Visual Question Answering","date":"2024-06-13","arxiv_id":"2406.09156","repositories_listed":1,"syntology":null},{"url":"/paper/towards-vision-language-geo-foundation-model","slug":"towards-vision-language-geo-foundation-model","title":"Towards Vision-Language Geo-Foundation Model: A Survey","date":"2024-06-13","arxiv_id":"2406.09385","repositories_listed":1,"syntology":null},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yo-llava-your-personalized-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.09400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09400"}},"official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-high-resolution-vision-language","slug":"advancing-high-resolution-vision-language","title":"Advancing High Resolution Vision-Language Models in Biomedicine","date":"2024-06-12","arxiv_id":"2406.09454","repositories_listed":1,"syntology":null},{"url":"/paper/flash-vstream-memory-based-real-time","slug":"flash-vstream-memory-based-real-time","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","date":"2024-06-12","arxiv_id":"2406.08085","repositories_listed":1,"syntology":null},{"url":"/paper/visionllm-v2-an-end-to-end-generalist","slug":"visionllm-v2-an-end-to-end-generalist","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","date":"2024-06-12","arxiv_id":"2406.08394","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-vision-language-contrastive","slug":"benchmarking-vision-language-contrastive","title":"Benchmarking Vision-Language Contrastive Methods for Medical Representation Learning","date":"2024-06-11","arxiv_id":"2406.07450","repositories_listed":1,"syntology":null},{"url":"/paper/dara-decomposition-alignment-reasoning","slug":"dara-decomposition-alignment-reasoning","title":"DARA: Decomposition-Alignment-Reasoning Autonomous Language Agent for Question Answering over Knowledge Graphs","date":"2024-06-11","arxiv_id":"2406.07080","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dara-decomposition-alignment-reasoning#ran","syntology_url":"https://syntology.ai/paper/2406.07080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07080"}},"official":{"repos":["UKPLab/acl2024-DARA"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mbbq-a-dataset-for-cross-lingual-comparison","slug":"mbbq-a-dataset-for-cross-lingual-comparison","title":"MBBQ: A Dataset for Cross-Lingual Comparison of Stereotypes in Generative LLMs","date":"2024-06-11","arxiv_id":"2406.07243","repositories_listed":1,"syntology":null},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scholarly-question-answering-using-large","slug":"scholarly-question-answering-using-large","title":"Scholarly Question Answering using Large Language Models in the NFDI4DataScience Gateway","date":"2024-06-11","arxiv_id":"2406.07257","repositories_listed":1,"syntology":null},{"url":"/paper/situational-awareness-matters-in-3d-vision","slug":"situational-awareness-matters-in-3d-vision","title":"Situational Awareness Matters in 3D Vision Language Reasoning","date":"2024-06-11","arxiv_id":"2406.07544","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/situational-awareness-matters-in-3d-vision#ran","syntology_url":"https://syntology.ai/paper/2406.07544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07544"}},"official":{"repos":["YunzeMan/Situation3D"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/husky-a-unified-open-source-language-agent","slug":"husky-a-unified-open-source-language-agent","title":"Husky: A Unified, Open-Source Language Agent for Multi-Step Reasoning","date":"2024-06-10","arxiv_id":"2406.06469","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/husky-a-unified-open-source-language-agent#ran","syntology_url":"https://syntology.ai/paper/2406.06469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06469"}},"official":{"repos":["agent-husky/husky-v1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/medexqa-medical-question-answering-benchmark","slug":"medexqa-medical-question-answering-benchmark","title":"MedExQA: Medical Question Answering Benchmark with Multiple Explanations","date":"2024-06-10","arxiv_id":"2406.06331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medexqa-medical-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2406.06331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06331"}},"official":{"repos":["knowlab/medexqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-context-compression-efficiently","slug":"recurrent-context-compression-efficiently","title":"Recurrent Context Compression: Efficiently Expanding the Context Window of LLM","date":"2024-06-10","arxiv_id":"2406.06110","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-context-compression-efficiently#ran","syntology_url":"https://syntology.ai/paper/2406.06110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06110"}},"official":{"repos":["WUHU-G/RCC_Transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sciriff-a-resource-to-enhance-language-model","slug":"sciriff-a-resource-to-enhance-language-model","title":"SciRIFF: A Resource to Enhance Language Model Instruction-Following over Scientific Literature","date":"2024-06-10","arxiv_id":"2406.07835","repositories_listed":1,"syntology":null},{"url":"/paper/should-we-fine-tune-or-rag-evaluating","slug":"should-we-fine-tune-or-rag-evaluating","title":"Should We Fine-Tune or RAG? Evaluating Different Techniques to Adapt LLMs for Dialogue","date":"2024-06-10","arxiv_id":"2406.06399","repositories_listed":1,"syntology":null},{"url":"/paper/vcr-visual-caption-restoration","slug":"vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","arxiv_id":"2406.06462","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vcr-visual-caption-restoration#ran","syntology_url":"https://syntology.ai/paper/2406.06462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06462"}},"official":{"repos":["tianyu-z/vcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/re-rag-improving-open-domain-qa-performance","slug":"re-rag-improving-open-domain-qa-performance","title":"RE-RAG: Improving Open-Domain QA Performance and Interpretability with Relevance Estimator in Retrieval-Augmented Generation","date":"2024-06-09","arxiv_id":"2406.05794","repositories_listed":1,"syntology":null},{"url":"/paper/ceret-cost-effective-extrinsic-refinement-for","slug":"ceret-cost-effective-extrinsic-refinement-for","title":"CERET: Cost-Effective Extrinsic Refinement for Text Generation","date":"2024-06-08","arxiv_id":"2406.05588","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-benchmark-for-causal-business","slug":"towards-a-benchmark-for-causal-business","title":"Towards a Benchmark for Causal Business Process Reasoning with LLMs","date":"2024-06-08","arxiv_id":"2406.05506","repositories_listed":1,"syntology":null},{"url":"/paper/complextempqa-a-large-scale-dataset-for","slug":"complextempqa-a-large-scale-dataset-for","title":"ComplexTempQA: A Large-Scale Dataset for Complex Temporal Question Answering","date":"2024-06-07","arxiv_id":"2406.04866","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/complextempqa-a-large-scale-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2406.04866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04866"}},"official":{"repos":["datascienceuibk/complextempqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/composition-vision-language-understanding-via","slug":"composition-vision-language-understanding-via","title":"Composition Vision-Language Understanding via Segment and Depth Anything Model","date":"2024-06-07","arxiv_id":"2406.18591","repositories_listed":1,"syntology":null},{"url":"/paper/corda-context-oriented-decomposition","slug":"corda-context-oriented-decomposition","title":"CorDA: Context-Oriented Decomposition Adaptation of Large Language Models for Task-Aware Parameter-Efficient Fine-tuning","date":"2024-06-07","arxiv_id":"2406.05223","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/corda-context-oriented-decomposition#ran","syntology_url":"https://syntology.ai/paper/2406.05223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05223"}},"official":{"repos":["iboing/corda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/linkq-an-llm-assisted-visual-interface-for","slug":"linkq-an-llm-assisted-visual-interface-for","title":"LinkQ: An LLM-Assisted Visual Interface for Knowledge Graph Question-Answering","date":"2024-06-07","arxiv_id":"2406.06621","repositories_listed":1,"syntology":null},{"url":"/paper/on-subjective-uncertainty-quantification-and","slug":"on-subjective-uncertainty-quantification-and","title":"On Subjective Uncertainty Quantification and Calibration in Natural Language Generation","date":"2024-06-07","arxiv_id":"2406.05213","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-subjective-uncertainty-quantification-and#ran","syntology_url":"https://syntology.ai/paper/2406.05213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05213"}},"official":{"repos":["meta-inf/suq-nlg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fairytaleqa-translated-enabling-educational","slug":"fairytaleqa-translated-enabling-educational","title":"FairytaleQA Translated: Enabling Educational Question and Answer Generation in Less-Resourced Languages","date":"2024-06-06","arxiv_id":"2406.04233","repositories_listed":1,"syntology":null},{"url":"/paper/m-qalm-a-benchmark-to-assess-clinical-reading","slug":"m-qalm-a-benchmark-to-assess-clinical-reading","title":"M-QALM: A Benchmark to Assess Clinical Reading Comprehension and Knowledge Recall in Large Language Models via Question Answering","date":"2024-06-06","arxiv_id":"2406.03699","repositories_listed":1,"syntology":null},{"url":"/paper/semantically-diverse-language-generation-for","slug":"semantically-diverse-language-generation-for","title":"Semantically Diverse Language Generation for Uncertainty Estimation in Language Models","date":"2024-06-06","arxiv_id":"2406.04306","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/semantically-diverse-language-generation-for#ran","syntology_url":"https://syntology.ai/paper/2406.04306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04306"}},"official":{"repos":["ml-jku/SDLG"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/css-contrastive-semantic-similarity-for","slug":"css-contrastive-semantic-similarity-for","title":"CSS: Contrastive Semantic Similarity for Uncertainty Quantification of LLMs","date":"2024-06-05","arxiv_id":"2406.03158","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/css-contrastive-semantic-similarity-for#ran","syntology_url":"https://syntology.ai/paper/2406.03158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03158"}},"official":{"repos":["aoshuang92/css_uq_llms"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-social-biases-in-japanese-large","slug":"analyzing-social-biases-in-japanese-large","title":"Analyzing Social Biases in Japanese Large Language Models","date":"2024-06-04","arxiv_id":"2406.02050","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-refined-vqa-annotations-for-semi","slug":"diffusion-refined-vqa-annotations-for-semi","title":"Diffusion-Refined VQA Annotations for Semi-Supervised Gaze Following","date":"2024-06-04","arxiv_id":"2406.02774","repositories_listed":1,"syntology":null},{"url":"/paper/from-redundancy-to-relevance-enhancing","slug":"from-redundancy-to-relevance-enhancing","title":"From Redundancy to Relevance: Information Flow in LVLMs Across Reasoning Tasks","date":"2024-06-04","arxiv_id":"2406.06579","repositories_listed":1,"syntology":null},{"url":"/paper/retaining-key-information-under-high","slug":"retaining-key-information-under-high","title":"Retaining Key Information under High Compression Ratios: Query-Guided Compressor for LLMs","date":"2024-06-04","arxiv_id":"2406.02376","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retaining-key-information-under-high#ran","syntology_url":"https://syntology.ai/paper/2406.02376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02376"}},"official":{"repos":["DeepLearnXMU/QGC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-accord-closing-the-commonsense","slug":"texttt-accord-closing-the-commonsense","title":"$\\texttt{ACCORD}$: Closing the Commonsense Measurability Gap","date":"2024-06-04","arxiv_id":"2406.02804","repositories_listed":1,"syntology":null},{"url":"/paper/an-information-bottleneck-perspective-for","slug":"an-information-bottleneck-perspective-for","title":"An Information Bottleneck Perspective for Effective Noise Filtering on Retrieval-Augmented Generation","date":"2024-06-03","arxiv_id":"2406.01549","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-information-bottleneck-perspective-for#ran","syntology_url":"https://syntology.ai/paper/2406.01549","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01549"}},"official":{"repos":["zhukun1020/noisefilter_ib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextualized-sequence-likelihood-enhanced","slug":"contextualized-sequence-likelihood-enhanced","title":"Contextualized Sequence Likelihood: Enhanced Confidence Scores for Natural Language Generation","date":"2024-06-03","arxiv_id":"2406.01806","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/contextualized-sequence-likelihood-enhanced#ran","syntology_url":"https://syntology.ai/paper/2406.01806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01806"}},"official":{"repos":["zlin7/contextsl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/reflection-reinforced-self-training-for","slug":"reflection-reinforced-self-training-for","title":"Re-ReST: Reflection-Reinforced Self-Training for Language Agents","date":"2024-06-03","arxiv_id":"2406.01495","repositories_listed":1,"syntology":null},{"url":"/paper/tabpedia-towards-comprehensive-visual-table","slug":"tabpedia-towards-comprehensive-visual-table","title":"TabPedia: Towards Comprehensive Visual Table Understanding with Concept Synergy","date":"2024-06-03","arxiv_id":"2406.01326","repositories_listed":1,"syntology":null},{"url":"/paper/tcmbench-a-comprehensive-benchmark-for","slug":"tcmbench-a-comprehensive-benchmark-for","title":"TCMBench: A Comprehensive Benchmark for Evaluating Large Language Models in Traditional Chinese Medicine","date":"2024-06-03","arxiv_id":"2406.01126","repositories_listed":1,"syntology":null},{"url":"/paper/cmdbench-a-benchmark-for-coarse-to-fine","slug":"cmdbench-a-benchmark-for-coarse-to-fine","title":"CMDBench: A Benchmark for Coarse-to-fine Multimodal Data Discovery in Compound AI Systems","date":"2024-06-02","arxiv_id":"2406.00583","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-4d-dynamic-scenes-understanding","slug":"compositional-4d-dynamic-scenes-understanding","title":"Compositional 4D Dynamic Scenes Understanding with Physics Priors for Video Question Answering","date":"2024-06-02","arxiv_id":"2406.00622","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compositional-4d-dynamic-scenes-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.00622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00622"}},"official":{"repos":["XingruiWang/SuperCLEVR-Physics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/passage-specific-prompt-tuning-for-passage","slug":"passage-specific-prompt-tuning-for-passage","title":"Passage-specific Prompt Tuning for Passage Reranking in Question Answering with Large Language Models","date":"2024-05-31","arxiv_id":"2405.20654","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-and-mitigating-retriever","slug":"unraveling-and-mitigating-retriever","title":"Unraveling and Mitigating Retriever Inconsistencies in Retrieval-Augmented Large Language Models","date":"2024-05-31","arxiv_id":"2405.20680","repositories_listed":1,"syntology":null},{"url":"/paper/anah-analytical-annotation-of-hallucinations","slug":"anah-analytical-annotation-of-hallucinations","title":"ANAH: Analytical Annotation of Hallucinations in Large Language Models","date":"2024-05-30","arxiv_id":"2405.20315","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anah-analytical-annotation-of-hallucinations#ran","syntology_url":"https://syntology.ai/paper/2405.20315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20315"}},"official":{"repos":["open-compass/anah"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/encoding-and-controlling-global-semantics-for","slug":"encoding-and-controlling-global-semantics-for","title":"Encoding and Controlling Global Semantics for Long-form Video Question Answering","date":"2024-05-30","arxiv_id":"2405.19723","repositories_listed":1,"syntology":null},{"url":"/paper/gnn-rag-graph-neural-retrieval-for-large","slug":"gnn-rag-graph-neural-retrieval-for-large","title":"GNN-RAG: Graph Neural Retrieval for Large Language Model Reasoning","date":"2024-05-30","arxiv_id":"2405.20139","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gnn-rag-graph-neural-retrieval-for-large#ran","syntology_url":"https://syntology.ai/paper/2405.20139","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20139"}},"official":{"repos":["cmavro/gnn-rag"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/worse-than-random-an-embarrassingly-simple","slug":"worse-than-random-an-embarrassingly-simple","title":"Worse than Random? An Embarrassingly Simple Probing Evaluation of Large Multimodal Models in Medical VQA","date":"2024-05-30","arxiv_id":"2405.20421","repositories_listed":1,"syntology":null},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reverse-image-retrieval-cues-parametric","slug":"reverse-image-retrieval-cues-parametric","title":"Reverse Image Retrieval Cues Parametric Memory in Multimodal LLMs","date":"2024-05-29","arxiv_id":"2405.18740","repositories_listed":1,"syntology":null},{"url":"/paper/atm-adversarial-tuning-multi-agent-system","slug":"atm-adversarial-tuning-multi-agent-system","title":"ATM: Adversarial Tuning Multi-agent System Makes a Robust Retrieval-Augmented Generator","date":"2024-05-28","arxiv_id":"2405.18111","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/atm-adversarial-tuning-multi-agent-system#ran","syntology_url":"https://syntology.ai/paper/2405.18111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18111"}},"official":{"repos":["chuhac/atm-rag"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/peering-into-the-mind-of-language-models-an","slug":"peering-into-the-mind-of-language-models-an","title":"Peering into the Mind of Language Models: An Approach for Attribution in Contextual Question Answering","date":"2024-05-28","arxiv_id":"2405.17980","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-large-language-models-to-set-up-a","slug":"empowering-large-language-models-to-set-up-a","title":"Empowering Large Language Models to Set up a Knowledge Retrieval Indexer via Self-Learning","date":"2024-05-27","arxiv_id":"2405.16933","repositories_listed":1,"syntology":null},{"url":"/paper/hawk-learning-to-understand-open-world-video","slug":"hawk-learning-to-understand-open-world-video","title":"Hawk: Learning to Understand Open-World Video Anomalies","date":"2024-05-27","arxiv_id":"2405.16886","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hawk-learning-to-understand-open-world-video#ran","syntology_url":"https://syntology.ai/paper/2405.16886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16886"}},"official":{"repos":["jqtangust/hawk"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reason3d-searching-and-reasoning-3d","slug":"reason3d-searching-and-reasoning-3d","title":"Reason3D: Searching and Reasoning 3D Segmentation via Large Language Model","date":"2024-05-27","arxiv_id":"2405.17427","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reason3d-searching-and-reasoning-3d#ran","syntology_url":"https://syntology.ai/paper/2405.17427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17427"}},"official":{"repos":["kuanchihhuang/reason3d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/thread-thinking-deeper-with-recursive","slug":"thread-thinking-deeper-with-recursive","title":"THREAD: Thinking Deeper with Recursive Spawning","date":"2024-05-27","arxiv_id":"2405.17402","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/thread-thinking-deeper-with-recursive#ran","syntology_url":"https://syntology.ai/paper/2405.17402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17402"}},"official":{"repos":["philipmit/thread"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/accurate-and-nuanced-open-qa-evaluation","slug":"accurate-and-nuanced-open-qa-evaluation","title":"Accurate and Nuanced Open-QA Evaluation Through Textual Entailment","date":"2024-05-26","arxiv_id":"2405.16702","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accurate-and-nuanced-open-qa-evaluation#ran","syntology_url":"https://syntology.ai/paper/2405.16702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16702"}},"official":{"repos":["U-Alberta/QA-partial-marks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/map-based-modular-approach-for-zero-shot","slug":"map-based-modular-approach-for-zero-shot","title":"Map-based Modular Approach for Zero-shot Embodied Question Answering","date":"2024-05-26","arxiv_id":"2405.16559","repositories_listed":1,"syntology":null},{"url":"/paper/on-bits-and-bandits-quantifying-the-regret","slug":"on-bits-and-bandits-quantifying-the-regret","title":"On Bits and Bandits: Quantifying the Regret-Information Trade-off","date":"2024-05-26","arxiv_id":"2405.16581","repositories_listed":1,"syntology":null},{"url":"/paper/irel-at-semeval-2024-task-9-improving","slug":"irel-at-semeval-2024-task-9-improving","title":"iREL at SemEval-2024 Task 9: Improving Conventional Prompting Methods for Brain Teasers","date":"2024-05-25","arxiv_id":"2405.16129","repositories_listed":1,"syntology":null},{"url":"/paper/optllm-optimal-assignment-of-queries-to-large","slug":"optllm-optimal-assignment-of-queries-to-large","title":"OptLLM: Optimal Assignment of Queries to Large Language Models","date":"2024-05-24","arxiv_id":"2405.15130","repositories_listed":1,"syntology":null},{"url":"/paper/text-generation-a-systematic-literature","slug":"text-generation-a-systematic-literature","title":"Text Generation: A Systematic Literature Review of Tasks, Evaluation, and Challenges","date":"2024-05-24","arxiv_id":"2405.15604","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-vision-language-action-models-for","slug":"a-survey-on-vision-language-action-models-for","title":"A Survey on Vision-Language-Action Models for Embodied AI","date":"2024-05-23","arxiv_id":"2405.14093","repositories_listed":1,"syntology":null},{"url":"/paper/agile-a-novel-framework-of-llm-agents","slug":"agile-a-novel-framework-of-llm-agents","title":"AGILE: A Novel Reinforcement Learning Framework of LLM Agents","date":"2024-05-23","arxiv_id":"2405.14751","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agile-a-novel-framework-of-llm-agents#ran","syntology_url":"https://syntology.ai/paper/2405.14751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14751"}},"official":{"repos":["bytarnish/agile"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-medical-question-answering-with","slug":"efficient-medical-question-answering-with","title":"Efficient Medical Question Answering with Knowledge-Augmented Question Generation","date":"2024-05-23","arxiv_id":"2405.14654","repositories_listed":1,"syntology":null},{"url":"/paper/lova3-learning-to-visual-question-answering","slug":"lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","arxiv_id":"2405.14974","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lova3-learning-to-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2405.14974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14974"}},"official":{"repos":["showlab/lova3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/wise-rethinking-the-knowledge-memory-for","slug":"wise-rethinking-the-knowledge-memory-for","title":"WISE: Rethinking the Knowledge Memory for Lifelong Model Editing of Large Language Models","date":"2024-05-23","arxiv_id":"2405.14768","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":4,"n_instrument":9,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/wise-rethinking-the-knowledge-memory-for#ran","syntology_url":"https://syntology.ai/paper/2405.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14768"}},"official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-evaluation-of-retrieval-augmented","slug":"automated-evaluation-of-retrieval-augmented","title":"Automated Evaluation of Retrieval-Augmented Language Models with Task-Specific Exam Generation","date":"2024-05-22","arxiv_id":"2405.13622","repositories_listed":1,"syntology":null},{"url":"/paper/pitvqa-image-grounded-text-embedding-llm-for","slug":"pitvqa-image-grounded-text-embedding-llm-for","title":"PitVQA: Image-grounded Text Embedding LLM for Visual Question Answering in Pituitary Surgery","date":"2024-05-22","arxiv_id":"2405.13949","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-density-uncertainty-quantification","slug":"semantic-density-uncertainty-quantification","title":"Semantic Density: Uncertainty Quantification for Large Language Models through Confidence Measurement in Semantic Space","date":"2024-05-22","arxiv_id":"2405.13845","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/semantic-density-uncertainty-quantification#ran","syntology_url":"https://syntology.ai/paper/2405.13845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13845"}},"official":{"repos":["cognizant-ai-labs/semantic-density-paper"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dataset-and-benchmark-for-urdu-natural-scenes","slug":"dataset-and-benchmark-for-urdu-natural-scenes","title":"Dataset and Benchmark for Urdu Natural Scenes Text Detection, Recognition and Visual Question Answering","date":"2024-05-21","arxiv_id":"2405.12533","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-interpretable-information","slug":"efficient-and-interpretable-information","title":"Efficient and Interpretable Information Retrieval for Product Question Answering with Heterogeneous Data","date":"2024-05-21","arxiv_id":"2405.13173","repositories_listed":1,"syntology":null},{"url":"/paper/olaph-improving-factuality-in-biomedical-long","slug":"olaph-improving-factuality-in-biomedical-long","title":"OLAPH: Improving Factuality in Biomedical Long-form Question Answering","date":"2024-05-21","arxiv_id":"2405.12701","repositories_listed":1,"syntology":null},{"url":"/paper/prott3-protein-to-text-generation-for-text","slug":"prott3-protein-to-text-generation-for-text","title":"ProtT3: Protein-to-Text Generation for Text-based Protein Understanding","date":"2024-05-21","arxiv_id":"2405.12564","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/prott3-protein-to-text-generation-for-text#ran","syntology_url":"https://syntology.ai/paper/2405.12564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12564"}},"official":{"repos":["acharkq/prott3"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-multimodal-large-language-models-a","slug":"efficient-multimodal-large-language-models-a","title":"Efficient Multimodal Large Language Models: A Survey","date":"2024-05-17","arxiv_id":"2405.10739","repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-question-generation-in-qa","slug":"towards-better-question-generation-in-qa","title":"Towards Better Question Generation in QA-based Event Extraction","date":"2024-05-17","arxiv_id":"2405.10517","repositories_listed":1,"syntology":null},{"url":"/paper/conformal-alignment-knowing-when-to-trust","slug":"conformal-alignment-knowing-when-to-trust","title":"Conformal Alignment: Knowing When to Trust Foundation Models with Guarantees","date":"2024-05-16","arxiv_id":"2405.10301","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conformal-alignment-knowing-when-to-trust#ran","syntology_url":"https://syntology.ai/paper/2405.10301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10301"}},"official":{"repos":["yugjerry/conformal-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-3d-llm-with-referent-tokens","slug":"grounded-3d-llm-with-referent-tokens","title":"Grounded 3D-LLM with Referent Tokens","date":"2024-05-16","arxiv_id":"2405.10370","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-3d-llm-with-referent-tokens#ran","syntology_url":"https://syntology.ai/paper/2405.10370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10370"}},"official":{"repos":["OpenRobotLab/Grounded_3D-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sciqag-a-framework-for-auto-generated","slug":"sciqag-a-framework-for-auto-generated","title":"SciQAG: A Framework for Auto-Generated Science Question Answering Dataset with Fine-grained Evaluation","date":"2024-05-16","arxiv_id":"2405.09939","repositories_listed":1,"syntology":null},{"url":"/paper/unirag-universal-retrieval-augmentation-for","slug":"unirag-universal-retrieval-augmentation-for","title":"UniRAG: Universal Retrieval Augmentation for Large Vision Language Models","date":"2024-05-16","arxiv_id":"2405.10311","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unirag-universal-retrieval-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2405.10311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10311"}},"official":{"repos":["castorini/unirag"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/when-llms-step-into-the-3d-world-a-survey-and","slug":"when-llms-step-into-the-3d-world-a-survey-and","title":"When LLMs step into the 3D World: A Survey and Meta-Analysis of 3D Tasks via Multi-modal Large Language Models","date":"2024-05-16","arxiv_id":"2405.10255","repositories_listed":1,"syntology":null},{"url":"/paper/prompting-based-synthetic-data-generation-for","slug":"prompting-based-synthetic-data-generation-for","title":"Prompting-based Synthetic Data Generation for Few-Shot Question Answering","date":"2024-05-15","arxiv_id":"2405.09335","repositories_listed":1,"syntology":null},{"url":"/paper/econlogicqa-a-question-answering-benchmark","slug":"econlogicqa-a-question-answering-benchmark","title":"EconLogicQA: A Question-Answering Benchmark for Evaluating Large Language Models in Economic Sequential Reasoning","date":"2024-05-13","arxiv_id":"2405.07938","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/econlogicqa-a-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2405.07938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07938"}},"official":{"repos":["yinzhu-quan/lm-evaluation-harness"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/freeva-offline-mllm-as-training-free-video","slug":"freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","arxiv_id":"2405.07798","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/freeva-offline-mllm-as-training-free-video#ran","syntology_url":"https://syntology.ai/paper/2405.07798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07798"}},"official":{"repos":["whwu95/freeva"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tanq-an-open-domain-dataset-of-table-answered","slug":"tanq-an-open-domain-dataset-of-table-answered","title":"TANQ: An open domain dataset of table answered questions","date":"2024-05-13","arxiv_id":"2405.07765","repositories_listed":1,"syntology":null},{"url":"/paper/limited-ability-of-llms-to-simulate-human","slug":"limited-ability-of-llms-to-simulate-human","title":"Limited Ability of LLMs to Simulate Human Psychological Behaviours: a Psychometric Analysis","date":"2024-05-12","arxiv_id":"2405.07248","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/limited-ability-of-llms-to-simulate-human#ran","syntology_url":"https://syntology.ai/paper/2405.07248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07248"}},"official":{"repos":["nikbpetrov/llms-simulate-humans"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/medconceptsqa-open-source-medical-concepts-qa","slug":"medconceptsqa-open-source-medical-concepts-qa","title":"MedConceptsQA: Open Source Medical Concepts QA Benchmark","date":"2024-05-12","arxiv_id":"2405.07348","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medconceptsqa-open-source-medical-concepts-qa#ran","syntology_url":"https://syntology.ai/paper/2405.07348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07348"}},"official":{"repos":["nadavlab/MedConceptsQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-document-visual-question-answering","slug":"federated-document-visual-question-answering","title":"Federated Document Visual Question Answering: A Pilot Study","date":"2024-05-10","arxiv_id":"2405.06636","repositories_listed":1,"syntology":null},{"url":"/paper/hmt-hierarchical-memory-transformer-for-long","slug":"hmt-hierarchical-memory-transformer-for-long","title":"HMT: Hierarchical Memory Transformer for Long Context Language Processing","date":"2024-05-09","arxiv_id":"2405.06067","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hmt-hierarchical-memory-transformer-for-long#ran","syntology_url":"https://syntology.ai/paper/2405.06067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06067"}},"official":{"repos":["OswaldHe/HMT-pytorch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dalk-dynamic-co-augmentation-of-llms-and-kg","slug":"dalk-dynamic-co-augmentation-of-llms-and-kg","title":"DALK: Dynamic Co-Augmentation of LLMs and KG to answer Alzheimer's Disease Questions with Scientific Literature","date":"2024-05-08","arxiv_id":"2405.04819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dalk-dynamic-co-augmentation-of-llms-and-kg#ran","syntology_url":"https://syntology.ai/paper/2405.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04819"}},"official":{"repos":["david-li0406/dalk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cleangraph-human-in-the-loop-knowledge-graph","slug":"cleangraph-human-in-the-loop-knowledge-graph","title":"CleanGraph: Human-in-the-loop Knowledge Graph Refinement and Completion","date":"2024-05-07","arxiv_id":"2405.03932","repositories_listed":1,"syntology":null},{"url":"/paper/eragent-enhancing-retrieval-augmented","slug":"eragent-enhancing-retrieval-augmented","title":"ERAGent: Enhancing Retrieval-Augmented Language Models with Improved Accuracy, Efficiency, and Personalization","date":"2024-05-06","arxiv_id":"2405.06683","repositories_listed":1,"syntology":null},{"url":"/paper/hire-me-or-not-examining-language-model-s","slug":"hire-me-or-not-examining-language-model-s","title":"Hire Me or Not? Examining Language Model's Behavior with Occupation Attributes","date":"2024-05-06","arxiv_id":"2405.06687","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hire-me-or-not-examining-language-model-s#ran","syntology_url":"https://syntology.ai/paper/2405.06687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06687"}},"official":{"repos":["daminz97/multi-step_gsv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-contextual-understanding-in-large","slug":"enhancing-contextual-understanding-in-large","title":"Enhancing Contextual Understanding in Large Language Models through Contrastive Decoding","date":"2024-05-04","arxiv_id":"2405.02750","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/enhancing-contextual-understanding-in-large#ran","syntology_url":"https://syntology.ai/paper/2405.02750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02750"}},"official":{"repos":["amazon-science/contextualunderstanding-contrastivedecoding"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}}],"record_sha256":"8b77eaec37bb4ca8015c8d30aa366c03951f544242c084dcc2eeb273ada2665f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}