{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/9","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":109,"rows_per_page":100,"rows":[801,900],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/8","next":"/task/question-answering/papers/10","papers":[{"url":"/paper/cxreasonbench-a-benchmark-for-evaluating","slug":"cxreasonbench-a-benchmark-for-evaluating","title":"CXReasonBench: A Benchmark for Evaluating Structured Diagnostic Reasoning in Chest X-rays","date":"2025-05-23","arxiv_id":"2505.18087","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cxreasonbench-a-benchmark-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2505.18087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18087"}},"official":{"repos":["ttumyche/cxreasonbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/danmakutppbench-a-multi-modal-benchmark-for","slug":"danmakutppbench-a-multi-modal-benchmark-for","title":"DanmakuTPPBench: A Multi-modal Benchmark for Temporal Point Process Modeling and Understanding","date":"2025-05-23","arxiv_id":"2505.18411","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/danmakutppbench-a-multi-modal-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2505.18411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18411"}},"official":{"repos":["frenkie-chiang/danmakutppbench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/metagen-blended-rag-higher-accuracy-for","slug":"metagen-blended-rag-higher-accuracy-for","title":"MetaGen Blended RAG: Higher Accuracy for Domain-Specific Q&A Without Fine-Tuning","date":"2025-05-23","arxiv_id":"2505.18247","repositories_listed":1,"syntology":null},{"url":"/paper/qwenlong-l1-towards-long-context-large","slug":"qwenlong-l1-towards-long-context-large","title":"QwenLong-L1: Towards Long-Context Large Reasoning Models with Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.17667","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-up-biomedical-vision-language-models","slug":"scaling-up-biomedical-vision-language-models","title":"Scaling Up Biomedical Vision-Language Models: Fine-Tuning, Instruction Tuning, and Multi-Modal Learning","date":"2025-05-23","arxiv_id":"2505.17436","repositories_listed":1,"syntology":null},{"url":"/paper/veattack-downstream-agnostic-vision-encoder","slug":"veattack-downstream-agnostic-vision-encoder","title":"VEAttack: Downstream-agnostic Vision Encoder Attack against Large Vision Language Models","date":"2025-05-23","arxiv_id":"2505.17440","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-retrieval-augmented-multimomal","slug":"benchmarking-retrieval-augmented-multimomal","title":"Benchmarking Retrieval-Augmented Multimomal Generation for Document Question Answering","date":"2025-05-22","arxiv_id":"2505.16470","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-multimomal#ran","syntology_url":"https://syntology.ai/paper/2505.16470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16470"}},"official":{"repos":["mmdocrag/mmdocrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucinations-in-vision-language","slug":"mitigating-hallucinations-in-vision-language","title":"Mitigating Hallucinations in Vision-Language Models through Image-Guided Head Suppression","date":"2025-05-22","arxiv_id":"2505.16411","repositories_listed":1,"syntology":null},{"url":"/paper/o-2-searcher-a-searching-based-agent-model","slug":"o-2-searcher-a-searching-based-agent-model","title":"O$^2$-Searcher: A Searching-based Agent Model for Open-Domain Open-Ended Question Answering","date":"2025-05-22","arxiv_id":"2505.16582","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/o-2-searcher-a-searching-based-agent-model#ran","syntology_url":"https://syntology.ai/paper/2505.16582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16582"}},"official":{"repos":["acade-mate/o2-searcher"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spatialscore-towards-unified-evaluation-for","slug":"spatialscore-towards-unified-evaluation-for","title":"SpatialScore: Towards Unified Evaluation for Multimodal Spatial Understanding","date":"2025-05-22","arxiv_id":"2505.17012","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatialscore-towards-unified-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.17012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17012"}},"official":{"repos":["haoningwu3639/SpatialScore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-large-language-models-to-maintain","slug":"teaching-large-language-models-to-maintain","title":"Teaching Large Language Models to Maintain Contextual Faithfulness via Synthetic Tasks and Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16483","repositories_listed":1,"syntology":null},{"url":"/paper/chartcards-a-chart-metadata-generation","slug":"chartcards-a-chart-metadata-generation","title":"ChartCards: A Chart-Metadata Generation Framework for Multi-Task Chart Understanding","date":"2025-05-21","arxiv_id":"2505.15046","repositories_listed":1,"syntology":null},{"url":"/paper/from-problem-solving-to-teaching-problem","slug":"from-problem-solving-to-teaching-problem","title":"From Problem-Solving to Teaching Problem-Solving: Aligning LLMs with Pedagogy using Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15607","repositories_listed":1,"syntology":null},{"url":"/paper/hopweaver-synthesizing-authentic-multi-hop","slug":"hopweaver-synthesizing-authentic-multi-hop","title":"HopWeaver: Synthesizing Authentic Multi-Hop Questions Across Text Corpora","date":"2025-05-21","arxiv_id":"2505.15087","repositories_listed":1,"syntology":null},{"url":"/paper/keep-security-benchmarking-security-policy","slug":"keep-security-benchmarking-security-policy","title":"Keep Security! Benchmarking Security Policy Preservation in Large Language Model Contexts Against Indirect Attacks in Question Answering","date":"2025-05-21","arxiv_id":"2505.15805","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-online-data-to-enhance-medical","slug":"leveraging-online-data-to-enhance-medical","title":"Leveraging Online Data to Enhance Medical Knowledge in a Small Persian Language Model","date":"2025-05-21","arxiv_id":"2505.16000","repositories_listed":1,"syntology":null},{"url":"/paper/raven-query-guided-representation-alignment","slug":"raven-query-guided-representation-alignment","title":"RAVEN: Query-Guided Representation Alignment for Question Answering over Audio, Video, Embedded Sensors, and Natural Language","date":"2025-05-21","arxiv_id":"2505.17114","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/raven-query-guided-representation-alignment#ran","syntology_url":"https://syntology.ai/paper/2505.17114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17114"}},"official":{"repos":["bashlab/raven"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/snap-a-benchmark-for-testing-the-effects-of","slug":"snap-a-benchmark-for-testing-the-effects-of","title":"SNAP: A Benchmark for Testing the Effects of Capture Conditions on Fundamental Vision Tasks","date":"2025-05-21","arxiv_id":"2505.15628","repositories_listed":1,"syntology":null},{"url":"/paper/the-atlas-of-in-context-learning-how","slug":"the-atlas-of-in-context-learning-how","title":"The Atlas of In-Context Learning: How Attention Heads Shape In-Context Retrieval Augmentation","date":"2025-05-21","arxiv_id":"2505.15807","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-atlas-of-in-context-learning-how#ran","syntology_url":"https://syntology.ai/paper/2505.15807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15807"}},"official":{"repos":["pkhdipraja/in-context-atlas"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/timecausality-evaluating-the-causal-ability","slug":"timecausality-evaluating-the-causal-ability","title":"TimeCausality: Evaluating the Causal Ability in Time Dimension for Vision Language Models","date":"2025-05-21","arxiv_id":"2505.15435","repositories_listed":1,"syntology":null},{"url":"/paper/traveling-across-languages-benchmarking-cross","slug":"traveling-across-languages-benchmarking-cross","title":"Traveling Across Languages: Benchmarking Cross-Lingual Consistency in Multimodal LLMs","date":"2025-05-21","arxiv_id":"2505.15075","repositories_listed":1,"syntology":null},{"url":"/paper/urdufactcheck-an-agentic-fact-checking","slug":"urdufactcheck-an-agentic-fact-checking","title":"UrduFactCheck: An Agentic Fact-Checking Framework for Urdu with Evidence Boosting and Benchmarking","date":"2025-05-21","arxiv_id":"2505.15063","repositories_listed":1,"syntology":null},{"url":"/paper/viqagent-zero-shot-video-question-answering","slug":"viqagent-zero-shot-video-question-answering","title":"ViQAgent: Zero-Shot Video Question Answering via Agent with Open-Vocabulary Grounding Validation","date":"2025-05-21","arxiv_id":"2505.15928","repositories_listed":1,"syntology":null},{"url":"/paper/qa-prompting-improving-summarization-with","slug":"qa-prompting-improving-summarization-with","title":"QA-prompting: Improving Summarization with Large Language Models using Question-Answering","date":"2025-05-20","arxiv_id":"2505.14347","repositories_listed":1,"syntology":null},{"url":"/paper/ravenea-a-benchmark-for-multimodal-retrieval","slug":"ravenea-a-benchmark-for-multimodal-retrieval","title":"RAVENEA: A Benchmark for Multimodal Retrieval-Augmented Visual Culture Understanding","date":"2025-05-20","arxiv_id":"2505.14462","repositories_listed":1,"syntology":null},{"url":"/paper/texts-or-images-a-fine-grained-analysis-on","slug":"texts-or-images-a-fine-grained-analysis-on","title":"Texts or Images? A Fine-grained Analysis on the Effectiveness of Input Representations and Models for Table Question Answering","date":"2025-05-20","arxiv_id":"2505.14131","repositories_listed":1,"syntology":null},{"url":"/paper/voqa-visual-only-question-answering","slug":"voqa-visual-only-question-answering","title":"VoQA: Visual-only Question Answering","date":"2025-05-20","arxiv_id":"2505.14227","repositories_listed":1,"syntology":null},{"url":"/paper/a-case-study-of-cross-lingual-zero-shot","slug":"a-case-study-of-cross-lingual-zero-shot","title":"A Case Study of Cross-Lingual Zero-Shot Generalization for Classical Languages in LLMs","date":"2025-05-19","arxiv_id":"2505.13173","repositories_listed":1,"syntology":null},{"url":"/paper/agi-elo-how-far-are-we-from-mastering-a-task","slug":"agi-elo-how-far-are-we-from-mastering-a-task","title":"AGI-Elo: How Far Are We From Mastering A Task?","date":"2025-05-19","arxiv_id":"2505.12844","repositories_listed":1,"syntology":null},{"url":"/paper/learnware-of-language-models-specialized","slug":"learnware-of-language-models-specialized","title":"Learnware of Language Models: Specialized Small Language Models Can Do Big","date":"2025-05-19","arxiv_id":"2505.13425","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-ocr-can-large-multimodal-models","slug":"reasoning-ocr-can-large-multimodal-models","title":"Reasoning-OCR: Can Large Multimodal Models Solve Complex Logical Reasoning Problems from OCR Cues?","date":"2025-05-19","arxiv_id":"2505.12766","repositories_listed":1,"syntology":null},{"url":"/paper/medagentboard-benchmarking-multi-agent","slug":"medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","arxiv_id":"2505.12371","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentboard-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.12371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12371"}},"official":{"repos":["yhzhu99/medagentboard"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/love-benchmarking-and-evaluating-text-to","slug":"love-benchmarking-and-evaluating-text-to","title":"LOVE: Benchmarking and Evaluating Text-to-Video Generation and Video-to-Text Interpretation","date":"2025-05-17","arxiv_id":"2505.12098","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10772","slug":"2505-10772","title":"Ranked Voting based Self-Consistency of Large Language Models","date":"2025-05-16","arxiv_id":"2505.10772","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10852","slug":"2505-10852","title":"MatTools: Benchmarking Large Language Models for Materials Science Tools","date":"2025-05-16","arxiv_id":"2505.10852","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10928","slug":"2505-10928","title":"A Dataset for Spatiotemporal-Sensitive POI Question Answering","date":"2025-05-16","arxiv_id":"2505.10928","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11140","slug":"2505-11140","title":"Scaling Reasoning can Improve Factuality in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11140","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11180","slug":"2505-11180","title":"mmRAG: A Modular Benchmark for Retrieval-Augmented Generation over Text, Tables, and Knowledge Graphs","date":"2025-05-16","arxiv_id":"2505.11180","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11326","slug":"2505-11326","title":"Temporally-Grounded Language Generation: A Benchmark for Real-Time Vision-Language Models","date":"2025-05-16","arxiv_id":"2505.11326","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11404","slug":"2505-11404","title":"Patho-R1: A Multimodal Reinforcement Learning-Based Pathology Expert Reasoner","date":"2025-05-16","arxiv_id":"2505.11404","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11454","slug":"2505-11454","title":"HumaniBench: A Human-Centric Framework for Large Multimodal Models Evaluation","date":"2025-05-16","arxiv_id":"2505.11454","repositories_listed":1,"syntology":null},{"url":"/paper/masking-in-multi-hop-qa-an-analysis-of-how","slug":"masking-in-multi-hop-qa-an-analysis-of-how","title":"Masking in Multi-hop QA: An Analysis of How Language Models Perform with Context Permutation","date":"2025-05-16","arxiv_id":"2505.11754","repositories_listed":1,"syntology":null},{"url":"/paper/tcc-bench-benchmarking-the-traditional","slug":"tcc-bench-benchmarking-the-traditional","title":"TCC-Bench: Benchmarking the Traditional Chinese Culture Understanding Capabilities of MLLMs","date":"2025-05-16","arxiv_id":"2505.11275","repositories_listed":1,"syntology":null},{"url":"/paper/time-r1-towards-comprehensive-temporal","slug":"time-r1-towards-comprehensive-temporal","title":"Time-R1: Towards Comprehensive Temporal Reasoning in LLMs","date":"2025-05-16","arxiv_id":"2505.13508","repositories_listed":1,"syntology":null},{"url":"/paper/do-rag-a-domain-specific-qa-framework-using","slug":"do-rag-a-domain-specific-qa-framework-using","title":"DO-RAG: A Domain-Specific QA Framework Using Knowledge Graph-Enhanced Retrieval-Augmented Generation","date":"2025-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-brain-inspired-mechanisms-for","slug":"incorporating-brain-inspired-mechanisms-for","title":"Incorporating brain-inspired mechanisms for multimodal learning in artificial intelligence","date":"2025-05-15","arxiv_id":"2505.10176","repositories_listed":1,"syntology":null},{"url":"/paper/atomic-consistency-preference-optimization","slug":"atomic-consistency-preference-optimization","title":"Atomic Consistency Preference Optimization for Long-Form Question Answering","date":"2025-05-14","arxiv_id":"2505.09039","repositories_listed":1,"syntology":null},{"url":"/paper/focus-merge-rank-improved-question-answering","slug":"focus-merge-rank-improved-question-answering","title":"Focus, Merge, Rank: Improved Question Answering Based on Semi-structured Knowledge Bases","date":"2025-05-14","arxiv_id":"2505.09246","repositories_listed":1,"syntology":null},{"url":"/paper/pt-moe-an-efficient-finetuning-framework-for","slug":"pt-moe-an-efficient-finetuning-framework-for","title":"PT-MoE: An Efficient Finetuning Framework for Integrating Mixture-of-Experts into Prompt Tuning","date":"2025-05-14","arxiv_id":"2505.09519","repositories_listed":1,"syntology":null},{"url":"/paper/scent-of-knowledge-optimizing-search-enhanced","slug":"scent-of-knowledge-optimizing-search-enhanced","title":"Scent of Knowledge: Optimizing Search-Enhanced Reasoning with Information Foraging","date":"2025-05-14","arxiv_id":"2505.09316","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scent-of-knowledge-optimizing-search-enhanced#ran","syntology_url":"https://syntology.ai/paper/2505.09316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.09316"}},"official":{"repos":["qhjqhj00/inforage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/judging-the-judges-can-large-vision-language","slug":"judging-the-judges-can-large-vision-language","title":"Judging the Judges: Can Large Vision-Language Models Fairly Evaluate Chart Comprehension and Reasoning?","date":"2025-05-13","arxiv_id":"2505.08468","repositories_listed":1,"syntology":null},{"url":"/paper/docvxqa-context-aware-visual-explanations-for","slug":"docvxqa-context-aware-visual-explanations-for","title":"DocVXQA: Context-Aware Visual Explanations for Document Question Answering","date":"2025-05-12","arxiv_id":"2505.07496","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-reproducible-biomedical","slug":"efficient-and-reproducible-biomedical","title":"Efficient and Reproducible Biomedical Question Answering using Retrieval Augmented Generation","date":"2025-05-12","arxiv_id":"2505.07917","repositories_listed":1,"syntology":null},{"url":"/paper/kalman-filter-enhanced-grpo-for-reinforcement","slug":"kalman-filter-enhanced-grpo-for-reinforcement","title":"Kalman Filter Enhanced GRPO for Reinforcement Learning-Based Language Model Reasoning","date":"2025-05-12","arxiv_id":"2505.07527","repositories_listed":1,"syntology":null},{"url":"/paper/recdap-relation-based-conditional-diffusion","slug":"recdap-relation-based-conditional-diffusion","title":"ReCDAP: Relation-Based Conditional Diffusion with Attention Pooling for Few-Shot Knowledge Graph Completion","date":"2025-05-12","arxiv_id":"2505.07171","repositories_listed":1,"syntology":null},{"url":"/paper/bioprobench-comprehensive-dataset-and","slug":"bioprobench-comprehensive-dataset-and","title":"BioProBench: Comprehensive Dataset and Benchmark in Biological Protocol Understanding and Reasoning","date":"2025-05-11","arxiv_id":"2505.07889","repositories_listed":1,"syntology":null},{"url":"/paper/smartpilot-a-multiagent-copilot-for-adaptive","slug":"smartpilot-a-multiagent-copilot-for-adaptive","title":"SmartPilot: A Multiagent CoPilot for Adaptive and Intelligent Manufacturing","date":"2025-05-10","arxiv_id":"2505.06492","repositories_listed":1,"syntology":null},{"url":"/paper/mm-skin-enhancing-dermatology-vision-language","slug":"mm-skin-enhancing-dermatology-vision-language","title":"MM-Skin: Enhancing Dermatology Vision-Language Model with an Image-Text Dataset Derived from Textbooks","date":"2025-05-09","arxiv_id":"2505.06152","repositories_listed":1,"syntology":null},{"url":"/paper/neoqa-evidence-based-question-answering-with","slug":"neoqa-evidence-based-question-answering-with","title":"NeoQA: Evidence-based Question Answering with Generated News Events","date":"2025-05-09","arxiv_id":"2505.05949","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-thought-machines","slug":"continuous-thought-machines","title":"Continuous Thought Machines","date":"2025-05-08","arxiv_id":"2505.05522","repositories_listed":1,"syntology":{"n":34,"n_ran":21,"n_constructed":9,"n_ran_checked":16,"n_instrument":5,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":1,"phrase":"21 ran (of which 9 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/continuous-thought-machines#ran","syntology_url":"https://syntology.ai/paper/2505.05522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05522"}},"official":{"repos":["SakanaAI/continuous-thought-machines"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":9,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/probabilistic-embeddings-for-frozen-vision","slug":"probabilistic-embeddings-for-frozen-vision","title":"Probabilistic Embeddings for Frozen Vision-Language Models: Uncertainty Quantification with Gaussian Process Latent Variable Models","date":"2025-05-08","arxiv_id":"2505.05163","repositories_listed":1,"syntology":null},{"url":"/paper/transproqa-an-llm-based-literary-translation","slug":"transproqa-an-llm-based-literary-translation","title":"LiTransProQA: an LLM-based Literary Translation evaluation metric with Professional Question Answering","date":"2025-05-08","arxiv_id":"2505.05423","repositories_listed":1,"syntology":null},{"url":"/paper/echoink-r1-exploring-audio-visual-reasoning","slug":"echoink-r1-exploring-audio-visual-reasoning","title":"EchoInk-R1: Exploring Audio-Visual Reasoning in Multimodal LLMs via Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04623","repositories_listed":1,"syntology":null},{"url":"/paper/characterising-topic-familiarity-and-query","slug":"characterising-topic-familiarity-and-query","title":"Characterising Topic Familiarity and Query Specificity Using Eye-Tracking Data","date":"2025-05-06","arxiv_id":"2505.03136","repositories_listed":1,"syntology":null},{"url":"/paper/dygenc-encoding-a-sequence-of-textual-scene","slug":"dygenc-encoding-a-sequence-of-textual-scene","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","date":"2025-05-06","arxiv_id":"2505.03581","repositories_listed":1,"syntology":null},{"url":"/paper/indicsquad-a-comprehensive-multilingual","slug":"indicsquad-a-comprehensive-multilingual","title":"IndicSQuAD: A Comprehensive Multilingual Question Answering Dataset for Indic Languages","date":"2025-05-06","arxiv_id":"2505.03688","repositories_listed":1,"syntology":null},{"url":"/paper/medarabiq-benchmarking-large-language-models","slug":"medarabiq-benchmarking-large-language-models","title":"MedArabiQ: Benchmarking Large Language Models on Arabic Medical Tasks","date":"2025-05-06","arxiv_id":"2505.03427","repositories_listed":1,"syntology":null},{"url":"/paper/vita-audio-fast-interleaved-cross-modal-token","slug":"vita-audio-fast-interleaved-cross-modal-token","title":"VITA-Audio: Fast Interleaved Cross-Modal Token Generation for Efficient Large Speech-Language Model","date":"2025-05-06","arxiv_id":"2505.03739","repositories_listed":1,"syntology":null},{"url":"/paper/invoke-interfaces-only-when-needed-adaptive","slug":"invoke-interfaces-only-when-needed-adaptive","title":"Invoke Interfaces Only When Needed: Adaptive Invocation for Large Language Models in Question Answering","date":"2025-05-05","arxiv_id":"2505.02311","repositories_listed":1,"syntology":null},{"url":"/paper/sim2real-transfer-for-vision-based-grasp","slug":"sim2real-transfer-for-vision-based-grasp","title":"Sim2Real Transfer for Vision-Based Grasp Verification","date":"2025-05-05","arxiv_id":"2505.03046","repositories_listed":1,"syntology":null},{"url":"/paper/rtv-bench-benchmarking-mllm-continuous","slug":"rtv-bench-benchmarking-mllm-continuous","title":"RTV-Bench: Benchmarking MLLM Continuous Perception, Understanding and Reasoning through Real-Time Video","date":"2025-05-04","arxiv_id":"2505.02064","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rtv-bench-benchmarking-mllm-continuous#ran","syntology_url":"https://syntology.ai/paper/2505.02064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02064"}},"official":{"repos":["ljungang/rtv-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adcare-vlm-leveraging-large-vision-language","slug":"adcare-vlm-leveraging-large-vision-language","title":"AdCare-VLM: Leveraging Large Vision Language Model (LVLM) to Monitor Long-Term Medication Adherence and Care","date":"2025-05-01","arxiv_id":"2505.00275","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/adcare-vlm-leveraging-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.00275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00275"}},"official":{"repos":["asad14053/AdCare-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/cse-sfp-enabling-unsupervised-sentence","slug":"cse-sfp-enabling-unsupervised-sentence","title":"CSE-SFP: Enabling Unsupervised Sentence Representation Learning via a Single Forward Pass","date":"2025-05-01","arxiv_id":"2505.00389","repositories_listed":1,"syntology":null},{"url":"/paper/unlearning-sensitive-information-in","slug":"unlearning-sensitive-information-in","title":"Unlearning Sensitive Information in Multimodal LLMs: Benchmark and Attack-Defense Evaluation","date":"2025-05-01","arxiv_id":"2505.01456","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unlearning-sensitive-information-in#ran","syntology_url":"https://syntology.ai/paper/2505.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.01456"}},"official":{"repos":["vaidehi99/unlok-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/talk-before-you-retrieve-agent-led","slug":"talk-before-you-retrieve-agent-led","title":"Talk Before You Retrieve: Agent-Led Discussions for Better RAG in Medical QA","date":"2025-04-30","arxiv_id":"2504.21252","repositories_listed":1,"syntology":null},{"url":"/paper/unibiomed-a-universal-foundation-model-for","slug":"unibiomed-a-universal-foundation-model-for","title":"UniBiomed: A Universal Foundation Model for Grounded Biomedical Image Interpretation","date":"2025-04-30","arxiv_id":"2504.21336","repositories_listed":1,"syntology":null},{"url":"/paper/chestx-reasoner-advancing-radiology","slug":"chestx-reasoner-advancing-radiology","title":"ChestX-Reasoner: Advancing Radiology Foundation Models with Reasoning through Step-by-Step Verification","date":"2025-04-29","arxiv_id":"2504.20930","repositories_listed":1,"syntology":null},{"url":"/paper/universalrag-retrieval-augmented-generation","slug":"universalrag-retrieval-augmented-generation","title":"UniversalRAG: Retrieval-Augmented Generation over Corpora of Diverse Modalities and Granularities","date":"2025-04-29","arxiv_id":"2504.20734","repositories_listed":1,"syntology":null},{"url":"/paper/toward-evaluative-thinking-meta-policy","slug":"toward-evaluative-thinking-meta-policy","title":"Toward Evaluative Thinking: Meta Policy Optimization with Evolving Reward Models","date":"2025-04-28","arxiv_id":"2504.20157","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/toward-evaluative-thinking-meta-policy#ran","syntology_url":"https://syntology.ai/paper/2504.20157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20157"}},"official":{"repos":["minnesotanlp/mpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/treehop-generate-and-filter-next-query","slug":"treehop-generate-and-filter-next-query","title":"TreeHop: Generate and Filter Next Query Embeddings Efficiently for Multi-hop Question Answering","date":"2025-04-28","arxiv_id":"2504.20114","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-quantification-for-language","slug":"uncertainty-quantification-for-language","title":"Uncertainty Quantification for Language Models: A Suite of Black-Box, White-Box, LLM Judge, and Ensemble Scorers","date":"2025-04-27","arxiv_id":"2504.19254","repositories_listed":1,"syntology":null},{"url":"/paper/test-it-before-you-trust-it-applying-software","slug":"test-it-before-you-trust-it-applying-software","title":"Test It Before You Trust It: Applying Software Testing for Trustworthy In-context Learning","date":"2025-04-26","arxiv_id":"2504.18827","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-evaluating-long-form","slug":"an-empirical-study-of-evaluating-long-form","title":"An Empirical Study of Evaluating Long-form Question Answering","date":"2025-04-25","arxiv_id":"2504.18413","repositories_listed":1,"syntology":null},{"url":"/paper/kimi-audio-technical-report","slug":"kimi-audio-technical-report","title":"Kimi-Audio Technical Report","date":"2025-04-25","arxiv_id":"2504.18425","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kimi-audio-technical-report#ran","syntology_url":"https://syntology.ai/paper/2504.18425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18425"}},"official":{"repos":["moonshotai/kimi-audio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videomultiagents-a-multi-agent-framework-for","slug":"videomultiagents-a-multi-agent-framework-for","title":"VideoMultiAgents: A Multi-Agent Framework for Video Question Answering","date":"2025-04-25","arxiv_id":"2504.20091","repositories_listed":1,"syntology":null},{"url":"/paper/2504-18589","slug":"2504-18589","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","date":"2025-04-24","arxiv_id":"2504.18589","repositories_listed":1,"syntology":null},{"url":"/paper/finbert-qa-financial-question-answering-with","slug":"finbert-qa-financial-question-answering-with","title":"FinBERT-QA: Financial Question Answering with pre-trained BERT Language Models","date":"2025-04-24","arxiv_id":"2505.00725","repositories_listed":1,"syntology":null},{"url":"/paper/survey-of-video-diffusion-models-foundations","slug":"survey-of-video-diffusion-models-foundations","title":"Survey of Video Diffusion Models: Foundations, Implementations, and Applications","date":"2025-04-22","arxiv_id":"2504.16081","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-document-retrieval-with-g-retriever","slug":"efficient-document-retrieval-with-g-retriever","title":"Efficient Document Retrieval with G-Retriever","date":"2025-04-21","arxiv_id":"2504.14955","repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-llms-road-ready-a-comprehensive","slug":"are-vision-llms-road-ready-a-comprehensive","title":"Are Vision LLMs Road-Ready? A Comprehensive Benchmark for Safety-Critical Driving Video Understanding","date":"2025-04-20","arxiv_id":"2504.14526","repositories_listed":1,"syntology":null},{"url":"/paper/walk-the-talk-measuring-the-faithfulness-of","slug":"walk-the-talk-measuring-the-faithfulness-of","title":"Walk the Talk? Measuring the Faithfulness of Large Language Model Explanations","date":"2025-04-19","arxiv_id":"2504.14150","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-attribute-with-attention","slug":"learning-to-attribute-with-attention","title":"Learning to Attribute with Attention","date":"2025-04-18","arxiv_id":"2504.13752","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-attribute-with-attention#ran","syntology_url":"https://syntology.ai/paper/2504.13752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13752"}},"official":{"repos":["madrylab/at2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/long-context-non-factoid-question-answering","slug":"long-context-non-factoid-question-answering","title":"Long-context Non-factoid Question Answering in Indic Languages","date":"2025-04-18","arxiv_id":"2504.13615","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llm-based-relevance-judgment","slug":"benchmarking-llm-based-relevance-judgment","title":"Benchmarking LLM-based Relevance Judgment Methods","date":"2025-04-17","arxiv_id":"2504.12558","repositories_listed":1,"syntology":null},{"url":"/paper/llm-as-a-judge-reassessing-the-performance-of","slug":"llm-as-a-judge-reassessing-the-performance-of","title":"LLM-as-a-Judge: Reassessing the Performance of LLMs in Extractive QA","date":"2025-04-16","arxiv_id":"2504.11972","repositories_listed":1,"syntology":null},{"url":"/paper/ai2-scholar-qa-organized-literature-synthesis","slug":"ai2-scholar-qa-organized-literature-synthesis","title":"Ai2 Scholar QA: Organized Literature Synthesis with Attribution","date":"2025-04-15","arxiv_id":"2504.10861","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai2-scholar-qa-organized-literature-synthesis#ran","syntology_url":"https://syntology.ai/paper/2504.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10861"}},"official":{"repos":["allenai/ai2-scholarqa-lib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qava-query-agnostic-visual-attack-to-large","slug":"qava-query-agnostic-visual-attack-to-large","title":"QAVA: Query-Agnostic Visual Attack to Large Vision-Language Models","date":"2025-04-15","arxiv_id":"2504.11038","repositories_listed":1,"syntology":null},{"url":"/paper/rankalign-a-ranking-view-of-the-generator","slug":"rankalign-a-ranking-view-of-the-generator","title":"RankAlign: A Ranking View of the Generator-Validator Gap in Large Language Models","date":"2025-04-15","arxiv_id":"2504.11381","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rankalign-a-ranking-view-of-the-generator#ran","syntology_url":"https://syntology.ai/paper/2504.11381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11381"}},"official":{"repos":["juand-r/rankalign"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pixel-sail-single-transformer-for-pixel","slug":"pixel-sail-single-transformer-for-pixel","title":"Pixel-SAIL: Single Transformer For Pixel-Grounded Understanding","date":"2025-04-14","arxiv_id":"2504.10465","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pixel-sail-single-transformer-for-pixel#ran","syntology_url":"https://syntology.ai/paper/2504.10465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10465"}},"official":{"repos":["magic-research/Sa2VA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reasondrive-efficient-visual-question","slug":"reasondrive-efficient-visual-question","title":"ReasonDrive: Efficient Visual Question Answering for Autonomous Vehicles with Reasoning-Enhanced Small Vision-Language Models","date":"2025-04-14","arxiv_id":"2504.10757","repositories_listed":1,"syntology":null}],"record_sha256":"f1617c11fa942a687bef53703472e089fad14e6a02a770b86a198dbe7863756d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}