{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/8","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":109,"rows_per_page":100,"rows":[701,800],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/7","next":"/task/question-answering/papers/9","papers":[{"url":"/paper/query-reduction-networks-for-question","slug":"query-reduction-networks-for-question","title":"Query-Reduction Networks for Question Answering","date":"2016-06-14","arxiv_id":"1606.04582","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-reduction-networks-for-question#ran","syntology_url":"https://syntology.ai/paper/1606.04582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.04582"}},"official":{"repos":["uwnlp/qrn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/key-value-memory-networks-for-directly","slug":"key-value-memory-networks-for-directly","title":"Key-Value Memory Networks for Directly Reading Documents","date":"2016-06-09","arxiv_id":"1606.03126","repositories_listed":2,"syntology":null},{"url":"/paper/text-understanding-with-the-attention-sum","slug":"text-understanding-with-the-attention-sum","title":"Text Understanding with the Attention Sum Reader Network","date":"2016-03-04","arxiv_id":"1603.01547","repositories_listed":2,"syntology":{"n":10,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/text-understanding-with-the-attention-sum#ran","syntology_url":"https://syntology.ai/paper/1603.01547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1603.01547"}},"official":{"repos":["rkadlec/asreader"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/visual-genome-connecting-language-and-vision","slug":"visual-genome-connecting-language-and-vision","title":"Visual Genome: Connecting Language and Vision Using Crowdsourced Dense Image Annotations","date":"2016-02-23","arxiv_id":"1602.07332","repositories_listed":2,"syntology":null},{"url":"/paper/applying-deep-learning-to-answer-selection-a","slug":"applying-deep-learning-to-answer-selection-a","title":"Applying Deep Learning to Answer Selection: A Study and An Open Task","date":"2015-08-07","arxiv_id":"1508.01585","repositories_listed":2,"syntology":null},{"url":"/paper/convolutional-neural-network-architectures-1","slug":"convolutional-neural-network-architectures-1","title":"Convolutional Neural Network Architectures for Matching Natural Language Sentences","date":"2015-03-11","arxiv_id":"1503.03244","repositories_listed":2,"syntology":null},{"url":"/paper/deep-learning-for-answer-sentence-selection","slug":"deep-learning-for-answer-sentence-selection","title":"Deep Learning for Answer Sentence Selection","date":"2014-12-04","arxiv_id":"1412.1632","repositories_listed":2,"syntology":null},{"url":"/paper/from-roots-to-rewards-dynamic-tree-reasoning","slug":"from-roots-to-rewards-dynamic-tree-reasoning","title":"From Roots to Rewards: Dynamic Tree Reasoning with RL","date":"2025-07-17","arxiv_id":"2507.13142","repositories_listed":1,"syntology":null},{"url":"/paper/describe-anything-model-for-visual-question","slug":"describe-anything-model-for-visual-question","title":"Describe Anything Model for Visual Question Answering on Text-rich Images","date":"2025-07-16","arxiv_id":"2507.12441","repositories_listed":1,"syntology":null},{"url":"/paper/warehouse-spatial-question-answering-with-llm","slug":"warehouse-spatial-question-answering-with-llm","title":"Warehouse Spatial Question Answering with LLM Agent","date":"2025-07-14","arxiv_id":"2507.10778","repositories_listed":1,"syntology":null},{"url":"/paper/l0-reinforcement-learning-to-become-general-1","slug":"l0-reinforcement-learning-to-become-general-1","title":"L0: Reinforcement Learning to Become General Agents","date":"2025-06-30","arxiv_id":"2506.23667","repositories_listed":1,"syntology":null},{"url":"/paper/rag-r1-incentivize-the-search-and-reasoning","slug":"rag-r1-incentivize-the-search-and-reasoning","title":"RAG-R1 : Incentivize the Search and Reasoning Capabilities of LLMs through Multi-query Parallelism","date":"2025-06-30","arxiv_id":"2507.02962","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rag-r1-incentivize-the-search-and-reasoning#ran","syntology_url":"https://syntology.ai/paper/2507.02962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02962"}},"official":null}},{"url":"/paper/decoupled-seg-tokens-make-stronger-reasoning","slug":"decoupled-seg-tokens-make-stronger-reasoning","title":"Decoupled Seg Tokens Make Stronger Reasoning Video Segmenter and Grounder","date":"2025-06-28","arxiv_id":"2506.22880","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-cropa-a-reproducibility-study-and","slug":"revisiting-cropa-a-reproducibility-study-and","title":"Revisiting CroPA: A Reproducibility Study and Enhancements for Cross-Prompt Adversarial Transferability in Vision-Language Models","date":"2025-06-28","arxiv_id":"2506.22982","repositories_listed":1,"syntology":null},{"url":"/paper/llava-scissor-token-compression-with-semantic","slug":"llava-scissor-token-compression-with-semantic","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","date":"2025-06-27","arxiv_id":"2506.21862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llava-scissor-token-compression-with-semantic#ran","syntology_url":"https://syntology.ai/paper/2506.21862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21862"}},"official":{"repos":["HumanMLLM/LLaVA-Scissor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/drishtikon-multi-granular-visual-grounding","slug":"drishtikon-multi-granular-visual-grounding","title":"DrishtiKon: Multi-Granular Visual Grounding for Text-Rich Document Images","date":"2025-06-26","arxiv_id":"2506.21316","repositories_listed":1,"syntology":null},{"url":"/paper/response-quality-assessment-for-retrieval","slug":"response-quality-assessment-for-retrieval","title":"Response Quality Assessment for Retrieval-Augmented Generation via Conditional Conformal Factuality","date":"2025-06-26","arxiv_id":"2506.20978","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/response-quality-assessment-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2506.20978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20978"}},"official":{"repos":["n4feng/ResponseQualityAssessment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hribench-benchmarking-vision-language-models","slug":"hribench-benchmarking-vision-language-models","title":"HRIBench: Benchmarking Vision-Language Models for Real-Time Human Perception in Human-Robot Interaction","date":"2025-06-25","arxiv_id":"2506.20566","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-prompts-in-context-an-open-ended","slug":"evolving-prompts-in-context-an-open-ended","title":"Evolving Prompts In-Context: An Open-ended, Self-replicating Perspective","date":"2025-06-22","arxiv_id":"2506.17930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evolving-prompts-in-context-an-open-ended#ran","syntology_url":"https://syntology.ai/paper/2506.17930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17930"}},"official":{"repos":["jianyu-cs/promptquine"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uprop-investigating-the-uncertainty","slug":"uprop-investigating-the-uncertainty","title":"UProp: Investigating the Uncertainty Propagation of LLMs in Multi-Step Agentic Decision-Making","date":"2025-06-20","arxiv_id":"2506.17419","repositories_listed":1,"syntology":null},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-lightweight-vision-language-models","slug":"adapting-lightweight-vision-language-models","title":"Adapting Lightweight Vision Language Models for Radiological Visual Question Answering","date":"2025-06-17","arxiv_id":"2506.14451","repositories_listed":1,"syntology":null},{"url":"/paper/generationprograms-fine-grained-attribution","slug":"generationprograms-fine-grained-attribution","title":"GenerationPrograms: Fine-grained Attribution with Executable Programs","date":"2025-06-17","arxiv_id":"2506.14580","repositories_listed":1,"syntology":null},{"url":"/paper/re-initialization-token-learning-for-tool","slug":"re-initialization-token-learning-for-tool","title":"Re-Initialization Token Learning for Tool-Augmented Large Language Models","date":"2025-06-17","arxiv_id":"2506.14248","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-omics-cohort-discovery-for-research","slug":"enhancing-omics-cohort-discovery-for-research","title":"Enhancing Omics Cohort Discovery for Research on Neurodegeneration through Ontology-Augmented Embedding Models","date":"2025-06-16","arxiv_id":"2506.13467","repositories_listed":1,"syntology":null},{"url":"/paper/seqpe-transformer-with-sequential-position","slug":"seqpe-transformer-with-sequential-position","title":"SeqPE: Transformer with Sequential Position Encoding","date":"2025-06-16","arxiv_id":"2506.13277","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seqpe-transformer-with-sequential-position#ran","syntology_url":"https://syntology.ai/paper/2506.13277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13277"}},"official":{"repos":["ghrua/seqpe"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/simpledoc-multi-modal-document-understanding","slug":"simpledoc-multi-modal-document-understanding","title":"SimpleDoc: Multi-Modal Document Understanding with Dual-Cue Page Retrieval and Iterative Refinement","date":"2025-06-16","arxiv_id":"2506.14035","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simpledoc-multi-modal-document-understanding#ran","syntology_url":"https://syntology.ai/paper/2506.14035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14035"}},"official":{"repos":["ag2ai/simpledoc"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-llm-merging-for-multi-task","slug":"training-free-llm-merging-for-multi-task","title":"Training-free LLM Merging for Multi-task Learning","date":"2025-06-14","arxiv_id":"2506.12379","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10467","slug":"2506-10467","title":"Specification and Evaluation of Multi-Agent LLM Systems -- Prototype and Cybersecurity Applications","date":"2025-06-12","arxiv_id":"2506.10467","repositories_listed":1,"syntology":null},{"url":"/paper/monitoring-decomposition-attacks-in-llms-with","slug":"monitoring-decomposition-attacks-in-llms-with","title":"Monitoring Decomposition Attacks in LLMs with Lightweight Sequential Monitors","date":"2025-06-12","arxiv_id":"2506.10949","repositories_listed":1,"syntology":null},{"url":"/paper/slotpi-physics-informed-object-centric","slug":"slotpi-physics-informed-object-centric","title":"SlotPi: Physics-informed Object-centric Reasoning Models","date":"2025-06-12","arxiv_id":"2506.10778","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/slotpi-physics-informed-object-centric#ran","syntology_url":"https://syntology.ai/paper/2506.10778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10778"}},"official":{"repos":["intell-sci-comput/slotpi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tablerag-a-retrieval-augmented-generation","slug":"tablerag-a-retrieval-augmented-generation","title":"TableRAG: A Retrieval Augmented Generation Framework for Heterogeneous Document Reasoning","date":"2025-06-12","arxiv_id":"2506.10380","repositories_listed":1,"syntology":null},{"url":"/paper/think-before-you-simulate-symbolic-reasoning-1","slug":"think-before-you-simulate-symbolic-reasoning-1","title":"Think before You Simulate: Symbolic Reasoning to Orchestrate Neural Computation for Counterfactual Question Answering","date":"2025-06-12","arxiv_id":"2506.10753","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-open-source-and","slug":"bridging-the-gap-between-open-source-and","title":"Bridging the Gap Between Open-Source and Proprietary LLMs in Table QA","date":"2025-06-11","arxiv_id":"2506.09657","repositories_listed":1,"syntology":null},{"url":"/paper/causalvqa-a-physically-grounded-causal","slug":"causalvqa-a-physically-grounded-causal","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","date":"2025-06-11","arxiv_id":"2506.09943","repositories_listed":1,"syntology":null},{"url":"/paper/learning-efficient-and-generalizable-graph","slug":"learning-efficient-and-generalizable-graph","title":"Learning Efficient and Generalizable Graph Retriever for Knowledge-Graph Question Answering","date":"2025-06-11","arxiv_id":"2506.09645","repositories_listed":1,"syntology":null},{"url":"/paper/med-refl-medical-reasoning-enhancement-via","slug":"med-refl-medical-reasoning-enhancement-via","title":"Med-REFL: Medical Reasoning Enhancement via Self-Corrected Fine-grained Reflection","date":"2025-06-11","arxiv_id":"2506.13793","repositories_listed":1,"syntology":null},{"url":"/paper/omnidrca-parallel-speech-text-foundation","slug":"omnidrca-parallel-speech-text-foundation","title":"OmniDRCA: Parallel Speech-Text Foundation Model via Dual-Resolution Speech Representations and Contrastive Alignment","date":"2025-06-11","arxiv_id":"2506.09349","repositories_listed":1,"syntology":null},{"url":"/paper/outside-knowledge-conversational-video-okcv","slug":"outside-knowledge-conversational-video-okcv","title":"Outside Knowledge Conversational Video (OKCV) Dataset -- Dialoguing over Videos","date":"2025-06-11","arxiv_id":"2506.09953","repositories_listed":1,"syntology":null},{"url":"/paper/reasonmed-a-370k-multi-agent-generated","slug":"reasonmed-a-370k-multi-agent-generated","title":"ReasonMed: A 370K Multi-Agent Generated Dataset for Advancing Medical Reasoning","date":"2025-06-11","arxiv_id":"2506.09513","repositories_listed":1,"syntology":null},{"url":"/paper/v-jepa-2-self-supervised-video-models-enable","slug":"v-jepa-2-self-supervised-video-models-enable","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","date":"2025-06-11","arxiv_id":"2506.09985","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/v-jepa-2-self-supervised-video-models-enable#ran","syntology_url":"https://syntology.ai/paper/2506.09985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09985"}},"official":{"repos":["facebookresearch/vjepa2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/flagevalmm-a-flexible-framework-for","slug":"flagevalmm-a-flexible-framework-for","title":"FlagEvalMM: A Flexible Framework for Comprehensive Multimodal Model Evaluation","date":"2025-06-10","arxiv_id":"2506.09081","repositories_listed":1,"syntology":null},{"url":"/paper/cognitive-weave-synthesizing-abstracted","slug":"cognitive-weave-synthesizing-abstracted","title":"Cognitive Weave: Synthesizing Abstracted Knowledge with a Spatio-Temporal Resonance Graph","date":"2025-06-09","arxiv_id":"2506.08098","repositories_listed":1,"syntology":null},{"url":"/paper/haibu-remud-reasoning-multimodal-ultrasound","slug":"haibu-remud-reasoning-multimodal-ultrasound","title":"HAIBU-ReMUD: Reasoning Multimodal Ultrasound Dataset and Model Bridging to General Specific Domains","date":"2025-06-09","arxiv_id":"2506.07837","repositories_listed":1,"syntology":null},{"url":"/paper/looking-beyond-visible-cues-implicit-video","slug":"looking-beyond-visible-cues-implicit-video","title":"Looking Beyond Visible Cues: Implicit Video Question Answering via Dual-Clue Reasoning","date":"2025-06-09","arxiv_id":"2506.07811","repositories_listed":1,"syntology":null},{"url":"/paper/multi-step-visual-reasoning-with-visual","slug":"multi-step-visual-reasoning-with-visual","title":"Multi-Step Visual Reasoning with Visual Tokens Scaling and Verification","date":"2025-06-08","arxiv_id":"2506.07235","repositories_listed":1,"syntology":null},{"url":"/paper/ecorag-evidentiality-guided-compression-for","slug":"ecorag-evidentiality-guided-compression-for","title":"ECoRAG: Evidentiality-guided Compression for Long Context RAG","date":"2025-06-05","arxiv_id":"2506.05167","repositories_listed":1,"syntology":null},{"url":"/paper/micro-act-mitigate-knowledge-conflict-in","slug":"micro-act-mitigate-knowledge-conflict-in","title":"Micro-Act: Mitigate Knowledge Conflict in Question Answering via Actionable Self-Reasoning","date":"2025-06-05","arxiv_id":"2506.05278","repositories_listed":1,"syntology":null},{"url":"/paper/egovlm-policy-optimization-for-egocentric","slug":"egovlm-policy-optimization-for-egocentric","title":"EgoVLM: Policy Optimization for Egocentric Video Understanding","date":"2025-06-03","arxiv_id":"2506.03097","repositories_listed":1,"syntology":null},{"url":"/paper/failuresensoriq-a-multi-choice-qa-dataset-for","slug":"failuresensoriq-a-multi-choice-qa-dataset-for","title":"FailureSensorIQ: A Multi-Choice QA Dataset for Understanding Sensor Relationships and Failure Modes","date":"2025-06-03","arxiv_id":"2506.03278","repositories_listed":1,"syntology":null},{"url":"/paper/othink-r1-intrinsic-fast-slow-thinking-mode","slug":"othink-r1-intrinsic-fast-slow-thinking-mode","title":"OThink-R1: Intrinsic Fast/Slow Thinking Mode Switching for Over-Reasoning Mitigation","date":"2025-06-03","arxiv_id":"2506.02397","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/othink-r1-intrinsic-fast-slow-thinking-mode#ran","syntology_url":"https://syntology.ai/paper/2506.02397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.02397"}},"official":{"repos":["agenticir-lab/othink-r1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-fine-tuning-llama-3-1-for","slug":"parameter-efficient-fine-tuning-llama-3-1-for","title":"Parameter Efficient Fine Tuning Llama 3.1 for Answering Arabic Legal Questions: A Case Study on Jordanian Laws","date":"2025-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-table-exploring-reinforcement","slug":"reasoning-table-exploring-reinforcement","title":"Reasoning-Table: Exploring Reinforcement Learning for Table Reasoning","date":"2025-06-02","arxiv_id":"2506.01710","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-chunking-and-selection-for-reading","slug":"dynamic-chunking-and-selection-for-reading","title":"Dynamic Chunking and Selection for Reading Comprehension of Ultra-Long Context in Large Language Models","date":"2025-06-01","arxiv_id":"2506.00773","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":1,"n_instrument":8,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-chunking-and-selection-for-reading#ran","syntology_url":"https://syntology.ai/paper/2506.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00773"}},"official":{"repos":["ecnu-text-computing/dcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probing-the-geometry-of-truth-consistency-and","slug":"probing-the-geometry-of-truth-consistency-and","title":"Probing the Geometry of Truth: Consistency and Generalization of Truth Directions in LLMs Across Logical Transformations and Question Answering Tasks","date":"2025-06-01","arxiv_id":"2506.00823","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probing-the-geometry-of-truth-consistency-and#ran","syntology_url":"https://syntology.ai/paper/2506.00823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00823"}},"official":{"repos":["colored-dye/truthfulness_probe_generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pakton-a-multi-agent-framework-for-question","slug":"pakton-a-multi-agent-framework-for-question","title":"PAKTON: A Multi-Agent Framework for Question Answering in Long Legal Agreements","date":"2025-05-31","arxiv_id":"2506.00608","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pakton-a-multi-agent-framework-for-question#ran","syntology_url":"https://syntology.ai/paper/2506.00608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00608"}},"official":{"repos":["petrosrapto/pakton"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/drop-dropout-on-single-epoch-language-model","slug":"drop-dropout-on-single-epoch-language-model","title":"Drop Dropout on Single-Epoch Language Model Pretraining","date":"2025-05-30","arxiv_id":"2505.24788","repositories_listed":1,"syntology":null},{"url":"/paper/lgar-zero-shot-llm-guided-neural-ranking-for","slug":"lgar-zero-shot-llm-guided-neural-ranking-for","title":"LGAR: Zero-Shot LLM-Guided Neural Ranking for Abstract Screening in Systematic Literature Reviews","date":"2025-05-30","arxiv_id":"2505.24757","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-epistemic-markers-in-confidence","slug":"revisiting-epistemic-markers-in-confidence","title":"Revisiting Epistemic Markers in Confidence Estimation: Can Markers Accurately Reflect Large Language Models' Uncertainty?","date":"2025-05-30","arxiv_id":"2505.24778","repositories_listed":1,"syntology":null},{"url":"/paper/videocad-a-large-scale-video-dataset-for","slug":"videocad-a-large-scale-video-dataset-for","title":"VideoCAD: A Large-Scale Video Dataset for Learning UI Interactions and 3D Reasoning from CAD Software","date":"2025-05-30","arxiv_id":"2505.24838","repositories_listed":1,"syntology":null},{"url":"/paper/impromptu-vla-open-weights-and-open-data-for","slug":"impromptu-vla-open-weights-and-open-data-for","title":"Impromptu VLA: Open Weights and Open Data for Driving Vision-Language-Action Models","date":"2025-05-29","arxiv_id":"2505.23757","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-chest-x-rays-like-a-radiologist","slug":"interpreting-chest-x-rays-like-a-radiologist","title":"Interpreting Chest X-rays Like a Radiologist: A Benchmark with Clinical Reasoning","date":"2025-05-29","arxiv_id":"2505.23143","repositories_listed":1,"syntology":null},{"url":"/paper/kvzip-query-agnostic-kv-cache-compression","slug":"kvzip-query-agnostic-kv-cache-compression","title":"KVzip: Query-Agnostic KV Cache Compression with Context Reconstruction","date":"2025-05-29","arxiv_id":"2505.23416","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kvzip-query-agnostic-kv-cache-compression#ran","syntology_url":"https://syntology.ai/paper/2505.23416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23416"}},"official":{"repos":["snu-mllab/kvzip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/multi-sourced-compositional-generalization-in","slug":"multi-sourced-compositional-generalization-in","title":"Multi-Sourced Compositional Generalization in Visual Question Answering","date":"2025-05-29","arxiv_id":"2505.23045","repositories_listed":1,"syntology":null},{"url":"/paper/puzzled-by-puzzles-when-vision-language","slug":"puzzled-by-puzzles-when-vision-language","title":"Puzzled by Puzzles: When Vision-Language Models Can't Take a Hint","date":"2025-05-29","arxiv_id":"2505.23759","repositories_listed":1,"syntology":null},{"url":"/paper/qlip-a-dynamic-quadtree-vision-prior-enhances","slug":"qlip-a-dynamic-quadtree-vision-prior-enhances","title":"QLIP: A Dynamic Quadtree Vision Prior Enhances MLLM Performance Without Retraining","date":"2025-05-29","arxiv_id":"2505.23004","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-document-question-answering-in","slug":"synthetic-document-question-answering-in","title":"Synthetic Document Question Answering in Hungarian","date":"2025-05-29","arxiv_id":"2505.23008","repositories_listed":1,"syntology":null},{"url":"/paper/vau-r1-advancing-video-anomaly-understanding","slug":"vau-r1-advancing-video-anomaly-understanding","title":"VAU-R1: Advancing Video Anomaly Understanding via Reinforcement Fine-Tuning","date":"2025-05-29","arxiv_id":"2505.23504","repositories_listed":1,"syntology":null},{"url":"/paper/vf-eval-evaluating-multimodal-llms-for","slug":"vf-eval-evaluating-multimodal-llms-for","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","date":"2025-05-29","arxiv_id":"2505.23693","repositories_listed":1,"syntology":null},{"url":"/paper/climate-finance-bench","slug":"climate-finance-bench","title":"Climate Finance Bench","date":"2025-05-28","arxiv_id":"2505.22752","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-paraphrase-type-generation-the","slug":"enhancing-paraphrase-type-generation-the","title":"Enhancing Paraphrase Type Generation: The Impact of DPO and RLHF Evaluated with Human-Ranked Data","date":"2025-05-28","arxiv_id":"2506.02018","repositories_listed":1,"syntology":null},{"url":"/paper/vignette-socially-grounded-bias-evaluation","slug":"vignette-socially-grounded-bias-evaluation","title":"VIGNETTE: Socially Grounded Bias Evaluation for Vision-Language Models","date":"2025-05-28","arxiv_id":"2505.22897","repositories_listed":1,"syntology":null},{"url":"/paper/frames-vqa-benchmarking-fine-tuning-1","slug":"frames-vqa-benchmarking-fine-tuning-1","title":"FRAMES-VQA: Benchmarking Fine-Tuning Robustness across Multi-Modal Shifts in Visual Question Answering","date":"2025-05-27","arxiv_id":"2505.21755","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-external-knowledge-input-beyond","slug":"scaling-external-knowledge-input-beyond","title":"Scaling External Knowledge Input Beyond Context Windows of LLMs via Multi-Agent Collaboration","date":"2025-05-27","arxiv_id":"2505.21471","repositories_listed":1,"syntology":null},{"url":"/paper/understand-think-and-answer-advancing-visual","slug":"understand-think-and-answer-advancing-visual","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","date":"2025-05-27","arxiv_id":"2505.20753","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/understand-think-and-answer-advancing-visual#ran","syntology_url":"https://syntology.ai/paper/2505.20753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20753"}},"official":{"repos":["jefferyzhan/griffon"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/amqa-an-adversarial-dataset-for-benchmarking","slug":"amqa-an-adversarial-dataset-for-benchmarking","title":"AMQA: An Adversarial Dataset for Benchmarking Bias of LLMs in Medicine and Healthcare","date":"2025-05-26","arxiv_id":"2505.19562","repositories_listed":1,"syntology":null},{"url":"/paper/automated-text-to-table-for-reasoning","slug":"automated-text-to-table-for-reasoning","title":"Automated Text-to-Table for Reasoning-Intensive Table QA: Pipeline Design and Benchmarking Insights","date":"2025-05-26","arxiv_id":"2505.19563","repositories_listed":1,"syntology":null},{"url":"/paper/bizfinbench-a-business-driven-real-world","slug":"bizfinbench-a-business-driven-real-world","title":"BizFinBench: A Business-Driven Real-World Financial Benchmark for Evaluating LLMs","date":"2025-05-26","arxiv_id":"2505.19457","repositories_listed":1,"syntology":null},{"url":"/paper/culfit-a-fine-grained-cultural-aware-llm","slug":"culfit-a-fine-grained-cultural-aware-llm","title":"CulFiT: A Fine-grained Cultural-aware LLM Training Paradigm via Multilingual Critique Data Synthesis","date":"2025-05-26","arxiv_id":"2505.19484","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/culfit-a-fine-grained-cultural-aware-llm#ran","syntology_url":"https://syntology.ai/paper/2505.19484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19484"}},"official":{"repos":["mmadmax/culfit"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/diagnosing-and-mitigating-modality","slug":"diagnosing-and-mitigating-modality","title":"Diagnosing and Mitigating Modality Interference in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diagnosing-and-mitigating-modality#ran","syntology_url":"https://syntology.ai/paper/2505.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19616"}},"official":null}},{"url":"/paper/doctoragent-rl-a-multi-agent-collaborative","slug":"doctoragent-rl-a-multi-agent-collaborative","title":"DoctorAgent-RL: A Multi-Agent Collaborative Reinforcement Learning System for Multi-Turn Clinical Dialogue","date":"2025-05-26","arxiv_id":"2505.19630","repositories_listed":1,"syntology":null},{"url":"/paper/exante-a-benchmark-for-ex-ante-inference-in","slug":"exante-a-benchmark-for-ex-ante-inference-in","title":"ExAnte: A Benchmark for Ex-Ante Inference in Large Language Models","date":"2025-05-26","arxiv_id":"2505.19533","repositories_listed":1,"syntology":null},{"url":"/paper/genki-enhancing-open-domain-question","slug":"genki-enhancing-open-domain-question","title":"GenKI: Enhancing Open-Domain Question Answering with Knowledge Integration and Controllable Generation in Large Language Models","date":"2025-05-26","arxiv_id":"2505.19660","repositories_listed":1,"syntology":null},{"url":"/paper/graphgen-enhancing-supervised-fine-tuning-for","slug":"graphgen-enhancing-supervised-fine-tuning-for","title":"GraphGen: Enhancing Supervised Fine-Tuning for LLMs with Knowledge-Driven Synthetic Data Generation","date":"2025-05-26","arxiv_id":"2505.20416","repositories_listed":1,"syntology":null},{"url":"/paper/knowtrace-bootstrapping-iterative-retrieval","slug":"knowtrace-bootstrapping-iterative-retrieval","title":"KnowTrace: Bootstrapping Iterative Retrieval-Augmented Generation with Structured Knowledge Tracing","date":"2025-05-26","arxiv_id":"2505.20245","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-meet-knowledge-graphs-1","slug":"large-language-models-meet-knowledge-graphs-1","title":"Large Language Models Meet Knowledge Graphs for Question Answering: Synthesis and Opportunities","date":"2025-05-26","arxiv_id":"2505.20099","repositories_listed":1,"syntology":null},{"url":"/paper/mangavqa-and-mangalmm-a-benchmark-and","slug":"mangavqa-and-mangalmm-a-benchmark-and","title":"MangaVQA and MangaLMM: A Benchmark and Specialized Model for Multimodal Manga Understanding","date":"2025-05-26","arxiv_id":"2505.20298","repositories_listed":1,"syntology":null},{"url":"/paper/masksearch-a-universal-pre-training-framework","slug":"masksearch-a-universal-pre-training-framework","title":"MASKSEARCH: A Universal Pre-Training Framework to Enhance Agentic Search Capability","date":"2025-05-26","arxiv_id":"2505.20285","repositories_listed":1,"syntology":null},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/mm-prompt-cross-modal-prompt-tuning-for","slug":"mm-prompt-cross-modal-prompt-tuning-for","title":"MM-Prompt: Cross-Modal Prompt Tuning for Continual Visual Question Answering","date":"2025-05-26","arxiv_id":"2505.19455","repositories_listed":1,"syntology":null},{"url":"/paper/neusym-rag-hybrid-neural-symbolic-retrieval","slug":"neusym-rag-hybrid-neural-symbolic-retrieval","title":"NeuSym-RAG: Hybrid Neural Symbolic Retrieval with Multiview Structuring for PDF Question Answering","date":"2025-05-26","arxiv_id":"2505.19754","repositories_listed":1,"syntology":null},{"url":"/paper/visualized-text-to-image-retrieval","slug":"visualized-text-to-image-retrieval","title":"Visualized Text-to-Image Retrieval","date":"2025-05-26","arxiv_id":"2505.20291","repositories_listed":1,"syntology":null},{"url":"/paper/wximpactbench-a-disruptive-weather-impact","slug":"wximpactbench-a-disruptive-weather-impact","title":"WXImpactBench: A Disruptive Weather Impact Understanding Benchmark for Evaluating Large Language Models","date":"2025-05-26","arxiv_id":"2505.20249","repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hypercube-rag-hypercube-based-retrieval","slug":"hypercube-rag-hypercube-based-retrieval","title":"Hypercube-RAG: Hypercube-Based Retrieval-Augmented Generation for In-domain Scientific Question-Answering","date":"2025-05-25","arxiv_id":"2505.19288","repositories_listed":1,"syntology":null},{"url":"/paper/medical-large-vision-language-models-with","slug":"medical-large-vision-language-models-with","title":"Medical Large Vision Language Models with Multi-Image Visual Ability","date":"2025-05-25","arxiv_id":"2505.19031","repositories_listed":1,"syntology":null},{"url":"/paper/satori-r1-incentivizing-multimodal-reasoning","slug":"satori-r1-incentivizing-multimodal-reasoning","title":"SATORI-R1: Incentivizing Multimodal Reasoning with Spatial Grounding and Verifiable Rewards","date":"2025-05-25","arxiv_id":"2505.19094","repositories_listed":1,"syntology":null},{"url":"/paper/self-critique-guided-iterative-reasoning-for","slug":"self-critique-guided-iterative-reasoning-for","title":"Self-Critique Guided Iterative Reasoning for Multi-hop Question Answering","date":"2025-05-25","arxiv_id":"2505.19112","repositories_listed":1,"syntology":null},{"url":"/paper/situatedthinker-grounding-llm-reasoning-with","slug":"situatedthinker-grounding-llm-reasoning-with","title":"SituatedThinker: Grounding LLM Reasoning with Real-World through Situated Thinking","date":"2025-05-25","arxiv_id":"2505.19300","repositories_listed":1,"syntology":null},{"url":"/paper/vtool-r1-vlms-learn-to-think-with-images-via","slug":"vtool-r1-vlms-learn-to-think-with-images-via","title":"VTool-R1: VLMs Learn to Think with Images via Reinforcement Learning on Multimodal Tool Use","date":"2025-05-25","arxiv_id":"2505.19255","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtool-r1-vlms-learn-to-think-with-images-via#ran","syntology_url":"https://syntology.ai/paper/2505.19255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19255"}},"official":null}}],"record_sha256":"520a1da586e2ac62d3540723a7f4582d8fbf51b1673c1b14f53c41356435e095","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}