{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/ran/1","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":13,"rows_per_page":100,"rows":[1,100],"of":1274,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering/papers/ran/1","prev":null,"next":"/task/question-answering/papers/ran/2","papers":[{"url":"/paper/rag-r1-incentivize-the-search-and-reasoning","slug":"rag-r1-incentivize-the-search-and-reasoning","title":"RAG-R1 : Incentivize the Search and Reasoning Capabilities of LLMs through Multi-query Parallelism","date":"2025-06-30","arxiv_id":"2507.02962","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rag-r1-incentivize-the-search-and-reasoning#ran","syntology_url":"https://syntology.ai/paper/2507.02962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02962"}},"official":null}},{"url":"/paper/llava-scissor-token-compression-with-semantic","slug":"llava-scissor-token-compression-with-semantic","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","date":"2025-06-27","arxiv_id":"2506.21862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llava-scissor-token-compression-with-semantic#ran","syntology_url":"https://syntology.ai/paper/2506.21862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21862"}},"official":{"repos":["HumanMLLM/LLaVA-Scissor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/response-quality-assessment-for-retrieval","slug":"response-quality-assessment-for-retrieval","title":"Response Quality Assessment for Retrieval-Augmented Generation via Conditional Conformal Factuality","date":"2025-06-26","arxiv_id":"2506.20978","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/response-quality-assessment-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2506.20978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20978"}},"official":{"repos":["n4feng/ResponseQualityAssessment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evolving-prompts-in-context-an-open-ended","slug":"evolving-prompts-in-context-an-open-ended","title":"Evolving Prompts In-Context: An Open-ended, Self-replicating Perspective","date":"2025-06-22","arxiv_id":"2506.17930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evolving-prompts-in-context-an-open-ended#ran","syntology_url":"https://syntology.ai/paper/2506.17930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17930"}},"official":{"repos":["jianyu-cs/promptquine"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/seqpe-transformer-with-sequential-position","slug":"seqpe-transformer-with-sequential-position","title":"SeqPE: Transformer with Sequential Position Encoding","date":"2025-06-16","arxiv_id":"2506.13277","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seqpe-transformer-with-sequential-position#ran","syntology_url":"https://syntology.ai/paper/2506.13277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13277"}},"official":{"repos":["ghrua/seqpe"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/simpledoc-multi-modal-document-understanding","slug":"simpledoc-multi-modal-document-understanding","title":"SimpleDoc: Multi-Modal Document Understanding with Dual-Cue Page Retrieval and Iterative Refinement","date":"2025-06-16","arxiv_id":"2506.14035","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simpledoc-multi-modal-document-understanding#ran","syntology_url":"https://syntology.ai/paper/2506.14035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14035"}},"official":{"repos":["ag2ai/simpledoc"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slotpi-physics-informed-object-centric","slug":"slotpi-physics-informed-object-centric","title":"SlotPi: Physics-informed Object-centric Reasoning Models","date":"2025-06-12","arxiv_id":"2506.10778","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/slotpi-physics-informed-object-centric#ran","syntology_url":"https://syntology.ai/paper/2506.10778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10778"}},"official":{"repos":["intell-sci-comput/slotpi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-jepa-2-self-supervised-video-models-enable","slug":"v-jepa-2-self-supervised-video-models-enable","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","date":"2025-06-11","arxiv_id":"2506.09985","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/v-jepa-2-self-supervised-video-models-enable#ran","syntology_url":"https://syntology.ai/paper/2506.09985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09985"}},"official":{"repos":["facebookresearch/vjepa2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-open-source-software-toolkit-benchmark","slug":"an-open-source-software-toolkit-benchmark","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","date":"2025-06-10","arxiv_id":"2506.09172","repositories_listed":0,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-open-source-software-toolkit-benchmark#ran","syntology_url":"https://syntology.ai/paper/2506.09172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09172"}},"official":null}},{"url":"/paper/memoir-lifelong-model-editing-with-minimal","slug":"memoir-lifelong-model-editing-with-minimal","title":"MEMOIR: Lifelong Model Editing with Minimal Overwrite and Informed Retention for LLMs","date":"2025-06-09","arxiv_id":"2506.07899","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memoir-lifelong-model-editing-with-minimal#ran","syntology_url":"https://syntology.ai/paper/2506.07899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07899"}},"official":null}},{"url":"/paper/othink-r1-intrinsic-fast-slow-thinking-mode","slug":"othink-r1-intrinsic-fast-slow-thinking-mode","title":"OThink-R1: Intrinsic Fast/Slow Thinking Mode Switching for Over-Reasoning Mitigation","date":"2025-06-03","arxiv_id":"2506.02397","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/othink-r1-intrinsic-fast-slow-thinking-mode#ran","syntology_url":"https://syntology.ai/paper/2506.02397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.02397"}},"official":{"repos":["agenticir-lab/othink-r1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-chunking-and-selection-for-reading","slug":"dynamic-chunking-and-selection-for-reading","title":"Dynamic Chunking and Selection for Reading Comprehension of Ultra-Long Context in Large Language Models","date":"2025-06-01","arxiv_id":"2506.00773","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":1,"n_instrument":8,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-chunking-and-selection-for-reading#ran","syntology_url":"https://syntology.ai/paper/2506.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00773"}},"official":{"repos":["ecnu-text-computing/dcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probing-the-geometry-of-truth-consistency-and","slug":"probing-the-geometry-of-truth-consistency-and","title":"Probing the Geometry of Truth: Consistency and Generalization of Truth Directions in LLMs Across Logical Transformations and Question Answering Tasks","date":"2025-06-01","arxiv_id":"2506.00823","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probing-the-geometry-of-truth-consistency-and#ran","syntology_url":"https://syntology.ai/paper/2506.00823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00823"}},"official":{"repos":["colored-dye/truthfulness_probe_generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pakton-a-multi-agent-framework-for-question","slug":"pakton-a-multi-agent-framework-for-question","title":"PAKTON: A Multi-Agent Framework for Question Answering in Long Legal Agreements","date":"2025-05-31","arxiv_id":"2506.00608","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pakton-a-multi-agent-framework-for-question#ran","syntology_url":"https://syntology.ai/paper/2506.00608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00608"}},"official":{"repos":["petrosrapto/pakton"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/kvzip-query-agnostic-kv-cache-compression","slug":"kvzip-query-agnostic-kv-cache-compression","title":"KVzip: Query-Agnostic KV Cache Compression with Context Reconstruction","date":"2025-05-29","arxiv_id":"2505.23416","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kvzip-query-agnostic-kv-cache-compression#ran","syntology_url":"https://syntology.ai/paper/2505.23416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23416"}},"official":{"repos":["snu-mllab/kvzip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/understand-think-and-answer-advancing-visual","slug":"understand-think-and-answer-advancing-visual","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","date":"2025-05-27","arxiv_id":"2505.20753","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/understand-think-and-answer-advancing-visual#ran","syntology_url":"https://syntology.ai/paper/2505.20753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20753"}},"official":{"repos":["jefferyzhan/griffon"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diagnosing-and-mitigating-modality","slug":"diagnosing-and-mitigating-modality","title":"Diagnosing and Mitigating Modality Interference in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diagnosing-and-mitigating-modality#ran","syntology_url":"https://syntology.ai/paper/2505.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19616"}},"official":null}},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infochartqa-a-benchmark-for-multimodal","slug":"infochartqa-a-benchmark-for-multimodal","title":"InfoChartQA: A Benchmark for Multimodal Question Answering on Infographic Charts","date":"2025-05-25","arxiv_id":"2505.19028","repositories_listed":3,"syntology":{"n":20,"n_ran":19,"n_constructed":0,"n_ran_checked":16,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":3,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infochartqa-a-benchmark-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2505.19028","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19028"}},"official":{"repos":["cooldawnant/infochartqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/vtool-r1-vlms-learn-to-think-with-images-via","slug":"vtool-r1-vlms-learn-to-think-with-images-via","title":"VTool-R1: VLMs Learn to Think with Images via Reinforcement Learning on Multimodal Tool Use","date":"2025-05-25","arxiv_id":"2505.19255","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtool-r1-vlms-learn-to-think-with-images-via#ran","syntology_url":"https://syntology.ai/paper/2505.19255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19255"}},"official":null}},{"url":"/paper/deep-video-discovery-agentic-search-with-tool","slug":"deep-video-discovery-agentic-search-with-tool","title":"Deep Video Discovery: Agentic Search with Tool Use for Long-form Video Understanding","date":"2025-05-23","arxiv_id":"2505.18079","repositories_listed":0,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-video-discovery-agentic-search-with-tool#ran","syntology_url":"https://syntology.ai/paper/2505.18079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18079"}},"official":null}},{"url":"/paper/cxreasonbench-a-benchmark-for-evaluating","slug":"cxreasonbench-a-benchmark-for-evaluating","title":"CXReasonBench: A Benchmark for Evaluating Structured Diagnostic Reasoning in Chest X-rays","date":"2025-05-23","arxiv_id":"2505.18087","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cxreasonbench-a-benchmark-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2505.18087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18087"}},"official":{"repos":["ttumyche/cxreasonbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/danmakutppbench-a-multi-modal-benchmark-for","slug":"danmakutppbench-a-multi-modal-benchmark-for","title":"DanmakuTPPBench: A Multi-modal Benchmark for Temporal Point Process Modeling and Understanding","date":"2025-05-23","arxiv_id":"2505.18411","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/danmakutppbench-a-multi-modal-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2505.18411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18411"}},"official":{"repos":["frenkie-chiang/danmakutppbench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/benchmarking-retrieval-augmented-multimomal","slug":"benchmarking-retrieval-augmented-multimomal","title":"Benchmarking Retrieval-Augmented Multimomal Generation for Document Question Answering","date":"2025-05-22","arxiv_id":"2505.16470","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-multimomal#ran","syntology_url":"https://syntology.ai/paper/2505.16470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16470"}},"official":{"repos":["mmdocrag/mmdocrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/o-2-searcher-a-searching-based-agent-model","slug":"o-2-searcher-a-searching-based-agent-model","title":"O$^2$-Searcher: A Searching-based Agent Model for Open-Domain Open-Ended Question Answering","date":"2025-05-22","arxiv_id":"2505.16582","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/o-2-searcher-a-searching-based-agent-model#ran","syntology_url":"https://syntology.ai/paper/2505.16582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16582"}},"official":{"repos":["acade-mate/o2-searcher"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spatialscore-towards-unified-evaluation-for","slug":"spatialscore-towards-unified-evaluation-for","title":"SpatialScore: Towards Unified Evaluation for Multimodal Spatial Understanding","date":"2025-05-22","arxiv_id":"2505.17012","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatialscore-towards-unified-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.17012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17012"}},"official":{"repos":["haoningwu3639/SpatialScore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-visual-feature-space-for","slug":"exploring-the-visual-feature-space-for","title":"Exploring The Visual Feature Space for Multimodal Neural Decoding","date":"2025-05-21","arxiv_id":"2505.15755","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-visual-feature-space-for#ran","syntology_url":"https://syntology.ai/paper/2505.15755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15755"}},"official":null}},{"url":"/paper/raven-query-guided-representation-alignment","slug":"raven-query-guided-representation-alignment","title":"RAVEN: Query-Guided Representation Alignment for Question Answering over Audio, Video, Embedded Sensors, and Natural Language","date":"2025-05-21","arxiv_id":"2505.17114","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/raven-query-guided-representation-alignment#ran","syntology_url":"https://syntology.ai/paper/2505.17114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17114"}},"official":{"repos":["bashlab/raven"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/medagentboard-benchmarking-multi-agent","slug":"medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","arxiv_id":"2505.12371","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentboard-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.12371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12371"}},"official":{"repos":["yhzhu99/medagentboard"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/scent-of-knowledge-optimizing-search-enhanced","slug":"scent-of-knowledge-optimizing-search-enhanced","title":"Scent of Knowledge: Optimizing Search-Enhanced Reasoning with Information Foraging","date":"2025-05-14","arxiv_id":"2505.09316","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scent-of-knowledge-optimizing-search-enhanced#ran","syntology_url":"https://syntology.ai/paper/2505.09316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.09316"}},"official":{"repos":["qhjqhj00/inforage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omgm-orchestrate-multiple-granularities-and","slug":"omgm-orchestrate-multiple-granularities-and","title":"OMGM: Orchestrate Multiple Granularities and Modalities for Efficient Multimodal Retrieval","date":"2025-05-10","arxiv_id":"2505.07879","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/omgm-orchestrate-multiple-granularities-and#ran","syntology_url":"https://syntology.ai/paper/2505.07879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07879"}},"official":null}},{"url":"/paper/continuous-thought-machines","slug":"continuous-thought-machines","title":"Continuous Thought Machines","date":"2025-05-08","arxiv_id":"2505.05522","repositories_listed":1,"syntology":{"n":34,"n_ran":21,"n_constructed":9,"n_ran_checked":16,"n_instrument":5,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":1,"phrase":"21 ran (of which 9 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/continuous-thought-machines#ran","syntology_url":"https://syntology.ai/paper/2505.05522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05522"}},"official":{"repos":["SakanaAI/continuous-thought-machines"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":9,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/llama-omni2-llm-based-real-time-spoken","slug":"llama-omni2-llm-based-real-time-spoken","title":"LLaMA-Omni2: LLM-based Real-time Spoken Chatbot with Autoregressive Streaming Speech Synthesis","date":"2025-05-05","arxiv_id":"2505.02625","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":3,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llama-omni2-llm-based-real-time-spoken#ran","syntology_url":"https://syntology.ai/paper/2505.02625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02625"}},"official":{"repos":["ictnlp/llama-omni2"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/rtv-bench-benchmarking-mllm-continuous","slug":"rtv-bench-benchmarking-mllm-continuous","title":"RTV-Bench: Benchmarking MLLM Continuous Perception, Understanding and Reasoning through Real-Time Video","date":"2025-05-04","arxiv_id":"2505.02064","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rtv-bench-benchmarking-mllm-continuous#ran","syntology_url":"https://syntology.ai/paper/2505.02064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02064"}},"official":{"repos":["ljungang/rtv-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adcare-vlm-leveraging-large-vision-language","slug":"adcare-vlm-leveraging-large-vision-language","title":"AdCare-VLM: Leveraging Large Vision Language Model (LVLM) to Monitor Long-Term Medication Adherence and Care","date":"2025-05-01","arxiv_id":"2505.00275","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/adcare-vlm-leveraging-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.00275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00275"}},"official":{"repos":["asad14053/AdCare-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/unlearning-sensitive-information-in","slug":"unlearning-sensitive-information-in","title":"Unlearning Sensitive Information in Multimodal LLMs: Benchmark and Attack-Defense Evaluation","date":"2025-05-01","arxiv_id":"2505.01456","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unlearning-sensitive-information-in#ran","syntology_url":"https://syntology.ai/paper/2505.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.01456"}},"official":{"repos":["vaidehi99/unlok-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-evaluative-thinking-meta-policy","slug":"toward-evaluative-thinking-meta-policy","title":"Toward Evaluative Thinking: Meta Policy Optimization with Evolving Reward Models","date":"2025-04-28","arxiv_id":"2504.20157","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/toward-evaluative-thinking-meta-policy#ran","syntology_url":"https://syntology.ai/paper/2504.20157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20157"}},"official":{"repos":["minnesotanlp/mpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-audio-technical-report","slug":"kimi-audio-technical-report","title":"Kimi-Audio Technical Report","date":"2025-04-25","arxiv_id":"2504.18425","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kimi-audio-technical-report#ran","syntology_url":"https://syntology.ai/paper/2504.18425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18425"}},"official":{"repos":["moonshotai/kimi-audio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-attribute-with-attention","slug":"learning-to-attribute-with-attention","title":"Learning to Attribute with Attention","date":"2025-04-18","arxiv_id":"2504.13752","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-attribute-with-attention#ran","syntology_url":"https://syntology.ai/paper/2504.13752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13752"}},"official":{"repos":["madrylab/at2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ai2-scholar-qa-organized-literature-synthesis","slug":"ai2-scholar-qa-organized-literature-synthesis","title":"Ai2 Scholar QA: Organized Literature Synthesis with Attribution","date":"2025-04-15","arxiv_id":"2504.10861","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai2-scholar-qa-organized-literature-synthesis#ran","syntology_url":"https://syntology.ai/paper/2504.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10861"}},"official":{"repos":["allenai/ai2-scholarqa-lib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rankalign-a-ranking-view-of-the-generator","slug":"rankalign-a-ranking-view-of-the-generator","title":"RankAlign: A Ranking View of the Generator-Validator Gap in Large Language Models","date":"2025-04-15","arxiv_id":"2504.11381","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rankalign-a-ranking-view-of-the-generator#ran","syntology_url":"https://syntology.ai/paper/2504.11381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11381"}},"official":{"repos":["juand-r/rankalign"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-biopharmaceuticals-retrieval","slug":"benchmarking-biopharmaceuticals-retrieval","title":"Benchmarking Biopharmaceuticals Retrieval-Augmented Generation Evaluation","date":"2025-04-15","arxiv_id":"2504.12342","repositories_listed":0,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-biopharmaceuticals-retrieval#ran","syntology_url":"https://syntology.ai/paper/2504.12342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.12342"}},"official":null}},{"url":"/paper/pixel-sail-single-transformer-for-pixel","slug":"pixel-sail-single-transformer-for-pixel","title":"Pixel-SAIL: Single Transformer For Pixel-Grounded Understanding","date":"2025-04-14","arxiv_id":"2504.10465","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pixel-sail-single-transformer-for-pixel#ran","syntology_url":"https://syntology.ai/paper/2504.10465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10465"}},"official":{"repos":["magic-research/Sa2VA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/out-of-style-rag-s-fragility-to-linguistic","slug":"out-of-style-rag-s-fragility-to-linguistic","title":"Out of Style: RAG's Fragility to Linguistic Variation","date":"2025-04-11","arxiv_id":"2504.08231","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/out-of-style-rag-s-fragility-to-linguistic#ran","syntology_url":"https://syntology.ai/paper/2504.08231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.08231"}},"official":{"repos":["springcty/rag-fragility-to-linguistic-variation"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chartqapro-a-more-diverse-and-challenging","slug":"chartqapro-a-more-diverse-and-challenging","title":"ChartQAPro: A More Diverse and Challenging Benchmark for Chart Question Answering","date":"2025-04-07","arxiv_id":"2504.05506","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartqapro-a-more-diverse-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.05506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05506"}},"official":{"repos":["vis-nlp/chartqapro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/single-pass-document-scanning-for-question","slug":"single-pass-document-scanning-for-question","title":"Single-Pass Document Scanning for Question Answering","date":"2025-04-04","arxiv_id":"2504.03101","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/single-pass-document-scanning-for-question#ran","syntology_url":"https://syntology.ai/paper/2504.03101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03101"}},"official":{"repos":["mambaretriever/mambaretriever"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/opendrivevla-towards-end-to-end-autonomous","slug":"opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","arxiv_id":"2503.23463","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/opendrivevla-towards-end-to-end-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.23463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23463"}},"official":{"repos":["DriveVLA/OpenDriveVLA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/facebench-a-multi-view-multi-level-facial","slug":"facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","arxiv_id":"2503.21457","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/facebench-a-multi-view-multi-level-facial#ran","syntology_url":"https://syntology.ai/paper/2503.21457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21457"}},"official":{"repos":["cvi-szu/facebench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/med3dvlm-an-efficient-vision-language-model","slug":"med3dvlm-an-efficient-vision-language-model","title":"Med3DVLM: An Efficient Vision-Language Model for 3D Medical Image Analysis","date":"2025-03-25","arxiv_id":"2503.20047","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/med3dvlm-an-efficient-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.20047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20047"}},"official":{"repos":["mirthai/med3dvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/graspcot-integrating-physical-property","slug":"graspcot-integrating-physical-property","title":"GraspCoT: Integrating Physical Property Reasoning for 6-DoF Grasping under Flexible Language Instructions","date":"2025-03-20","arxiv_id":"2503.16013","repositories_listed":0,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graspcot-integrating-physical-property#ran","syntology_url":"https://syntology.ai/paper/2503.16013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16013"}},"official":null}},{"url":"/paper/agentic-keyframe-search-for-video-question","slug":"agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","arxiv_id":"2503.16032","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-keyframe-search-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2503.16032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16032"}},"official":{"repos":["fansunqi/akeys"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mdocagent-a-multi-modal-multi-agent-framework","slug":"mdocagent-a-multi-modal-multi-agent-framework","title":"MDocAgent: A Multi-Modal Multi-Agent Framework for Document Understanding","date":"2025-03-18","arxiv_id":"2503.13964","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdocagent-a-multi-modal-multi-agent-framework#ran","syntology_url":"https://syntology.ai/paper/2503.13964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13964"}},"official":{"repos":["aiming-lab/mdocagent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/marten-visual-question-answering-with-mask","slug":"marten-visual-question-answering-with-mask","title":"Marten: Visual Question Answering with Mask Generation for Multi-modal Document Understanding","date":"2025-03-18","arxiv_id":"2503.14140","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/marten-visual-question-answering-with-mask#ran","syntology_url":"https://syntology.ai/paper/2503.14140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14140"}},"official":null}},{"url":"/paper/videomind-a-chain-of-lora-agent-for-long","slug":"videomind-a-chain-of-lora-agent-for-long","title":"VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning","date":"2025-03-17","arxiv_id":"2503.13444","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomind-a-chain-of-lora-agent-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.13444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13444"}},"official":{"repos":["yeliudev/VideoMind"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-outperforms-supervised","slug":"reinforcement-learning-outperforms-supervised","title":"Reinforcement Learning Outperforms Supervised Fine-Tuning: A Case Study on Audio Question Answering","date":"2025-03-14","arxiv_id":"2503.11197","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-outperforms-supervised#ran","syntology_url":"https://syntology.ai/paper/2503.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11197"}},"official":{"repos":["xiaomi-research/r1-aqa","huggingface.co/mispeech/r1-aqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-augmented-generation-with-1","slug":"retrieval-augmented-generation-with-1","title":"Retrieval-Augmented Generation with Hierarchical Knowledge","date":"2025-03-13","arxiv_id":"2503.10150","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retrieval-augmented-generation-with-1#ran","syntology_url":"https://syntology.ai/paper/2503.10150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10150"}},"official":{"repos":["hhy-huang/HiRAG"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/search-r1-training-llms-to-reason-and","slug":"search-r1-training-llms-to-reason-and","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","date":"2025-03-12","arxiv_id":"2503.09516","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/search-r1-training-llms-to-reason-and#ran","syntology_url":"https://syntology.ai/paper/2503.09516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09516"}},"official":{"repos":["petergriffinjin/search-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/simlingo-vision-only-closed-loop-autonomous","slug":"simlingo-vision-only-closed-loop-autonomous","title":"SimLingo: Vision-Only Closed-Loop Autonomous Driving with Language-Action Alignment","date":"2025-03-12","arxiv_id":"2503.09594","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/simlingo-vision-only-closed-loop-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.09594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09594"}},"official":null}},{"url":"/paper/medagentsbench-benchmarking-thinking-models","slug":"medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","arxiv_id":"2503.07459","repositories_listed":1,"syntology":{"n":23,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":4,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentsbench-benchmarking-thinking-models#ran","syntology_url":"https://syntology.ai/paper/2503.07459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07459"}},"official":{"repos":["gersteinlab/medagents-benchmark"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/dspnet-dual-vision-scene-perception-for","slug":"dspnet-dual-vision-scene-perception-for","title":"DSPNet: Dual-vision Scene Perception for Robust 3D Question Answering","date":"2025-03-05","arxiv_id":"2503.03190","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dspnet-dual-vision-scene-perception-for#ran","syntology_url":"https://syntology.ai/paper/2503.03190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03190"}},"official":{"repos":["LZ-CH/DSPNet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/addressing-overprescribing-challenges-fine","slug":"addressing-overprescribing-challenges-fine","title":"Addressing Overprescribing Challenges: Fine-Tuning Large Language Models for Medication Recommendation Tasks","date":"2025-03-05","arxiv_id":"2503.03687","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/addressing-overprescribing-challenges-fine#ran","syntology_url":"https://syntology.ai/paper/2503.03687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03687"}},"official":{"repos":["zzhustc2016/lamo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/egolife-towards-egocentric-life-assistant","slug":"egolife-towards-egocentric-life-assistant","title":"EgoLife: Towards Egocentric Life Assistant","date":"2025-03-05","arxiv_id":"2503.03803","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":9,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/egolife-towards-egocentric-life-assistant#ran","syntology_url":"https://syntology.ai/paper/2503.03803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03803"}},"official":{"repos":["evolvinglmms-lab/egolife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-causal-relation-alignment-for-1","slug":"cross-modal-causal-relation-alignment-for-1","title":"Cross-modal Causal Relation Alignment for Video Question Grounding","date":"2025-03-05","arxiv_id":"2503.07635","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":12,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":16,"phrase":"13 ran (of which 12 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cross-modal-causal-relation-alignment-for-1#ran","syntology_url":"https://syntology.ai/paper/2503.07635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07635"}},"official":{"repos":["wissingchen/cra-gqa"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":12,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-video-question-answering-with-in","slug":"streaming-video-question-answering-with-in","title":"Streaming Video Question-Answering with In-context Video KV-Cache Retrieval","date":"2025-03-01","arxiv_id":"2503.00540","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/streaming-video-question-answering-with-in#ran","syntology_url":"https://syntology.ai/paper/2503.00540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00540"}},"official":{"repos":["becomebright/rekv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vidorag-visual-document-retrieval-augmented","slug":"vidorag-visual-document-retrieval-augmented","title":"ViDoRAG: Visual Document Retrieval-Augmented Generation via Dynamic Iterative Reasoning Agents","date":"2025-02-25","arxiv_id":"2502.18017","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/vidorag-visual-document-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2502.18017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18017"}},"official":{"repos":["Alibaba-NLP/ViDoRAG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/baichuan-audio-a-unified-framework-for-end-to","slug":"baichuan-audio-a-unified-framework-for-end-to","title":"Baichuan-Audio: A Unified Framework for End-to-End Speech Interaction","date":"2025-02-24","arxiv_id":"2502.17239","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/baichuan-audio-a-unified-framework-for-end-to#ran","syntology_url":"https://syntology.ai/paper/2502.17239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17239"}},"official":{"repos":["baichuan-inc/baichuan-audio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hippo-enhancing-the-table-understanding","slug":"hippo-enhancing-the-table-understanding","title":"HIPPO: Enhancing the Table Understanding Capability of Large Language Models through Hybrid-Modal Preference Optimization","date":"2025-02-24","arxiv_id":"2502.17315","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hippo-enhancing-the-table-understanding#ran","syntology_url":"https://syntology.ai/paper/2502.17315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17315"}},"official":{"repos":["neuir/hippo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-faithfulness-of-chains-of-thought","slug":"measuring-faithfulness-of-chains-of-thought","title":"Measuring Faithfulness of Chains of Thought by Unlearning Reasoning Steps","date":"2025-02-20","arxiv_id":"2502.14829","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-faithfulness-of-chains-of-thought#ran","syntology_url":"https://syntology.ai/paper/2502.14829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14829"}},"official":{"repos":["technion-cs-nlp/parametric-faithfulness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/peerqa-a-scientific-question-answering","slug":"peerqa-a-scientific-question-answering","title":"PeerQA: A Scientific Question Answering Dataset from Peer Reviews","date":"2025-02-19","arxiv_id":"2502.13668","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/peerqa-a-scientific-question-answering#ran","syntology_url":"https://syntology.ai/paper/2502.13668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13668"}},"official":{"repos":["ukplab/peerqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mudaf-long-context-multi-document-attention","slug":"mudaf-long-context-multi-document-attention","title":"MuDAF: Long-Context Multi-Document Attention Focusing through Contrastive Learning on Attention Heads","date":"2025-02-19","arxiv_id":"2502.13963","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mudaf-long-context-multi-document-attention#ran","syntology_url":"https://syntology.ai/paper/2502.13963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13963"}},"official":{"repos":["NeosKnight233/MuDAF"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied","slug":"cityeqa-a-hierarchical-llm-agent-on-embodied","title":"CityEQA: A Hierarchical LLM Agent on Embodied Question Answering Benchmark in City Space","date":"2025-02-18","arxiv_id":"2502.12532","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied#ran","syntology_url":"https://syntology.ai/paper/2502.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12532"}},"official":{"repos":["biluyong/cityeqa","tsinghua-fib-lab/CityEQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/re-align-aligning-vision-language-models-via","slug":"re-align-aligning-vision-language-models-via","title":"Re-Align: Aligning Vision Language Models via Retrieval-Augmented Direct Preference Optimization","date":"2025-02-18","arxiv_id":"2502.13146","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/re-align-aligning-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2502.13146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13146"}},"official":{"repos":["taco-group/re-align"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-mirage-of-model-editing-revisiting","slug":"the-mirage-of-model-editing-revisiting","title":"The Mirage of Model Editing: Revisiting Evaluation in the Wild","date":"2025-02-16","arxiv_id":"2502.11177","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-mirage-of-model-editing-revisiting#ran","syntology_url":"https://syntology.ai/paper/2502.11177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11177"}},"official":{"repos":["wanliyoung/revisit-editing-evaluation","zjunlp/easyedit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/svbench-a-benchmark-with-temporal-multi-turn","slug":"svbench-a-benchmark-with-temporal-multi-turn","title":"SVBench: A Benchmark with Temporal Multi-Turn Dialogues for Streaming Video Understanding","date":"2025-02-15","arxiv_id":"2502.10810","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svbench-a-benchmark-with-temporal-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2502.10810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10810"}},"official":{"repos":["yzy-bupt/SVBench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/egotextvqa-towards-egocentric-scene-text","slug":"egotextvqa-towards-egocentric-scene-text","title":"EgoTextVQA: Towards Egocentric Scene-Text Aware Video Question Answering","date":"2025-02-11","arxiv_id":"2502.07411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egotextvqa-towards-egocentric-scene-text#ran","syntology_url":"https://syntology.ai/paper/2502.07411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07411"}},"official":{"repos":["zhousheng97/egotextvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/docmia-document-level-membership-inference","slug":"docmia-document-level-membership-inference","title":"DocMIA: Document-Level Membership Inference Attacks against DocVQA Models","date":"2025-02-06","arxiv_id":"2502.03692","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/docmia-document-level-membership-inference#ran","syntology_url":"https://syntology.ai/paper/2502.03692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03692"}},"official":{"repos":["khanhnguyen21006/mia_docvqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/kbqa-o1-agentic-knowledge-base-question","slug":"kbqa-o1-agentic-knowledge-base-question","title":"KBQA-o1: Agentic Knowledge Base Question Answering with Monte Carlo Tree Search","date":"2025-01-31","arxiv_id":"2501.18922","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/kbqa-o1-agentic-knowledge-base-question#ran","syntology_url":"https://syntology.ai/paper/2501.18922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.18922"}},"official":{"repos":["lhrlab/kbqa-o1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/infty-video-a-training-free-approach-to-long","slug":"infty-video-a-training-free-approach-to-long","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","date":"2025-01-31","arxiv_id":"2501.19098","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infty-video-a-training-free-approach-to-long#ran","syntology_url":"https://syntology.ai/paper/2501.19098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19098"}},"official":{"repos":["deep-spin/infinite-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-and-boosting-the-power-of-fine","slug":"analyzing-and-boosting-the-power-of-fine","title":"Analyzing and Boosting the Power of Fine-Grained Visual Recognition for Multi-modal Large Language Models","date":"2025-01-25","arxiv_id":"2501.15140","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/analyzing-and-boosting-the-power-of-fine#ran","syntology_url":"https://syntology.ai/paper/2501.15140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15140"}},"official":{"repos":["pku-icst-mipl/finedefics_iclr2025"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-retrieval-without-self-knowledge","slug":"adaptive-retrieval-without-self-knowledge","title":"Adaptive Retrieval Without Self-Knowledge? Bringing Uncertainty Back Home","date":"2025-01-22","arxiv_id":"2501.12835","repositories_listed":0,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-retrieval-without-self-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.12835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12835"}},"official":null}},{"url":"/paper/embodiedeval-evaluate-multimodal-llms-as","slug":"embodiedeval-evaluate-multimodal-llms-as","title":"EmbodiedEval: Evaluate Multimodal LLMs as Embodied Agents","date":"2025-01-21","arxiv_id":"2501.11858","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/embodiedeval-evaluate-multimodal-llms-as#ran","syntology_url":"https://syntology.ai/paper/2501.11858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11858"}},"official":{"repos":["thunlp/embodiedeval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/peace-empowering-geologic-map-holistic","slug":"peace-empowering-geologic-map-holistic","title":"PEACE: Empowering Geologic Map Holistic Understanding with MLLMs","date":"2025-01-10","arxiv_id":"2501.06184","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/peace-empowering-geologic-map-holistic#ran","syntology_url":"https://syntology.ai/paper/2501.06184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.06184"}},"official":null}},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ecbench-can-multi-modal-foundation-models","slug":"ecbench-can-multi-modal-foundation-models","title":"ECBench: Can Multi-modal Foundation Models Understand the Egocentric World? A Holistic Embodied Cognition Benchmark","date":"2025-01-09","arxiv_id":"2501.05031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ecbench-can-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2501.05031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05031"}},"official":{"repos":["rh-dang/ecbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/predicting-the-performance-of-black-box-llms","slug":"predicting-the-performance-of-black-box-llms","title":"Predicting the Performance of Black-box LLMs through Self-Queries","date":"2025-01-02","arxiv_id":"2501.01558","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/predicting-the-performance-of-black-box-llms#ran","syntology_url":"https://syntology.ai/paper/2501.01558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01558"}},"official":{"repos":["dsam99/quere"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-diffusion-for-unified-image-generation","slug":"dual-diffusion-for-unified-image-generation","title":"Dual Diffusion for Unified Image Generation and Understanding","date":"2024-12-31","arxiv_id":"2501.00289","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dual-diffusion-for-unified-image-generation#ran","syntology_url":"https://syntology.ai/paper/2501.00289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00289"}},"official":null}},{"url":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial","slug":"mapeval-a-map-based-evaluation-of-geo-spatial","title":"MapEval: A Map-Based Evaluation of Geo-Spatial Reasoning in Foundation Models","date":"2024-12-31","arxiv_id":"2501.00316","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial#ran","syntology_url":"https://syntology.ai/paper/2501.00316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00316"}},"official":{"repos":["MapEval/MapEval-API","MapEval/MapEval-Textual","MapEval/MapEval-Visual"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/framefusion-combining-similarity-and","slug":"framefusion-combining-similarity-and","title":"FrameFusion: Combining Similarity and Importance for Video Token Reduction on Large Visual Language Models","date":"2024-12-30","arxiv_id":"2501.01986","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/framefusion-combining-similarity-and#ran","syntology_url":"https://syntology.ai/paper/2501.01986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01986"}},"official":{"repos":["thu-nics/framefusion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/longdocurl-a-comprehensive-multimodal-long","slug":"longdocurl-a-comprehensive-multimodal-long","title":"LongDocURL: a Comprehensive Multimodal Long Document Benchmark Integrating Understanding, Reasoning, and Locating","date":"2024-12-24","arxiv_id":"2412.18424","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longdocurl-a-comprehensive-multimodal-long#ran","syntology_url":"https://syntology.ai/paper/2412.18424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18424"}},"official":{"repos":["dengc2023/longdocurl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-large-language-models-for-1","slug":"harnessing-large-language-models-for-1","title":"Harnessing Large Language Models for Knowledge Graph Question Answering via Adaptive Multi-Aspect Retrieval-Augmentation","date":"2024-12-24","arxiv_id":"2412.18537","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2412.18537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18537"}},"official":{"repos":["Applied-Machine-Learning-Lab/AMAR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autotrust-benchmarking-trustworthiness-in","slug":"autotrust-benchmarking-trustworthiness-in","title":"AutoTrust: Benchmarking Trustworthiness in Large Vision Language Models for Autonomous Driving","date":"2024-12-19","arxiv_id":"2412.15206","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autotrust-benchmarking-trustworthiness-in#ran","syntology_url":"https://syntology.ai/paper/2412.15206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15206"}},"official":{"repos":["taco-group/autotrust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/medcot-medical-chain-of-thought-via","slug":"medcot-medical-chain-of-thought-via","title":"MedCoT: Medical Chain of Thought via Hierarchical Expert","date":"2024-12-18","arxiv_id":"2412.13736","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medcot-medical-chain-of-thought-via#ran","syntology_url":"https://syntology.ai/paper/2412.13736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13736"}},"official":{"repos":["jxliu-ai/medcot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thinking-in-space-how-multimodal-large","slug":"thinking-in-space-how-multimodal-large","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","date":"2024-12-18","arxiv_id":"2412.14171","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thinking-in-space-how-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14171"}},"official":{"repos":["vision-x-nyu/thinking-in-space"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"f42113b4bd2f6d5869adc8f0eefa48c821f58cb63f5819dfa0d50849722379c7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}