{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/11","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":109,"rows_per_page":100,"rows":[1001,1100],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/10","next":"/task/question-answering/papers/12","papers":[{"url":"/paper/levelrag-enhancing-retrieval-augmented","slug":"levelrag-enhancing-retrieval-augmented","title":"LevelRAG: Enhancing Retrieval-Augmented Generation with Multi-hop Logic Planning over Rewriting Augmented Searchers","date":"2025-02-25","arxiv_id":"2502.18139","repositories_listed":1,"syntology":null},{"url":"/paper/mm-poisonrag-disrupting-multimodal-rag-with","slug":"mm-poisonrag-disrupting-multimodal-rag-with","title":"MM-PoisonRAG: Disrupting Multimodal RAG with Local and Global Poisoning Attacks","date":"2025-02-25","arxiv_id":"2502.17832","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-quantification-in-retrieval","slug":"uncertainty-quantification-in-retrieval","title":"Uncertainty Quantification in Retrieval Augmented Question Answering","date":"2025-02-25","arxiv_id":"2502.18108","repositories_listed":1,"syntology":null},{"url":"/paper/vidorag-visual-document-retrieval-augmented","slug":"vidorag-visual-document-retrieval-augmented","title":"ViDoRAG: Visual Document Retrieval-Augmented Generation via Dynamic Iterative Reasoning Agents","date":"2025-02-25","arxiv_id":"2502.18017","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/vidorag-visual-document-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2502.18017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18017"}},"official":{"repos":["Alibaba-NLP/ViDoRAG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/baichuan-audio-a-unified-framework-for-end-to","slug":"baichuan-audio-a-unified-framework-for-end-to","title":"Baichuan-Audio: A Unified Framework for End-to-End Speech Interaction","date":"2025-02-24","arxiv_id":"2502.17239","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/baichuan-audio-a-unified-framework-for-end-to#ran","syntology_url":"https://syntology.ai/paper/2502.17239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17239"}},"official":{"repos":["baichuan-inc/baichuan-audio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hippo-enhancing-the-table-understanding","slug":"hippo-enhancing-the-table-understanding","title":"HIPPO: Enhancing the Table Understanding Capability of Large Language Models through Hybrid-Modal Preference Optimization","date":"2025-02-24","arxiv_id":"2502.17315","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hippo-enhancing-the-table-understanding#ran","syntology_url":"https://syntology.ai/paper/2502.17315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17315"}},"official":{"repos":["neuir/hippo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mllms-know-where-to-look-training-free","slug":"mllms-know-where-to-look-training-free","title":"MLLMs Know Where to Look: Training-free Perception of Small Visual Details with Multimodal LLMs","date":"2025-02-24","arxiv_id":"2502.17422","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"0 ran · 4 unverified","sample_list":"/paper/mllms-know-where-to-look-training-free#ran","syntology_url":"https://syntology.ai/paper/2502.17422","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17422"}},"official":{"repos":["saccharomycetes/mllms_know"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/multiocr-qa-dataset-for-evaluating-robustness","slug":"multiocr-qa-dataset-for-evaluating-robustness","title":"MultiOCR-QA: Dataset for Evaluating Robustness of LLMs in Question Answering on Multilingual OCR Texts","date":"2025-02-24","arxiv_id":"2502.16781","repositories_listed":1,"syntology":null},{"url":"/paper/multitat-benchmarking-multilingual-table-and","slug":"multitat-benchmarking-multilingual-table-and","title":"MULTITAT: Benchmarking Multilingual Table-and-Text Question Answering","date":"2025-02-24","arxiv_id":"2502.17253","repositories_listed":1,"syntology":null},{"url":"/paper/visual-rag-benchmarking-text-to-image","slug":"visual-rag-benchmarking-text-to-image","title":"Visual-RAG: Benchmarking Text-to-Image Retrieval Augmented Generation for Visual Knowledge Intensive Queries","date":"2025-02-23","arxiv_id":"2502.16636","repositories_listed":1,"syntology":null},{"url":"/paper/wrong-answers-can-also-be-useful-plausibleqa","slug":"wrong-answers-can-also-be-useful-plausibleqa","title":"Wrong Answers Can Also Be Useful: PlausibleQA -- A Large-Scale QA Dataset with Answer Plausibility Scores","date":"2025-02-22","arxiv_id":"2502.16358","repositories_listed":1,"syntology":null},{"url":"/paper/improving-consistency-in-large-language","slug":"improving-consistency-in-large-language","title":"Improving Consistency in Large Language Models through Chain of Guidance","date":"2025-02-21","arxiv_id":"2502.15924","repositories_listed":1,"syntology":null},{"url":"/paper/kvlink-accelerating-large-language-models-via","slug":"kvlink-accelerating-large-language-models-via","title":"KVLink: Accelerating Large Language Models via Efficient KV Cache Reuse","date":"2025-02-21","arxiv_id":"2502.16002","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-rag-through-a-chart","slug":"benchmarking-multimodal-rag-through-a-chart","title":"Benchmarking Multimodal RAG through a Chart-based Document Question-Answering Generation Framework","date":"2025-02-20","arxiv_id":"2502.14864","repositories_listed":1,"syntology":null},{"url":"/paper/chatvla-unified-multimodal-understanding-and","slug":"chatvla-unified-multimodal-understanding-and","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","date":"2025-02-20","arxiv_id":"2502.14420","repositories_listed":1,"syntology":null},{"url":"/paper/does-time-have-its-place-temporal-heads-where","slug":"does-time-have-its-place-temporal-heads-where","title":"Does Time Have Its Place? Temporal Heads: Where Language Models Recall Time-specific Information","date":"2025-02-20","arxiv_id":"2502.14258","repositories_listed":1,"syntology":null},{"url":"/paper/how-much-knowledge-can-you-pack-into-a-lora-1","slug":"how-much-knowledge-can-you-pack-into-a-lora-1","title":"How Much Knowledge Can You Pack into a LoRA Adapter without Harming LLM?","date":"2025-02-20","arxiv_id":"2502.14502","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-get-your-llm-to-generate-challenging","slug":"how-to-get-your-llm-to-generate-challenging","title":"How to Get Your LLM to Generate Challenging Problems for Evaluation","date":"2025-02-20","arxiv_id":"2502.14678","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-faithfulness-of-chains-of-thought","slug":"measuring-faithfulness-of-chains-of-thought","title":"Measuring Faithfulness of Chains of Thought by Unlearning Reasoning Steps","date":"2025-02-20","arxiv_id":"2502.14829","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-faithfulness-of-chains-of-thought#ran","syntology_url":"https://syntology.ai/paper/2502.14829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14829"}},"official":{"repos":["technion-cs-nlp/parametric-faithfulness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-rewardbench-holistic-evaluation-of","slug":"multimodal-rewardbench-holistic-evaluation-of","title":"Multimodal RewardBench: Holistic Evaluation of Reward Models for Vision Language Models","date":"2025-02-20","arxiv_id":"2502.14191","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-influence-of-context-size-and-model","slug":"on-the-influence-of-context-size-and-model","title":"On the Influence of Context Size and Model Choice in Retrieval-Augmented Generation Systems","date":"2025-02-20","arxiv_id":"2502.14759","repositories_listed":1,"syntology":null},{"url":"/paper/is-that-your-final-answer-test-time-scaling","slug":"is-that-your-final-answer-test-time-scaling","title":"Is That Your Final Answer? Test-Time Scaling Improves Selective Question Answering","date":"2025-02-19","arxiv_id":"2502.13962","repositories_listed":1,"syntology":null},{"url":"/paper/mudaf-long-context-multi-document-attention","slug":"mudaf-long-context-multi-document-attention","title":"MuDAF: Long-Context Multi-Document Attention Focusing through Contrastive Learning on Attention Heads","date":"2025-02-19","arxiv_id":"2502.13963","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mudaf-long-context-multi-document-attention#ran","syntology_url":"https://syntology.ai/paper/2502.13963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13963"}},"official":{"repos":["NeosKnight233/MuDAF"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/peerqa-a-scientific-question-answering","slug":"peerqa-a-scientific-question-answering","title":"PeerQA: A Scientific Question Answering Dataset from Peer Reviews","date":"2025-02-19","arxiv_id":"2502.13668","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/peerqa-a-scientific-question-answering#ran","syntology_url":"https://syntology.ai/paper/2502.13668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13668"}},"official":{"repos":["ukplab/peerqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pitvqa-vector-matrix-low-rank-adaptation-for","slug":"pitvqa-vector-matrix-low-rank-adaptation-for","title":"PitVQA++: Vector Matrix-Low-Rank Adaptation for Open-Ended Visual Question Answering in Pituitary Surgery","date":"2025-02-19","arxiv_id":"2502.14149","repositories_listed":1,"syntology":null},{"url":"/paper/trustrag-an-information-assistant-with","slug":"trustrag-an-information-assistant-with","title":"TrustRAG: An Information Assistant with Retrieval Augmented Generation","date":"2025-02-19","arxiv_id":"2502.13719","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-seen-data-improving-kbqa","slug":"beyond-seen-data-improving-kbqa","title":"Beyond Seen Data: Improving KBQA Generalization Through Schema-Guided Logical Form Generation","date":"2025-02-18","arxiv_id":"2502.12737","repositories_listed":1,"syntology":null},{"url":"/paper/how-much-do-llms-hallucinate-across-languages","slug":"how-much-do-llms-hallucinate-across-languages","title":"How Much Do LLMs Hallucinate across Languages? On Multilingual Estimation of LLM Hallucination in the Wild","date":"2025-02-18","arxiv_id":"2502.12769","repositories_listed":1,"syntology":null},{"url":"/paper/longfaith-enhancing-long-context-reasoning-in","slug":"longfaith-enhancing-long-context-reasoning-in","title":"LongFaith: Enhancing Long-Context Reasoning in LLMs with Faithful Synthetic Data","date":"2025-02-18","arxiv_id":"2502.12583","repositories_listed":1,"syntology":null},{"url":"/paper/re-align-aligning-vision-language-models-via","slug":"re-align-aligning-vision-language-models-via","title":"Re-Align: Aligning Vision Language Models via Retrieval-Augmented Direct Preference Optimization","date":"2025-02-18","arxiv_id":"2502.13146","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/re-align-aligning-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2502.13146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13146"}},"official":{"repos":["taco-group/re-align"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-can-see-better-visual","slug":"language-models-can-see-better-visual","title":"Language Models Can See Better: Visual Contrastive Decoding For LLM Multimodal Reasoning","date":"2025-02-17","arxiv_id":"2502.11751","repositories_listed":1,"syntology":null},{"url":"/paper/mmxu-a-multi-modal-and-multi-x-ray","slug":"mmxu-a-multi-modal-and-multi-x-ray","title":"MMXU: A Multi-Modal and Multi-X-ray Understanding Dataset for Disease Progression","date":"2025-02-17","arxiv_id":"2502.11651","repositories_listed":1,"syntology":null},{"url":"/paper/ra-mtr-a-retrieval-augmented-multi-task","slug":"ra-mtr-a-retrieval-augmented-multi-task","title":"RA-MTR: A Retrieval Augmented Multi-Task Reader based Approach for Inspirational Quote Extraction from Long Documents","date":"2025-02-17","arxiv_id":"2502.12124","repositories_listed":1,"syntology":null},{"url":"/paper/talk-structurally-act-hierarchically-a","slug":"talk-structurally-act-hierarchically-a","title":"Talk Structurally, Act Hierarchically: A Collaborative Framework for LLM Multi-Agent Systems","date":"2025-02-16","arxiv_id":"2502.11098","repositories_listed":1,"syntology":null},{"url":"/paper/nitibench-a-comprehensive-studies-of-llm","slug":"nitibench-a-comprehensive-studies-of-llm","title":"NitiBench: A Comprehensive Studies of LLM Frameworks Capabilities for Thai Legal Question Answering","date":"2025-02-15","arxiv_id":"2502.10868","repositories_listed":1,"syntology":null},{"url":"/paper/svbench-a-benchmark-with-temporal-multi-turn","slug":"svbench-a-benchmark-with-temporal-multi-turn","title":"SVBench: A Benchmark with Temporal Multi-Turn Dialogues for Streaming Video Understanding","date":"2025-02-15","arxiv_id":"2502.10810","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svbench-a-benchmark-with-temporal-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2502.10810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10810"}},"official":{"repos":["yzy-bupt/SVBench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/ket-rag-a-cost-efficient-multi-granular","slug":"ket-rag-a-cost-efficient-multi-granular","title":"KET-RAG: A Cost-Efficient Multi-Granular Indexing Framework for Graph-RAG","date":"2025-02-13","arxiv_id":"2502.09304","repositories_listed":1,"syntology":null},{"url":"/paper/lp-lm-no-hallucinations-in-question-answering","slug":"lp-lm-no-hallucinations-in-question-answering","title":"LP-LM: No Hallucinations in Question Answering with Logic Programming","date":"2025-02-13","arxiv_id":"2502.09212","repositories_listed":1,"syntology":null},{"url":"/paper/selfcite-self-supervised-alignment-for","slug":"selfcite-self-supervised-alignment-for","title":"SelfCite: Self-Supervised Alignment for Context Attribution in Large Language Models","date":"2025-02-13","arxiv_id":"2502.09604","repositories_listed":1,"syntology":null},{"url":"/paper/square-sequential-question-answering","slug":"square-sequential-question-answering","title":"SQuARE: Sequential Question Answering Reasoning Engine for Enhanced Chain-of-Thought in Large Language Models","date":"2025-02-13","arxiv_id":"2502.09390","repositories_listed":1,"syntology":null},{"url":"/paper/egotextvqa-towards-egocentric-scene-text","slug":"egotextvqa-towards-egocentric-scene-text","title":"EgoTextVQA: Towards Egocentric Scene-Text Aware Video Question Answering","date":"2025-02-11","arxiv_id":"2502.07411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egotextvqa-towards-egocentric-scene-text#ran","syntology_url":"https://syntology.ai/paper/2502.07411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07411"}},"official":{"repos":["zhousheng97/egotextvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/elevating-legal-llm-responses-harnessing","slug":"elevating-legal-llm-responses-harnessing","title":"Elevating Legal LLM Responses: Harnessing Trainable Logical Structures and Semantic Knowledge with Legal Reasoning","date":"2025-02-11","arxiv_id":"2502.07912","repositories_listed":1,"syntology":null},{"url":"/paper/rotor-towards-more-reliable-responses-for","slug":"rotor-towards-more-reliable-responses-for","title":"RoToR: Towards More Reliable Responses for Order-Invariant Inputs","date":"2025-02-10","arxiv_id":"2502.08662","repositories_listed":1,"syntology":null},{"url":"/paper/clinkd-cross-modal-clinical-knowledge","slug":"clinkd-cross-modal-clinical-knowledge","title":"ClinKD: Cross-Modal Clinical Knowledge Distiller For Multi-Task Medical Images","date":"2025-02-09","arxiv_id":"2502.05928","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-working-memory-query-guided-segment","slug":"temporal-working-memory-query-guided-segment","title":"Temporal Working Memory: Query-Guided Segment Refinement for Enhanced Multimodal Understanding","date":"2025-02-09","arxiv_id":"2502.06020","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-shortcomings-of-llms-in","slug":"investigating-the-shortcomings-of-llms-in","title":"Investigating the Shortcomings of LLMs in Step-by-Step Legal Reasoning","date":"2025-02-08","arxiv_id":"2502.05675","repositories_listed":1,"syntology":null},{"url":"/paper/arr-question-answering-with-large-language","slug":"arr-question-answering-with-large-language","title":"ARR: Question Answering with Large Language Models via Analyzing, Retrieving, and Reasoning","date":"2025-02-07","arxiv_id":"2502.04689","repositories_listed":1,"syntology":null},{"url":"/paper/cocoa-a-generalized-approach-to-uncertainty","slug":"cocoa-a-generalized-approach-to-uncertainty","title":"Uncertainty Quantification for LLMs through Minimum Bayes Risk: Bridging Confidence and Consistency","date":"2025-02-07","arxiv_id":"2502.04964","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-unintended-memorization-with-lora","slug":"mitigating-unintended-memorization-with-lora","title":"Mitigating Unintended Memorization with LoRA in Federated Learning for LLMs","date":"2025-02-07","arxiv_id":"2502.05087","repositories_listed":1,"syntology":null},{"url":"/paper/docmia-document-level-membership-inference","slug":"docmia-document-level-membership-inference","title":"DocMIA: Document-Level Membership Inference Attacks against DocVQA Models","date":"2025-02-06","arxiv_id":"2502.03692","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/docmia-document-level-membership-inference#ran","syntology_url":"https://syntology.ai/paper/2502.03692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03692"}},"official":{"repos":["khanhnguyen21006/mia_docvqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/no-images-no-problem-retaining-knowledge-in","slug":"no-images-no-problem-retaining-knowledge-in","title":"No Images, No Problem: Retaining Knowledge in Continual VQA with Questions-Only Memory","date":"2025-02-06","arxiv_id":"2502.04469","repositories_listed":1,"syntology":null},{"url":"/paper/ontology-guided-hybrid-prompt-learning-for","slug":"ontology-guided-hybrid-prompt-learning-for","title":"Ontology-Guided, Hybrid Prompt Learning for Generalization in Knowledge Graph Question Answering","date":"2025-02-06","arxiv_id":"2502.03992","repositories_listed":1,"syntology":null},{"url":"/paper/pixfoundation-are-we-heading-in-the-right","slug":"pixfoundation-are-we-heading-in-the-right","title":"PixFoundation: Are We Heading in the Right Direction with Pixel-level Vision Foundation Models?","date":"2025-02-06","arxiv_id":"2502.04192","repositories_listed":1,"syntology":null},{"url":"/paper/scoreflow-mastering-llm-agent-workflows-via","slug":"scoreflow-mastering-llm-agent-workflows-via","title":"ScoreFlow: Mastering LLM Agent Workflows via Score-based Preference Optimization","date":"2025-02-06","arxiv_id":"2502.04306","repositories_listed":1,"syntology":null},{"url":"/paper/smi-an-information-theoretic-metric-for","slug":"smi-an-information-theoretic-metric-for","title":"SMI: An Information-Theoretic Metric for Predicting Model Knowledge Solely from Pre-Training Signals","date":"2025-02-06","arxiv_id":"2502.04066","repositories_listed":1,"syntology":null},{"url":"/paper/rankify-a-comprehensive-python-toolkit-for","slug":"rankify-a-comprehensive-python-toolkit-for","title":"Rankify: A Comprehensive Python Toolkit for Retrieval, Re-Ranking, and Retrieval-Augmented Generation","date":"2025-02-04","arxiv_id":"2502.02464","repositories_listed":1,"syntology":null},{"url":"/paper/tumtraffic-videoqa-a-benchmark-for-unified","slug":"tumtraffic-videoqa-a-benchmark-for-unified","title":"TUMTraffic-VideoQA: A Benchmark for Unified Spatio-Temporal Video Understanding in Traffic Scenes","date":"2025-02-04","arxiv_id":"2502.02449","repositories_listed":1,"syntology":null},{"url":"/paper/robust-llava-on-the-effectiveness-of-large","slug":"robust-llava-on-the-effectiveness-of-large","title":"Robust-LLaVA: On the Effectiveness of Large-Scale Robust Image Encoders for Multi-modal Large Language Models","date":"2025-02-03","arxiv_id":"2502.01576","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-state-space-models-for","slug":"multilingual-state-space-models-for","title":"Multilingual State Space Models for Structured Question Answering in Indic Languages","date":"2025-02-01","arxiv_id":"2502.01673","repositories_listed":1,"syntology":null},{"url":"/paper/infty-video-a-training-free-approach-to-long","slug":"infty-video-a-training-free-approach-to-long","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","date":"2025-01-31","arxiv_id":"2501.19098","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infty-video-a-training-free-approach-to-long#ran","syntology_url":"https://syntology.ai/paper/2501.19098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19098"}},"official":{"repos":["deep-spin/infinite-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kbqa-o1-agentic-knowledge-base-question","slug":"kbqa-o1-agentic-knowledge-base-question","title":"KBQA-o1: Agentic Knowledge Base Question Answering with Monte Carlo Tree Search","date":"2025-01-31","arxiv_id":"2501.18922","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/kbqa-o1-agentic-knowledge-base-question#ran","syntology_url":"https://syntology.ai/paper/2501.18922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.18922"}},"official":{"repos":["lhrlab/kbqa-o1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/o3-mini-vs-deepseek-r1-which-one-is-safer","slug":"o3-mini-vs-deepseek-r1-which-one-is-safer","title":"o3-mini vs DeepSeek-R1: Which One is Safer?","date":"2025-01-30","arxiv_id":"2501.18438","repositories_listed":1,"syntology":null},{"url":"/paper/auto-differentiating-any-llm-workflow-a","slug":"auto-differentiating-any-llm-workflow-a","title":"LLM-AutoDiff: Auto-Differentiate Any LLM Workflow","date":"2025-01-28","arxiv_id":"2501.16673","repositories_listed":1,"syntology":null},{"url":"/paper/large-models-in-dialogue-for-active","slug":"large-models-in-dialogue-for-active","title":"Large Models in Dialogue for Active Perception and Anomaly Detection","date":"2025-01-27","arxiv_id":"2501.16300","repositories_listed":1,"syntology":null},{"url":"/paper/lucy-linguistic-understanding-and-control","slug":"lucy-linguistic-understanding-and-control","title":"LUCY: Linguistic Understanding and Control Yielding Early Stage of Her","date":"2025-01-27","arxiv_id":"2501.16327","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-and-boosting-the-power-of-fine","slug":"analyzing-and-boosting-the-power-of-fine","title":"Analyzing and Boosting the Power of Fine-Grained Visual Recognition for Multi-modal Large Language Models","date":"2025-01-25","arxiv_id":"2501.15140","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/analyzing-and-boosting-the-power-of-fine#ran","syntology_url":"https://syntology.ai/paper/2501.15140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15140"}},"official":{"repos":["pku-icst-mipl/finedefics_iclr2025"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-graphs-meet-thoughts-enhancing-complex","slug":"causal-graphs-meet-thoughts-enhancing-complex","title":"Causal Graphs Meet Thoughts: Enhancing Complex Reasoning in Graph-Augmented LLMs","date":"2025-01-24","arxiv_id":"2501.14892","repositories_listed":1,"syntology":null},{"url":"/paper/dressing-up-llm-efficient-stylized-question","slug":"dressing-up-llm-efficient-stylized-question","title":"DRESSing Up LLM: Efficient Stylized Question-Answering via Style Subspace Editing","date":"2025-01-24","arxiv_id":"2501.14371","repositories_listed":1,"syntology":null},{"url":"/paper/k-comp-retrieval-augmented-medical-domain","slug":"k-comp-retrieval-augmented-medical-domain","title":"K-COMP: Retrieval-Augmented Medical Domain Question Answering With Knowledge-Injected Compressor","date":"2025-01-23","arxiv_id":"2501.13567","repositories_listed":1,"syntology":null},{"url":"/paper/ramqa-a-unified-framework-for-retrieval","slug":"ramqa-a-unified-framework-for-retrieval","title":"RAMQA: A Unified Framework for Retrieval-Augmented Multi-Modal Question Answering","date":"2025-01-23","arxiv_id":"2501.13297","repositories_listed":1,"syntology":null},{"url":"/paper/patent-figure-classification-using-large","slug":"patent-figure-classification-using-large","title":"Patent Figure Classification using Large Vision-language Models","date":"2025-01-22","arxiv_id":"2501.12751","repositories_listed":1,"syntology":null},{"url":"/paper/embodiedeval-evaluate-multimodal-llms-as","slug":"embodiedeval-evaluate-multimodal-llms-as","title":"EmbodiedEval: Evaluate Multimodal LLMs as Embodied Agents","date":"2025-01-21","arxiv_id":"2501.11858","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/embodiedeval-evaluate-multimodal-llms-as#ran","syntology_url":"https://syntology.ai/paper/2501.11858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11858"}},"official":{"repos":["thunlp/embodiedeval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/insqabench-benchmarking-chinese-insurance","slug":"insqabench-benchmarking-chinese-insurance","title":"InsQABench: Benchmarking Chinese Insurance Domain Question Answering with Large Language Models","date":"2025-01-19","arxiv_id":"2501.10943","repositories_listed":1,"syntology":null},{"url":"/paper/dialogue-benchmark-generation-from-knowledge","slug":"dialogue-benchmark-generation-from-knowledge","title":"Dialogue Benchmark Generation from Knowledge Graphs with Cost-Effective Retrieval-Augmented LLMs","date":"2025-01-17","arxiv_id":"2501.09928","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-aerial-detection-baseline-of","slug":"a-simple-aerial-detection-baseline-of","title":"A Simple Aerial Detection Baseline of Multimodal Language Models","date":"2025-01-16","arxiv_id":"2501.09720","repositories_listed":1,"syntology":null},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/adkgd-anomaly-detection-in-knowledge-graphs","slug":"adkgd-anomaly-detection-in-knowledge-graphs","title":"ADKGD: Anomaly Detection in Knowledge Graphs with Dual-Channel Training","date":"2025-01-13","arxiv_id":"2501.07078","repositories_listed":1,"syntology":null},{"url":"/paper/mecd-unlocking-event-level-causal-graph","slug":"mecd-unlocking-event-level-causal-graph","title":"MECD+: Unlocking Event-Level Causal Graph Discovery for Video Reasoning","date":"2025-01-13","arxiv_id":"2501.07227","repositories_listed":1,"syntology":null},{"url":"/paper/language-fusion-for-parameter-efficient-cross","slug":"language-fusion-for-parameter-efficient-cross","title":"Language Fusion for Parameter-Efficient Cross-lingual Transfer","date":"2025-01-12","arxiv_id":"2501.06892","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-large-language-models-for-6","slug":"fine-tuning-large-language-models-for-6","title":"Fine-tuning Large Language Models for Improving Factuality in Legal Question Answering","date":"2025-01-11","arxiv_id":"2501.06521","repositories_listed":1,"syntology":null},{"url":"/paper/fleurs-slu-a-massively-multilingual-benchmark","slug":"fleurs-slu-a-massively-multilingual-benchmark","title":"Fleurs-SLU: A Massively Multilingual Benchmark for Spoken Language Understanding","date":"2025-01-10","arxiv_id":"2501.06117","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-tune-a-multilingual-encoder-model-for","slug":"how-to-tune-a-multilingual-encoder-model-for","title":"How to Tune a Multilingual Encoder Model for Germanic Languages: A Study of PEFT, Full Fine-Tuning, and Language Adapters","date":"2025-01-10","arxiv_id":"2501.06025","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-collaboration-mechanisms-a-survey","slug":"multi-agent-collaboration-mechanisms-a-survey","title":"Multi-Agent Collaboration Mechanisms: A Survey of LLMs","date":"2025-01-10","arxiv_id":"2501.06322","repositories_listed":1,"syntology":null},{"url":"/paper/ecbench-can-multi-modal-foundation-models","slug":"ecbench-can-multi-modal-foundation-models","title":"ECBench: Can Multi-modal Foundation Models Understand the Egocentric World? A Holistic Embodied Cognition Benchmark","date":"2025-01-09","arxiv_id":"2501.05031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ecbench-can-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2501.05031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05031"}},"official":{"repos":["rh-dang/ecbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sensorqa-a-question-answering-benchmark-for","slug":"sensorqa-a-question-answering-benchmark-for","title":"SensorQA: A Question Answering Benchmark for Daily-Life Monitoring","date":"2025-01-09","arxiv_id":"2501.04974","repositories_listed":1,"syntology":null},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timelinekgqa-a-comprehensive-question-answer","slug":"timelinekgqa-a-comprehensive-question-answer","title":"TimelineKGQA: A Comprehensive Question-Answer Pair Generator for Temporal Knowledge Graphs","date":"2025-01-08","arxiv_id":"2501.04343","repositories_listed":1,"syntology":null},{"url":"/paper/automated-generation-of-challenging-multiple","slug":"automated-generation-of-challenging-multiple","title":"Automated Generation of Challenging Multiple-Choice Questions for Vision Language Model Evaluation","date":"2025-01-06","arxiv_id":"2501.03225","repositories_listed":1,"syntology":null},{"url":"/paper/redit-re-evaluating-large-visual-question","slug":"redit-re-evaluating-large-visual-question","title":"ReDiT: Re‑evaluating large visual question answering model confidence by defining input scenario Difficulty and applying Temperature mapping","date":"2025-01-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/socratic-questioning-learn-to-self-guide","slug":"socratic-questioning-learn-to-self-guide","title":"Socratic Questioning: Learn to Self-guide Multimodal Reasoning in the Wild","date":"2025-01-06","arxiv_id":"2501.02964","repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hlv-1k-a-large-scale-hour-long-video","slug":"hlv-1k-a-large-scale-hour-long-video","title":"HLV-1K: A Large-scale Hour-Long Video Benchmark for Time-Specific Long Video Understanding","date":"2025-01-03","arxiv_id":"2501.01645","repositories_listed":1,"syntology":null},{"url":"/paper/whyphi-fine-tuning-phi-3-for-multiple-choice","slug":"whyphi-fine-tuning-phi-3-for-multiple-choice","title":"(WhyPHI) Fine-Tuning PHI-3 for Multiple-Choice Question Answering: Methodology, Results, and Challenges","date":"2025-01-03","arxiv_id":"2501.01588","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-singlish-understanding-bridging-the","slug":"advancing-singlish-understanding-bridging-the","title":"Advancing Singlish Understanding: Bridging the Gap with Datasets and Multimodal Models","date":"2025-01-02","arxiv_id":"2501.01034","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-the-performance-of-black-box-llms","slug":"predicting-the-performance-of-black-box-llms","title":"Predicting the Performance of Black-box LLMs through Self-Queries","date":"2025-01-02","arxiv_id":"2501.01558","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/predicting-the-performance-of-black-box-llms#ran","syntology_url":"https://syntology.ai/paper/2501.01558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01558"}},"official":{"repos":["dsam99/quere"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/avqacl-a-novel-benchmark-for-audio-visual","slug":"avqacl-a-novel-benchmark-for-audio-visual","title":"AVQACL: A Novel Benchmark for Audio-Visual Question Answering Continual Learning","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/notes-guided-mllm-reasoning-enhancing-mllm","slug":"notes-guided-mllm-reasoning-enhancing-mllm","title":"Notes-guided MLLM Reasoning: Enhancing MLLM with Knowledge and Visual Notes for Visual Question Answering","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dual-diffusion-for-unified-image-generation","slug":"dual-diffusion-for-unified-image-generation","title":"Dual Diffusion for Unified Image Generation and Understanding","date":"2024-12-31","arxiv_id":"2501.00289","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dual-diffusion-for-unified-image-generation#ran","syntology_url":"https://syntology.ai/paper/2501.00289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00289"}},"official":null}},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"f132d0746fc42af404bfa48e4b2a869ec89442e8363581814769d07abb92a086","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}