{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/4","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":109,"rows_per_page":100,"rows":[301,400],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/3","next":"/task/question-answering/papers/5","papers":[{"url":"/paper/learning-to-compose-neural-networks-for","slug":"learning-to-compose-neural-networks-for","title":"Learning to Compose Neural Networks for Question Answering","date":"2016-01-07","arxiv_id":"1601.01705","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-compose-neural-networks-for#ran","syntology_url":"https://syntology.ai/paper/1601.01705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1601.01705"}},"official":{"repos":["jacobandreas/nmn2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-simple-question-answering-with","slug":"large-scale-simple-question-answering-with","title":"Large-scale Simple Question Answering with Memory Networks","date":"2015-06-05","arxiv_id":"1506.02075","repositories_listed":3,"syntology":null},{"url":"/paper/exploring-models-and-data-for-image-question","slug":"exploring-models-and-data-for-image-question","title":"Exploring Models and Data for Image Question Answering","date":"2015-05-08","arxiv_id":"1505.02074","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-models-and-data-for-image-question#ran","syntology_url":"https://syntology.ai/paper/1505.02074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1505.02074"}},"official":{"repos":["renmengye/imageqa-public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/llama-omni2-llm-based-real-time-spoken","slug":"llama-omni2-llm-based-real-time-spoken","title":"LLaMA-Omni2: LLM-based Real-time Spoken Chatbot with Autoregressive Streaming Speech Synthesis","date":"2025-05-05","arxiv_id":"2505.02625","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":3,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llama-omni2-llm-based-real-time-spoken#ran","syntology_url":"https://syntology.ai/paper/2505.02625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02625"}},"official":{"repos":["ictnlp/llama-omni2"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/lmm4lmm-benchmarking-and-evaluating-large","slug":"lmm4lmm-benchmarking-and-evaluating-large","title":"LMM4LMM: Benchmarking and Evaluating Large-multimodal Image Generation with LMMs","date":"2025-04-11","arxiv_id":"2504.08358","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-outperforms-supervised","slug":"reinforcement-learning-outperforms-supervised","title":"Reinforcement Learning Outperforms Supervised Fine-Tuning: A Case Study on Audio Question Answering","date":"2025-03-14","arxiv_id":"2503.11197","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-outperforms-supervised#ran","syntology_url":"https://syntology.ai/paper/2503.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11197"}},"official":{"repos":["xiaomi-research/r1-aqa","huggingface.co/mispeech/r1-aqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tip-of-the-tongue-query-elicitation-for","slug":"tip-of-the-tongue-query-elicitation-for","title":"Tip of the Tongue Query Elicitation for Simulated Evaluation","date":"2025-02-25","arxiv_id":"2502.17776","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-retrieval-augmented-generation-1","slug":"benchmarking-retrieval-augmented-generation-1","title":"Benchmarking Retrieval-Augmented Generation in Multi-Modal Contexts","date":"2025-02-24","arxiv_id":"2502.17297","repositories_listed":2,"syntology":null},{"url":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied","slug":"cityeqa-a-hierarchical-llm-agent-on-embodied","title":"CityEQA: A Hierarchical LLM Agent on Embodied Question Answering Benchmark in City Space","date":"2025-02-18","arxiv_id":"2502.12532","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied#ran","syntology_url":"https://syntology.ai/paper/2502.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12532"}},"official":{"repos":["biluyong/cityeqa","tsinghua-fib-lab/CityEQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-mirage-of-model-editing-revisiting","slug":"the-mirage-of-model-editing-revisiting","title":"The Mirage of Model Editing: Revisiting Evaluation in the Wild","date":"2025-02-16","arxiv_id":"2502.11177","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-mirage-of-model-editing-revisiting#ran","syntology_url":"https://syntology.ai/paper/2502.11177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11177"}},"official":{"repos":["wanliyoung/revisit-editing-evaluation","zjunlp/easyedit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lm2-large-memory-models","slug":"lm2-large-memory-models","title":"LM2: Large Memory Models","date":"2025-02-09","arxiv_id":"2502.06049","repositories_listed":2,"syntology":null},{"url":"/paper/vargpt-unified-understanding-and-generation","slug":"vargpt-unified-understanding-and-generation","title":"VARGPT: Unified Understanding and Generation in a Visual Autoregressive Multimodal Large Language Model","date":"2025-01-21","arxiv_id":"2501.12327","repositories_listed":2,"syntology":null},{"url":"/paper/webwalker-benchmarking-llms-in-web-traversal","slug":"webwalker-benchmarking-llms-in-web-traversal","title":"WebWalker: Benchmarking LLMs in Web Traversal","date":"2025-01-13","arxiv_id":"2501.07572","repositories_listed":2,"syntology":null},{"url":"/paper/search-o1-agentic-search-enhanced-large","slug":"search-o1-agentic-search-enhanced-large","title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","date":"2025-01-09","arxiv_id":"2501.05366","repositories_listed":2,"syntology":null},{"url":"/paper/evaluating-llm-reasoning-in-the-operations","slug":"evaluating-llm-reasoning-in-the-operations","title":"Evaluating LLM Reasoning in the Operations Research Domain with ORQA","date":"2024-12-22","arxiv_id":"2412.17874","repositories_listed":2,"syntology":null},{"url":"/paper/retqa-a-large-scale-open-domain-tabular","slug":"retqa-a-large-scale-open-domain-tabular","title":"RETQA: A Large-Scale Open-Domain Tabular Question Answering Dataset for Real Estate Sector","date":"2024-12-13","arxiv_id":"2412.10104","repositories_listed":2,"syntology":null},{"url":"/paper/learn-from-downstream-and-be-yourself-in","slug":"learn-from-downstream-and-be-yourself-in","title":"Learn from Downstream and Be Yourself in Multimodal Large Language Model Fine-Tuning","date":"2024-11-17","arxiv_id":"2411.10928","repositories_listed":2,"syntology":null},{"url":"/paper/llava-o1-let-vision-language-models-reason","slug":"llava-o1-let-vision-language-models-reason","title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","date":"2024-11-15","arxiv_id":"2411.10440","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-o1-let-vision-language-models-reason#ran","syntology_url":"https://syntology.ai/paper/2411.10440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.10440"}},"official":{"repos":["PKU-YuanGroup/LLaVA-CoT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/initial-nugget-evaluation-results-for-the","slug":"initial-nugget-evaluation-results-for-the","title":"Initial Nugget Evaluation Results for the TREC 2024 RAG Track with the AutoNuggetizer Framework","date":"2024-11-14","arxiv_id":"2411.09607","repositories_listed":2,"syntology":null},{"url":"/paper/ichigo-mixed-modal-early-fusion-realtime","slug":"ichigo-mixed-modal-early-fusion-realtime","title":"Ichigo: Mixed-Modal Early-Fusion Realtime Voice Assistant","date":"2024-10-20","arxiv_id":"2410.15316","repositories_listed":2,"syntology":null},{"url":"/paper/towards-foundation-models-for-3d-vision-how","slug":"towards-foundation-models-for-3d-vision-how","title":"Towards Foundation Models for 3D Vision: How Close Are We?","date":"2024-10-14","arxiv_id":"2410.10799","repositories_listed":2,"syntology":{"n":29,"n_ran":20,"n_constructed":0,"n_ran_checked":19,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":1,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/towards-foundation-models-for-3d-vision-how#ran","syntology_url":"https://syntology.ai/paper/2410.10799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10799"}},"official":{"repos":["princeton-vl/uniqa-3d"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/differential-transformer","slug":"differential-transformer","title":"Differential Transformer","date":"2024-10-07","arxiv_id":"2410.05258","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differential-transformer#ran","syntology_url":"https://syntology.ai/paper/2410.05258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05258"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mediconfusion-can-you-trust-your-ai","slug":"mediconfusion-can-you-trust-your-ai","title":"MediConfusion: Can you trust your AI radiologist? Probing the reliability of multimodal medical foundation models","date":"2024-09-23","arxiv_id":"2409.15477","repositories_listed":2,"syntology":{"n":28,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":16,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":28,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 16 unverified","sample_list":"/paper/mediconfusion-can-you-trust-your-ai#ran","syntology_url":"https://syntology.ai/paper/2409.15477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15477"}},"official":{"repos":["mshahabsepehri/mediconfusion","AIF4S/MediConfusion"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":16,"ran_from_kinds":["official"]}}},{"url":"/paper/lime-m-less-is-more-for-evaluation-of-mllms","slug":"lime-m-less-is-more-for-evaluation-of-mllms","title":"LIME: Less Is More for MLLM Evaluation","date":"2024-09-10","arxiv_id":"2409.06851","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":7,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lime-m-less-is-more-for-evaluation-of-mllms#ran","syntology_url":"https://syntology.ai/paper/2409.06851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.06851"}},"official":{"repos":["kangreen0210/lime","kangreen0210/lime-m"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-large-language-models-with-human","slug":"fine-tuning-large-language-models-with-human","title":"Evaluating Fine-Tuning Efficiency of Human-Inspired Learning Strategies in Medical Question Answering","date":"2024-08-15","arxiv_id":"2408.07888","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/fine-tuning-large-language-models-with-human#ran","syntology_url":"https://syntology.ai/paper/2408.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07888"}},"official":{"repos":["Oxford-AI-for-Society/human-learning-strategies"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/w-rag-weakly-supervised-dense-retrieval-in","slug":"w-rag-weakly-supervised-dense-retrieval-in","title":"W-RAG: Weakly Supervised Dense Retrieval in RAG for Open-domain Question Answering","date":"2024-08-15","arxiv_id":"2408.08444","repositories_listed":2,"syntology":null},{"url":"/paper/2408-02657","slug":"2408-02657","title":"Lumina-mGPT: Illuminate Flexible Photorealistic Text-to-Image Generation with Multimodal Generative Pretraining","date":"2024-08-05","arxiv_id":"2408.02657","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2408-02657#ran","syntology_url":"https://syntology.ai/paper/2408.02657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02657"}},"official":{"repos":["alpha-vllm/lumina-mgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/spinach-sparql-based-information-navigation","slug":"spinach-sparql-based-information-navigation","title":"SPINACH: SPARQL-Based Information Navigation for Challenging Real-World Questions","date":"2024-07-16","arxiv_id":"2407.11417","repositories_listed":2,"syntology":null},{"url":"/paper/mm-instruct-generated-visual-instructions-for","slug":"mm-instruct-generated-visual-instructions-for","title":"MM-Instruct: Generated Visual Instructions for Large Multimodal Model Alignment","date":"2024-06-28","arxiv_id":"2406.19736","repositories_listed":2,"syntology":null},{"url":"/paper/torchspatial-a-location-encoding-framework","slug":"torchspatial-a-location-encoding-framework","title":"TorchSpatial: A Location Encoding Framework and Benchmark for Spatial Representation Learning","date":"2024-06-21","arxiv_id":"2406.15658","repositories_listed":2,"syntology":{"n":22,"n_ran":21,"n_constructed":0,"n_ran_checked":17,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":16,"n_pointer_only":4,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/torchspatial-a-location-encoding-framework#ran","syntology_url":"https://syntology.ai/paper/2406.15658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15658"}},"official":{"repos":["seai-lab/torchspatial","seai-lab/pygbs"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learnable-in-context-vector-for-visual","slug":"learnable-in-context-vector-for-visual","title":"LIVE: Learnable In-Context Vector for Visual Question Answering","date":"2024-06-19","arxiv_id":"2406.13185","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learnable-in-context-vector-for-visual#ran","syntology_url":"https://syntology.ai/paper/2406.13185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13185"}},"official":{"repos":["forjadeforest/live-learnable-in-context-vector"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gama-a-large-audio-language-model-with","slug":"gama-a-large-audio-language-model-with","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","date":"2024-06-17","arxiv_id":"2406.11768","repositories_listed":2,"syntology":null},{"url":"/paper/trace-the-evidence-constructing-knowledge","slug":"trace-the-evidence-constructing-knowledge","title":"TRACE the Evidence: Constructing Knowledge-Grounded Reasoning Chains for Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11460","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trace-the-evidence-constructing-knowledge#ran","syntology_url":"https://syntology.ai/paper/2406.11460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11460"}},"official":{"repos":["jyfang6/trace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intentionqa-a-benchmark-for-evaluating","slug":"intentionqa-a-benchmark-for-evaluating","title":"IntentionQA: A Benchmark for Evaluating Purchase Intention Comprehension Abilities of Language Models in E-commerce","date":"2024-06-14","arxiv_id":"2406.10173","repositories_listed":2,"syntology":null},{"url":"/paper/vane-bench-video-anomaly-evaluation-benchmark","slug":"vane-bench-video-anomaly-evaluation-benchmark","title":"VANE-Bench: Video Anomaly Evaluation Benchmark for Conversational LMMs","date":"2024-06-14","arxiv_id":"2406.10326","repositories_listed":2,"syntology":null},{"url":"/paper/cudrt-benchmarking-the-detection-of-human-vs","slug":"cudrt-benchmarking-the-detection-of-human-vs","title":"Towards Reliable Detection of LLM-Generated Texts: A Comprehensive Evaluation Framework with CUDRT","date":"2024-06-13","arxiv_id":"2406.09056","repositories_listed":2,"syntology":null},{"url":"/paper/textgrad-automatic-differentiation-via-text","slug":"textgrad-automatic-differentiation-via-text","title":"TextGrad: Automatic \"Differentiation\" via Text","date":"2024-06-11","arxiv_id":"2406.07496","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textgrad-automatic-differentiation-via-text#ran","syntology_url":"https://syntology.ai/paper/2406.07496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07496"}},"official":{"repos":["zou-group/textgrad"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/f-lmm-grounding-frozen-large-multimodal","slug":"f-lmm-grounding-frozen-large-multimodal","title":"F-LMM: Grounding Frozen Large Multimodal Models","date":"2024-06-09","arxiv_id":"2406.05821","repositories_listed":2,"syntology":null},{"url":"/paper/crag-comprehensive-rag-benchmark","slug":"crag-comprehensive-rag-benchmark","title":"CRAG -- Comprehensive RAG Benchmark","date":"2024-06-07","arxiv_id":"2406.04744","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crag-comprehensive-rag-benchmark#ran","syntology_url":"https://syntology.ai/paper/2406.04744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04744"}},"official":{"repos":["facebookresearch/crag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wings-learning-multimodal-llms-without-text","slug":"wings-learning-multimodal-llms-without-text","title":"Wings: Learning Multimodal LLMs without Text-only Forgetting","date":"2024-06-05","arxiv_id":"2406.03496","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wings-learning-multimodal-llms-without-text#ran","syntology_url":"https://syntology.ai/paper/2406.03496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03496"}},"official":null}},{"url":"/paper/mediq-question-asking-llms-for-adaptive-and","slug":"mediq-question-asking-llms-for-adaptive-and","title":"MediQ: Question-Asking LLMs and a Benchmark for Reliable Interactive Clinical Reasoning","date":"2024-06-03","arxiv_id":"2406.00922","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mediq-question-asking-llms-for-adaptive-and#ran","syntology_url":"https://syntology.ai/paper/2406.00922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00922"}},"official":{"repos":["stellali7/mediq","stellalisy/mediq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/one-token-can-help-learning-scalable-and","slug":"one-token-can-help-learning-scalable-and","title":"One Token Can Help! Learning Scalable and Pluggable Virtual Tokens for Retrieval-Augmented Large Language Models","date":"2024-05-30","arxiv_id":"2405.19670","repositories_listed":2,"syntology":null},{"url":"/paper/enhancing-visual-language-modality-alignment","slug":"enhancing-visual-language-modality-alignment","title":"Enhancing Visual-Language Modality Alignment in Large Vision Language Models via Self-Improvement","date":"2024-05-24","arxiv_id":"2405.15973","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-visual-language-modality-alignment#ran","syntology_url":"https://syntology.ai/paper/2405.15973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15973"}},"official":{"repos":["umd-huang-lab/sima"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hipporag-neurobiologically-inspired-long-term","slug":"hipporag-neurobiologically-inspired-long-term","title":"HippoRAG: Neurobiologically Inspired Long-Term Memory for Large Language Models","date":"2024-05-23","arxiv_id":"2405.14831","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hipporag-neurobiologically-inspired-long-term#ran","syntology_url":"https://syntology.ai/paper/2405.14831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14831"}},"official":{"repos":["osu-nlp-group/hipporag"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/freb-tqa-a-fine-grained-robustness-evaluation","slug":"freb-tqa-a-fine-grained-robustness-evaluation","title":"FREB-TQA: A Fine-Grained Robustness Evaluation Benchmark for Table Question Answering","date":"2024-04-29","arxiv_id":"2404.18585","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/freb-tqa-a-fine-grained-robustness-evaluation#ran","syntology_url":"https://syntology.ai/paper/2404.18585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18585"}},"official":{"repos":["hiyouga/llama-factory","boschresearch/freb-tqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/viocrvqa-novel-benchmark-dataset-and-vision","slug":"viocrvqa-novel-benchmark-dataset-and-vision","title":"ViOCRVQA: Novel Benchmark Dataset and Vision Reader for Visual Question Answering by Understanding Vietnamese Text in Images","date":"2024-04-29","arxiv_id":"2404.18397","repositories_listed":2,"syntology":null},{"url":"/paper/mixlora-enhancing-large-language-models-fine","slug":"mixlora-enhancing-large-language-models-fine","title":"MixLoRA: Enhancing Large Language Models Fine-Tuning with LoRA-based Mixture of Experts","date":"2024-04-22","arxiv_id":"2404.15159","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mixlora-enhancing-large-language-models-fine#ran","syntology_url":"https://syntology.ai/paper/2404.15159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15159"}},"official":{"repos":["TUDB-Labs/MixLoRA","mikecovlee/mLoRA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/sure-summarizing-retrievals-using-answer","slug":"sure-summarizing-retrievals-using-answer","title":"SuRe: Summarizing Retrievals using Answer Candidates for Open-domain QA of LLMs","date":"2024-04-17","arxiv_id":"2404.13081","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sure-summarizing-retrievals-using-answer#ran","syntology_url":"https://syntology.ai/paper/2404.13081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13081"}},"official":{"repos":["bbuing9/iclr24_sure"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/moe-tinymed-mixture-of-experts-for-tiny","slug":"moe-tinymed-mixture-of-experts-for-tiny","title":"Med-MoE: Mixture of Domain-Specific Experts for Lightweight Medical Vision-Language Models","date":"2024-04-16","arxiv_id":"2404.10237","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/moe-tinymed-mixture-of-experts-for-tiny#ran","syntology_url":"https://syntology.ai/paper/2404.10237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10237"}},"official":{"repos":["jiangsongtao/med-moe","jiangsongtao/tinymed"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vitextvqa-a-large-scale-visual-question","slug":"vitextvqa-a-large-scale-visual-question","title":"ViTextVQA: A Large-Scale Visual Question Answering Dataset for Evaluating Vietnamese Text Comprehension in Images","date":"2024-04-16","arxiv_id":"2404.10652","repositories_listed":2,"syntology":null},{"url":"/paper/tabsqlify-enhancing-reasoning-capabilities-of","slug":"tabsqlify-enhancing-reasoning-capabilities-of","title":"TabSQLify: Enhancing Reasoning Capabilities of LLMs Through Table Decomposition","date":"2024-04-15","arxiv_id":"2404.10150","repositories_listed":2,"syntology":{"n":15,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":8,"n_honours":6,"n_violates":0,"n_no_contract":0,"n_pointer_only":15,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 6 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/tabsqlify-enhancing-reasoning-capabilities-of#ran","syntology_url":"https://syntology.ai/paper/2404.10150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10150"}},"official":{"repos":["mahadi-nahid/tabsqlify"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-dataset-creation-and-fine-tuning-of","slug":"synthetic-dataset-creation-and-fine-tuning-of","title":"Synthetic Dataset Creation and Fine-Tuning of Transformer Models for Question Answering in Serbian","date":"2024-04-12","arxiv_id":"2404.08617","repositories_listed":2,"syntology":null},{"url":"/paper/view-selection-for-3d-captioning-via","slug":"view-selection-for-3d-captioning-via","title":"View Selection for 3D Captioning via Diffusion Ranking","date":"2024-04-11","arxiv_id":"2404.07984","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/view-selection-for-3d-captioning-via#ran","syntology_url":"https://syntology.ai/paper/2404.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07984"}},"official":null}},{"url":"/paper/m3d-advancing-3d-medical-image-analysis-with","slug":"m3d-advancing-3d-medical-image-analysis-with","title":"M3D: Advancing 3D Medical Image Analysis with Multi-Modal Large Language Models","date":"2024-03-31","arxiv_id":"2404.00578","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m3d-advancing-3d-medical-image-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2404.00578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00578"}},"official":{"repos":["baai-dcai/m3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/design-as-desired-utilizing-visual-question","slug":"design-as-desired-utilizing-visual-question","title":"Design as Desired: Utilizing Visual Question Answering for Multimodal Pre-training","date":"2024-03-30","arxiv_id":"2404.00226","repositories_listed":2,"syntology":null},{"url":"/paper/h2rsvlm-towards-helpful-and-honest-remote","slug":"h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","arxiv_id":"2403.20213","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h2rsvlm-towards-helpful-and-honest-remote#ran","syntology_url":"https://syntology.ai/paper/2403.20213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20213"}},"official":{"repos":["opendatalab/h2rsvlm","opendatalab/vhm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-rag-learning-to-adapt-retrieval","slug":"adaptive-rag-learning-to-adapt-retrieval","title":"Adaptive-RAG: Learning to Adapt Retrieval-Augmented Large Language Models through Question Complexity","date":"2024-03-21","arxiv_id":"2403.14403","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-rag-learning-to-adapt-retrieval#ran","syntology_url":"https://syntology.ai/paper/2403.14403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14403"}},"official":{"repos":["starsuzi/adaptive-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sq-llava-self-questioning-for-large-vision","slug":"sq-llava-self-questioning-for-large-vision","title":"SQ-LLaVA: Self-Questioning for Large Vision-Language Assistant","date":"2024-03-17","arxiv_id":"2403.11299","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sq-llava-self-questioning-for-large-vision#ran","syntology_url":"https://syntology.ai/paper/2403.11299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11299"}},"official":{"repos":["heliossun/sq-llava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/quiet-star-language-models-can-teach","slug":"quiet-star-language-models-can-teach","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","date":"2024-03-14","arxiv_id":"2403.09629","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quiet-star-language-models-can-teach#ran","syntology_url":"https://syntology.ai/paper/2403.09629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09629"}},"official":{"repos":["ezelikman/quiet-star"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/inficoder-eval-systematically-evaluating-the","slug":"inficoder-eval-systematically-evaluating-the","title":"InfiBench: Evaluating the Question-Answering Capabilities of Code Large Language Models","date":"2024-03-11","arxiv_id":"2404.07940","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/inficoder-eval-systematically-evaluating-the#ran","syntology_url":"https://syntology.ai/paper/2404.07940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07940"}},"official":{"repos":["infi-coder/infibench-evaluation-harness","infi-coder/infibench-evaluator"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/are-language-models-puzzle-prodigies","slug":"are-language-models-puzzle-prodigies","title":"Are Language Models Puzzle Prodigies? Algorithmic Puzzles Unveil Serious Challenges in Multimodal Reasoning","date":"2024-03-06","arxiv_id":"2403.03864","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-generate-instruction-tuning","slug":"learning-to-generate-instruction-tuning","title":"Learning to Generate Instruction Tuning Datasets for Zero-Shot Task Adaptation","date":"2024-02-28","arxiv_id":"2402.18334","repositories_listed":2,"syntology":null},{"url":"/paper/chain-of-discussion-a-multi-model-framework","slug":"chain-of-discussion-a-multi-model-framework","title":"Chain-of-Discussion: A Multi-Model Framework for Complex Evidence-Based Question Answering","date":"2024-02-26","arxiv_id":"2402.16313","repositories_listed":2,"syntology":null},{"url":"/paper/lstp-language-guided-spatial-temporal-prompt","slug":"lstp-language-guided-spatial-temporal-prompt","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","date":"2024-02-25","arxiv_id":"2402.16050","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lstp-language-guided-spatial-temporal-prompt#ran","syntology_url":"https://syntology.ai/paper/2402.16050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16050"}},"official":{"repos":["bigai-nlco/lstp-chat","bigai-nlco/videotgb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2402.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13178"}},"official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/formulaqa-a-question-answering-dataset-for","slug":"formulaqa-a-question-answering-dataset-for","title":"FormulaReasoning: A Dataset for Formula-Based Numerical Reasoning","date":"2024-02-20","arxiv_id":"2402.12692","repositories_listed":2,"syntology":null},{"url":"/paper/the-finben-an-holistic-financial-benchmark","slug":"the-finben-an-holistic-financial-benchmark","title":"FinBen: A Holistic Financial Benchmark for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12659","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-finben-an-holistic-financial-benchmark#ran","syntology_url":"https://syntology.ai/paper/2402.12659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12659"}},"official":{"repos":["the-finai/pixiu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/besa-pruning-large-language-models-with","slug":"besa-pruning-large-language-models-with","title":"BESA: Pruning Large Language Models with Blockwise Parameter-Efficient Sparsity Allocation","date":"2024-02-18","arxiv_id":"2402.16880","repositories_listed":2,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/besa-pruning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2402.16880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16880"}},"official":{"repos":["linkanonymous/besa","opengvlab/llmprune-besa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/knowtuning-knowledge-aware-fine-tuning-for","slug":"knowtuning-knowledge-aware-fine-tuning-for","title":"KnowTuning: Knowledge-aware Fine-tuning for Large Language Models","date":"2024-02-17","arxiv_id":"2402.11176","repositories_listed":2,"syntology":null},{"url":"/paper/multi-modal-preference-alignment-remedies","slug":"multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","arxiv_id":"2402.10884","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-preference-alignment-remedies#ran","syntology_url":"https://syntology.ai/paper/2402.10884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10884"}},"official":{"repos":["findalexli/mllm-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/g-retriever-retrieval-augmented-generation","slug":"g-retriever-retrieval-augmented-generation","title":"G-Retriever: Retrieval-Augmented Generation for Textual Graph Understanding and Question Answering","date":"2024-02-12","arxiv_id":"2402.07630","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/g-retriever-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2402.07630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07630"}},"official":{"repos":["xiaoxinhe/g-retriever"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grillbot-in-practice-lessons-and-tradeoffs","slug":"grillbot-in-practice-lessons-and-tradeoffs","title":"GRILLBot In Practice: Lessons and Tradeoffs Deploying Large Language Models for Adaptable Conversational Task Assistants","date":"2024-02-12","arxiv_id":"2402.07647","repositories_listed":2,"syntology":null},{"url":"/paper/screenai-a-vision-language-model-for-ui-and","slug":"screenai-a-vision-language-model-for-ui-and","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","date":"2024-02-07","arxiv_id":"2402.04615","repositories_listed":2,"syntology":null},{"url":"/paper/position-paper-what-can-large-language-models","slug":"position-paper-what-can-large-language-models","title":"Position: What Can Large Language Models Tell Us about Time Series Analysis","date":"2024-02-05","arxiv_id":"2402.02713","repositories_listed":2,"syntology":null},{"url":"/paper/deepedit-knowledge-editing-as-decoding-with","slug":"deepedit-knowledge-editing-as-decoding-with","title":"DeepEdit: Knowledge Editing as Decoding with Constraints","date":"2024-01-19","arxiv_id":"2401.10471","repositories_listed":2,"syntology":null},{"url":"/paper/tuning-language-models-by-proxy","slug":"tuning-language-models-by-proxy","title":"Tuning Language Models by Proxy","date":"2024-01-16","arxiv_id":"2401.08565","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tuning-language-models-by-proxy#ran","syntology_url":"https://syntology.ai/paper/2401.08565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08565"}},"official":{"repos":["alisawuffles/proxy-tuning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/divide-and-conquer-for-large-language-models","slug":"divide-and-conquer-for-large-language-models","title":"DCR: Divide-and-Conquer Reasoning for Multi-choice Question Answering with LLMs","date":"2024-01-10","arxiv_id":"2401.05190","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divide-and-conquer-for-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.05190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05190"}},"official":{"repos":["aimijie/dcr","aimijie/divide-and-conquer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-table-evolving-tables-in-the","slug":"chain-of-table-evolving-tables-in-the","title":"Chain-of-Table: Evolving Tables in the Reasoning Chain for Table Understanding","date":"2024-01-09","arxiv_id":"2401.04398","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-table-evolving-tables-in-the#ran","syntology_url":"https://syntology.ai/paper/2401.04398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04398"}},"official":null}},{"url":"/paper/parameter-efficient-sparsity-crafting-from","slug":"parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","arxiv_id":"2401.02731","repositories_listed":2,"syntology":null},{"url":"/paper/challenge-llms-to-reason-about-reasoning-a","slug":"challenge-llms-to-reason-about-reasoning-a","title":"MR-GSM8K: A Meta-Reasoning Benchmark for Large Language Model Evaluation","date":"2023-12-28","arxiv_id":"2312.17080","repositories_listed":2,"syntology":null},{"url":"/paper/tinygpt-v-efficient-multimodal-large-language","slug":"tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tinygpt-v-efficient-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.16862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16862"}},"official":{"repos":["dlyuangod/tinygpt-v"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lingoqa-video-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2312.14115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14115"}},"official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/advancing-surgical-vqa-with-scene-graph","slug":"advancing-surgical-vqa-with-scene-graph","title":"Advancing Surgical VQA with Scene Graph Knowledge","date":"2023-12-15","arxiv_id":"2312.10251","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/advancing-surgical-vqa-with-scene-graph#ran","syntology_url":"https://syntology.ai/paper/2312.10251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10251"}},"official":{"repos":["camma-public/ssg-qa","camma-public/ssg-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/icd-lm-configuring-vision-language-in-context","slug":"icd-lm-configuring-vision-language-in-context","title":"Lever LM: Configuring In-Context Sequence to Lever Large Vision Language Models","date":"2023-12-15","arxiv_id":"2312.10104","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/icd-lm-configuring-vision-language-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.10104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10104"}},"official":{"repos":["forjadeforest/icd-lm","forjadeforest/lever-lm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chat-3d-v2-bridging-3d-scene-and-large","slug":"chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","arxiv_id":"2312.08168","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chat-3d-v2-bridging-3d-scene-and-large#ran","syntology_url":"https://syntology.ai/paper/2312.08168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08168"}},"official":{"repos":["chat-3d/chat-3d-v2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pcoqa-persian-conversational-question","slug":"pcoqa-persian-conversational-question","title":"PCoQA: Persian Conversational Question Answering Dataset","date":"2023-12-07","arxiv_id":"2312.04362","repositories_listed":2,"syntology":null},{"url":"/paper/towards-learning-a-generalist-model-for","slug":"towards-learning-a-generalist-model-for","title":"Towards Learning a Generalist Model for Embodied Navigation","date":"2023-12-04","arxiv_id":"2312.02010","repositories_listed":2,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/towards-learning-a-generalist-model-for#ran","syntology_url":"https://syntology.ai/paper/2312.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02010"}},"official":{"repos":["lavi-lab/navillm","zd11024/NaviLLM"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/can-generalist-foundation-models-outcompete","slug":"can-generalist-foundation-models-outcompete","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","date":"2023-11-28","arxiv_id":"2311.16452","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-generalist-foundation-models-outcompete#ran","syntology_url":"https://syntology.ai/paper/2311.16452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16452"}},"official":null}},{"url":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","slug":"llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","arxiv_id":"2311.17043","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.17043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17043"}},"official":{"repos":["dvlab-research/llama-vid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finme-a-performance-enhanced-large-language","slug":"finme-a-performance-enhanced-large-language","title":"FinMem: A Performance-Enhanced LLM Trading Agent with Layered Memory and Character Design","date":"2023-11-23","arxiv_id":"2311.13743","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/finme-a-performance-enhanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2311.13743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13743"}},"official":{"repos":["pipiku915/finmem-llm-stocktrading"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/extracting-definienda-in-mathematical","slug":"extracting-definienda-in-mathematical","title":"Extracting Definienda in Mathematical Scholarly Articles with Transformers","date":"2023-11-21","arxiv_id":"2311.12448","repositories_listed":2,"syntology":null},{"url":"/paper/financebench-a-new-benchmark-for-financial","slug":"financebench-a-new-benchmark-for-financial","title":"FinanceBench: A New Benchmark for Financial Question Answering","date":"2023-11-20","arxiv_id":"2311.11944","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/financebench-a-new-benchmark-for-financial#ran","syntology_url":"https://syntology.ai/paper/2311.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.11944"}},"official":{"repos":["patronus-ai/financebench"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dynapipe-optimizing-multi-task-training","slug":"dynapipe-optimizing-multi-task-training","title":"DynaPipe: Optimizing Multi-task Training through Dynamic Pipelines","date":"2023-11-17","arxiv_id":"2311.10418","repositories_listed":2,"syntology":null},{"url":"/paper/pachinko-patching-interpretable-qa-models","slug":"pachinko-patching-interpretable-qa-models","title":"What if you said that differently?: How Explanation Formats Affect Human Feedback Efficacy and User Perception","date":"2023-11-16","arxiv_id":"2311.09558","repositories_listed":2,"syntology":null},{"url":"/paper/never-lost-in-the-middle-improving-large","slug":"never-lost-in-the-middle-improving-large","title":"Never Lost in the Middle: Mastering Long-Context Question Answering with Position-Agnostic Decompositional Training","date":"2023-11-15","arxiv_id":"2311.09198","repositories_listed":2,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":10,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 1 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/never-lost-in-the-middle-improving-large#ran","syntology_url":"https://syntology.ai/paper/2311.09198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09198"}},"official":{"repos":["hejunqing/never-lost-in-the-middle"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bring-your-own-kg-self-supervised-program","slug":"bring-your-own-kg-self-supervised-program","title":"Bring Your Own KG: Self-Supervised Program Synthesis for Zero-Shot KGQA","date":"2023-11-14","arxiv_id":"2311.07850","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-filter-context-for-retrieval","slug":"learning-to-filter-context-for-retrieval","title":"Learning to Filter Context for Retrieval-Augmented Generation","date":"2023-11-14","arxiv_id":"2311.08377","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-filter-context-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2311.08377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08377"}},"official":{"repos":["zorazrw/filco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infmllm-a-unified-framework-for-visual","slug":"infmllm-a-unified-framework-for-visual","title":"InfMLLM: A Unified Framework for Visual-Language Tasks","date":"2023-11-12","arxiv_id":"2311.06791","repositories_listed":2,"syntology":null},{"url":"/paper/lumos-learning-agents-with-unified-data","slug":"lumos-learning-agents-with-unified-data","title":"Agent Lumos: Unified and Modular Training for Open-Source Language Agents","date":"2023-11-09","arxiv_id":"2311.05657","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lumos-learning-agents-with-unified-data#ran","syntology_url":"https://syntology.ai/paper/2311.05657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05657"}},"official":{"repos":["allenai/lumos"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semqa-semi-extractive-multi-source-question","slug":"semqa-semi-extractive-multi-source-question","title":"SEMQA: Semi-Extractive Multi-Source Question Answering","date":"2023-11-08","arxiv_id":"2311.04886","repositories_listed":2,"syntology":null}],"record_sha256":"1fd8d2dab1e484c87ec7cb28d034b6fc5c4bfce5f3ce4f6c269711a770a7eb5e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}