{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/ran/3","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1274,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering/papers/ran/1","prev":"/task/question-answering/papers/ran/2","next":"/task/question-answering/papers/ran/4","papers":[{"url":"/paper/2408-02657","slug":"2408-02657","title":"Lumina-mGPT: Illuminate Flexible Photorealistic Text-to-Image Generation with Multimodal Generative Pretraining","date":"2024-08-05","arxiv_id":"2408.02657","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2408-02657#ran","syntology_url":"https://syntology.ai/paper/2408.02657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02657"}},"official":{"repos":["alpha-vllm/lumina-mgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/2408-01933","slug":"2408-01933","title":"DiReCT: Diagnostic Reasoning for Clinical Notes via Large Language Models","date":"2024-08-04","arxiv_id":"2408.01933","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-01933#ran","syntology_url":"https://syntology.ai/paper/2408.01933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01933"}},"official":{"repos":["wbw520/direct"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01419","slug":"2408-01419","title":"DebateQA: Evaluating Question Answering on Debatable Knowledge","date":"2024-08-02","arxiv_id":"2408.01419","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-01419#ran","syntology_url":"https://syntology.ai/paper/2408.01419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01419"}},"official":{"repos":["pillowsofwind/debateqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-llama-3-herd-of-models#ran","syntology_url":"https://syntology.ai/paper/2407.21783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21783"}},"official":null}},{"url":"/paper/boosting-audio-visual-question-answering-via","slug":"boosting-audio-visual-question-answering-via","title":"Boosting Audio Visual Question Answering via Key Semantic-Aware Cues","date":"2024-07-30","arxiv_id":"2407.20693","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/boosting-audio-visual-question-answering-via#ran","syntology_url":"https://syntology.ai/paper/2407.20693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20693"}},"official":{"repos":["gewu-lab/tspm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-generalizable-pathology-foundation","slug":"towards-a-generalizable-pathology-foundation","title":"Towards A Generalizable Pathology Foundation Model via Unified Knowledge Distillation","date":"2024-07-26","arxiv_id":"2407.18449","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-a-generalizable-pathology-foundation#ran","syntology_url":"https://syntology.ai/paper/2407.18449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18449"}},"official":null}},{"url":"/paper/learning-trimodal-relation-for-avqa-with","slug":"learning-trimodal-relation-for-avqa-with","title":"Learning Trimodal Relation for AVQA with Missing Modality","date":"2024-07-23","arxiv_id":"2407.16171","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":14,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":4,"n_violates":0,"n_no_contract":14,"n_pointer_only":3,"phrase":"20 ran (of which 14 constructed an object rather than computing a result; 18 with no instrument failure: 4 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-trimodal-relation-for-avqa-with#ran","syntology_url":"https://syntology.ai/paper/2407.16171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16171"}},"official":{"repos":["visualaikhu/missing-avqa"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/enhancing-llm-s-cognition-via-structurization","slug":"enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","arxiv_id":"2407.16434","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-llm-s-cognition-via-structurization#ran","syntology_url":"https://syntology.ai/paper/2407.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16434"}},"official":{"repos":["alibaba/struxgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mminstruct-a-high-quality-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2407.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15838"}},"official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-assemble-ace-understanding-how","slug":"answer-assemble-ace-understanding-how","title":"Answer, Assemble, Ace: Understanding How Transformers Answer Multiple Choice Questions","date":"2024-07-21","arxiv_id":"2407.15018","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-assemble-ace-understanding-how#ran","syntology_url":"https://syntology.ai/paper/2407.15018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15018"}},"official":null}},{"url":"/paper/generalization-v-s-memorization-tracing","slug":"generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","repositories_listed":0,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/generalization-v-s-memorization-tracing#ran","syntology_url":"https://syntology.ai/paper/2407.14985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14985"}},"official":null}},{"url":"/paper/rag-qa-arena-evaluating-domain-robustness-for","slug":"rag-qa-arena-evaluating-domain-robustness-for","title":"RAG-QA Arena: Evaluating Domain Robustness for Long-form Retrieval Augmented Question Answering","date":"2024-07-19","arxiv_id":"2407.13998","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rag-qa-arena-evaluating-domain-robustness-for#ran","syntology_url":"https://syntology.ai/paper/2407.13998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13998"}},"official":{"repos":["awslabs/rag-qa-arena"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-language-models-as-risk-scores","slug":"evaluating-language-models-as-risk-scores","title":"Evaluating language models as risk scores","date":"2024-07-19","arxiv_id":"2407.14614","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/evaluating-language-models-as-risk-scores#ran","syntology_url":"https://syntology.ai/paper/2407.14614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14614"}},"official":{"repos":["socialfoundations/folktexts"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-haystacks-answering-harder-questions","slug":"visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","date":"2024-07-18","arxiv_id":"2407.13766","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/visual-haystacks-answering-harder-questions#ran","syntology_url":"https://syntology.ai/paper/2407.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13766"}},"official":{"repos":["visual-haystacks/vhs_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/compact-compressing-retrieved-documents","slug":"compact-compressing-retrieved-documents","title":"CompAct: Compressing Retrieved Documents Actively for Question Answering","date":"2024-07-12","arxiv_id":"2407.09014","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compact-compressing-retrieved-documents#ran","syntology_url":"https://syntology.ai/paper/2407.09014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09014"}},"official":{"repos":["dmis-lab/compact"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spiqa-a-dataset-for-multimodal-question","slug":"spiqa-a-dataset-for-multimodal-question","title":"SPIQA: A Dataset for Multimodal Question Answering on Scientific Papers","date":"2024-07-12","arxiv_id":"2407.09413","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiqa-a-dataset-for-multimodal-question#ran","syntology_url":"https://syntology.ai/paper/2407.09413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09413"}},"official":{"repos":["google/spiqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gofa-a-generative-one-for-all-model-for-joint","slug":"gofa-a-generative-one-for-all-model-for-joint","title":"GOFA: A Generative One-For-All Model for Joint Graph Language Modeling","date":"2024-07-12","arxiv_id":"2407.09709","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gofa-a-generative-one-for-all-model-for-joint#ran","syntology_url":"https://syntology.ai/paper/2407.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09709"}},"official":{"repos":["jiaruifeng/gofa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/autobencher-creating-salient-novel-difficult","slug":"autobencher-creating-salient-novel-difficult","title":"AutoBencher: Creating Salient, Novel, Difficult Datasets for Language Models","date":"2024-07-11","arxiv_id":"2407.08351","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobencher-creating-salient-novel-difficult#ran","syntology_url":"https://syntology.ai/paper/2407.08351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08351"}},"official":{"repos":["XiangLi1999/AutoBencher"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-surgery-modulating-llm-s-behavior-via","slug":"model-surgery-modulating-llm-s-behavior-via","title":"Model Surgery: Modulating LLM's Behavior Via Simple Parameter Editing","date":"2024-07-11","arxiv_id":"2407.08770","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-surgery-modulating-llm-s-behavior-via#ran","syntology_url":"https://syntology.ai/paper/2407.08770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08770"}},"official":{"repos":["lucywang720/model-surgery"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ida-vlm-towards-movie-understanding-via-id","slug":"ida-vlm-towards-movie-understanding-via-id","title":"IDA-VLM: Towards Movie Understanding via ID-Aware Large Vision-Language Model","date":"2024-07-10","arxiv_id":"2407.07577","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ida-vlm-towards-movie-understanding-via-id#ran","syntology_url":"https://syntology.ai/paper/2407.07577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07577"}},"official":{"repos":["jiyt17/ida-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wsi-vqa-interpreting-whole-slide-images-by","slug":"wsi-vqa-interpreting-whole-slide-images-by","title":"WSI-VQA: Interpreting Whole Slide Images by Generative Visual Question Answering","date":"2024-07-08","arxiv_id":"2407.05603","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wsi-vqa-interpreting-whole-slide-images-by#ran","syntology_url":"https://syntology.ai/paper/2407.05603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05603"}},"official":{"repos":["cpystan/wsi-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-vision-and-language-pretraining-with-large","slug":"3d-vision-and-language-pretraining-with-large","title":"3D Vision and Language Pretraining with Large-Scale Synthetic Data","date":"2024-07-08","arxiv_id":"2407.06084","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-vision-and-language-pretraining-with-large#ran","syntology_url":"https://syntology.ai/paper/2407.06084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06084"}},"official":{"repos":["idejie/3DSyn"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/arigraph-learning-knowledge-graph-world","slug":"arigraph-learning-knowledge-graph-world","title":"AriGraph: Learning Knowledge Graph World Models with Episodic Memory for LLM Agents","date":"2024-07-05","arxiv_id":"2407.04363","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arigraph-learning-knowledge-graph-world#ran","syntology_url":"https://syntology.ai/paper/2407.04363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04363"}},"official":{"repos":["airi-institute/arigraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/anah-v2-scaling-analytical-hallucination","slug":"anah-v2-scaling-analytical-hallucination","title":"ANAH-v2: Scaling Analytical Hallucination Annotation of Large Language Models","date":"2024-07-05","arxiv_id":"2407.04693","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/anah-v2-scaling-analytical-hallucination#ran","syntology_url":"https://syntology.ai/paper/2407.04693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04693"}},"official":{"repos":["open-compass/anah"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/me-myself-and-ai-the-situational-awareness","slug":"me-myself-and-ai-the-situational-awareness","title":"Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs","date":"2024-07-05","arxiv_id":"2407.04694","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/me-myself-and-ai-the-situational-awareness#ran","syntology_url":"https://syntology.ai/paper/2407.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04694"}},"official":{"repos":["lrudl/sad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eliminating-position-bias-of-language-models","slug":"eliminating-position-bias-of-language-models","title":"Eliminating Position Bias of Language Models: A Mechanistic Approach","date":"2024-07-01","arxiv_id":"2407.01100","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eliminating-position-bias-of-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.01100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01100"}},"official":{"repos":["wzq016/pine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/searching-for-best-practices-in-retrieval","slug":"searching-for-best-practices-in-retrieval","title":"Searching for Best Practices in Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01219","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/searching-for-best-practices-in-retrieval#ran","syntology_url":"https://syntology.ai/paper/2407.01219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01219"}},"official":{"repos":["FudanDNN-NLP/RAG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/increasing-model-capacity-for-free-a-simple","slug":"increasing-model-capacity-for-free-a-simple","title":"Increasing Model Capacity for Free: A Simple Strategy for Parameter Efficient Fine-tuning","date":"2024-07-01","arxiv_id":"2407.01320","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/increasing-model-capacity-for-free-a-simple#ran","syntology_url":"https://syntology.ai/paper/2407.01320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01320"}},"official":{"repos":["lins-lab/capaboost"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive","slug":"h-star-llm-driven-hybrid-sql-text-adaptive","title":"H-STAR: LLM-driven Hybrid SQL-Text Adaptive Reasoning on Tables","date":"2024-06-29","arxiv_id":"2407.05952","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive#ran","syntology_url":"https://syntology.ai/paper/2407.05952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05952"}},"official":{"repos":["nikhilsab/h-star"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","repositories_listed":1,"syntology":{"n":19,"n_ran":9,"n_constructed":4,"n_ran_checked":5,"n_instrument":4,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/stllava-med-self-training-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.19973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19973"}},"official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/the-sifo-benchmark-investigating-the","slug":"the-sifo-benchmark-investigating-the","title":"The SIFo Benchmark: Investigating the Sequential Instruction Following Ability of Large Language Models","date":"2024-06-28","arxiv_id":"2406.19999","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-sifo-benchmark-investigating-the#ran","syntology_url":"https://syntology.ai/paper/2406.19999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19999"}},"official":{"repos":["shin-ee-chen/SIFo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llavolta-efficient-multi-modal-models-via","slug":"llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","arxiv_id":"2406.20092","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llavolta-efficient-multi-modal-models-via#ran","syntology_url":"https://syntology.ai/paper/2406.20092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20092"}},"official":{"repos":["beckschen/llavolta"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trustuqa-a-trustful-framework-for-unified","slug":"trustuqa-a-trustful-framework-for-unified","title":"TrustUQA: A Trustful Framework for Unified Structured Data Question Answering","date":"2024-06-27","arxiv_id":"2406.18916","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trustuqa-a-trustful-framework-for-unified#ran","syntology_url":"https://syntology.ai/paper/2406.18916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18916"}},"official":{"repos":["zjukg/trustuqa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seakr-self-aware-knowledge-retrieval-for","slug":"seakr-self-aware-knowledge-retrieval-for","title":"SeaKR: Self-aware Knowledge Retrieval for Adaptive Retrieval Augmented Generation","date":"2024-06-27","arxiv_id":"2406.19215","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/seakr-self-aware-knowledge-retrieval-for#ran","syntology_url":"https://syntology.ai/paper/2406.19215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19215"}},"official":{"repos":["thu-keg/seakr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/understand-what-llm-needs-dual-preference","slug":"understand-what-llm-needs-dual-preference","title":"Understand What LLM Needs: Dual Preference Alignment for Retrieval-Augmented Generation","date":"2024-06-26","arxiv_id":"2406.18676","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/understand-what-llm-needs-dual-preference#ran","syntology_url":"https://syntology.ai/paper/2406.18676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18676"}},"official":{"repos":["dongguanting/dpa-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leave-no-document-behind-benchmarking-long","slug":"leave-no-document-behind-benchmarking-long","title":"Leave No Document Behind: Benchmarking Long-Context LLMs with Extended Multi-Doc QA","date":"2024-06-25","arxiv_id":"2406.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/leave-no-document-behind-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2406.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17419"}},"official":{"repos":["mozerwang/loong"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-exponential-extension-of","slug":"training-free-exponential-extension-of","title":"Training-Free Exponential Context Extension via Cascading KV Cache","date":"2024-06-24","arxiv_id":"2406.17808","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-exponential-extension-of#ran","syntology_url":"https://syntology.ai/paper/2406.17808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17808"}},"official":{"repos":["jeffwillette/cascading_kv_cache"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hcqa-ego4d-egoschema-challenge-2024","slug":"hcqa-ego4d-egoschema-challenge-2024","title":"HCQA @ Ego4D EgoSchema Challenge 2024","date":"2024-06-22","arxiv_id":"2406.15771","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hcqa-ego4d-egoschema-challenge-2024#ran","syntology_url":"https://syntology.ai/paper/2406.15771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15771"}},"official":{"repos":["hyu-zhang/hcqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uda-a-benchmark-suite-for-retrieval-augmented","slug":"uda-a-benchmark-suite-for-retrieval-augmented","title":"UDA: A Benchmark Suite for Retrieval Augmented Generation in Real-world Document Analysis","date":"2024-06-21","arxiv_id":"2406.15187","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uda-a-benchmark-suite-for-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.15187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15187"}},"official":{"repos":["qinchuanhui/uda-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/torchspatial-a-location-encoding-framework","slug":"torchspatial-a-location-encoding-framework","title":"TorchSpatial: A Location Encoding Framework and Benchmark for Spatial Representation Learning","date":"2024-06-21","arxiv_id":"2406.15658","repositories_listed":2,"syntology":{"n":22,"n_ran":21,"n_constructed":0,"n_ran_checked":17,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":16,"n_pointer_only":4,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/torchspatial-a-location-encoding-framework#ran","syntology_url":"https://syntology.ai/paper/2406.15658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15658"}},"official":{"repos":["seai-lab/torchspatial","seai-lab/pygbs"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/timo-towards-better-temporal-reasoning-for","slug":"timo-towards-better-temporal-reasoning-for","title":"Timo: Towards Better Temporal Reasoning for Language Models","date":"2024-06-20","arxiv_id":"2406.14192","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timo-towards-better-temporal-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2406.14192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14192"}},"official":{"repos":["zhaochen0110/timo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/augmenting-query-and-passage-for-retrieval","slug":"augmenting-query-and-passage-for-retrieval","title":"QPaug: Question and Passage Augmentation for Open-Domain Question Answering of LLMs","date":"2024-06-20","arxiv_id":"2406.14277","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/augmenting-query-and-passage-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2406.14277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14277"}},"official":{"repos":["kmswin1/qpaug"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-plan-for-retrieval-augmented","slug":"learning-to-plan-for-retrieval-augmented","title":"Learning to Plan for Retrieval-Augmented Large Language Models from Knowledge Graphs","date":"2024-06-20","arxiv_id":"2406.14282","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-plan-for-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.14282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14282"}},"official":{"repos":["zjukg/lpkg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llasa-large-multimodal-agent-for-human","slug":"llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","arxiv_id":"2406.14498","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llasa-large-multimodal-agent-for-human#ran","syntology_url":"https://syntology.ai/paper/2406.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14498"}},"official":{"repos":["bashlab/llasa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/taglas-an-atlas-of-text-attributed-graph","slug":"taglas-an-atlas-of-text-attributed-graph","title":"TAGLAS: An atlas of text-attributed graph datasets in the era of large graph and language models","date":"2024-06-20","arxiv_id":"2406.14683","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/taglas-an-atlas-of-text-attributed-graph#ran","syntology_url":"https://syntology.ai/paper/2406.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14683"}},"official":{"repos":["jiaruifeng/taglas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-rag-fusion-with-ragelo-an","slug":"evaluating-rag-fusion-with-ragelo-an","title":"Evaluating RAG-Fusion with RAGElo: an Automated Elo-based Framework","date":"2024-06-20","arxiv_id":"2406.14783","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-rag-fusion-with-ragelo-an#ran","syntology_url":"https://syntology.ai/paper/2406.14783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14783"}},"official":{"repos":["zetaalphavector/ragelo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dialsim-a-real-time-simulator-for-evaluating","slug":"dialsim-a-real-time-simulator-for-evaluating","title":"DialSim: A Real-Time Simulator for Evaluating Long-Term Multi-Party Dialogue Understanding of Conversational Agents","date":"2024-06-19","arxiv_id":"2406.13144","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dialsim-a-real-time-simulator-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2406.13144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13144"}},"official":{"repos":["jiho283/simulator"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learnable-in-context-vector-for-visual","slug":"learnable-in-context-vector-for-visual","title":"LIVE: Learnable In-Context Vector for Visual Question Answering","date":"2024-06-19","arxiv_id":"2406.13185","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learnable-in-context-vector-for-visual#ran","syntology_url":"https://syntology.ai/paper/2406.13185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13185"}},"official":{"repos":["forjadeforest/live-learnable-in-context-vector"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/factual-confidence-of-llms-on-reliability-and","slug":"factual-confidence-of-llms-on-reliability-and","title":"Factual Confidence of LLMs: on Reliability and Robustness of Current Estimators","date":"2024-06-19","arxiv_id":"2406.13415","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/factual-confidence-of-llms-on-reliability-and#ran","syntology_url":"https://syntology.ai/paper/2406.13415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13415"}},"official":{"repos":["amazon-science/factual-confidence-of-llms"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/model-internals-based-answer-attribution-for","slug":"model-internals-based-answer-attribution-for","title":"Model Internals-based Answer Attribution for Trustworthy Retrieval-Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13663","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-internals-based-answer-attribution-for#ran","syntology_url":"https://syntology.ai/paper/2406.13663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13663"}},"official":{"repos":["betswish/mirage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voco-llama-towards-vision-compression-with","slug":"voco-llama-towards-vision-compression-with","title":"VoCo-LLaMA: Towards Vision Compression with Large Language Models","date":"2024-06-18","arxiv_id":"2406.12275","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":5,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voco-llama-towards-vision-compression-with#ran","syntology_url":"https://syntology.ai/paper/2406.12275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12275"}},"official":{"repos":["Yxxxb/VoCo-LLaMA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vrsbench-a-versatile-vision-language","slug":"vrsbench-a-versatile-vision-language","title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","date":"2024-06-18","arxiv_id":"2406.12384","repositories_listed":3,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":5,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vrsbench-a-versatile-vision-language#ran","syntology_url":"https://syntology.ai/paper/2406.12384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12384"}},"official":{"repos":["lx709/vrsbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pslm-parallel-generation-of-text-and-speech","slug":"pslm-parallel-generation-of-text-and-speech","title":"PSLM: Parallel Generation of Text and Speech with LLMs for Low-Latency Spoken Dialogue Systems","date":"2024-06-18","arxiv_id":"2406.12428","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pslm-parallel-generation-of-text-and-speech#ran","syntology_url":"https://syntology.ai/paper/2406.12428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12428"}},"official":{"repos":["eleutherai/gpt-neox"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/breaking-the-ceiling-of-the-llm-community-by","slug":"breaking-the-ceiling-of-the-llm-community-by","title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","date":"2024-06-18","arxiv_id":"2406.12585","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/breaking-the-ceiling-of-the-llm-community-by#ran","syntology_url":"https://syntology.ai/paper/2406.12585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12585"}},"official":{"repos":["yaoching0/gac"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/nash-cot-multi-path-inference-with-preference","slug":"nash-cot-multi-path-inference-with-preference","title":"Nash CoT: Multi-Path Inference with Preference Equilibrium","date":"2024-06-18","arxiv_id":"2407.07099","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/nash-cot-multi-path-inference-with-preference#ran","syntology_url":"https://syntology.ai/paper/2407.07099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07099"}},"official":{"repos":["stevezhangza/nash-chain-of-thought"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/avatar-optimizing-llm-agents-for-tool","slug":"avatar-optimizing-llm-agents-for-tool","title":"AvaTaR: Optimizing LLM Agents for Tool Usage via Contrastive Reasoning","date":"2024-06-17","arxiv_id":"2406.11200","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/avatar-optimizing-llm-agents-for-tool#ran","syntology_url":"https://syntology.ai/paper/2406.11200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11200"}},"official":{"repos":["zou-group/avatar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/i-srt-aligning-large-multimodal-models-for","slug":"i-srt-aligning-large-multimodal-models-for","title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","date":"2024-06-17","arxiv_id":"2406.11280","repositories_listed":4,"syntology":{"n":30,"n_ran":26,"n_constructed":0,"n_ran_checked":18,"n_instrument":8,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":16,"n_pointer_only":15,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 2 violated, 16 with no contract checked; 8 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/i-srt-aligning-large-multimodal-models-for#ran","syntology_url":"https://syntology.ai/paper/2406.11280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11280"}},"official":{"repos":["snumprlab/SRT","snumprlab/isr-dpo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/textit-refiner-restructure-retrieval-content","slug":"textit-refiner-restructure-retrieval-content","title":"Refiner: Restructure Retrieval Content Efficiently to Advance Question-Answering Capabilities","date":"2024-06-17","arxiv_id":"2406.11357","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/textit-refiner-restructure-retrieval-content#ran","syntology_url":"https://syntology.ai/paper/2406.11357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11357"}},"official":{"repos":["allen-li1231/refiner-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-scientific-concepts-understanding","slug":"boosting-scientific-concepts-understanding","title":"Boosting Scientific Concepts Understanding: Can Analogy from Teacher Models Empower Student Models?","date":"2024-06-17","arxiv_id":"2406.11375","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/boosting-scientific-concepts-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.11375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11375"}},"official":{"repos":["siyuyuan/scua"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/trace-the-evidence-constructing-knowledge","slug":"trace-the-evidence-constructing-knowledge","title":"TRACE the Evidence: Constructing Knowledge-Grounded Reasoning Chains for Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11460","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trace-the-evidence-constructing-knowledge#ran","syntology_url":"https://syntology.ai/paper/2406.11460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11460"}},"official":{"repos":["jyfang6/trace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-arithmetic-a-framework-for-test-time","slug":"safety-arithmetic-a-framework-for-test-time","title":"Safety Arithmetic: A Framework for Test-time Safety Alignment of Language Models by Steering Parameters and Activations","date":"2024-06-17","arxiv_id":"2406.11801","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safety-arithmetic-a-framework-for-test-time#ran","syntology_url":"https://syntology.ai/paper/2406.11801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11801"}},"official":{"repos":["declare-lab/safety-arithmetic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medcalc-bench-evaluating-large-language","slug":"medcalc-bench-evaluating-large-language","title":"MedCalc-Bench: Evaluating Large Language Models for Medical Calculations","date":"2024-06-17","arxiv_id":"2406.12036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medcalc-bench-evaluating-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.12036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12036"}},"official":{"repos":["ncbi-nlp/medcalc-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-prompting-for-unlearning-in-large","slug":"soft-prompting-for-unlearning-in-large","title":"Soft Prompting for Unlearning in Large Language Models","date":"2024-06-17","arxiv_id":"2406.12038","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/soft-prompting-for-unlearning-in-large#ran","syntology_url":"https://syntology.ai/paper/2406.12038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12038"}},"official":{"repos":["karuna-bhaila/llm_unlearning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-key-neurons-in-large-language","slug":"analyzing-key-neurons-in-large-language","title":"Identifying Query-Relevant Neurons in Large Language Models for Long-Form Texts","date":"2024-06-16","arxiv_id":"2406.10868","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analyzing-key-neurons-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10868"}},"official":{"repos":["tigerchen52/qrneuron"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scar-efficient-instruction-tuning-for-large","slug":"scar-efficient-instruction-tuning-for-large","title":"SCAR: Efficient Instruction-Tuning for Large Language Models via Style Consistency-Aware Response Ranking","date":"2024-06-16","arxiv_id":"2406.10882","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scar-efficient-instruction-tuning-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.10882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10882"}},"official":{"repos":["zhuang-li/scar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/foodieqa-a-multimodal-dataset-for-fine","slug":"foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","arxiv_id":"2406.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foodieqa-a-multimodal-dataset-for-fine#ran","syntology_url":"https://syntology.ai/paper/2406.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11030"}},"official":{"repos":["lyan62/FoodieQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}},"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/color-filter-conditional-loss-reduction","slug":"color-filter-conditional-loss-reduction","title":"CoLoR-Filter: Conditional Loss Reduction Filtering for Targeted Language Model Pre-training","date":"2024-06-15","arxiv_id":"2406.10670","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/color-filter-conditional-loss-reduction#ran","syntology_url":"https://syntology.ai/paper/2406.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10670"}},"official":{"repos":["davidbrandfonbrener/color-filter-olmo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-validity-via-enhanced","slug":"large-language-model-validity-via-enhanced","title":"Large language model validity via enhanced conformal prediction methods","date":"2024-06-14","arxiv_id":"2406.09714","repositories_listed":3,"syntology":{"n":32,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":22,"n_pointer_only":12,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/large-language-model-validity-via-enhanced#ran","syntology_url":"https://syntology.ai/paper/2406.09714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09714"}},"official":{"repos":["jjcherian/conformal-safety"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["found_in_text","listed","official"]}}},{"url":"/paper/babilong-testing-the-limits-of-llms-with-long","slug":"babilong-testing-the-limits-of-llms-with-long","title":"BABILong: Testing the Limits of LLMs with Long Context Reasoning-in-a-Haystack","date":"2024-06-14","arxiv_id":"2406.10149","repositories_listed":4,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/babilong-testing-the-limits-of-llms-with-long#ran","syntology_url":"https://syntology.ai/paper/2406.10149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10149"}},"official":{"repos":["booydar/babilong"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/living-in-the-moment-can-large-language","slug":"living-in-the-moment-can-large-language","title":"Living in the Moment: Can Large Language Models Grasp Co-Temporal Reasoning?","date":"2024-06-13","arxiv_id":"2406.09072","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/living-in-the-moment-can-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.09072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09072"}},"official":{"repos":["zhaochen0110/cotempqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-preference-optimization-improving","slug":"chain-of-preference-optimization-improving","title":"Chain of Preference Optimization: Improving Chain-of-Thought Reasoning in LLMs","date":"2024-06-13","arxiv_id":"2406.09136","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-preference-optimization-improving#ran","syntology_url":"https://syntology.ai/paper/2406.09136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09136"}},"official":{"repos":["sail-sg/cpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/needle-in-a-video-haystack-a-scalable","slug":"needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","arxiv_id":"2406.09367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/needle-in-a-video-haystack-a-scalable#ran","syntology_url":"https://syntology.ai/paper/2406.09367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09367"}},"official":{"repos":["joez17/videoniah"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/too-many-frames-not-all-useful-efficient","slug":"too-many-frames-not-all-useful-efficient","title":"Too Many Frames, Not All Useful: Efficient Strategies for Long-Form Video QA","date":"2024-06-13","arxiv_id":"2406.09396","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/too-many-frames-not-all-useful-efficient#ran","syntology_url":"https://syntology.ai/paper/2406.09396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09396"}},"official":{"repos":["jongwoopark7978/LVNet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yo-llava-your-personalized-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.09400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09400"}},"official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dara-decomposition-alignment-reasoning","slug":"dara-decomposition-alignment-reasoning","title":"DARA: Decomposition-Alignment-Reasoning Autonomous Language Agent for Question Answering over Knowledge Graphs","date":"2024-06-11","arxiv_id":"2406.07080","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dara-decomposition-alignment-reasoning#ran","syntology_url":"https://syntology.ai/paper/2406.07080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07080"}},"official":{"repos":["UKPLab/acl2024-DARA"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-2-advancing-spatial-temporal","slug":"videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","arxiv_id":"2406.07476","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/videollama-2-advancing-spatial-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.07476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07476"}},"official":{"repos":["damo-nlp-sg/videollama2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/textgrad-automatic-differentiation-via-text","slug":"textgrad-automatic-differentiation-via-text","title":"TextGrad: Automatic \"Differentiation\" via Text","date":"2024-06-11","arxiv_id":"2406.07496","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textgrad-automatic-differentiation-via-text#ran","syntology_url":"https://syntology.ai/paper/2406.07496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07496"}},"official":{"repos":["zou-group/textgrad"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/situational-awareness-matters-in-3d-vision","slug":"situational-awareness-matters-in-3d-vision","title":"Situational Awareness Matters in 3D Vision Language Reasoning","date":"2024-06-11","arxiv_id":"2406.07544","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/situational-awareness-matters-in-3d-vision#ran","syntology_url":"https://syntology.ai/paper/2406.07544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07544"}},"official":{"repos":["YunzeMan/Situation3D"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-context-compression-efficiently","slug":"recurrent-context-compression-efficiently","title":"Recurrent Context Compression: Efficiently Expanding the Context Window of LLM","date":"2024-06-10","arxiv_id":"2406.06110","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-context-compression-efficiently#ran","syntology_url":"https://syntology.ai/paper/2406.06110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06110"}},"official":{"repos":["WUHU-G/RCC_Transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medexqa-medical-question-answering-benchmark","slug":"medexqa-medical-question-answering-benchmark","title":"MedExQA: Medical Question Answering Benchmark with Multiple Explanations","date":"2024-06-10","arxiv_id":"2406.06331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medexqa-medical-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2406.06331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06331"}},"official":{"repos":["knowlab/medexqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vcr-visual-caption-restoration","slug":"vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","arxiv_id":"2406.06462","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vcr-visual-caption-restoration#ran","syntology_url":"https://syntology.ai/paper/2406.06462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06462"}},"official":{"repos":["tianyu-z/vcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/husky-a-unified-open-source-language-agent","slug":"husky-a-unified-open-source-language-agent","title":"Husky: A Unified, Open-Source Language Agent for Multi-Step Reasoning","date":"2024-06-10","arxiv_id":"2406.06469","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/husky-a-unified-open-source-language-agent#ran","syntology_url":"https://syntology.ai/paper/2406.06469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06469"}},"official":{"repos":["agent-husky/husky-v1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/crag-comprehensive-rag-benchmark","slug":"crag-comprehensive-rag-benchmark","title":"CRAG -- Comprehensive RAG Benchmark","date":"2024-06-07","arxiv_id":"2406.04744","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crag-comprehensive-rag-benchmark#ran","syntology_url":"https://syntology.ai/paper/2406.04744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04744"}},"official":{"repos":["facebookresearch/crag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/complextempqa-a-large-scale-dataset-for","slug":"complextempqa-a-large-scale-dataset-for","title":"ComplexTempQA: A Large-Scale Dataset for Complex Temporal Question Answering","date":"2024-06-07","arxiv_id":"2406.04866","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/complextempqa-a-large-scale-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2406.04866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04866"}},"official":{"repos":["datascienceuibk/complextempqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-subjective-uncertainty-quantification-and","slug":"on-subjective-uncertainty-quantification-and","title":"On Subjective Uncertainty Quantification and Calibration in Natural Language Generation","date":"2024-06-07","arxiv_id":"2406.05213","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-subjective-uncertainty-quantification-and#ran","syntology_url":"https://syntology.ai/paper/2406.05213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05213"}},"official":{"repos":["meta-inf/suq-nlg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/corda-context-oriented-decomposition","slug":"corda-context-oriented-decomposition","title":"CorDA: Context-Oriented Decomposition Adaptation of Large Language Models for Task-Aware Parameter-Efficient Fine-tuning","date":"2024-06-07","arxiv_id":"2406.05223","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/corda-context-oriented-decomposition#ran","syntology_url":"https://syntology.ai/paper/2406.05223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05223"}},"official":{"repos":["iboing/corda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/semantically-diverse-language-generation-for","slug":"semantically-diverse-language-generation-for","title":"Semantically Diverse Language Generation for Uncertainty Estimation in Language Models","date":"2024-06-06","arxiv_id":"2406.04306","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/semantically-diverse-language-generation-for#ran","syntology_url":"https://syntology.ai/paper/2406.04306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04306"}},"official":{"repos":["ml-jku/SDLG"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/css-contrastive-semantic-similarity-for","slug":"css-contrastive-semantic-similarity-for","title":"CSS: Contrastive Semantic Similarity for Uncertainty Quantification of LLMs","date":"2024-06-05","arxiv_id":"2406.03158","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/css-contrastive-semantic-similarity-for#ran","syntology_url":"https://syntology.ai/paper/2406.03158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03158"}},"official":{"repos":["aoshuang92/css_uq_llms"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/wings-learning-multimodal-llms-without-text","slug":"wings-learning-multimodal-llms-without-text","title":"Wings: Learning Multimodal LLMs without Text-only Forgetting","date":"2024-06-05","arxiv_id":"2406.03496","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wings-learning-multimodal-llms-without-text#ran","syntology_url":"https://syntology.ai/paper/2406.03496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03496"}},"official":null}},{"url":"/paper/retaining-key-information-under-high","slug":"retaining-key-information-under-high","title":"Retaining Key Information under High Compression Ratios: Query-Guided Compressor for LLMs","date":"2024-06-04","arxiv_id":"2406.02376","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retaining-key-information-under-high#ran","syntology_url":"https://syntology.ai/paper/2406.02376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02376"}},"official":{"repos":["DeepLearnXMU/QGC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mediq-question-asking-llms-for-adaptive-and","slug":"mediq-question-asking-llms-for-adaptive-and","title":"MediQ: Question-Asking LLMs and a Benchmark for Reliable Interactive Clinical Reasoning","date":"2024-06-03","arxiv_id":"2406.00922","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mediq-question-asking-llms-for-adaptive-and#ran","syntology_url":"https://syntology.ai/paper/2406.00922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00922"}},"official":{"repos":["stellali7/mediq","stellalisy/mediq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-information-bottleneck-perspective-for","slug":"an-information-bottleneck-perspective-for","title":"An Information Bottleneck Perspective for Effective Noise Filtering on Retrieval-Augmented Generation","date":"2024-06-03","arxiv_id":"2406.01549","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-information-bottleneck-perspective-for#ran","syntology_url":"https://syntology.ai/paper/2406.01549","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01549"}},"official":{"repos":["zhukun1020/noisefilter_ib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"70ca81aad3a59cac531eea0e1030973a2f72a0fe35b6e472df73ec77bf2a751c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}