{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/16","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":16,"pages_in_order":109,"rows_per_page":100,"rows":[1501,1600],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/15","next":"/task/question-answering/papers/17","papers":[{"url":"/paper/listen-and-speak-fairly-a-study-on-semantic","slug":"listen-and-speak-fairly-a-study-on-semantic","title":"Listen and Speak Fairly: A Study on Semantic Gender Bias in Speech Integrated Large Language Models","date":"2024-07-09","arxiv_id":"2407.06957","repositories_listed":1,"syntology":null},{"url":"/paper/peer-expertizing-domain-specific-tasks-with-a","slug":"peer-expertizing-domain-specific-tasks-with-a","title":"PEER: Expertizing Domain-Specific Tasks with a Multi-Agent Framework and Tuning Methods","date":"2024-07-09","arxiv_id":"2407.06985","repositories_listed":1,"syntology":null},{"url":"/paper/3d-vision-and-language-pretraining-with-large","slug":"3d-vision-and-language-pretraining-with-large","title":"3D Vision and Language Pretraining with Large-Scale Synthetic Data","date":"2024-07-08","arxiv_id":"2407.06084","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-vision-and-language-pretraining-with-large#ran","syntology_url":"https://syntology.ai/paper/2407.06084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06084"}},"official":{"repos":["idejie/3DSyn"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-understand-layouts","slug":"large-language-models-understand-layouts","title":"Large Language Models Understand Layout","date":"2024-07-08","arxiv_id":"2407.05750","repositories_listed":1,"syntology":null},{"url":"/paper/mst5-multilingual-question-answering-over","slug":"mst5-multilingual-question-answering-over","title":"MST5 -- Multilingual Question Answering over Knowledge Graphs","date":"2024-07-08","arxiv_id":"2407.06041","repositories_listed":1,"syntology":null},{"url":"/paper/wsi-vqa-interpreting-whole-slide-images-by","slug":"wsi-vqa-interpreting-whole-slide-images-by","title":"WSI-VQA: Interpreting Whole Slide Images by Generative Visual Question Answering","date":"2024-07-08","arxiv_id":"2407.05603","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wsi-vqa-interpreting-whole-slide-images-by#ran","syntology_url":"https://syntology.ai/paper/2407.05603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05603"}},"official":{"repos":["cpystan/wsi-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-do-you-know-that-teaching-generative","slug":"how-do-you-know-that-teaching-generative","title":"How do you know that? Teaching Generative Language Models to Reference Answers to Biomedical Questions","date":"2024-07-06","arxiv_id":"2407.05015","repositories_listed":1,"syntology":null},{"url":"/paper/mfe-etp-a-comprehensive-evaluation-benchmark","slug":"mfe-etp-a-comprehensive-evaluation-benchmark","title":"MFE-ETP: A Comprehensive Evaluation Benchmark for Multi-modal Foundation Models on Embodied Task Planning","date":"2024-07-06","arxiv_id":"2407.05047","repositories_listed":1,"syntology":null},{"url":"/paper/anah-v2-scaling-analytical-hallucination","slug":"anah-v2-scaling-analytical-hallucination","title":"ANAH-v2: Scaling Analytical Hallucination Annotation of Large Language Models","date":"2024-07-05","arxiv_id":"2407.04693","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/anah-v2-scaling-analytical-hallucination#ran","syntology_url":"https://syntology.ai/paper/2407.04693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04693"}},"official":{"repos":["open-compass/anah"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/arigraph-learning-knowledge-graph-world","slug":"arigraph-learning-knowledge-graph-world","title":"AriGraph: Learning Knowledge Graph World Models with Episodic Memory for LLM Agents","date":"2024-07-05","arxiv_id":"2407.04363","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arigraph-learning-knowledge-graph-world#ran","syntology_url":"https://syntology.ai/paper/2407.04363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04363"}},"official":{"repos":["airi-institute/arigraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/me-myself-and-ai-the-situational-awareness","slug":"me-myself-and-ai-the-situational-awareness","title":"Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs","date":"2024-07-05","arxiv_id":"2407.04694","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/me-myself-and-ai-the-situational-awareness#ran","syntology_url":"https://syntology.ai/paper/2407.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04694"}},"official":{"repos":["lrudl/sad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chartgemma-visual-instruction-tuning-for","slug":"chartgemma-visual-instruction-tuning-for","title":"ChartGemma: Visual Instruction-tuning for Chart Reasoning in the Wild","date":"2024-07-04","arxiv_id":"2407.04172","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-topic-specificity-and-social","slug":"leveraging-topic-specificity-and-social","title":"Leveraging Topic Specificity and Social Relationships for Expert Finding in Community Question Answering Platforms","date":"2024-07-04","arxiv_id":"2407.04018","repositories_listed":1,"syntology":null},{"url":"/paper/meta-optimized-angular-margin-contrastive","slug":"meta-optimized-angular-margin-contrastive","title":"MAMA: Meta-optimized Angular Margin Contrastive Framework for Video-Language Representation Learning","date":"2024-07-04","arxiv_id":"2407.03788","repositories_listed":1,"syntology":null},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-to-do-if-language-models-disagree-black","slug":"what-to-do-if-language-models-disagree-black","title":"Black-box Model Ensembling for Textual and Visual Question Answering via Information Fusion","date":"2024-07-04","arxiv_id":"2407.12841","repositories_listed":1,"syntology":null},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logeval-a-comprehensive-benchmark-suite-for","slug":"logeval-a-comprehensive-benchmark-suite-for","title":"LogEval: A Comprehensive Benchmark Suite for Large Language Models In Log Analysis","date":"2024-07-02","arxiv_id":"2407.01896","repositories_listed":1,"syntology":null},{"url":"/paper/neurocache-efficient-vector-retrieval-for","slug":"neurocache-efficient-vector-retrieval-for","title":"Neurocache: Efficient Vector Retrieval for Long-range Language Modeling","date":"2024-07-02","arxiv_id":"2407.02486","repositories_listed":1,"syntology":null},{"url":"/paper/referring-atomic-video-action-recognition","slug":"referring-atomic-video-action-recognition","title":"Referring Atomic Video Action Recognition","date":"2024-07-02","arxiv_id":"2407.01872","repositories_listed":1,"syntology":null},{"url":"/paper/cvlue-a-new-benchmark-dataset-for-chinese","slug":"cvlue-a-new-benchmark-dataset-for-chinese","title":"CVLUE: A New Benchmark Dataset for Chinese Vision-Language Understanding Evaluation","date":"2024-07-01","arxiv_id":"2407.01081","repositories_listed":1,"syntology":null},{"url":"/paper/eliminating-position-bias-of-language-models","slug":"eliminating-position-bias-of-language-models","title":"Eliminating Position Bias of Language Models: A Mechanistic Approach","date":"2024-07-01","arxiv_id":"2407.01100","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eliminating-position-bias-of-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.01100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01100"}},"official":{"repos":["wzq016/pine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/increasing-model-capacity-for-free-a-simple","slug":"increasing-model-capacity-for-free-a-simple","title":"Increasing Model Capacity for Free: A Simple Strategy for Parameter Efficient Fine-tuning","date":"2024-07-01","arxiv_id":"2407.01320","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/increasing-model-capacity-for-free-a-simple#ran","syntology_url":"https://syntology.ai/paper/2407.01320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01320"}},"official":{"repos":["lins-lab/capaboost"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/m2qa-multi-domain-multilingual-question","slug":"m2qa-multi-domain-multilingual-question","title":"M2QA: Multi-domain Multilingual Question Answering","date":"2024-07-01","arxiv_id":"2407.01091","repositories_listed":1,"syntology":null},{"url":"/paper/searching-for-best-practices-in-retrieval","slug":"searching-for-best-practices-in-retrieval","title":"Searching for Best Practices in Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01219","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/searching-for-best-practices-in-retrieval#ran","syntology_url":"https://syntology.ai/paper/2407.01219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01219"}},"official":{"repos":["FudanDNN-NLP/RAG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/polygongnn-representation-learning-for","slug":"polygongnn-representation-learning-for","title":"PolygonGNN: Representation Learning for Polygonal Geometries with Heterogeneous Visibility Graph","date":"2024-06-30","arxiv_id":"2407.00742","repositories_listed":1,"syntology":null},{"url":"/paper/biokgbench-a-knowledge-graph-checking","slug":"biokgbench-a-knowledge-graph-checking","title":"BioKGBench: A Knowledge Graph Checking Benchmark of AI Agent for Biomedical Science","date":"2024-06-29","arxiv_id":"2407.00466","repositories_listed":1,"syntology":null},{"url":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive","slug":"h-star-llm-driven-hybrid-sql-text-adaptive","title":"H-STAR: LLM-driven Hybrid SQL-Text Adaptive Reasoning on Tables","date":"2024-06-29","arxiv_id":"2407.05952","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive#ran","syntology_url":"https://syntology.ai/paper/2407.05952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05952"}},"official":{"repos":["nikhilsab/h-star"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llavolta-efficient-multi-modal-models-via","slug":"llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","arxiv_id":"2406.20092","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llavolta-efficient-multi-modal-models-via#ran","syntology_url":"https://syntology.ai/paper/2406.20092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20092"}},"official":{"repos":["beckschen/llavolta"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","repositories_listed":1,"syntology":{"n":19,"n_ran":9,"n_constructed":4,"n_ran_checked":5,"n_instrument":4,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/stllava-med-self-training-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.19973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19973"}},"official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/the-sifo-benchmark-investigating-the","slug":"the-sifo-benchmark-investigating-the","title":"The SIFo Benchmark: Investigating the Sequential Instruction Following Ability of Large Language Models","date":"2024-06-28","arxiv_id":"2406.19999","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-sifo-benchmark-investigating-the#ran","syntology_url":"https://syntology.ai/paper/2406.19999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19999"}},"official":{"repos":["shin-ee-chen/SIFo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-continual-learning-in-visual","slug":"enhancing-continual-learning-in-visual","title":"Enhancing Continual Learning in Visual Question Answering with Modality-Aware Feature Distillation","date":"2024-06-27","arxiv_id":"2406.19297","repositories_listed":1,"syntology":null},{"url":"/paper/handling-ontology-gaps-in-semantic-parsing","slug":"handling-ontology-gaps-in-semantic-parsing","title":"Handling Ontology Gaps in Semantic Parsing","date":"2024-06-27","arxiv_id":"2406.19537","repositories_listed":1,"syntology":null},{"url":"/paper/length-optimization-in-conformal-prediction","slug":"length-optimization-in-conformal-prediction","title":"Length Optimization in Conformal Prediction","date":"2024-06-27","arxiv_id":"2406.18814","repositories_listed":1,"syntology":null},{"url":"/paper/seakr-self-aware-knowledge-retrieval-for","slug":"seakr-self-aware-knowledge-retrieval-for","title":"SeaKR: Self-aware Knowledge Retrieval for Adaptive Retrieval Augmented Generation","date":"2024-06-27","arxiv_id":"2406.19215","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/seakr-self-aware-knowledge-retrieval-for#ran","syntology_url":"https://syntology.ai/paper/2406.19215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19215"}},"official":{"repos":["thu-keg/seakr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-illusion-of-competence-evaluating-the","slug":"the-illusion-of-competence-evaluating-the","title":"The Illusion of Competence: Evaluating the Effect of Explanations on Users' Mental Models of Visual Question Answering Systems","date":"2024-06-27","arxiv_id":"2406.19170","repositories_listed":1,"syntology":null},{"url":"/paper/trustuqa-a-trustful-framework-for-unified","slug":"trustuqa-a-trustful-framework-for-unified","title":"TrustUQA: A Trustful Framework for Unified Structured Data Question Answering","date":"2024-06-27","arxiv_id":"2406.18916","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trustuqa-a-trustful-framework-for-unified#ran","syntology_url":"https://syntology.ai/paper/2406.18916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18916"}},"official":{"repos":["zjukg/trustuqa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-graph-enhanced-retrieval-augmented","slug":"knowledge-graph-enhanced-retrieval-augmented","title":"Knowledge graph enhanced retrieval-augmented generation for failure mode and effects analysis","date":"2024-06-26","arxiv_id":"2406.18114","repositories_listed":1,"syntology":null},{"url":"/paper/understand-what-llm-needs-dual-preference","slug":"understand-what-llm-needs-dual-preference","title":"Understand What LLM Needs: Dual Preference Alignment for Retrieval-Augmented Generation","date":"2024-06-26","arxiv_id":"2406.18676","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/understand-what-llm-needs-dual-preference#ran","syntology_url":"https://syntology.ai/paper/2406.18676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18676"}},"official":{"repos":["dongguanting/dpa-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/calmqa-exploring-culturally-specific-long","slug":"calmqa-exploring-culturally-specific-long","title":"CaLMQA: Exploring culturally specific long-form question answering across 23 languages","date":"2024-06-25","arxiv_id":"2406.17761","repositories_listed":1,"syntology":null},{"url":"/paper/cogmg-collaborative-augmentation-between","slug":"cogmg-collaborative-augmentation-between","title":"CogMG: Collaborative Augmentation Between Large Language Model and Knowledge Graph","date":"2024-06-25","arxiv_id":"2406.17231","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-based-decoding-for-retrieval","slug":"entropy-based-decoding-for-retrieval","title":"Entropy-Based Decoding for Retrieval-Augmented Large Language Models","date":"2024-06-25","arxiv_id":"2406.17519","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-fairness-in-large-vision-language","slug":"evaluating-fairness-in-large-vision-language","title":"Evaluating Fairness in Large Vision-Language Models Across Diverse Demographic Attributes and Prompts","date":"2024-06-25","arxiv_id":"2406.17974","repositories_listed":1,"syntology":null},{"url":"/paper/leave-no-document-behind-benchmarking-long","slug":"leave-no-document-behind-benchmarking-long","title":"Leave No Document Behind: Benchmarking Long-Context LLMs with Extended Multi-Doc QA","date":"2024-06-25","arxiv_id":"2406.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/leave-no-document-behind-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2406.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17419"}},"official":{"repos":["mozerwang/loong"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-instruction-amplifying-attention-in","slug":"attention-instruction-amplifying-attention-in","title":"Attention Instruction: Amplifying Attention in the Middle via Prompting","date":"2024-06-24","arxiv_id":"2406.17095","repositories_listed":1,"syntology":null},{"url":"/paper/dexter-a-benchmark-for-open-domain-complex","slug":"dexter-a-benchmark-for-open-domain-complex","title":"DEXTER: A Benchmark for open-domain Complex Question Answering using LLMs","date":"2024-06-24","arxiv_id":"2406.17158","repositories_listed":1,"syntology":null},{"url":"/paper/llms-assist-nlp-researchers-critique-paper","slug":"llms-assist-nlp-researchers-critique-paper","title":"LLMs Assist NLP Researchers: Critique Paper (Meta-)Reviewing","date":"2024-06-24","arxiv_id":"2406.16253","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-exponential-extension-of","slug":"training-free-exponential-extension-of","title":"Training-Free Exponential Context Extension via Cascading KV Cache","date":"2024-06-24","arxiv_id":"2406.17808","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-exponential-extension-of#ran","syntology_url":"https://syntology.ai/paper/2406.17808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17808"}},"official":{"repos":["jeffwillette/cascading_kv_cache"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unipsda-unsupervised-pseudo-semantic-data","slug":"unipsda-unsupervised-pseudo-semantic-data","title":"UniPSDA: Unsupervised Pseudo Semantic Data Augmentation for Zero-Shot Cross-Lingual Natural Language Understanding","date":"2024-06-24","arxiv_id":"2406.16372","repositories_listed":1,"syntology":null},{"url":"/paper/hcqa-ego4d-egoschema-challenge-2024","slug":"hcqa-ego4d-egoschema-challenge-2024","title":"HCQA @ Ego4D EgoSchema Challenge 2024","date":"2024-06-22","arxiv_id":"2406.15771","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hcqa-ego4d-egoschema-challenge-2024#ran","syntology_url":"https://syntology.ai/paper/2406.15771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15771"}},"official":{"repos":["hyu-zhang/hcqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uda-a-benchmark-suite-for-retrieval-augmented","slug":"uda-a-benchmark-suite-for-retrieval-augmented","title":"UDA: A Benchmark Suite for Retrieval Augmented Generation in Real-world Document Analysis","date":"2024-06-21","arxiv_id":"2406.15187","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uda-a-benchmark-suite-for-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.15187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15187"}},"official":{"repos":["qinchuanhui/uda-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/augmenting-query-and-passage-for-retrieval","slug":"augmenting-query-and-passage-for-retrieval","title":"QPaug: Question and Passage Augmentation for Open-Domain Question Answering of LLMs","date":"2024-06-20","arxiv_id":"2406.14277","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/augmenting-query-and-passage-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2406.14277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14277"}},"official":{"repos":["kmswin1/qpaug"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-rag-fusion-with-ragelo-an","slug":"evaluating-rag-fusion-with-ragelo-an","title":"Evaluating RAG-Fusion with RAGElo: an Automated Elo-based Framework","date":"2024-06-20","arxiv_id":"2406.14783","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-rag-fusion-with-ragelo-an#ran","syntology_url":"https://syntology.ai/paper/2406.14783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14783"}},"official":{"repos":["zetaalphavector/ragelo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-plan-for-retrieval-augmented","slug":"learning-to-plan-for-retrieval-augmented","title":"Learning to Plan for Retrieval-Augmented Large Language Models from Knowledge Graphs","date":"2024-06-20","arxiv_id":"2406.14282","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-plan-for-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.14282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14282"}},"official":{"repos":["zjukg/lpkg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llasa-large-multimodal-agent-for-human","slug":"llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","arxiv_id":"2406.14498","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llasa-large-multimodal-agent-for-human#ran","syntology_url":"https://syntology.ai/paper/2406.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14498"}},"official":{"repos":["bashlab/llasa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/supergleber-german-language-understanding","slug":"supergleber-german-language-understanding","title":"SuperGLEBer: German Language Understanding Evaluation Benchmark","date":"2024-06-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/taglas-an-atlas-of-text-attributed-graph","slug":"taglas-an-atlas-of-text-attributed-graph","title":"TAGLAS: An atlas of text-attributed graph datasets in the era of large graph and language models","date":"2024-06-20","arxiv_id":"2406.14683","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/taglas-an-atlas-of-text-attributed-graph#ran","syntology_url":"https://syntology.ai/paper/2406.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14683"}},"official":{"repos":["jiaruifeng/taglas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timo-towards-better-temporal-reasoning-for","slug":"timo-towards-better-temporal-reasoning-for","title":"Timo: Towards Better Temporal Reasoning for Language Models","date":"2024-06-20","arxiv_id":"2406.14192","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timo-towards-better-temporal-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2406.14192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14192"}},"official":{"repos":["zhaochen0110/timo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vga-vision-gui-assistant-minimizing","slug":"vga-vision-gui-assistant-minimizing","title":"VGA: Vision GUI Assistant -- Minimizing Hallucinations through Image-Centric Fine-Tuning","date":"2024-06-20","arxiv_id":"2406.14056","repositories_listed":1,"syntology":null},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-hallucinations-in-large-language-1","slug":"detecting-hallucinations-in-large-language-1","title":"Detecting hallucinations in large language models using semantic entropy","date":"2024-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dialsim-a-real-time-simulator-for-evaluating","slug":"dialsim-a-real-time-simulator-for-evaluating","title":"DialSim: A Real-Time Simulator for Evaluating Long-Term Multi-Party Dialogue Understanding of Conversational Agents","date":"2024-06-19","arxiv_id":"2406.13144","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dialsim-a-real-time-simulator-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2406.13144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13144"}},"official":{"repos":["jiho283/simulator"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-cross-prompt-transferability-in","slug":"enhancing-cross-prompt-transferability-in","title":"Enhancing Cross-Prompt Transferability in Vision-Language Models through Contextual Injection of Target Tokens","date":"2024-06-19","arxiv_id":"2406.13294","repositories_listed":1,"syntology":null},{"url":"/paper/factual-confidence-of-llms-on-reliability-and","slug":"factual-confidence-of-llms-on-reliability-and","title":"Factual Confidence of LLMs: on Reliability and Robustness of Current Estimators","date":"2024-06-19","arxiv_id":"2406.13415","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/factual-confidence-of-llms-on-reliability-and#ran","syntology_url":"https://syntology.ai/paper/2406.13415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13415"}},"official":{"repos":["amazon-science/factual-confidence-of-llms"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/model-internals-based-answer-attribution-for","slug":"model-internals-based-answer-attribution-for","title":"Model Internals-based Answer Attribution for Trustworthy Retrieval-Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13663","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-internals-based-answer-attribution-for#ran","syntology_url":"https://syntology.ai/paper/2406.13663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13663"}},"official":{"repos":["betswish/mirage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/morehopqa-more-than-multi-hop-reasoning","slug":"morehopqa-more-than-multi-hop-reasoning","title":"MoreHopQA: More Than Multi-hop Reasoning","date":"2024-06-19","arxiv_id":"2406.13397","repositories_listed":1,"syntology":null},{"url":"/paper/breaking-the-ceiling-of-the-llm-community-by","slug":"breaking-the-ceiling-of-the-llm-community-by","title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","date":"2024-06-18","arxiv_id":"2406.12585","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/breaking-the-ceiling-of-the-llm-community-by#ran","syntology_url":"https://syntology.ai/paper/2406.12585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12585"}},"official":{"repos":["yaoching0/gac"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gw-moe-resolving-uncertainty-in-moe-router","slug":"gw-moe-resolving-uncertainty-in-moe-router","title":"GW-MoE: Resolving Uncertainty in MoE Router with Global Workspace Theory","date":"2024-06-18","arxiv_id":"2406.12375","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prompting-taxonomy-a-universal","slug":"hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","arxiv_id":"2406.12644","repositories_listed":1,"syntology":null},{"url":"/paper/nash-cot-multi-path-inference-with-preference","slug":"nash-cot-multi-path-inference-with-preference","title":"Nash CoT: Multi-Path Inference with Preference Equilibrium","date":"2024-06-18","arxiv_id":"2407.07099","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/nash-cot-multi-path-inference-with-preference#ran","syntology_url":"https://syntology.ai/paper/2407.07099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07099"}},"official":{"repos":["stevezhangza/nash-chain-of-thought"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/problem-solving-in-language-model-networks","slug":"problem-solving-in-language-model-networks","title":"Problem-Solving in Language Model Networks","date":"2024-06-18","arxiv_id":"2406.12374","repositories_listed":1,"syntology":null},{"url":"/paper/pslm-parallel-generation-of-text-and-speech","slug":"pslm-parallel-generation-of-text-and-speech","title":"PSLM: Parallel Generation of Text and Speech with LLMs for Low-Latency Spoken Dialogue Systems","date":"2024-06-18","arxiv_id":"2406.12428","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pslm-parallel-generation-of-text-and-speech#ran","syntology_url":"https://syntology.ai/paper/2406.12428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12428"}},"official":{"repos":["eleutherai/gpt-neox"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rationale-based-ensemble-of-multiple-qa","slug":"rationale-based-ensemble-of-multiple-qa","title":"Diversify, Rationalize, and Combine: Ensembling Multiple QA Strategies for Zero-shot Knowledge-based VQA","date":"2024-06-18","arxiv_id":"2406.12746","repositories_listed":1,"syntology":null},{"url":"/paper/voco-llama-towards-vision-compression-with","slug":"voco-llama-towards-vision-compression-with","title":"VoCo-LLaMA: Towards Vision Compression with Large Language Models","date":"2024-06-18","arxiv_id":"2406.12275","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":5,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voco-llama-towards-vision-compression-with#ran","syntology_url":"https://syntology.ai/paper/2406.12275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12275"}},"official":{"repos":["Yxxxb/VoCo-LLaMA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/avatar-optimizing-llm-agents-for-tool","slug":"avatar-optimizing-llm-agents-for-tool","title":"AvaTaR: Optimizing LLM Agents for Tool Usage via Contrastive Reasoning","date":"2024-06-17","arxiv_id":"2406.11200","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/avatar-optimizing-llm-agents-for-tool#ran","syntology_url":"https://syntology.ai/paper/2406.11200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11200"}},"official":{"repos":["zou-group/avatar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-scientific-concepts-understanding","slug":"boosting-scientific-concepts-understanding","title":"Boosting Scientific Concepts Understanding: Can Analogy from Teacher Models Empower Student Models?","date":"2024-06-17","arxiv_id":"2406.11375","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/boosting-scientific-concepts-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.11375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11375"}},"official":{"repos":["siyuyuan/scua"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/extrinsic-evaluation-of-cultural-competence","slug":"extrinsic-evaluation-of-cultural-competence","title":"Extrinsic Evaluation of Cultural Competence in Large Language Models","date":"2024-06-17","arxiv_id":"2406.11565","repositories_listed":1,"syntology":null},{"url":"/paper/learn-beyond-the-answer-training-language","slug":"learn-beyond-the-answer-training-language","title":"Learn Beyond The Answer: Training Language Models with Reflection for Mathematical Reasoning","date":"2024-06-17","arxiv_id":"2406.12050","repositories_listed":1,"syntology":null},{"url":"/paper/medcalc-bench-evaluating-large-language","slug":"medcalc-bench-evaluating-large-language","title":"MedCalc-Bench: Evaluating Large Language Models for Medical Calculations","date":"2024-06-17","arxiv_id":"2406.12036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medcalc-bench-evaluating-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.12036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12036"}},"official":{"repos":["ncbi-nlp/medcalc-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mfc-bench-benchmarking-multimodal-fact","slug":"mfc-bench-benchmarking-multimodal-fact","title":"MFC-Bench: Benchmarking Multimodal Fact-Checking with Large Vision-Language Models","date":"2024-06-17","arxiv_id":"2406.11288","repositories_listed":1,"syntology":null},{"url":"/paper/mmneuron-discovering-neuron-level-domain","slug":"mmneuron-discovering-neuron-level-domain","title":"MMNeuron: Discovering Neuron-Level Domain-Specific Interpretation in Multimodal Large Language Model","date":"2024-06-17","arxiv_id":"2406.11193","repositories_listed":1,"syntology":null},{"url":"/paper/repliqa-a-question-answering-dataset-for","slug":"repliqa-a-question-answering-dataset-for","title":"RepLiQA: A Question-Answering Dataset for Benchmarking LLMs on Unseen Reference Content","date":"2024-06-17","arxiv_id":"2406.11811","repositories_listed":1,"syntology":null},{"url":"/paper/safety-arithmetic-a-framework-for-test-time","slug":"safety-arithmetic-a-framework-for-test-time","title":"Safety Arithmetic: A Framework for Test-time Safety Alignment of Language Models by Steering Parameters and Activations","date":"2024-06-17","arxiv_id":"2406.11801","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safety-arithmetic-a-framework-for-test-time#ran","syntology_url":"https://syntology.ai/paper/2406.11801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11801"}},"official":{"repos":["declare-lab/safety-arithmetic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-prompting-for-unlearning-in-large","slug":"soft-prompting-for-unlearning-in-large","title":"Soft Prompting for Unlearning in Large Language Models","date":"2024-06-17","arxiv_id":"2406.12038","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/soft-prompting-for-unlearning-in-large#ran","syntology_url":"https://syntology.ai/paper/2406.12038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12038"}},"official":{"repos":["karuna-bhaila/llm_unlearning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/task-me-anything","slug":"task-me-anything","title":"Task Me Anything","date":"2024-06-17","arxiv_id":"2406.11775","repositories_listed":1,"syntology":null},{"url":"/paper/textit-refiner-restructure-retrieval-content","slug":"textit-refiner-restructure-retrieval-content","title":"Refiner: Restructure Retrieval Content Efficiently to Advance Question-Answering Capabilities","date":"2024-06-17","arxiv_id":"2406.11357","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/textit-refiner-restructure-retrieval-content#ran","syntology_url":"https://syntology.ai/paper/2406.11357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11357"}},"official":{"repos":["allen-li1231/refiner-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-key-neurons-in-large-language","slug":"analyzing-key-neurons-in-large-language","title":"Identifying Query-Relevant Neurons in Large Language Models for Long-Form Texts","date":"2024-06-16","arxiv_id":"2406.10868","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analyzing-key-neurons-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10868"}},"official":{"repos":["tigerchen52/qrneuron"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/foodieqa-a-multimodal-dataset-for-fine","slug":"foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","arxiv_id":"2406.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foodieqa-a-multimodal-dataset-for-fine#ran","syntology_url":"https://syntology.ai/paper/2406.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11030"}},"official":{"repos":["lyan62/FoodieQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}},"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scar-efficient-instruction-tuning-for-large","slug":"scar-efficient-instruction-tuning-for-large","title":"SCAR: Efficient Instruction-Tuning for Large Language Models via Style Consistency-Aware Response Ranking","date":"2024-06-16","arxiv_id":"2406.10882","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scar-efficient-instruction-tuning-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.10882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10882"}},"official":{"repos":["zhuang-li/scar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-raw-videos-understanding-edited-videos","slug":"beyond-raw-videos-understanding-edited-videos","title":"Beyond Raw Videos: Understanding Edited Videos with Large Multimodal Model","date":"2024-06-15","arxiv_id":"2406.10484","repositories_listed":1,"syntology":null},{"url":"/paper/color-filter-conditional-loss-reduction","slug":"color-filter-conditional-loss-reduction","title":"CoLoR-Filter: Conditional Loss Reduction Filtering for Targeted Language Model Pre-training","date":"2024-06-15","arxiv_id":"2406.10670","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/color-filter-conditional-loss-reduction#ran","syntology_url":"https://syntology.ai/paper/2406.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10670"}},"official":{"repos":["davidbrandfonbrener/color-filter-olmo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-as-event-forecasters","slug":"large-language-models-as-event-forecasters","title":"Large Language Models as Interpolated and Extrapolated Event Predictors","date":"2024-06-15","arxiv_id":"2406.10492","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-video-datasets-for-grounded-event","slug":"a-survey-of-video-datasets-for-grounded-event","title":"A Survey of Video Datasets for Grounded Event Understanding","date":"2024-06-14","arxiv_id":"2406.09646","repositories_listed":1,"syntology":null},{"url":"/paper/chiron-rich-character-representations-in-long","slug":"chiron-rich-character-representations-in-long","title":"CHIRON: Rich Character Representations in Long-Form Narratives","date":"2024-06-14","arxiv_id":"2406.10190","repositories_listed":1,"syntology":null},{"url":"/paper/chisafetybench-a-chinese-hierarchical-safety","slug":"chisafetybench-a-chinese-hierarchical-safety","title":"CHiSafetyBench: A Chinese Hierarchical Safety Benchmark for Large Language Models","date":"2024-06-14","arxiv_id":"2406.10311","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-meet-meteorology","slug":"vision-language-models-meet-meteorology","title":"Vision-Language Models Meet Meteorology: Developing Models for Extreme Weather Events Detection with Heatmaps","date":"2024-06-14","arxiv_id":"2406.09838","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-preference-optimization-improving","slug":"chain-of-preference-optimization-improving","title":"Chain of Preference Optimization: Improving Chain-of-Thought Reasoning in LLMs","date":"2024-06-13","arxiv_id":"2406.09136","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-preference-optimization-improving#ran","syntology_url":"https://syntology.ai/paper/2406.09136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09136"}},"official":{"repos":["sail-sg/cpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-the-limits-of-omni-modal-pretraining","slug":"explore-the-limits-of-omni-modal-pretraining","title":"Explore the Limits of Omni-modal Pretraining at Scale","date":"2024-06-13","arxiv_id":"2406.09412","repositories_listed":1,"syntology":null},{"url":"/paper/living-in-the-moment-can-large-language","slug":"living-in-the-moment-can-large-language","title":"Living in the Moment: Can Large Language Models Grasp Co-Temporal Reasoning?","date":"2024-06-13","arxiv_id":"2406.09072","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/living-in-the-moment-can-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.09072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09072"}},"official":{"repos":["zhaochen0110/cotempqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"375242d3c8020671bcb9a488e73006e9f442777f22666277e3783f3d05a024ec","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}