{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/ran/7","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":7,"pages_in_order":13,"rows_per_page":100,"rows":[601,700],"of":1274,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering/papers/ran/1","prev":"/task/question-answering/papers/ran/6","next":"/task/question-answering/papers/ran/8","papers":[{"url":"/paper/vdc-versatile-data-cleanser-for-detecting","slug":"vdc-versatile-data-cleanser-for-detecting","title":"VDC: Versatile Data Cleanser based on Visual-Linguistic Inconsistency by Multimodal Large Language Models","date":"2023-09-28","arxiv_id":"2309.16211","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":4,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/vdc-versatile-data-cleanser-for-detecting#ran","syntology_url":"https://syntology.ai/paper/2309.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16211"}},"official":{"repos":["zihao-ai/vdc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/at-which-training-stage-does-cocde-data-help","slug":"at-which-training-stage-does-cocde-data-help","title":"At Which Training Stage Does Code Data Help LLMs Reasoning?","date":"2023-09-28","arxiv_id":"2309.16298","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/at-which-training-stage-does-cocde-data-help#ran","syntology_url":"https://syntology.ai/paper/2309.16298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16298"}},"official":{"repos":["yingweima2022/codellm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bamboo-a-comprehensive-benchmark-for","slug":"bamboo-a-comprehensive-benchmark-for","title":"BAMBOO: A Comprehensive Benchmark for Evaluating Long Text Modeling Capacities of Large Language Models","date":"2023-09-23","arxiv_id":"2309.13345","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bamboo-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2309.13345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13345"}},"official":{"repos":["rucaibox/bamboo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-sanitization-of-large-language","slug":"knowledge-sanitization-of-large-language","title":"Knowledge Sanitization of Large Language Models","date":"2023-09-21","arxiv_id":"2309.11852","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/knowledge-sanitization-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2309.11852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11852"}},"official":{"repos":["yoichi1484/knowledge-sanitization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longlora-efficient-fine-tuning-of-long","slug":"longlora-efficient-fine-tuning-of-long","title":"LongLoRA: Efficient Fine-tuning of Long-Context Large Language Models","date":"2023-09-21","arxiv_id":"2309.12307","repositories_listed":4,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/longlora-efficient-fine-tuning-of-long#ran","syntology_url":"https://syntology.ai/paper/2309.12307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12307"}},"official":{"repos":["dvlab-research/longlora"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/retrieve-rewrite-answer-a-kg-to-text-enhanced","slug":"retrieve-rewrite-answer-a-kg-to-text-enhanced","title":"Retrieve-Rewrite-Answer: A KG-to-Text Enhanced LLMs Framework for Knowledge Graph Question Answering","date":"2023-09-20","arxiv_id":"2309.11206","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retrieve-rewrite-answer-a-kg-to-text-enhanced#ran","syntology_url":"https://syntology.ai/paper/2309.11206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11206"}},"official":{"repos":["wuyike2000/retrieve-rewrite-answer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-large-language-models-via-reading","slug":"adapting-large-language-models-via-reading","title":"Adapting Large Language Models to Domains via Reading Comprehension","date":"2023-09-18","arxiv_id":"2309.09530","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adapting-large-language-models-via-reading#ran","syntology_url":"https://syntology.ai/paper/2309.09530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09530"}},"official":{"repos":["microsoft/lmops"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fabricator-an-open-source-toolkit-for","slug":"fabricator-an-open-source-toolkit-for","title":"Fabricator: An Open Source Toolkit for Generating Labeled Training Data with Teacher LLMs","date":"2023-09-18","arxiv_id":"2309.09582","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fabricator-an-open-source-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2309.09582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09582"}},"official":{"repos":["flairnlp/fabricator"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/are-multilingual-llms-culturally-diverse","slug":"are-multilingual-llms-culturally-diverse","title":"Are Multilingual LLMs Culturally-Diverse Reasoners? An Investigation into Multicultural Proverbs and Sayings","date":"2023-09-15","arxiv_id":"2309.08591","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/are-multilingual-llms-culturally-diverse#ran","syntology_url":"https://syntology.ai/paper/2309.08591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.08591"}},"official":{"repos":["UKPLab/maps"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/catfood-counterfactual-augmented-training-for","slug":"catfood-counterfactual-augmented-training-for","title":"CATfOOD: Counterfactual Augmented Training for Improving Out-of-Domain Performance and Calibration","date":"2023-09-14","arxiv_id":"2309.07822","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catfood-counterfactual-augmented-training-for#ran","syntology_url":"https://syntology.ai/paper/2309.07822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07822"}},"official":{"repos":["ukplab/catfood"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code-style-in-context-learning-for-knowledge","slug":"code-style-in-context-learning-for-knowledge","title":"Code-Style In-Context Learning for Knowledge-Based Question Answering","date":"2023-09-09","arxiv_id":"2309.04695","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-style-in-context-learning-for-knowledge#ran","syntology_url":"https://syntology.ai/paper/2309.04695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04695"}},"official":{"repos":["arthurizijar/kb-coder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-tuning-large-language-models-with","slug":"knowledge-tuning-large-language-models-with","title":"Knowledge-tuning Large Language Models with Structured Medical Knowledge Bases for Reliable Response Generation in Chinese","date":"2023-09-08","arxiv_id":"2309.04175","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-tuning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2309.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04175"}},"official":null}},{"url":"/paper/aligning-large-language-models-for-clinical","slug":"aligning-large-language-models-for-clinical","title":"Aligning Large Language Models for Clinical Tasks","date":"2023-09-06","arxiv_id":"2309.02884","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-large-language-models-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2309.02884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02884"}},"official":{"repos":["ssm123ssm/medGPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/augmenting-black-box-llms-with-medical","slug":"augmenting-black-box-llms-with-medical","title":"Augmenting Black-box LLMs with Medical Textbooks for Biomedical Question Answering (Published in Findings of EMNLP 2024)","date":"2023-09-05","arxiv_id":"2309.02233","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/augmenting-black-box-llms-with-medical#ran","syntology_url":"https://syntology.ai/paper/2309.02233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02233"}},"official":{"repos":["TIGER-AI-Lab/LLM-AMT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-i-trust-your-answer-visually-grounded","slug":"can-i-trust-your-answer-visually-grounded","title":"Can I Trust Your Answer? Visually Grounded Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-i-trust-your-answer-visually-grounded#ran","syntology_url":"https://syntology.ai/paper/2309.01327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01327"}},"official":{"repos":["doc-doc/next-gqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cruise-screening-living-literature-reviews","slug":"cruise-screening-living-literature-reviews","title":"CRUISE-Screening: Living Literature Reviews Toolbox","date":"2023-09-04","arxiv_id":"2309.01684","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cruise-screening-living-literature-reviews#ran","syntology_url":"https://syntology.ai/paper/2309.01684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01684"}},"official":{"repos":["projectdossier/cruise-screening"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/point-bind-point-llm-aligning-point-cloud","slug":"point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","arxiv_id":"2309.00615","repositories_listed":5,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/point-bind-point-llm-aligning-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2309.00615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00615"}},"official":{"repos":["ziyuguo99/point-bind_point-llm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bridging-the-kb-text-gap-leveraging","slug":"bridging-the-kb-text-gap-leveraging","title":"Bridging the KB-Text Gap: Leveraging Structured Knowledge-aware Pre-training for KBQA","date":"2023-08-28","arxiv_id":"2308.14436","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bridging-the-kb-text-gap-leveraging#ran","syntology_url":"https://syntology.ai/paper/2308.14436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14436"}},"official":{"repos":["dongguanting/skp-for-kbqa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-driven-cot-exploring-faithful","slug":"knowledge-driven-cot-exploring-faithful","title":"Knowledge-Driven CoT: Exploring Faithful Reasoning in LLMs for Knowledge-intensive Question Answering","date":"2023-08-25","arxiv_id":"2308.13259","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-driven-cot-exploring-faithful#ran","syntology_url":"https://syntology.ai/paper/2308.13259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13259"}},"official":{"repos":["adelwang/kd-cot"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/calm-a-multi-task-benchmark-for-comprehensive","slug":"calm-a-multi-task-benchmark-for-comprehensive","title":"CALM : A Multi-task Benchmark for Comprehensive Assessment of Language Model Bias","date":"2023-08-24","arxiv_id":"2308.12539","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calm-a-multi-task-benchmark-for-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2308.12539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12539"}},"official":{"repos":["vipulgupta1011/calm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flexkbqa-a-flexible-llm-powered-framework-for","slug":"flexkbqa-a-flexible-llm-powered-framework-for","title":"FlexKBQA: A Flexible LLM-Powered Framework for Few-Shot Knowledge Base Question Answering","date":"2023-08-23","arxiv_id":"2308.12060","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flexkbqa-a-flexible-llm-powered-framework-for#ran","syntology_url":"https://syntology.ai/paper/2308.12060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12060"}},"official":{"repos":["leezythu/flexkbqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instructiongpt-4-a-200-instruction-paradigm","slug":"instructiongpt-4-a-200-instruction-paradigm","title":"InstructionGPT-4: A 200-Instruction Paradigm for Fine-Tuning MiniGPT-4","date":"2023-08-23","arxiv_id":"2308.12067","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instructiongpt-4-a-200-instruction-paradigm#ran","syntology_url":"https://syntology.ai/paper/2308.12067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12067"}},"official":{"repos":["waltonfuture/InstructionGPT-4"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-graph-prompting-for-multi-document","slug":"knowledge-graph-prompting-for-multi-document","title":"Knowledge Graph Prompting for Multi-Document Question Answering","date":"2023-08-22","arxiv_id":"2308.11730","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-graph-prompting-for-multi-document#ran","syntology_url":"https://syntology.ai/paper/2308.11730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11730"}},"official":{"repos":["yuwvandy/kg-llm-mdqa"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-baselines-for-interactive-video","slug":"simple-baselines-for-interactive-video","title":"Simple Baselines for Interactive Video Retrieval with Questions and Answers","date":"2023-08-21","arxiv_id":"2308.10402","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/simple-baselines-for-interactive-video#ran","syntology_url":"https://syntology.ai/paper/2308.10402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10402"}},"official":{"repos":["kevinliang888/ivr-qa-baselines"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ralle-a-framework-for-developing-and","slug":"ralle-a-framework-for-developing-and","title":"RaLLe: A Framework for Developing and Evaluating Retrieval-Augmented Large Language Models","date":"2023-08-21","arxiv_id":"2308.10633","repositories_listed":1,"syntology":{"n":15,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/ralle-a-framework-for-developing-and#ran","syntology_url":"https://syntology.ai/paper/2308.10633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10633"}},"official":{"repos":["yhoshi3/ralle"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gameeval-evaluating-llms-on-conversational","slug":"gameeval-evaluating-llms-on-conversational","title":"GameEval: Evaluating LLMs on Conversational Games","date":"2023-08-19","arxiv_id":"2308.10032","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gameeval-evaluating-llms-on-conversational#ran","syntology_url":"https://syntology.ai/paper/2308.10032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10032"}},"official":{"repos":["gameeval/gameeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/beam-retrieval-general-end-to-end-retrieval","slug":"beam-retrieval-general-end-to-end-retrieval","title":"End-to-End Beam Retrieval for Multi-Hop Question Answering","date":"2023-08-17","arxiv_id":"2308.08973","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beam-retrieval-general-end-to-end-retrieval#ran","syntology_url":"https://syntology.ai/paper/2308.08973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08973"}},"official":{"repos":["Alab-NII/2wikimultihop","canghongjian/beam_retriever"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-nlx-unifying-textual-explanations-for","slug":"uni-nlx-unifying-textual-explanations-for","title":"Uni-NLX: Unifying Textual Explanations for Vision and Vision-Language Tasks","date":"2023-08-17","arxiv_id":"2308.09033","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-nlx-unifying-textual-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2308.09033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09033"}},"official":{"repos":["fawazsammani/uni-nlx","fawazsammani/nlxgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoschema-a-diagnostic-benchmark-for-very-1#ran","syntology_url":"https://syntology.ai/paper/2308.09126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09126"}},"official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mindmap-knowledge-graph-prompting-sparks","slug":"mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mindmap-knowledge-graph-prompting-sparks#ran","syntology_url":"https://syntology.ai/paper/2308.09729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09729"}},"official":{"repos":["wyl-willing/MindMap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pro-cap-leveraging-a-frozen-vision-language","slug":"pro-cap-leveraging-a-frozen-vision-language","title":"Pro-Cap: Leveraging a Frozen Vision-Language Model for Hateful Meme Detection","date":"2023-08-16","arxiv_id":"2308.08088","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pro-cap-leveraging-a-frozen-vision-language#ran","syntology_url":"https://syntology.ai/paper/2308.08088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08088"}},"official":{"repos":["social-ai-studio/pro-cap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/autogen-enabling-next-gen-llm-applications","slug":"autogen-enabling-next-gen-llm-applications","title":"AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation","date":"2023-08-16","arxiv_id":"2308.08155","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/autogen-enabling-next-gen-llm-applications#ran","syntology_url":"https://syntology.ai/paper/2308.08155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08155"}},"official":null}},{"url":"/paper/tech-text-guided-reconstruction-of-lifelike","slug":"tech-text-guided-reconstruction-of-lifelike","title":"TeCH: Text-guided Reconstruction of Lifelike Clothed Humans","date":"2023-08-16","arxiv_id":"2308.08545","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tech-text-guided-reconstruction-of-lifelike#ran","syntology_url":"https://syntology.ai/paper/2308.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08545"}},"official":{"repos":["huangyangyi/tech"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/foundation-model-is-efficient-multimodal-1","slug":"foundation-model-is-efficient-multimodal-1","title":"Foundation Model is Efficient Multimodal Multitask Model Selector","date":"2023-08-11","arxiv_id":"2308.06262","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foundation-model-is-efficient-multimodal-1#ran","syntology_url":"https://syntology.ai/paper/2308.06262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06262"}},"official":{"repos":["opengvlab/multitask-model-selector"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/towards-an-ai-to-win-ghana-s-national-science","slug":"towards-an-ai-to-win-ghana-s-national-science","title":"Towards an AI to Win Ghana's National Science and Maths Quiz","date":"2023-08-08","arxiv_id":"2308.04333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-an-ai-to-win-ghana-s-national-science#ran","syntology_url":"https://syntology.ai/paper/2308.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04333"}},"official":{"repos":["nsmq-ai/nsmqai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-vista-pre-trained-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2308.04352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04352"}},"official":null}},{"url":"/paper/paniniqa-enhancing-patient-education-through","slug":"paniniqa-enhancing-patient-education-through","title":"PaniniQA: Enhancing Patient Education Through Interactive Question Answering","date":"2023-08-07","arxiv_id":"2308.03253","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/paniniqa-enhancing-patient-education-through#ran","syntology_url":"https://syntology.ai/paper/2308.03253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03253"}},"official":{"repos":["pengshancai/paniniqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2308.03349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03349"}},"official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/educhat-a-large-scale-language-model-based","slug":"educhat-a-large-scale-language-model-based","title":"EduChat: A Large-Scale Language Model-based Chatbot System for Intelligent Education","date":"2023-08-05","arxiv_id":"2308.02773","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/educhat-a-large-scale-language-model-based#ran","syntology_url":"https://syntology.ai/paper/2308.02773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02773"}},"official":{"repos":["icalk-nlp/educhat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generalist-foundation-model-for","slug":"towards-generalist-foundation-model-for","title":"Towards Generalist Foundation Model for Radiology by Leveraging Web-scale 2D&3D Medical Data","date":"2023-08-04","arxiv_id":"2308.02463","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generalist-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2308.02463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02463"}},"official":{"repos":["chaoyi-wu/radfm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conceptlab-creative-generation-using","slug":"conceptlab-creative-generation-using","title":"ConceptLab: Creative Concept Generation using VLM-Guided Diffusion Prior Constraints","date":"2023-08-03","arxiv_id":"2308.02669","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conceptlab-creative-generation-using#ran","syntology_url":"https://syntology.ai/paper/2308.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02669"}},"official":{"repos":["kfirgoldberg/ConceptLab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selfcheck-using-llms-to-zero-shot-check-their","slug":"selfcheck-using-llms-to-zero-shot-check-their","title":"SelfCheck: Using LLMs to Zero-Shot Check Their Own Step-by-Step Reasoning","date":"2023-08-01","arxiv_id":"2308.00436","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/selfcheck-using-llms-to-zero-shot-check-their#ran","syntology_url":"https://syntology.ai/paper/2308.00436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00436"}},"official":{"repos":["ningmiao/selfcheck"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-correctness-and-faithfulness-of","slug":"evaluating-correctness-and-faithfulness-of","title":"Evaluating Correctness and Faithfulness of Instruction-Following Models for Question Answering","date":"2023-07-31","arxiv_id":"2307.16877","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-correctness-and-faithfulness-of#ran","syntology_url":"https://syntology.ai/paper/2307.16877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16877"}},"official":{"repos":["mcgill-nlp/instruct-qa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-llm-injecting-the-3d-world-into-large","slug":"3d-llm-injecting-the-3d-world-into-large","title":"3D-LLM: Injecting the 3D World into Large Language Models","date":"2023-07-24","arxiv_id":"2307.12981","repositories_listed":5,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/3d-llm-injecting-the-3d-world-into-large#ran","syntology_url":"https://syntology.ai/paper/2307.12981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12981"}},"official":null}},{"url":"/paper/expert-knowledge-aware-image-difference-graph","slug":"expert-knowledge-aware-image-difference-graph","title":"Expert Knowledge-Aware Image Difference Graph Representation Learning for Difference-Aware Medical Visual Question Answering","date":"2023-07-22","arxiv_id":"2307.11986","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-knowledge-aware-image-difference-graph#ran","syntology_url":"https://syntology.ai/paper/2307.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11986"}},"official":{"repos":["holipori/mimic-diff-vqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-spatio-temporal-rationales-for","slug":"discovering-spatio-temporal-rationales-for","title":"Discovering Spatio-Temporal Rationales for Video Question Answering","date":"2023-07-22","arxiv_id":"2307.12058","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discovering-spatio-temporal-rationales-for#ran","syntology_url":"https://syntology.ai/paper/2307.12058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12058"}},"official":{"repos":["yl3800/transtr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-the-factual-knowledge-boundary","slug":"investigating-the-factual-knowledge-boundary","title":"Investigating the Factual Knowledge Boundary of Large Language Models with Retrieval Augmentation","date":"2023-07-20","arxiv_id":"2307.11019","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-the-factual-knowledge-boundary#ran","syntology_url":"https://syntology.ai/paper/2307.11019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11019"}},"official":{"repos":["rucaibox/llm-knowledge-boundary"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-2-open-foundation-and-fine-tuned-chat","slug":"llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","repositories_listed":19,"syntology":{"n":52,"n_ran":33,"n_constructed":7,"n_ran_checked":22,"n_instrument":11,"n_unverified":19,"n_honours":1,"n_violates":1,"n_no_contract":20,"n_pointer_only":20,"phrase":"33 ran (of which 7 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 1 violated, 20 with no contract checked; 11 where Syntology's instrument failed) · 19 unverified","sample_list":"/paper/llama-2-open-foundation-and-fine-tuned-chat#ran","syntology_url":"https://syntology.ai/paper/2307.09288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09288"}},"official":{"repos":["facebookresearch/llama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/coupling-large-language-models-with-logic","slug":"coupling-large-language-models-with-logic","title":"Coupling Large Language Models with Logic Programming for Robust and General Reasoning from Text","date":"2023-07-15","arxiv_id":"2307.07696","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coupling-large-language-models-with-logic#ran","syntology_url":"https://syntology.ai/paper/2307.07696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07696"}},"official":{"repos":["azreasoners/llm-asp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/think-on-graph-deep-and-responsible-reasoning","slug":"think-on-graph-deep-and-responsible-reasoning","title":"Think-on-Graph: Deep and Responsible Reasoning of Large Language Model on Knowledge Graph","date":"2023-07-15","arxiv_id":"2307.07697","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/think-on-graph-deep-and-responsible-reasoning#ran","syntology_url":"https://syntology.ai/paper/2307.07697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07697"}},"official":{"repos":["gasolsun36/tog","idea-finai/tog"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["community","official"]}}},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/event-extraction-as-question-generation-and","slug":"event-extraction-as-question-generation-and","title":"Event Extraction as Question Generation and Answering","date":"2023-07-10","arxiv_id":"2307.05567","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/event-extraction-as-question-generation-and#ran","syntology_url":"https://syntology.ai/paper/2307.05567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05567"}},"official":{"repos":["dataminr-ai/event-extraction-as-question-generation-and-answering"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prd-peer-rank-and-discussion-improve-large","slug":"prd-peer-rank-and-discussion-improve-large","title":"PRD: Peer Rank and Discussion Improve Large Language Model based Evaluations","date":"2023-07-06","arxiv_id":"2307.02762","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prd-peer-rank-and-discussion-improve-large#ran","syntology_url":"https://syntology.ai/paper/2307.02762","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02762"}},"official":{"repos":["bcdnlp/prd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-retrieval-augmented-large-language","slug":"improving-retrieval-augmented-large-language","title":"Improving Retrieval-Augmented Large Language Models via Data Importance Learning","date":"2023-07-06","arxiv_id":"2307.03027","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-retrieval-augmented-large-language#ran","syntology_url":"https://syntology.ai/paper/2307.03027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03027"}},"official":{"repos":["amsterdata/ragbooster"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lost-in-the-middle-how-language-models-use","slug":"lost-in-the-middle-how-language-models-use","title":"Lost in the Middle: How Language Models Use Long Contexts","date":"2023-07-06","arxiv_id":"2307.03172","repositories_listed":6,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lost-in-the-middle-how-language-models-use#ran","syntology_url":"https://syntology.ai/paper/2307.03172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03172"}},"official":{"repos":["nelson-liu/lost-in-the-middle"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/won-t-get-fooled-again-answering-questions","slug":"won-t-get-fooled-again-answering-questions","title":"Won't Get Fooled Again: Answering Questions with False Premises","date":"2023-07-05","arxiv_id":"2307.02394","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/won-t-get-fooled-again-answering-questions#ran","syntology_url":"https://syntology.ai/paper/2307.02394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02394"}},"official":{"repos":["thunlp/falseqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/shifting-attention-to-relevance-towards-the","slug":"shifting-attention-to-relevance-towards-the","title":"Shifting Attention to Relevance: Towards the Predictive Uncertainty Quantification of Free-Form Large Language Models","date":"2023-07-03","arxiv_id":"2307.01379","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shifting-attention-to-relevance-towards-the#ran","syntology_url":"https://syntology.ai/paper/2307.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.01379"}},"official":{"repos":["jinhaoduan/sar","jinhaoduan/shifting-attention-to-relevance"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-mining-from-a-pool-of-images-towards","slug":"answer-mining-from-a-pool-of-images-towards","title":"Answer Mining from a Pool of Images: Towards Retrieval-Based Visual Question Answering","date":"2023-06-29","arxiv_id":"2306.16713","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-mining-from-a-pool-of-images-towards#ran","syntology_url":"https://syntology.ai/paper/2306.16713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16713"}},"official":{"repos":["Abhiram4572/mi_bart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/funqa-towards-surprising-video-comprehension","slug":"funqa-towards-surprising-video-comprehension","title":"FunQA: Towards Surprising Video Comprehension","date":"2023-06-26","arxiv_id":"2306.14899","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/funqa-towards-surprising-video-comprehension#ran","syntology_url":"https://syntology.ai/paper/2306.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14899"}},"official":{"repos":["jingkang50/funqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robut-a-systematic-study-of-table-qa","slug":"robut-a-systematic-study-of-table-qa","title":"RobuT: A Systematic Study of Table QA Robustness Against Human-Annotated Adversarial Perturbations","date":"2023-06-25","arxiv_id":"2306.14321","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robut-a-systematic-study-of-table-qa#ran","syntology_url":"https://syntology.ai/paper/2306.14321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14321"}},"official":{"repos":["yilunzhao/robut"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toolqa-a-dataset-for-llm-question-answering-1","slug":"toolqa-a-dataset-for-llm-question-answering-1","title":"ToolQA: A Dataset for LLM Question Answering with External Tools","date":"2023-06-23","arxiv_id":"2306.13304","repositories_listed":2,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toolqa-a-dataset-for-llm-question-answering-1#ran","syntology_url":"https://syntology.ai/paper/2306.13304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13304"}},"official":{"repos":["night-chen/toolqa"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conformal-language-modeling","slug":"conformal-language-modeling","title":"Conformal Language Modeling","date":"2023-06-16","arxiv_id":"2306.10193","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/conformal-language-modeling#ran","syntology_url":"https://syntology.ai/paper/2306.10193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10193"}},"official":{"repos":["varal7/conformal-language-modeling"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/scalable-neural-probabilistic-answer-set","slug":"scalable-neural-probabilistic-answer-set","title":"Scalable Neural-Probabilistic Answer Set Programming","date":"2023-06-14","arxiv_id":"2306.08397","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-neural-probabilistic-answer-set#ran","syntology_url":"https://syntology.ai/paper/2306.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08397"}},"official":{"repos":["ml-research/slash"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pixiu-a-large-language-model-instruction-data","slug":"pixiu-a-large-language-model-instruction-data","title":"PIXIU: A Large Language Model, Instruction Data and Evaluation Benchmark for Finance","date":"2023-06-08","arxiv_id":"2306.05443","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pixiu-a-large-language-model-instruction-data#ran","syntology_url":"https://syntology.ai/paper/2306.05443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05443"}},"official":{"repos":["chancefocus/pixiu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/prompt-space-optimizing-few-shot-reasoning","slug":"prompt-space-optimizing-few-shot-reasoning","title":"Prompt Space Optimizing Few-shot Reasoning Success with Large Language Models","date":"2023-06-06","arxiv_id":"2306.03799","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-space-optimizing-few-shot-reasoning#ran","syntology_url":"https://syntology.ai/paper/2306.03799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03799"}},"official":{"repos":["youblei/prompt-space"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-on-cmexam#ran","syntology_url":"https://syntology.ai/paper/2306.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03030"}},"official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/i-3-retriever-incorporating-implicit","slug":"i-3-retriever-incorporating-implicit","title":"I^3 Retriever: Incorporating Implicit Interaction in Pre-trained Language Models for Passage Retrieval","date":"2023-06-04","arxiv_id":"2306.02371","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/i-3-retriever-incorporating-implicit#ran","syntology_url":"https://syntology.ai/paper/2306.02371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02371"}},"official":{"repos":["deriq-qian-dong/iii-retriever"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fine-grained-human-feedback-gives-better","slug":"fine-grained-human-feedback-gives-better","title":"Fine-Grained Human Feedback Gives Better Rewards for Language Model Training","date":"2023-06-02","arxiv_id":"2306.01693","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fine-grained-human-feedback-gives-better#ran","syntology_url":"https://syntology.ai/paper/2306.01693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01693"}},"official":null}},{"url":"/paper/make-pre-trained-model-reversible-from-1","slug":"make-pre-trained-model-reversible-from-1","title":"Make Pre-trained Model Reversible: From Parameter to Memory Efficient Fine-Tuning","date":"2023-06-01","arxiv_id":"2306.00477","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/make-pre-trained-model-reversible-from-1#ran","syntology_url":"https://syntology.ai/paper/2306.00477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00477"}},"official":{"repos":["baohaoliao/mefts"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/layout-and-task-aware-instruction-prompt-for","slug":"layout-and-task-aware-instruction-prompt-for","title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","date":"2023-06-01","arxiv_id":"2306.00526","repositories_listed":3,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/layout-and-task-aware-instruction-prompt-for#ran","syntology_url":"https://syntology.ai/paper/2306.00526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00526"}},"official":{"repos":["wenjinw/latin-prompt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/generating-with-confidence-uncertainty","slug":"generating-with-confidence-uncertainty","title":"Generating with Confidence: Uncertainty Quantification for Black-box Large Language Models","date":"2023-05-30","arxiv_id":"2305.19187","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-with-confidence-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2305.19187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19187"}},"official":{"repos":["zlin7/uq-nlg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/concise-answers-to-complex-questions","slug":"concise-answers-to-complex-questions","title":"Concise Answers to Complex Questions: Summarization of Long-form Answers","date":"2023-05-30","arxiv_id":"2305.19271","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/concise-answers-to-complex-questions#ran","syntology_url":"https://syntology.ai/paper/2305.19271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19271"}},"official":{"repos":["acpotluri/lfqa_summary"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-object-detection-with-multimodal","slug":"contextual-object-detection-with-multimodal","title":"Contextual Object Detection with Multimodal Large Language Models","date":"2023-05-29","arxiv_id":"2305.18279","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-object-detection-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2305.18279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18279"}},"official":{"repos":["yuhangzang/contextdet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pali-x-on-scaling-up-a-multilingual-vision","slug":"pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","arxiv_id":"2305.18565","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":4,"n_no_contract":1,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 4 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-x-on-scaling-up-a-multilingual-vision#ran","syntology_url":"https://syntology.ai/paper/2305.18565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18565"}},"official":null}},{"url":"/paper/conformal-prediction-with-large-language","slug":"conformal-prediction-with-large-language","title":"Conformal Prediction with Large Language Models for Multi-Choice Question Answering","date":"2023-05-28","arxiv_id":"2305.18404","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/conformal-prediction-with-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.18404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18404"}},"official":{"repos":["bhaweshiitk/conformalllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/biomedgpt-a-unified-and-generalist-biomedical#ran","syntology_url":"https://syntology.ai/paper/2305.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17100"}},"official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-contradictory-hallucinations-of-large","slug":"self-contradictory-hallucinations-of-large","title":"Self-contradictory Hallucinations of Large Language Models: Evaluation, Detection and Mitigation","date":"2023-05-25","arxiv_id":"2305.15852","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-contradictory-hallucinations-of-large#ran","syntology_url":"https://syntology.ai/paper/2305.15852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15852"}},"official":{"repos":["eth-sri/chatprotect"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/unichart-a-universal-vision-language","slug":"unichart-a-universal-vision-language","title":"UniChart: A Universal Vision-language Pretrained Model for Chart Comprehension and Reasoning","date":"2023-05-24","arxiv_id":"2305.14761","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/unichart-a-universal-vision-language#ran","syntology_url":"https://syntology.ai/paper/2305.14761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14761"}},"official":{"repos":["vis-nlp/unichart"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beamsearchqa-large-language-models-are-strong","slug":"beamsearchqa-large-language-models-are-strong","title":"Allies: Prompting Large Language Model with Beam Search","date":"2023-05-24","arxiv_id":"2305.14766","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/beamsearchqa-large-language-models-are-strong#ran","syntology_url":"https://syntology.ai/paper/2305.14766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14766"}},"official":{"repos":["microsoft/simxns"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/using-natural-language-explanations-to","slug":"using-natural-language-explanations-to","title":"Using Natural Language Explanations to Rescale Human Judgments","date":"2023-05-24","arxiv_id":"2305.14770","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/using-natural-language-explanations-to#ran","syntology_url":"https://syntology.ai/paper/2305.14770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14770"}},"official":{"repos":["manyawadhwa/explanation_based_rescaling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nuscenes-qa-a-multi-modal-visual-question","slug":"nuscenes-qa-a-multi-modal-visual-question","title":"NuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario","date":"2023-05-24","arxiv_id":"2305.14836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuscenes-qa-a-multi-modal-visual-question#ran","syntology_url":"https://syntology.ai/paper/2305.14836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14836"}},"official":{"repos":["qiantianwen/nuscenes-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-art-of-socratic-questioning-zero-shot","slug":"the-art-of-socratic-questioning-zero-shot","title":"The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14999","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-art-of-socratic-questioning-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2305.14999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14999"}},"official":{"repos":["vt-nlp/socratic-questioning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-learning-online-adaptation-of-language","slug":"meta-learning-online-adaptation-of-language","title":"Meta-Learning Online Adaptation of Language Models","date":"2023-05-24","arxiv_id":"2305.15076","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-learning-online-adaptation-of-language#ran","syntology_url":"https://syntology.ai/paper/2305.15076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15076"}},"official":{"repos":["nathanhu0/CaMeLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/csts-conditional-semantic-textual-similarity","slug":"csts-conditional-semantic-textual-similarity","title":"C-STS: Conditional Semantic Textual Similarity","date":"2023-05-24","arxiv_id":"2305.15093","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/csts-conditional-semantic-textual-similarity#ran","syntology_url":"https://syntology.ai/paper/2305.15093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15093"}},"official":{"repos":["princeton-nlp/c-sts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/peek-across-improving-multi-document-modeling","slug":"peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","arxiv_id":"2305.15387","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/peek-across-improving-multi-document-modeling#ran","syntology_url":"https://syntology.ai/paper/2305.15387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15387"}},"official":{"repos":["aviclu/peekacross"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-risk-of-misinformation-pollution-with","slug":"on-the-risk-of-misinformation-pollution-with","title":"On the Risk of Misinformation Pollution with Large Language Models","date":"2023-05-23","arxiv_id":"2305.13661","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-risk-of-misinformation-pollution-with#ran","syntology_url":"https://syntology.ai/paper/2305.13661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13661"}},"official":{"repos":["MexicanLemonade/LLM-Misinfo-QA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-of-knowledge-exploring-known","slug":"knowledge-of-knowledge-exploring-known","title":"Knowledge of Knowledge: Exploring Known-Unknowns Uncertainty with Large Language Models","date":"2023-05-23","arxiv_id":"2305.13712","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowledge-of-knowledge-exploring-known#ran","syntology_url":"https://syntology.ai/paper/2305.13712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13712"}},"official":{"repos":["amayuelas/knowledge-of-knowledge"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multitabqa-generating-tabular-answers-for","slug":"multitabqa-generating-tabular-answers-for","title":"MultiTabQA: Generating Tabular Answers for Multi-Table Question Answering","date":"2023-05-22","arxiv_id":"2305.12820","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multitabqa-generating-tabular-answers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12820"}},"official":{"repos":["kolk/multitabqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/averitec-a-dataset-for-real-world-claim-1","slug":"averitec-a-dataset-for-real-world-claim-1","title":"AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web","date":"2023-05-22","arxiv_id":"2305.13117","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/averitec-a-dataset-for-real-world-claim-1#ran","syntology_url":"https://syntology.ai/paper/2305.13117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13117"}},"official":{"repos":["michschli/averitec"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pruning-pre-trained-language-models-with","slug":"pruning-pre-trained-language-models-with","title":"Pruning Pre-trained Language Models with Principled Importance and Self-regularization","date":"2023-05-21","arxiv_id":"2305.12394","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pruning-pre-trained-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2305.12394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12394"}},"official":{"repos":["drsy/pins"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/theoremqa-a-theorem-driven-question-answering","slug":"theoremqa-a-theorem-driven-question-answering","title":"TheoremQA: A Theorem-driven Question Answering dataset","date":"2023-05-21","arxiv_id":"2305.12524","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/theoremqa-a-theorem-driven-question-answering#ran","syntology_url":"https://syntology.ai/paper/2305.12524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12524"}},"official":{"repos":["wenhuchen/theoremqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-makes-for-good-visual-tokenizers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12223"}},"official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-itself-can-read-and-generate-cxr-images","slug":"llm-itself-can-read-and-generate-cxr-images","title":"LLM-CXR: Instruction-Finetuned LLM for CXR Image Understanding and Generation","date":"2023-05-19","arxiv_id":"2305.11490","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/llm-itself-can-read-and-generate-cxr-images#ran","syntology_url":"https://syntology.ai/paper/2305.11490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11490"}},"official":{"repos":["hyn2028/llm-cxr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/toolkengpt-augmenting-frozen-language-models-1","slug":"toolkengpt-augmenting-frozen-language-models-1","title":"ToolkenGPT: Augmenting Frozen Language Models with Massive Tools via Tool Embeddings","date":"2023-05-19","arxiv_id":"2305.11554","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/toolkengpt-augmenting-frozen-language-models-1#ran","syntology_url":"https://syntology.ai/paper/2305.11554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11554"}},"official":{"repos":["Ber666/ToolkenGPT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/surgical-vqla-transformer-with-gated-vision","slug":"surgical-vqla-transformer-with-gated-vision","title":"Surgical-VQLA: Transformer with Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-05-19","arxiv_id":"2305.11692","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgical-vqla-transformer-with-gated-vision#ran","syntology_url":"https://syntology.ai/paper/2305.11692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11692"}},"official":{"repos":["longbai1006/surgical-vqla"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critic-large-language-models-can-self-correct","slug":"critic-large-language-models-can-self-correct","title":"CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing","date":"2023-05-19","arxiv_id":"2305.11738","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/critic-large-language-models-can-self-correct#ran","syntology_url":"https://syntology.ai/paper/2305.11738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11738"}},"official":{"repos":["microsoft/ProphetNet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"725f69a5bc5fa272006658395ec10301cf96e69b7c203d2ddd44ef8b56559fc3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}