{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/6","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":12,"rows_per_page":100,"rows":[501,600],"of":1135,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following","prev":"/task/instruction-following/papers/5","next":"/task/instruction-following/papers/7","papers":[{"url":"/paper/understanding-the-effects-of-rlhf-on-llm","slug":"understanding-the-effects-of-rlhf-on-llm","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","date":"2023-10-10","arxiv_id":"2310.06452","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/understanding-the-effects-of-rlhf-on-llm#ran","syntology_url":"https://syntology.ai/paper/2310.06452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06452"}},"official":{"repos":["facebookresearch/rlfh-gen-div"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/back-to-the-future-towards-explainable","slug":"back-to-the-future-towards-explainable","title":"Back to the Future: Towards Explainable Temporal Reasoning with Large Language Models","date":"2023-10-02","arxiv_id":"2310.01074","repositories_listed":1,"syntology":null},{"url":"/paper/fool-your-vision-and-language-model-with","slug":"fool-your-vision-and-language-model-with","title":"Fool Your (Vision and) Language Model With Embarrassingly Simple Permutations","date":"2023-10-02","arxiv_id":"2310.01651","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fool-your-vision-and-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2310.01651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01651"}},"official":{"repos":["ys-zong/foolyourvllms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/tadis-steering-models-for-deep-thinking-about","slug":"tadis-steering-models-for-deep-thinking-about","title":"PACIT: Unlocking the Power of Examples for Better In-Context Instruction Tuning","date":"2023-10-02","arxiv_id":"2310.00901","repositories_listed":1,"syntology":null},{"url":"/paper/use-your-instinct-instruction-optimization","slug":"use-your-instinct-instruction-optimization","title":"Use Your INSTINCT: INSTruction optimization for LLMs usIng Neural bandits Coupled with Transformers","date":"2023-10-02","arxiv_id":"2310.02905","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/use-your-instinct-instruction-optimization#ran","syntology_url":"https://syntology.ai/paper/2310.02905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02905"}},"official":{"repos":["xqlin98/INSTINCT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-task-performance-evaluating-and","slug":"beyond-task-performance-evaluating-and","title":"Beyond Task Performance: Evaluating and Reducing the Flaws of Large Multimodal Models with In-Context Learning","date":"2023-10-01","arxiv_id":"2310.00647","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-task-performance-evaluating-and#ran","syntology_url":"https://syntology.ai/paper/2310.00647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00647"}},"official":{"repos":["mshukor/EvALign-ICL"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-language-modeling-to-instruction","slug":"from-language-modeling-to-instruction","title":"From Language Modeling to Instruction Following: Understanding the Behavior Shift in LLMs after Instruction Tuning","date":"2023-09-30","arxiv_id":"2310.00492","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-language-modeling-to-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.00492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00492"}},"official":{"repos":["jacksonwuxs/interpret_instruction_tuning_llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/acegpt-localizing-large-language-models-in","slug":"acegpt-localizing-large-language-models-in","title":"AceGPT, Localizing Large Language Models in Arabic","date":"2023-09-21","arxiv_id":"2309.12053","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-embedded-programs-for-hybrid","slug":"natural-language-embedded-programs-for-hybrid","title":"Natural Language Embedded Programs for Hybrid Language Symbolic Reasoning","date":"2023-09-19","arxiv_id":"2309.10814","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/natural-language-embedded-programs-for-hybrid#ran","syntology_url":"https://syntology.ai/paper/2309.10814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10814"}},"official":{"repos":["luohongyin/langcode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/monolingual-or-multilingual-instruction","slug":"monolingual-or-multilingual-instruction","title":"Monolingual or Multilingual Instruction Tuning: Which Makes a Better Alpaca","date":"2023-09-16","arxiv_id":"2309.08958","repositories_listed":1,"syntology":null},{"url":"/paper/textbind-multi-turn-interleaved-multimodal","slug":"textbind-multi-turn-interleaved-multimodal","title":"TextBind: Multi-turn Interleaved Multimodal Instruction-following in the Wild","date":"2023-09-14","arxiv_id":"2309.08637","repositories_listed":1,"syntology":null},{"url":"/paper/tegit-generating-high-quality-instruction","slug":"tegit-generating-high-quality-instruction","title":"DoG-Instruct: Towards Premium Instruction-Tuning Data via Text-Grounded Instruction Wrapping","date":"2023-09-11","arxiv_id":"2309.05447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tegit-generating-high-quality-instruction#ran","syntology_url":"https://syntology.ai/paper/2309.05447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05447"}},"official":{"repos":["bahuia/dog-instruct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-emergent-abilities-in-large-language","slug":"are-emergent-abilities-in-large-language","title":"Are Emergent Abilities in Large Language Models just In-Context Learning?","date":"2023-09-04","arxiv_id":"2309.01809","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-emergent-abilities-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2309.01809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01809"}},"official":{"repos":["ukplab/on-emergence"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-conditioned-change-point-detection","slug":"language-conditioned-change-point-detection","title":"Language-Conditioned Change-point Detection to Identify Sub-Tasks in Robotics Domains","date":"2023-09-01","arxiv_id":"2309.00743","repositories_listed":1,"syntology":null},{"url":"/paper/sparkles-unlocking-chats-across-multiple","slug":"sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","arxiv_id":"2308.16463","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":3,"n_instrument":7,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sparkles-unlocking-chats-across-multiple#ran","syntology_url":"https://syntology.ai/paper/2308.16463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16463"}},"official":{"repos":["hypjudy/sparkles"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llasm-large-language-and-speech-model","slug":"llasm-large-language-and-speech-model","title":"LLaSM: Large Language and Speech Model","date":"2023-08-30","arxiv_id":"2308.15930","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llasm-large-language-and-speech-model#ran","syntology_url":"https://syntology.ai/paper/2308.15930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15930"}},"official":{"repos":["linksoul-ai/llasm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-cross-lingual-abilities-of","slug":"empowering-cross-lingual-abilities-of","title":"Empowering Cross-lingual Abilities of Instruction-tuned Large Language Models by Translation-following demonstrations","date":"2023-08-27","arxiv_id":"2308.14186","repositories_listed":1,"syntology":null},{"url":"/paper/improving-translation-faithfulness-of-large","slug":"improving-translation-faithfulness-of-large","title":"Improving Translation Faithfulness of Large Language Models via Augmenting Instructions","date":"2023-08-24","arxiv_id":"2308.12674","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-translation-faithfulness-of-large#ran","syntology_url":"https://syntology.ai/paper/2308.12674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12674"}},"official":{"repos":["pppa2019/swie_overmiss_llm4mt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-position-matters-in-sequence","slug":"instruction-position-matters-in-sequence","title":"Instruction Position Matters in Sequence Generation with Large Language Models","date":"2023-08-23","arxiv_id":"2308.12097","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instruction-position-matters-in-sequence#ran","syntology_url":"https://syntology.ai/paper/2308.12097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12097"}},"official":{"repos":["adaxry/post-instruction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-level-compositional-reasoning-for","slug":"multi-level-compositional-reasoning-for","title":"Multi-Level Compositional Reasoning for Interactive Instruction Following","date":"2023-08-18","arxiv_id":"2308.09387","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/multi-level-compositional-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2308.09387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09387"}},"official":{"repos":["yonseivnl/mcr-agent"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/context-aware-planning-and-environment-aware","slug":"context-aware-planning-and-environment-aware","title":"Context-Aware Planning and Environment-Aware Memory for Instruction Following Embodied Agents","date":"2023-08-14","arxiv_id":"2308.07241","repositories_listed":1,"syntology":null},{"url":"/paper/ecomgpt-instruction-tuning-large-language","slug":"ecomgpt-instruction-tuning-large-language","title":"EcomGPT: Instruction-tuning Large Language Models with Chain-of-Task Tasks for E-commerce","date":"2023-08-14","arxiv_id":"2308.06966","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ecomgpt-instruction-tuning-large-language#ran","syntology_url":"https://syntology.ai/paper/2308.06966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06966"}},"official":{"repos":["Alibaba-NLP/EcomGPT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/instag-instruction-tagging-for-diversity-and","slug":"instag-instruction-tagging-for-diversity-and","title":"#InsTag: Instruction Tagging for Analyzing Supervised Fine-tuning of Large Language Models","date":"2023-08-14","arxiv_id":"2308.07074","repositories_listed":1,"syntology":null},{"url":"/paper/visit-bench-a-benchmark-for-vision-language","slug":"visit-bench-a-benchmark-for-vision-language","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","date":"2023-08-12","arxiv_id":"2308.06595","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":3,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visit-bench-a-benchmark-for-vision-language#ran","syntology_url":"https://syntology.ai/paper/2308.06595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06595"}},"official":{"repos":["mlfoundations/VisIT-Bench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-vision-language-models-to-follow","slug":"empowering-vision-language-models-to-follow","title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","date":"2023-08-08","arxiv_id":"2308.04152","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/empowering-vision-language-models-to-follow#ran","syntology_url":"https://syntology.ai/paper/2308.04152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04152"}},"official":{"repos":["dcdmllm/cheetah"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/agentbench-evaluating-llms-as-agents","slug":"agentbench-evaluating-llms-as-agents","title":"AgentBench: Evaluating LLMs as Agents","date":"2023-08-07","arxiv_id":"2308.03688","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agentbench-evaluating-llms-as-agents#ran","syntology_url":"https://syntology.ai/paper/2308.03688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03688"}},"official":{"repos":["thudm/agentbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zhongjing-enhancing-the-chinese-medical","slug":"zhongjing-enhancing-the-chinese-medical","title":"Zhongjing: Enhancing the Chinese Medical Capabilities of Large Language Model through Expert Feedback and Real-world Multi-turn Dialogue","date":"2023-08-07","arxiv_id":"2308.03549","repositories_listed":1,"syntology":null},{"url":"/paper/forget-demonstrations-focus-on-learning-from","slug":"forget-demonstrations-focus-on-learning-from","title":"Toward Zero-Shot Instruction Following","date":"2023-08-04","arxiv_id":"2308.03795","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-correctness-and-faithfulness-of","slug":"evaluating-correctness-and-faithfulness-of","title":"Evaluating Correctness and Faithfulness of Instruction-Following Models for Question Answering","date":"2023-07-31","arxiv_id":"2307.16877","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-correctness-and-faithfulness-of#ran","syntology_url":"https://syntology.ai/paper/2307.16877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16877"}},"official":{"repos":["mcgill-nlp/instruct-qa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flask-fine-grained-language-model-evaluation","slug":"flask-fine-grained-language-model-evaluation","title":"FLASK: Fine-grained Language Model Evaluation based on Alignment Skill Sets","date":"2023-07-20","arxiv_id":"2307.10928","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flask-fine-grained-language-model-evaluation#ran","syntology_url":"https://syntology.ai/paper/2307.10928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10928"}},"official":{"repos":["kaistai/flask"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bubogpt-enabling-visual-grounding-in-multi","slug":"bubogpt-enabling-visual-grounding-in-multi","title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","date":"2023-07-17","arxiv_id":"2307.08581","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bubogpt-enabling-visual-grounding-in-multi#ran","syntology_url":"https://syntology.ai/paper/2307.08581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08581"}},"official":null}},{"url":"/paper/do-emergent-abilities-exist-in-quantized","slug":"do-emergent-abilities-exist-in-quantized","title":"Do Emergent Abilities Exist in Quantized Large Language Models: An Empirical Study","date":"2023-07-16","arxiv_id":"2307.08072","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":5,"n_instrument":7,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/do-emergent-abilities-exist-in-quantized#ran","syntology_url":"https://syntology.ai/paper/2307.08072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08072"}},"official":{"repos":["rucaibox/quantizedempirical"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/opening-up-chatgpt-tracking-openness","slug":"opening-up-chatgpt-tracking-openness","title":"Opening up ChatGPT: Tracking openness, transparency, and accountability in instruction-tuned text generators","date":"2023-07-08","arxiv_id":"2307.05532","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-exploitability-of-instruction-tuning-1","slug":"on-the-exploitability-of-instruction-tuning-1","title":"On the Exploitability of Instruction Tuning","date":"2023-06-28","arxiv_id":"2306.17194","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/on-the-exploitability-of-instruction-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2306.17194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17194"}},"official":{"repos":["azshue/autopoison"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ophglm-training-an-ophthalmology-large","slug":"ophglm-training-an-ophthalmology-large","title":"OphGLM: Training an Ophthalmology Large Language-and-Vision Assistant based on Instructions and Dialogue","date":"2023-06-21","arxiv_id":"2306.12174","repositories_listed":1,"syntology":null},{"url":"/paper/bayling-bridging-cross-lingual-alignment-and","slug":"bayling-bridging-cross-lingual-alignment-and","title":"BayLing: Bridging Cross-lingual Alignment and Instruction Following through Interactive Translation for Large Language Models","date":"2023-06-19","arxiv_id":"2306.10968","repositories_listed":1,"syntology":null},{"url":"/paper/lvlm-ehub-a-comprehensive-evaluation","slug":"lvlm-ehub-a-comprehensive-evaluation","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","date":"2023-06-15","arxiv_id":"2306.09265","repositories_listed":1,"syntology":null},{"url":"/paper/valley-video-assistant-with-large-language","slug":"valley-video-assistant-with-large-language","title":"Valley: Video Assistant with Large Language model Enhanced abilitY","date":"2023-06-12","arxiv_id":"2306.07207","repositories_listed":1,"syntology":null},{"url":"/paper/how-can-recommender-systems-benefit-from","slug":"how-can-recommender-systems-benefit-from","title":"How Can Recommender Systems Benefit from Large Language Models: A Survey","date":"2023-06-09","arxiv_id":"2306.05817","repositories_listed":1,"syntology":null},{"url":"/paper/llava-med-training-a-large-language-and","slug":"llava-med-training-a-large-language-and","title":"LLaVA-Med: Training a Large Language-and-Vision Assistant for Biomedicine in One Day","date":"2023-06-01","arxiv_id":"2306.00890","repositories_listed":1,"syntology":null},{"url":"/paper/steve-1-a-generative-model-for-text-to","slug":"steve-1-a-generative-model-for-text-to","title":"STEVE-1: A Generative Model for Text-to-Behavior in Minecraft","date":"2023-06-01","arxiv_id":"2306.00937","repositories_listed":1,"syntology":null},{"url":"/paper/from-pixels-to-ui-actions-learning-to-follow","slug":"from-pixels-to-ui-actions-learning-to-follow","title":"From Pixels to UI Actions: Learning to Follow Instructions via Graphical User Interfaces","date":"2023-05-31","arxiv_id":"2306.00245","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-pixels-to-ui-actions-learning-to-follow#ran","syntology_url":"https://syntology.ai/paper/2306.00245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00245"}},"official":{"repos":["google-deepmind/pix2act"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt4tools-teaching-large-language-model-to","slug":"gpt4tools-teaching-large-language-model-to","title":"GPT4Tools: Teaching Large Language Model to Use Tools via Self-instruction","date":"2023-05-30","arxiv_id":"2305.18752","repositories_listed":1,"syntology":null},{"url":"/paper/a-reminder-of-its-brittleness-language-reward","slug":"a-reminder-of-its-brittleness-language-reward","title":"A Reminder of its Brittleness: Language Reward Shaping May Hinder Learning for Instruction Following Agents","date":"2023-05-26","arxiv_id":"2305.16621","repositories_listed":1,"syntology":null},{"url":"/paper/pandagpt-one-model-to-instruction-follow-them","slug":"pandagpt-one-model-to-instruction-follow-them","title":"PandaGPT: One Model To Instruction-Follow Them All","date":"2023-05-25","arxiv_id":"2305.16355","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pandagpt-one-model-to-instruction-follow-them#ran","syntology_url":"https://syntology.ai/paper/2305.16355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16355"}},"official":null}},{"url":"/paper/bactrian-x-a-multilingual-replicable","slug":"bactrian-x-a-multilingual-replicable","title":"Bactrian-X: Multilingual Replicable Instruction-Following Models with Low-Rank Adaptation","date":"2023-05-24","arxiv_id":"2305.15011","repositories_listed":1,"syntology":null},{"url":"/paper/pathasst-redefining-pathology-through","slug":"pathasst-redefining-pathology-through","title":"PathAsst: A Generative Foundation AI Assistant Towards Artificial General Intelligence of Pathology","date":"2023-05-24","arxiv_id":"2305.15072","repositories_listed":1,"syntology":null},{"url":"/paper/pivoine-instruction-tuning-for-open-world","slug":"pivoine-instruction-tuning-for-open-world","title":"PIVOINE: Instruction Tuning for Open-world Information Extraction","date":"2023-05-24","arxiv_id":"2305.14898","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pivoine-instruction-tuning-for-open-world#ran","syntology_url":"https://syntology.ai/paper/2305.14898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14898"}},"official":{"repos":["lukeming-tsinghua/instruction-tuning-for-open-world-ie"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/refocusing-is-key-to-transfer-learning","slug":"refocusing-is-key-to-transfer-learning","title":"TOAST: Transfer Learning via Attention Steering","date":"2023-05-24","arxiv_id":"2305.15542","repositories_listed":1,"syntology":null},{"url":"/paper/schema-driven-information-extraction-from","slug":"schema-driven-information-extraction-from","title":"Schema-Driven Information Extraction from Heterogeneous Tables","date":"2023-05-23","arxiv_id":"2305.14336","repositories_listed":1,"syntology":null},{"url":"/paper/lion-adversarial-distillation-of-closed","slug":"lion-adversarial-distillation-of-closed","title":"Lion: Adversarial Distillation of Proprietary Large Language Models","date":"2023-05-22","arxiv_id":"2305.12870","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lion-adversarial-distillation-of-closed#ran","syntology_url":"https://syntology.ai/paper/2305.12870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12870"}},"official":{"repos":["yjiangcm/lion"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-itself-can-read-and-generate-cxr-images","slug":"llm-itself-can-read-and-generate-cxr-images","title":"LLM-CXR: Instruction-Finetuned LLM for CXR Image Understanding and Generation","date":"2023-05-19","arxiv_id":"2305.11490","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/llm-itself-can-read-and-generate-cxr-images#ran","syntology_url":"https://syntology.ai/paper/2305.11490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11490"}},"official":{"repos":["hyn2028/llm-cxr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-instruction-tasks-unlocks-large","slug":"aligning-instruction-tasks-unlocks-large","title":"Aligning Instruction Tasks Unlocks Large Language Models as Zero-Shot Relation Extractors","date":"2023-05-18","arxiv_id":"2305.11159","repositories_listed":1,"syntology":null},{"url":"/paper/m3ke-a-massive-multi-level-multi-subject","slug":"m3ke-a-massive-multi-level-multi-subject","title":"M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models","date":"2023-05-17","arxiv_id":"2305.10263","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-gpt-a-vision-and-language-model","slug":"multimodal-gpt-a-vision-and-language-model","title":"MultiModal-GPT: A Vision and Language Model for Dialogue with Humans","date":"2023-05-08","arxiv_id":"2305.04790","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/multimodal-gpt-a-vision-and-language-model#ran","syntology_url":"https://syntology.ai/paper/2305.04790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04790"}},"official":{"repos":["open-mmlab/multimodal-gpt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/otter-a-multi-modal-model-with-in-context","slug":"otter-a-multi-modal-model-with-in-context","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","date":"2023-05-05","arxiv_id":"2305.03726","repositories_listed":1,"syntology":null},{"url":"/paper/caption-anything-interactive-image","slug":"caption-anything-interactive-image","title":"Caption Anything: Interactive Image Description with Diverse Multimodal Controls","date":"2023-05-04","arxiv_id":"2305.02677","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/caption-anything-interactive-image#ran","syntology_url":"https://syntology.ai/paper/2305.02677","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02677"}},"official":{"repos":["ttengwang/caption-anything"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/panda-llm-training-data-and-evaluation-for","slug":"panda-llm-training-data-and-evaluation-for","title":"Panda LLM: Training Data and Evaluation for Open-Sourced Chinese Instruction-Following Large Language Models","date":"2023-05-04","arxiv_id":"2305.03025","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-infinite-length-input-capacity-for","slug":"unleashing-infinite-length-input-capacity-for","title":"Enhancing Large Language Model with Self-Controlled Memory Framework","date":"2023-04-26","arxiv_id":"2304.13343","repositories_listed":1,"syntology":{"n":14,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/unleashing-infinite-length-input-capacity-for#ran","syntology_url":"https://syntology.ai/paper/2304.13343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13343"}},"official":{"repos":["wbbeyourself/scm4llms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/generation-driven-contrastive-self-training","slug":"generation-driven-contrastive-self-training","title":"Generation-driven Contrastive Self-training for Zero-shot Text Classification with Instruction-following LLM","date":"2023-04-24","arxiv_id":"2304.11872","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-between-full-parameter","slug":"a-comparative-study-between-full-parameter","title":"A Comparative Study between Full-Parameter and LoRA-based Fine-Tuning on Chinese Instruction Data for Instruction Following Large Language Model","date":"2023-04-17","arxiv_id":"2304.08109","repositories_listed":1,"syntology":null},{"url":"/paper/parrot-translating-during-chat-using-large","slug":"parrot-translating-during-chat-using-large","title":"ParroT: Translating during Chat using Large Language Models tuned with Human Translation and Feedback","date":"2023-04-05","arxiv_id":"2304.02426","repositories_listed":1,"syntology":null},{"url":"/paper/is-prompt-all-you-need-no-a-comprehensive-and","slug":"is-prompt-all-you-need-no-a-comprehensive-and","title":"Large Language Model Instruction Following: A Survey of Progresses and Challenges","date":"2023-03-18","arxiv_id":"2303.10475","repositories_listed":1,"syntology":null},{"url":"/paper/lana-a-language-capable-navigator-for","slug":"lana-a-language-capable-navigator-for","title":"Lana: A Language-Capable Navigator for Instruction Following and Generation","date":"2023-03-15","arxiv_id":"2303.08409","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":8,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lana-a-language-capable-navigator-for#ran","syntology_url":"https://syntology.ai/paper/2303.08409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08409"}},"official":{"repos":["wxh1996/lana-vln"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/cb2-collaborative-natural-language","slug":"cb2-collaborative-natural-language","title":"CB2: Collaborative Natural Language Interaction Research Platform","date":"2023-03-14","arxiv_id":"2303.08127","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cb2-collaborative-natural-language#ran","syntology_url":"https://syntology.ai/paper/2303.08127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08127"}},"official":{"repos":["lil-lab/cb2"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-may-pass-the-bar-exam-soon-but-has-a","slug":"chatgpt-may-pass-the-bar-exam-soon-but-has-a","title":"ChatGPT may Pass the Bar Exam soon, but has a Long Way to Go for the LexGLUE benchmark","date":"2023-03-09","arxiv_id":"2304.12202","repositories_listed":1,"syntology":null},{"url":"/paper/alexa-arena-a-user-centric-interactive-1","slug":"alexa-arena-a-user-centric-interactive-1","title":"Alexa Arena: A User-Centric Interactive Platform for Embodied AI","date":"2023-03-02","arxiv_id":"2303.01586","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-clarification-requests-in","slug":"instruction-clarification-requests-in","title":"Instruction Clarification Requests in Multimodal Collaborative Dialogue Games: Tasks, and an Analysis of the CoDraw Dataset","date":"2023-02-28","arxiv_id":"2302.14406","repositories_listed":1,"syntology":null},{"url":"/paper/no-to-the-right-online-language-corrections","slug":"no-to-the-right-online-language-corrections","title":"\"No, to the Right\" -- Online Language Corrections for Robotic Manipulation via Shared Autonomy","date":"2023-01-06","arxiv_id":"2301.02555","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/no-to-the-right-online-language-corrections#ran","syntology_url":"https://syntology.ai/paper/2301.02555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02555"}},"official":{"repos":["stanford-iliad/lilac"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-learning-for-instruction-following","slug":"continual-learning-for-instruction-following","title":"Continual Learning for Instruction Following from Realtime Feedback","date":"2022-12-19","arxiv_id":"2212.09710","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-learning-for-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2212.09710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09710"}},"official":{"repos":["lil-lab/clif_cb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-second-thought-let-s-not-think-step-by","slug":"on-second-thought-let-s-not-think-step-by","title":"On Second Thought, Let's Not Think Step by Step! Bias and Toxicity in Zero-Shot Reasoning","date":"2022-12-15","arxiv_id":"2212.08061","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-second-thought-let-s-not-think-step-by#ran","syntology_url":"https://syntology.ai/paper/2212.08061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08061"}},"official":{"repos":["salt-nlp/chain-of-thought-bias"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-conditioned-reinforcement-learning","slug":"language-conditioned-reinforcement-learning","title":"Language-Conditioned Reinforcement Learning to Solve Misunderstandings with Action Corrections","date":"2022-11-18","arxiv_id":"2211.10168","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-conditioned-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.10168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10168"}},"official":{"repos":["frankroeder/lanro-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-follow-instructions-in-text-based","slug":"learning-to-follow-instructions-in-text-based","title":"Learning to Follow Instructions in Text-Based Games","date":"2022-11-08","arxiv_id":"2211.04591","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-follow-instructions-in-text-based#ran","syntology_url":"https://syntology.ai/paper/2211.04591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.04591"}},"official":{"repos":["mathieutuli/ltl-gata"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-following-agents-with-jointly-pre","slug":"instruction-following-agents-with-jointly-pre","title":"Instruction-Following Agents with Multimodal Transformer","date":"2022-10-24","arxiv_id":"2210.13431","repositories_listed":1,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/instruction-following-agents-with-jointly-pre#ran","syntology_url":"https://syntology.ai/paper/2210.13431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13431"}},"official":{"repos":["lhao499/instructrl"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/danli-deliberative-agent-for-following","slug":"danli-deliberative-agent-for-following","title":"DANLI: Deliberative Agent for Following Natural Language Instructions","date":"2022-10-22","arxiv_id":"2210.12485","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/danli-deliberative-agent-for-following#ran","syntology_url":"https://syntology.ai/paper/2210.12485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12485"}},"official":{"repos":["sled-group/danli"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-copy-the-teacher-data-and-model","slug":"don-t-copy-the-teacher-data-and-model","title":"Don't Copy the Teacher: Data and Model Challenges in Embodied Dialogue","date":"2022-10-10","arxiv_id":"2210.04443","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-of-soft-prompt-enhances-zero-shot","slug":"retrieval-of-soft-prompt-enhances-zero-shot","title":"Efficiently Enhancing Zero-Shot Performance of Instruction Following Model via Retrieval of Soft Prompt","date":"2022-10-06","arxiv_id":"2210.03029","repositories_listed":1,"syntology":null},{"url":"/paper/lm-nav-robotic-navigation-with-large-pre","slug":"lm-nav-robotic-navigation-with-large-pre","title":"LM-Nav: Robotic Navigation with Large Pre-Trained Models of Language, Vision, and Action","date":"2022-07-10","arxiv_id":"2207.04429","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lm-nav-robotic-navigation-with-large-pre#ran","syntology_url":"https://syntology.ai/paper/2207.04429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.04429"}},"official":{"repos":["blazejosinski/lm_nav"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-are-general-purpose","slug":"language-models-are-general-purpose","title":"Language Models are General-Purpose Interfaces","date":"2022-06-13","arxiv_id":"2206.06336","repositories_listed":1,"syntology":null},{"url":"/paper/goalnet-inferring-conjunctive-goal-predicates","slug":"goalnet-inferring-conjunctive-goal-predicates","title":"GoalNet: Inferring Conjunctive Goal Predicates from Human Plan Demonstrations for Robot Instruction Following","date":"2022-05-14","arxiv_id":"2205.07081","repositories_listed":1,"syntology":null},{"url":"/paper/engineering-flexible-machine-learning-systems","slug":"engineering-flexible-machine-learning-systems","title":"Engineering flexible machine learning systems by traversing functionally-invariant paths","date":"2022-04-30","arxiv_id":"2205.00334","repositories_listed":1,"syntology":null},{"url":"/paper/inferring-rewards-from-language-in-context","slug":"inferring-rewards-from-language-in-context","title":"Inferring Rewards from Language in Context","date":"2022-04-05","arxiv_id":"2204.02515","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-cycle-consistent-learning-for","slug":"counterfactual-cycle-consistent-learning-for","title":"Counterfactual Cycle-Consistent Learning for Instruction Following and Generation in Vision-Language Navigation","date":"2022-03-30","arxiv_id":"2203.16586","repositories_listed":1,"syntology":null},{"url":"/paper/compositionality-as-lexical-symmetry","slug":"compositionality-as-lexical-symmetry","title":"Compositionality as Lexical Symmetry","date":"2022-01-30","arxiv_id":"2201.12926","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compositionality-as-lexical-symmetry#ran","syntology_url":"https://syntology.ai/paper/2201.12926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12926"}},"official":{"repos":["ekinakyurek/lexsym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/film-following-instructions-in-language-with-1","slug":"film-following-instructions-in-language-with-1","title":"FILM: Following Instructions in Language with Modular Methods","date":"2021-10-12","arxiv_id":"2110.07342","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":1,"n_ran_checked":2,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":10,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/film-following-instructions-in-language-with-1#ran","syntology_url":"https://syntology.ai/paper/2110.07342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07342"}},"official":{"repos":["soyeonm/film"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/waypoint-models-for-instruction-guided-1","slug":"waypoint-models-for-instruction-guided-1","title":"Waypoint Models for Instruction-guided Navigation in Continuous Environments","date":"2021-10-05","arxiv_id":"2110.02207","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/waypoint-models-for-instruction-guided-1#ran","syntology_url":"https://syntology.ai/paper/2110.02207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02207"}},"official":{"repos":["jacobkrantz/VLN-CE"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-modular-framework-for-long","slug":"hierarchical-modular-framework-for-long","title":"Hierarchical Modular Framework for Long Horizon Instruction Following","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-language-change-in-collaborative","slug":"analysis-of-language-change-in-collaborative","title":"Analysis of Language Change in Collaborative Instruction Following","date":"2021-09-09","arxiv_id":"2109.04452","repositories_listed":1,"syntology":null},{"url":"/paper/lexicon-learning-for-few-shot-sequence","slug":"lexicon-learning-for-few-shot-sequence","title":"Lexicon Learning for Few Shot Sequence Modeling","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/room-and-object-aware-knowledge-reasoning-for","slug":"room-and-object-aware-knowledge-reasoning-for","title":"Room-and-Object Aware Knowledge Reasoning for Remote Embodied Referring Expression","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lexicon-learning-for-few-shot-neural-sequence","slug":"lexicon-learning-for-few-shot-neural-sequence","title":"Lexicon Learning for Few-Shot Neural Sequence Modeling","date":"2021-06-07","arxiv_id":"2106.03993","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lexicon-learning-for-few-shot-neural-sequence#ran","syntology_url":"https://syntology.ai/paper/2106.03993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03993"}},"official":{"repos":["ekinakyurek/lexical"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/look-wide-and-interpret-twice-improving","slug":"look-wide-and-interpret-twice-improving","title":"Look Wide and Interpret Twice: Improving Performance on Interactive Instruction-following Tasks","date":"2021-06-01","arxiv_id":"2106.00596","repositories_listed":1,"syntology":null},{"url":"/paper/a-modular-vision-language-navigation-and","slug":"a-modular-vision-language-navigation-and","title":"A modular vision language navigation and manipulation framework for long horizon compositional tasks in indoor environment","date":"2021-01-19","arxiv_id":"2101.07891","repositories_listed":1,"syntology":null},{"url":"/paper/moca-a-modular-object-centric-approach-for","slug":"moca-a-modular-object-centric-approach-for","title":"Factorizing Perception and Policy for Interactive Instruction Following","date":"2020-12-06","arxiv_id":"2012.03208","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-language-understanding-for-object","slug":"spatial-language-understanding-for-object","title":"Spatial Language Understanding for Object Search in Partially Observed City-scale Environments","date":"2020-12-04","arxiv_id":"2012.02705","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-object-grounding-and-mapping-for","slug":"few-shot-object-grounding-and-mapping-for","title":"Few-shot Object Grounding and Mapping for Natural Language Robot Instruction Following","date":"2020-11-14","arxiv_id":"2011.07384","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/few-shot-object-grounding-and-mapping-for#ran","syntology_url":"https://syntology.ai/paper/2011.07384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07384"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rmm-a-recursive-mental-model-for-dialogue","slug":"rmm-a-recursive-mental-model-for-dialogue","title":"RMM: A Recursive Mental Model for Dialogue Navigation","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-recombine-and-resample-data-for-1","slug":"learning-to-recombine-and-resample-data-for-1","title":"Learning to Recombine and Resample Data for Compositional Generalization","date":"2020-10-08","arxiv_id":"2010.03706","repositories_listed":1,"syntology":null},{"url":"/paper/allenact-a-framework-for-embodied-ai-research","slug":"allenact-a-framework-for-embodied-ai-research","title":"AllenAct: A Framework for Embodied AI Research","date":"2020-08-28","arxiv_id":"2008.12760","repositories_listed":1,"syntology":null},{"url":"/paper/rmm-a-recursive-mental-model-for-dialog","slug":"rmm-a-recursive-mental-model-for-dialog","title":"RMM: A Recursive Mental Model for Dialog Navigation","date":"2020-05-02","arxiv_id":"2005.00728","repositories_listed":1,"syntology":null}],"record_sha256":"ffd4add04ad007c9944b46b5a1f3468dfc82f0d8ac846d7ec14acb129aa53fb1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}