{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/5","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":12,"rows_per_page":100,"rows":[401,500],"of":1135,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following","prev":"/task/instruction-following/papers/4","next":"/task/instruction-following/papers/6","papers":[{"url":"/paper/lumen-unleashing-versatile-vision-centric","slug":"lumen-unleashing-versatile-vision-centric","title":"Lumen: Unleashing Versatile Vision-Centric Capabilities of Large Multimodal Models","date":"2024-03-12","arxiv_id":"2403.07304","repositories_listed":1,"syntology":null},{"url":"/paper/online-continual-learning-for-interactive","slug":"online-continual-learning-for-interactive","title":"Online Continual Learning For Interactive Instruction Following Agents","date":"2024-03-12","arxiv_id":"2403.07548","repositories_listed":1,"syntology":null},{"url":"/paper/ircoder-intermediate-representations-make","slug":"ircoder-intermediate-representations-make","title":"IRCoder: Intermediate Representations Make Language Models Robust Multilingual Code Generators","date":"2024-03-06","arxiv_id":"2403.03894","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-decode-collaboratively-with","slug":"learning-to-decode-collaboratively-with","title":"Learning to Decode Collaboratively with Multiple Language Models","date":"2024-03-06","arxiv_id":"2403.03870","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-decode-collaboratively-with#ran","syntology_url":"https://syntology.ai/paper/2403.03870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03870"}},"official":{"repos":["clinicalml/co-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-shot-a-unified-system-to-handle-frequent","slug":"x-shot-a-unified-system-to-handle-frequent","title":"X-Shot: A Unified System to Handle Frequent, Few-shot and Zero-shot Learning Simultaneously in Classification","date":"2024-03-06","arxiv_id":"2403.03863","repositories_listed":1,"syntology":null},{"url":"/paper/nphardeval4v-a-dynamic-reasoning-benchmark-of","slug":"nphardeval4v-a-dynamic-reasoning-benchmark-of","title":"NPHardEval4V: A Dynamic Reasoning Benchmark of Multimodal Large Language Models","date":"2024-03-04","arxiv_id":"2403.01777","repositories_listed":1,"syntology":null},{"url":"/paper/autodefense-multi-agent-llm-defense-against","slug":"autodefense-multi-agent-llm-defense-against","title":"AutoDefense: Multi-Agent LLM Defense against Jailbreak Attacks","date":"2024-03-02","arxiv_id":"2403.04783","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autodefense-multi-agent-llm-defense-against#ran","syntology_url":"https://syntology.ai/paper/2403.04783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04783"}},"official":{"repos":["xhmy/autodefense"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lab-large-scale-alignment-for-chatbots","slug":"lab-large-scale-alignment-for-chatbots","title":"LAB: Large-Scale Alignment for ChatBots","date":"2024-03-02","arxiv_id":"2403.01081","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lab-large-scale-alignment-for-chatbots#ran","syntology_url":"https://syntology.ai/paper/2403.01081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01081"}},"official":null}},{"url":"/paper/follow-my-instruction-and-spill-the-beans","slug":"follow-my-instruction-and-spill-the-beans","title":"Follow My Instruction and Spill the Beans: Scalable Data Extraction from Retrieval-Augmented Generation Systems","date":"2024-02-27","arxiv_id":"2402.17840","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/follow-my-instruction-and-spill-the-beans#ran","syntology_url":"https://syntology.ai/paper/2402.17840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17840"}},"official":{"repos":["zhentingqi/rag-privacy"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mini-ensemble-low-rank-adapters-for-parameter","slug":"mini-ensemble-low-rank-adapters-for-parameter","title":"MELoRA: Mini-Ensemble Low-Rank Adapters for Parameter-Efficient Fine-Tuning","date":"2024-02-27","arxiv_id":"2402.17263","repositories_listed":1,"syntology":null},{"url":"/paper/pragmatic-instruction-following-and-goal","slug":"pragmatic-instruction-following-and-goal","title":"Pragmatic Instruction Following and Goal Assistance via Cooperative Language-Guided Inverse Planning","date":"2024-02-27","arxiv_id":"2402.17930","repositories_listed":1,"syntology":null},{"url":"/paper/songcomposer-a-large-language-model-for-lyric","slug":"songcomposer-a-large-language-model-for-lyric","title":"SongComposer: A Large Language Model for Lyric and Melody Generation in Song Composition","date":"2024-02-27","arxiv_id":"2402.17645","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/songcomposer-a-large-language-model-for-lyric#ran","syntology_url":"https://syntology.ai/paper/2402.17645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17645"}},"official":{"repos":["pjlab-songcomposer/songcomposer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/long-context-language-modeling-with-parallel","slug":"long-context-language-modeling-with-parallel","title":"Long-Context Language Modeling with Parallel Context Encoding","date":"2024-02-26","arxiv_id":"2402.16617","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/long-context-language-modeling-with-parallel#ran","syntology_url":"https://syntology.ai/paper/2402.16617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16617"}},"official":{"repos":["princeton-nlp/cepe"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/defending-large-language-models-against-1","slug":"defending-large-language-models-against-1","title":"Defending Large Language Models against Jailbreak Attacks via Semantic Smoothing","date":"2024-02-25","arxiv_id":"2402.16192","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/defending-large-language-models-against-1#ran","syntology_url":"https://syntology.ai/paper/2402.16192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16192"}},"official":{"repos":["ucsb-nlp-chang/semanticsmooth"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/graphwiz-an-instruction-following-language","slug":"graphwiz-an-instruction-following-language","title":"GraphWiz: An Instruction-Following Language Model for Graph Problems","date":"2024-02-25","arxiv_id":"2402.16029","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/graphwiz-an-instruction-following-language#ran","syntology_url":"https://syntology.ai/paper/2402.16029","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16029"}},"official":{"repos":["nuochenpku/Graph-Reasoning-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-multi-turn-instruction-following-for","slug":"on-the-multi-turn-instruction-following-for","title":"On the Multi-turn Instruction Following for Conversational Web Agents","date":"2024-02-23","arxiv_id":"2402.15057","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-multi-turn-instruction-following-for#ran","syntology_url":"https://syntology.ai/paper/2402.15057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15057"}},"official":{"repos":["magicgh/self-map"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instraug-automatic-instruction-augmentation","slug":"instraug-automatic-instruction-augmentation","title":"Towards Robust Instruction Tuning on Multimodal Large Language Models","date":"2024-02-22","arxiv_id":"2402.14492","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instraug-automatic-instruction-augmentation#ran","syntology_url":"https://syntology.ai/paper/2402.14492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14492"}},"official":{"repos":["declare-lab/instraug"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instructir-a-benchmark-for-instruction","slug":"instructir-a-benchmark-for-instruction","title":"INSTRUCTIR: A Benchmark for Instruction Following of Information Retrieval Models","date":"2024-02-22","arxiv_id":"2402.14334","repositories_listed":1,"syntology":null},{"url":"/paper/unintended-impacts-of-llm-alignment-on-global","slug":"unintended-impacts-of-llm-alignment-on-global","title":"Unintended Impacts of LLM Alignment on Global Representation","date":"2024-02-22","arxiv_id":"2402.15018","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/unintended-impacts-of-llm-alignment-on-global#ran","syntology_url":"https://syntology.ai/paper/2402.15018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15018"}},"official":{"repos":["salt-nlp/unintended-impacts-of-alignment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/refutebench-evaluating-refuting-instruction","slug":"refutebench-evaluating-refuting-instruction","title":"RefuteBench: Evaluating Refuting Instruction-Following for Large Language Models","date":"2024-02-21","arxiv_id":"2402.13463","repositories_listed":1,"syntology":null},{"url":"/paper/self-distillation-bridges-distribution-gap-in","slug":"self-distillation-bridges-distribution-gap-in","title":"Self-Distillation Bridges Distribution Gap in Language Model Fine-Tuning","date":"2024-02-21","arxiv_id":"2402.13669","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-distillation-bridges-distribution-gap-in#ran","syntology_url":"https://syntology.ai/paper/2402.13669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13669"}},"official":{"repos":["sail-sg/sdft"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/promptkd-distilling-student-friendly","slug":"promptkd-distilling-student-friendly","title":"PromptKD: Distilling Student-Friendly Knowledge for Generative Language Models via Prompt Tuning","date":"2024-02-20","arxiv_id":"2402.12842","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/promptkd-distilling-student-friendly#ran","syntology_url":"https://syntology.ai/paper/2402.12842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12842"}},"official":{"repos":["gmkim-ai/promptkd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-impact-of-demonstrations-on-multilingual","slug":"the-impact-of-demonstrations-on-multilingual","title":"The Impact of Demonstrations on Multilingual In-Context Learning: A Multidimensional Analysis","date":"2024-02-20","arxiv_id":"2402.12976","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-critical-evaluation-of-ai-feedback-for#ran","syntology_url":"https://syntology.ai/paper/2402.12366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12366"}},"official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-r-evolution-of-multimodal-large-language","slug":"the-r-evolution-of-multimodal-large-language","title":"The Revolution of Multimodal Large Language Models: A Survey","date":"2024-02-19","arxiv_id":"2402.12451","repositories_listed":1,"syntology":null},{"url":"/paper/your-vision-language-model-itself-is-a-strong","slug":"your-vision-language-model-itself-is-a-strong","title":"Your Vision-Language Model Itself Is a Strong Filter: Towards High-Quality Instruction Tuning with Data Selection","date":"2024-02-19","arxiv_id":"2402.12501","repositories_listed":1,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/your-vision-language-model-itself-is-a-strong#ran","syntology_url":"https://syntology.ai/paper/2402.12501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12501"}},"official":{"repos":["rayruibochen/self-filter"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-modalities-in-vision-large-language","slug":"aligning-modalities-in-vision-large-language","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","date":"2024-02-18","arxiv_id":"2402.11411","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-modalities-in-vision-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.11411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11411"}},"official":{"repos":["yiyangzhou/povid"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eventrl-enhancing-event-extraction-with","slug":"eventrl-enhancing-event-extraction-with","title":"EventRL: Enhancing Event Extraction with Outcome Supervision for Large Language Models","date":"2024-02-18","arxiv_id":"2402.11430","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-large-language-models-by-on-policy","slug":"aligning-large-language-models-by-on-policy","title":"Aligning Large Language Models by On-Policy Self-Judgment","date":"2024-02-17","arxiv_id":"2402.11253","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/aligning-large-language-models-by-on-policy#ran","syntology_url":"https://syntology.ai/paper/2402.11253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11253"}},"official":{"repos":["oddqueue/self-judge"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/absinstruct-eliciting-abstraction-ability","slug":"absinstruct-eliciting-abstraction-ability","title":"AbsInstruct: Eliciting Abstraction Ability from LLMs through Explanation Tuning with Plausibility Estimation","date":"2024-02-16","arxiv_id":"2402.10646","repositories_listed":1,"syntology":null},{"url":"/paper/answer-is-all-you-need-instruction-following","slug":"answer-is-all-you-need-instruction-following","title":"Answer is All You Need: Instruction-following Text Embedding via Answering the Question","date":"2024-02-15","arxiv_id":"2402.09642","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/answer-is-all-you-need-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2402.09642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09642"}},"official":{"repos":["zhang-yu-wei/inbedder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/air-bench-benchmarking-large-audio-language","slug":"air-bench-benchmarking-large-audio-language","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","date":"2024-02-12","arxiv_id":"2402.07729","repositories_listed":1,"syntology":null},{"url":"/paper/policy-improvement-using-language-feedback","slug":"policy-improvement-using-language-feedback","title":"Policy Improvement using Language Feedback Models","date":"2024-02-12","arxiv_id":"2402.07876","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-improvement-using-language-feedback#ran","syntology_url":"https://syntology.ai/paper/2402.07876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07876"}},"official":{"repos":["vzhong/language_feedback_models"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/graphtranslator-aligning-graph-model-to-large","slug":"graphtranslator-aligning-graph-model-to-large","title":"GraphTranslator: Aligning Graph Model to Large Language Model for Open-ended Tasks","date":"2024-02-11","arxiv_id":"2402.07197","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graphtranslator-aligning-graph-model-to-large#ran","syntology_url":"https://syntology.ai/paper/2402.07197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07197"}},"official":{"repos":["alibaba/graphtranslator"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/aya-dataset-an-open-access-collection-for","slug":"aya-dataset-an-open-access-collection-for","title":"Aya Dataset: An Open-Access Collection for Multilingual Instruction Tuning","date":"2024-02-09","arxiv_id":"2402.06619","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-es-gradient-free-planning-with","slug":"diffusion-es-gradient-free-planning-with","title":"Diffusion-ES: Gradient-free Planning with Diffusion for Autonomous Driving and Zero-Shot Instruction Following","date":"2024-02-09","arxiv_id":"2402.06559","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffusion-es-gradient-free-planning-with#ran","syntology_url":"https://syntology.ai/paper/2402.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06559"}},"official":{"repos":["bhyang/diffusion-es"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/personalized-language-modeling-from","slug":"personalized-language-modeling-from","title":"Personalized Language Modeling from Personalized Human Feedback","date":"2024-02-06","arxiv_id":"2402.05133","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/personalized-language-modeling-from#ran","syntology_url":"https://syntology.ai/paper/2402.05133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05133"}},"official":{"repos":["humainlab/personalized_rlhf"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-data-selection-for-llm","slug":"a-survey-on-data-selection-for-llm","title":"A Survey on Data Selection for LLM Instruction Tuning","date":"2024-02-04","arxiv_id":"2402.05123","repositories_listed":1,"syntology":null},{"url":"/paper/safety-fine-tuning-at-almost-no-cost-a","slug":"safety-fine-tuning-at-almost-no-cost-a","title":"Safety Fine-Tuning at (Almost) No Cost: A Baseline for Vision Large Language Models","date":"2024-02-03","arxiv_id":"2402.02207","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safety-fine-tuning-at-almost-no-cost-a#ran","syntology_url":"https://syntology.ai/paper/2402.02207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02207"}},"official":{"repos":["ys-zong/vlguard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/indivec-an-exploration-of-leveraging-large","slug":"indivec-an-exploration-of-leveraging-large","title":"IndiVec: An Exploration of Leveraging Large Language Models for Media Bias Detection with Fine-Grained Bias Indicators","date":"2024-02-01","arxiv_id":"2402.00345","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-makes-a-difference","slug":"instruction-makes-a-difference","title":"Instruction Makes a Difference","date":"2024-02-01","arxiv_id":"2402.00453","repositories_listed":1,"syntology":null},{"url":"/paper/longalign-a-recipe-for-long-context-alignment","slug":"longalign-a-recipe-for-long-context-alignment","title":"LongAlign: A Recipe for Long Context Alignment of Large Language Models","date":"2024-01-31","arxiv_id":"2401.18058","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/longalign-a-recipe-for-long-context-alignment#ran","syntology_url":"https://syntology.ai/paper/2401.18058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.18058"}},"official":{"repos":["thudm/longalign"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/earthgpt-a-universal-multi-modal-large","slug":"earthgpt-a-universal-multi-modal-large","title":"EarthGPT: A Universal Multi-modal Large Language Model for Multi-sensor Image Comprehension in Remote Sensing Domain","date":"2024-01-30","arxiv_id":"2401.16822","repositories_listed":1,"syntology":null},{"url":"/paper/taking-action-towards-graceful-interaction","slug":"taking-action-towards-graceful-interaction","title":"Taking Action Towards Graceful Interaction: The Effects of Performing Actions on Modelling Policies for Instruction Clarification Requests","date":"2024-01-30","arxiv_id":"2401.17039","repositories_listed":1,"syntology":null},{"url":"/paper/selectllm-can-llms-select-important","slug":"selectllm-can-llms-select-important","title":"SelectLLM: Can LLMs Select Important Instructions to Annotate?","date":"2024-01-29","arxiv_id":"2401.16553","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-speculative-sampling-requires","slug":"eagle-speculative-sampling-requires","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","date":"2024-01-26","arxiv_id":"2401.15077","repositories_listed":1,"syntology":null},{"url":"/paper/towards-3d-molecule-text-interpretation-in","slug":"towards-3d-molecule-text-interpretation-in","title":"Towards 3D Molecule-Text Interpretation in Language Models","date":"2024-01-25","arxiv_id":"2401.13923","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-3d-molecule-text-interpretation-in#ran","syntology_url":"https://syntology.ai/paper/2401.13923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13923"}},"official":{"repos":["lsh0520/3d-molm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-superpositions-of","slug":"large-language-models-are-superpositions-of","title":"Large Language Models are Superpositions of All Characters: Attaining Arbitrary Role-play via Self-Alignment","date":"2024-01-23","arxiv_id":"2401.12474","repositories_listed":1,"syntology":null},{"url":"/paper/skyeyegpt-unifying-remote-sensing-vision","slug":"skyeyegpt-unifying-remote-sensing-vision","title":"SkyEyeGPT: Unifying Remote Sensing Vision-Language Tasks via Instruction Tuning with Large Language Model","date":"2024-01-18","arxiv_id":"2401.09712","repositories_listed":1,"syntology":null},{"url":"/paper/emollms-a-series-of-emotional-large-language","slug":"emollms-a-series-of-emotional-large-language","title":"EmoLLMs: A Series of Emotional Large Language Models and Annotation Tools for Comprehensive Affective Analysis","date":"2024-01-16","arxiv_id":"2401.08508","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/emollms-a-series-of-emotional-large-language#ran","syntology_url":"https://syntology.ai/paper/2401.08508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08508"}},"official":{"repos":["lzw108/emollms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/kun-answer-polishment-for-chinese-self","slug":"kun-answer-polishment-for-chinese-self","title":"Kun: Answer Polishment for Chinese Self-Alignment with Instruction Back-Translation","date":"2024-01-12","arxiv_id":"2401.06477","repositories_listed":1,"syntology":null},{"url":"/paper/infobench-evaluating-instruction-following","slug":"infobench-evaluating-instruction-following","title":"InFoBench: Evaluating Instruction Following Ability in Large Language Models","date":"2024-01-07","arxiv_id":"2401.03601","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/infobench-evaluating-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2401.03601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03601"}},"official":{"repos":["qinyiwei/infobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chartassisstant-a-universal-chart-multimodal","slug":"chartassisstant-a-universal-chart-multimodal","title":"ChartAssisstant: A Universal Chart Multimodal Language Model via Chart-to-Table Pre-training and Multitask Instruction Tuning","date":"2024-01-04","arxiv_id":"2401.02384","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartassisstant-a-universal-chart-multimodal#ran","syntology_url":"https://syntology.ai/paper/2401.02384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02384"}},"official":{"repos":["opengvlab/chartast"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-pro-progressive-llama-with-block","slug":"llama-pro-progressive-llama-with-block","title":"LLaMA Pro: Progressive LLaMA with Block Expansion","date":"2024-01-04","arxiv_id":"2401.02415","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-on","slug":"benchmarking-large-language-models-on","title":"Benchmarking Large Language Models on Controllable Generation under Diversified Instructions","date":"2024-01-01","arxiv_id":"2401.00690","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2401.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00690"}},"official":{"repos":["xt-cyh/codi-eval"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/maplm-a-real-world-large-scale-vision","slug":"maplm-a-real-world-large-scale-vision","title":"MAPLM: A Real-World Large-Scale Vision-Language Benchmark for Map and Traffic Scene Understanding","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/jatmo-prompt-injection-defense-by-task","slug":"jatmo-prompt-injection-defense-by-task","title":"Jatmo: Prompt Injection Defense by Task-Specific Finetuning","date":"2023-12-29","arxiv_id":"2312.17673","repositories_listed":1,"syntology":{"n":18,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":18,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/jatmo-prompt-injection-defense-by-task#ran","syntology_url":"https://syntology.ai/paper/2312.17673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17673"}},"official":{"repos":["wagner-group/prompt-injection-defense"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/aurora-activating-chinese-chat-capability-for","slug":"aurora-activating-chinese-chat-capability-for","title":"Aurora:Activating Chinese chat capability for Mixtral-8x7B sparse Mixture-of-Experts through Instruction-Tuning","date":"2023-12-22","arxiv_id":"2312.14557","repositories_listed":1,"syntology":null},{"url":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-in-depth-look-at-gemini-s-language#ran","syntology_url":"https://syntology.ai/paper/2312.11444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11444"}},"official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instructany2pix-flexible-visual-editing-via","slug":"instructany2pix-flexible-visual-editing-via","title":"InstructAny2Pix: Flexible Visual Editing via Multimodal Instruction Following","date":"2023-12-11","arxiv_id":"2312.06738","repositories_listed":1,"syntology":null},{"url":"/paper/creative-agents-empowering-agents-with","slug":"creative-agents-empowering-agents-with","title":"Creative Agents: Empowering Agents with Imagination for Creative Tasks","date":"2023-12-05","arxiv_id":"2312.02519","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/creative-agents-empowering-agents-with#ran","syntology_url":"https://syntology.ai/paper/2312.02519","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02519"}},"official":{"repos":["pku-rl/creative-agents"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gift-generative-interpretable-fine-tuning","slug":"gift-generative-interpretable-fine-tuning","title":"Generative Parameter-Efficient Fine-Tuning","date":"2023-12-01","arxiv_id":"2312.00700","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":10,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gift-generative-interpretable-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2312.00700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00700"}},"official":{"repos":["savadikarc/gift"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fft-towards-harmlessness-evaluation-and","slug":"fft-towards-harmlessness-evaluation-and","title":"FFT: Towards Harmlessness Evaluation and Analysis for LLMs with Factuality, Fairness, Toxicity","date":"2023-11-30","arxiv_id":"2311.18580","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-vision-language-alignment-makes","slug":"contrastive-vision-language-alignment-makes","title":"Contrastive Vision-Language Alignment Makes Efficient Instruction Learner","date":"2023-11-29","arxiv_id":"2311.17945","repositories_listed":1,"syntology":null},{"url":"/paper/vim-probing-multimodal-large-language-models","slug":"vim-probing-multimodal-large-language-models","title":"Text as Images: Can Multimodal Large Language Models Follow Printed Instructions in Pixels?","date":"2023-11-29","arxiv_id":"2311.17647","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vim-probing-multimodal-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2311.17647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17647"}},"official":{"repos":["vim-bench/vim_tool"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ranni-taming-text-to-image-diffusion-for","slug":"ranni-taming-text-to-image-diffusion-for","title":"Ranni: Taming Text-to-Image Diffusion for Accurate Instruction Following","date":"2023-11-28","arxiv_id":"2311.17002","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ranni-taming-text-to-image-diffusion-for#ran","syntology_url":"https://syntology.ai/paper/2311.17002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17002"}},"official":null}},{"url":"/paper/mods-model-oriented-data-selection-for","slug":"mods-model-oriented-data-selection-for","title":"MoDS: Model-oriented Data Selection for Instruction Tuning","date":"2023-11-27","arxiv_id":"2311.15653","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mods-model-oriented-data-selection-for#ran","syntology_url":"https://syntology.ai/paper/2311.15653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15653"}},"official":{"repos":["casia-lm/mods"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geochat-grounded-large-vision-language-model","slug":"geochat-grounded-large-vision-language-model","title":"GeoChat: Grounded Large Vision-Language Model for Remote Sensing","date":"2023-11-24","arxiv_id":"2311.15826","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/geochat-grounded-large-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2311.15826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15826"}},"official":{"repos":["mbzuai-oryx/geochat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/hallucidoctor-mitigating-hallucinatory","slug":"hallucidoctor-mitigating-hallucinatory","title":"HalluciDoctor: Mitigating Hallucinatory Toxicity in Visual Instruction Data","date":"2023-11-22","arxiv_id":"2311.13614","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/hallucidoctor-mitigating-hallucinatory#ran","syntology_url":"https://syntology.ai/paper/2311.13614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13614"}},"official":{"repos":["yuqifan1117/hallucidoctor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-improving-document-understanding-an","slug":"towards-improving-document-understanding-an","title":"Towards Improving Document Understanding: An Exploration on Text-Grounding via MLLMs","date":"2023-11-22","arxiv_id":"2311.13194","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-improving-document-understanding-an#ran","syntology_url":"https://syntology.ai/paper/2311.13194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13194"}},"official":{"repos":["harrytea/tgdoc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/recexplainer-aligning-large-language-models","slug":"recexplainer-aligning-large-language-models","title":"RecExplainer: Aligning Large Language Models for Explaining Recommendation Models","date":"2023-11-18","arxiv_id":"2311.10947","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recexplainer-aligning-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2311.10947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10947"}},"official":{"repos":["microsoft/recai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-generation-and-evaluation","slug":"benchmarking-generation-and-evaluation","title":"Benchmarking Generation and Evaluation Capabilities of Large Language Models for Instruction Controllable Summarization","date":"2023-11-15","arxiv_id":"2311.09184","repositories_listed":1,"syntology":null},{"url":"/paper/defending-large-language-models-against","slug":"defending-large-language-models-against","title":"Defending Large Language Models Against Jailbreaking Attacks Through Goal Prioritization","date":"2023-11-15","arxiv_id":"2311.09096","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/defending-large-language-models-against#ran","syntology_url":"https://syntology.ai/paper/2311.09096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09096"}},"official":{"repos":["thu-coai/jailbreakdefense_goalpriority"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/how-trustworthy-are-open-source-llms-an","slug":"how-trustworthy-are-open-source-llms-an","title":"How Trustworthy are Open-Source LLMs? An Assessment under Malicious Demonstrations Shows their Vulnerabilities","date":"2023-11-15","arxiv_id":"2311.09447","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/how-trustworthy-are-open-source-llms-an#ran","syntology_url":"https://syntology.ai/paper/2311.09447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09447"}},"official":{"repos":["osu-nlp-group/eval-llm-trust"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/plug-leveraging-pivot-language-in-cross","slug":"plug-leveraging-pivot-language-in-cross","title":"PLUG: Leveraging Pivot Language in Cross-Lingual Instruction Tuning","date":"2023-11-15","arxiv_id":"2311.08711","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plug-leveraging-pivot-language-in-cross#ran","syntology_url":"https://syntology.ai/paper/2311.08711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08711"}},"official":{"repos":["ytyz1307zzh/plug"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-you-prompt-matters-even-task-oriented","slug":"how-you-prompt-matters-even-task-oriented","title":"How You Prompt Matters! Even Task-Oriented Constraints in Instructions Affect LLM-Generated Text Detection","date":"2023-11-14","arxiv_id":"2311.08369","repositories_listed":1,"syntology":null},{"url":"/paper/self-evolved-diverse-data-sampling-for","slug":"self-evolved-diverse-data-sampling-for","title":"Self-Evolved Diverse Data Sampling for Efficient Instruction Tuning","date":"2023-11-14","arxiv_id":"2311.08182","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-evolved-diverse-data-sampling-for#ran","syntology_url":"https://syntology.ai/paper/2311.08182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08182"}},"official":{"repos":["ofa-sys/diverseevol"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-analogies-genies-a-testbed-for","slug":"generalization-analogies-genies-a-testbed-for","title":"Generalization Analogies: A Testbed for Generalizing AI Oversight to Hard-To-Measure Domains","date":"2023-11-13","arxiv_id":"2311.07723","repositories_listed":1,"syntology":null},{"url":"/paper/cappy-outperforming-and-boosting-large-multi","slug":"cappy-outperforming-and-boosting-large-multi","title":"Cappy: Outperforming and Boosting Large Multi-Task LMs with a Small Scorer","date":"2023-11-12","arxiv_id":"2311.06720","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cappy-outperforming-and-boosting-large-multi#ran","syntology_url":"https://syntology.ai/paper/2311.06720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06720"}},"official":null}},{"url":"/paper/dialmat-dialogue-enabled-transformer-with","slug":"dialmat-dialogue-enabled-transformer-with","title":"DialMAT: Dialogue-Enabled Transformer with Moment-Based Adversarial Training","date":"2023-11-12","arxiv_id":"2311.06855","repositories_listed":1,"syntology":null},{"url":"/paper/llava-plus-learning-to-use-tools-for-creating","slug":"llava-plus-learning-to-use-tools-for-creating","title":"LLaVA-Plus: Learning to Use Tools for Creating Multimodal Agents","date":"2023-11-09","arxiv_id":"2311.05437","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-plus-learning-to-use-tools-for-creating#ran","syntology_url":"https://syntology.ai/paper/2311.05437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05437"}},"official":{"repos":["LLaVA-VL/LLaVA-Plus-Codebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/u-llava-unifying-multi-modal-tasks-via-large","slug":"u-llava-unifying-multi-modal-tasks-via-large","title":"u-LLaVA: Unifying Multi-Modal Tasks via Large Language Model","date":"2023-11-09","arxiv_id":"2311.05348","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/u-llava-unifying-multi-modal-tasks-via-large#ran","syntology_url":"https://syntology.ai/paper/2311.05348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05348"}},"official":{"repos":["OPPOMKLab/u-LLaVA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/phogpt-generative-pre-training-for-vietnamese","slug":"phogpt-generative-pre-training-for-vietnamese","title":"PhoGPT: Generative Pre-training for Vietnamese","date":"2023-11-06","arxiv_id":"2311.02945","repositories_listed":1,"syntology":null},{"url":"/paper/chef-a-comprehensive-evaluation-framework-for","slug":"chef-a-comprehensive-evaluation-framework-for","title":"ChEF: A Comprehensive Evaluation Framework for Standardized Assessment of Multimodal Large Language Models","date":"2023-11-05","arxiv_id":"2311.02692","repositories_listed":1,"syntology":null},{"url":"/paper/faithscore-evaluating-hallucinations-in-large","slug":"faithscore-evaluating-hallucinations-in-large","title":"FaithScore: Fine-grained Evaluations of Hallucinations in Large Vision-Language Models","date":"2023-11-02","arxiv_id":"2311.01477","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/faithscore-evaluating-hallucinations-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.01477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01477"}},"official":{"repos":["bcdnlp/faithscore"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/distort-distract-decode-instruction-tuned","slug":"distort-distract-decode-instruction-tuned","title":"Instructive Decoding: Instruction-Tuned Large Language Models are Self-Refiner from Noisy Instructions","date":"2023-11-01","arxiv_id":"2311.00233","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/distort-distract-decode-instruction-tuned#ran","syntology_url":"https://syntology.ai/paper/2311.00233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00233"}},"official":{"repos":["joonkeekim/Instructive-Decoding"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/followbench-a-multi-level-fine-grained","slug":"followbench-a-multi-level-fine-grained","title":"FollowBench: A Multi-level Fine-grained Constraints Following Benchmark for Large Language Models","date":"2023-10-31","arxiv_id":"2310.20410","repositories_listed":1,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/followbench-a-multi-level-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2310.20410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20410"}},"official":{"repos":["yjiangcm/followbench"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/making-large-language-models-better-data","slug":"making-large-language-models-better-data","title":"Making Large Language Models Better Data Creators","date":"2023-10-31","arxiv_id":"2310.20111","repositories_listed":1,"syntology":null},{"url":"/paper/myriad-large-multimodal-model-by-applying","slug":"myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","arxiv_id":"2310.19070","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/myriad-large-multimodal-model-by-applying#ran","syntology_url":"https://syntology.ai/paper/2310.19070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19070"}},"official":{"repos":["tzjtatata/myriad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/instruct-and-extract-instruction-tuning-for","slug":"instruct-and-extract-instruction-tuning-for","title":"Instruct and Extract: Instruction Tuning for On-Demand Information Extraction","date":"2023-10-24","arxiv_id":"2310.16040","repositories_listed":1,"syntology":{"n":21,"n_ran":12,"n_constructed":1,"n_ran_checked":11,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":21,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/instruct-and-extract-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2310.16040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16040"}},"official":{"repos":["yzjiao/on-demand-ie"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":1,"n_ran_no_instrument_failure":11,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/alpacare-instruction-tuned-large-language","slug":"alpacare-instruction-tuned-large-language","title":"AlpaCare:Instruction-tuned Large Language Models for Medical Application","date":"2023-10-23","arxiv_id":"2310.14558","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alpacare-instruction-tuned-large-language#ran","syntology_url":"https://syntology.ai/paper/2310.14558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14558"}},"official":{"repos":["xzhang97666/alpacare"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monte-carlo-thought-search-large-language","slug":"monte-carlo-thought-search-large-language","title":"Monte Carlo Thought Search: Large Language Model Querying for Complex Scientific Reasoning in Catalyst Design","date":"2023-10-22","arxiv_id":"2310.14420","repositories_listed":1,"syntology":null},{"url":"/paper/botchat-evaluating-llms-capabilities-of","slug":"botchat-evaluating-llms-capabilities-of","title":"BotChat: Evaluating LLMs' Capabilities of Having Multi-Turn Dialogues","date":"2023-10-20","arxiv_id":"2310.13650","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/botchat-evaluating-llms-capabilities-of#ran","syntology_url":"https://syntology.ai/paper/2310.13650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13650"}},"official":{"repos":["open-compass/botchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/democratizing-reasoning-ability-tailored","slug":"democratizing-reasoning-ability-tailored","title":"Democratizing Reasoning Ability: Tailored Learning from Large Language Model","date":"2023-10-20","arxiv_id":"2310.13332","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/democratizing-reasoning-ability-tailored#ran","syntology_url":"https://syntology.ai/paper/2310.13332","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13332"}},"official":{"repos":["raibows/learn-to-reason"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-emulator-for-fine-tuning-large-language","slug":"an-emulator-for-fine-tuning-large-language","title":"An Emulator for Fine-Tuning Large Language Models using Small Language Models","date":"2023-10-19","arxiv_id":"2310.12962","repositories_listed":1,"syntology":null},{"url":"/paper/lacma-language-aligning-contrastive-learning","slug":"lacma-language-aligning-contrastive-learning","title":"LACMA: Language-Aligning Contrastive Learning with Meta-Actions for Embodied Instruction Following","date":"2023-10-18","arxiv_id":"2310.12344","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lacma-language-aligning-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2310.12344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12344"}},"official":{"repos":["joeyy5588/lacma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/quantify-health-related-atomic-knowledge-in","slug":"quantify-health-related-atomic-knowledge-in","title":"Quantifying Self-diagnostic Atomic Knowledge in Chinese Medical Foundation Model: A Computational Analysis","date":"2023-10-18","arxiv_id":"2310.11722","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-large-language-models-at","slug":"evaluating-large-language-models-at","title":"Evaluating Large Language Models at Evaluating Instruction Following","date":"2023-10-11","arxiv_id":"2310.07641","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-large-language-models-at#ran","syntology_url":"https://syntology.ai/paper/2310.07641","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07641"}},"official":{"repos":["princeton-nlp/llmbar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llark-a-multimodal-foundation-model-for-music","slug":"llark-a-multimodal-foundation-model-for-music","title":"LLark: A Multimodal Instruction-Following Language Model for Music","date":"2023-10-11","arxiv_id":"2310.07160","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":1,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llark-a-multimodal-foundation-model-for-music#ran","syntology_url":"https://syntology.ai/paper/2310.07160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07160"}},"official":{"repos":["spotify-research/llark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/trace-a-comprehensive-benchmark-for-continual","slug":"trace-a-comprehensive-benchmark-for-continual","title":"TRACE: A Comprehensive Benchmark for Continual Learning in Large Language Models","date":"2023-10-10","arxiv_id":"2310.06762","repositories_listed":1,"syntology":null}],"record_sha256":"6d516de7307b195c8161e8e1e0d4f69d99b2e119a3996eef98450f458c40cb2e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}