{"url":"/task/instruction-following","name":"Instruction Following","slug":"instruction-following","description_markdown":"Instruction following is the basic task of the model. This task is dedicated to evaluating the ability of the large model to follow human instructions. It is hoped that the model can generate controllable and safe answers.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1135,"papers_with_code":609,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":14,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/instruction-following-on-ifeval","slug":"instruction-following-on-ifeval","dataset":"IFEval","dataset_url":"/dataset/ifeval","rows_in_archive":4,"metrics":["Inst-level loose-accuracy","Inst-level strict-accuracy","Prompt-level loose-accuracy","Prompt-level strict-accuracy"],"first_row_in_archive_order":{"model":"AutoIF (Llama3 70B)","paper_title":"Self-play with Execution Feedback: Improving Instruction-following Capabilities of Large Language Models","paper_url":"/paper/self-play-with-execution-feedback-improving","paper_date":"2024-06-19","arxiv_id":"2406.13542","code_links":[{"title":"QwenLM/AutoIF","url":"https://github.com/QwenLM/AutoIF"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/ifeval","name":"IFEval","full_name":"Instruction Following Evaluation Datset","num_papers_in_archive":180},{"url":"/dataset/gscan","name":"GSCAN","full_name":"Grounded SCAN","num_papers_in_archive":22},{"url":"/dataset/mimic-it","name":"MIMIC-IT","full_name":"","num_papers_in_archive":11},{"url":"/dataset/ugif","name":"UGIF","full_name":"","num_papers_in_archive":10},{"url":"/dataset/bactrian-x","name":"Bactrian-X","full_name":"","num_papers_in_archive":9},{"url":"/dataset/cidar","name":"CIDAR","full_name":"","num_papers_in_archive":4},{"url":"/dataset/alpaca-data-galician","name":"Alpaca Data Galician","full_name":"Alpaca Data Galician","num_papers_in_archive":1},{"url":"/dataset/instructopenwiki","name":"InstructOpenWiki","full_name":"","num_papers_in_archive":1},{"url":"/dataset/sequential-instructions","name":"Sequential Instructions","full_name":"","num_papers_in_archive":1},{"url":"/dataset/surgeglobal-evol-instruct","name":"SurgeGlobal/Evol-Instruct","full_name":"","num_papers_in_archive":1},{"url":"/dataset/surgeglobal-lamini","name":"SurgeGlobal/LaMini","full_name":"","num_papers_in_archive":1},{"url":"/dataset/surgeglobal-orca","name":"SurgeGlobal/Orca","full_name":"","num_papers_in_archive":1},{"url":"/dataset/tamil-alpaca","name":"Tamil Alpaca","full_name":"","num_papers_in_archive":1},{"url":"/dataset/tamil-alpaca-orca","name":"Tamil Alpaca Orca","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/visual-instruction-following","name":"visual instruction following"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":609,"tagged_in_all":1135,"items":[{"url":"/paper/qlora-efficient-finetuning-of-quantized-llms","title":"QLoRA: Efficient Finetuning of Quantized LLMs","date":"2023-05-23","arxiv_id":"2305.14314","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_unverified":9,"n_pointer_only":17}},{"url":"/paper/self-instruct-aligning-language-model-with","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","date":"2022-12-20","arxiv_id":"2212.10560","repositories_listed":19,"syntology":{"n":17,"n_ran":7,"n_unverified":10,"n_pointer_only":1}},{"url":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_unverified":35,"n_pointer_only":0}},{"url":"/paper/habitat-a-platform-for-embodied-ai-research","title":"Habitat: A Platform for Embodied AI Research","date":"2019-04-02","arxiv_id":"1904.01201","repositories_listed":13,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":15}},{"url":"/paper/benchmarking-generalization-via-in-context","title":"Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ NLP Tasks","date":"2022-04-16","arxiv_id":"2204.07705","repositories_listed":10,"syntology":{"n":28,"n_ran":7,"n_unverified":21,"n_pointer_only":4}},{"url":"/paper/chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":15,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/llama-adapter-efficient-fine-tuning-of","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","date":"2023-03-28","arxiv_id":"2303.16199","repositories_listed":7,"syntology":null},{"url":"/paper/qwen2-5-technical-report","title":"Qwen2.5 Technical Report","date":"2024-12-19","arxiv_id":"2412.15115","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/distillm-towards-streamlined-distillation-for","title":"DistiLLM: Towards Streamlined Distillation for Large Language Models","date":"2024-02-06","arxiv_id":"2402.03898","repositories_listed":5,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/lmsys-chat-1m-a-large-scale-real-world-llm","title":"LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset","date":"2023-09-21","arxiv_id":"2309.11998","repositories_listed":5,"syntology":null},{"url":"/paper/point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","arxiv_id":"2309.00615","repositories_listed":5,"syntology":{"n":20,"n_ran":13,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/mapping-instructions-to-actions-in-3d","title":"Mapping Instructions to Actions in 3D Environments with Visual Goal Prediction","date":"2018-09-04","arxiv_id":"1809.00786","repositories_listed":5,"syntology":null},{"url":"/paper/instruction-following-evaluation-for-large","title":"Instruction-Following Evaluation for Large Language Models","date":"2023-11-14","arxiv_id":"2311.07911","repositories_listed":4,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/longlora-efficient-fine-tuning-of-long","title":"LongLoRA: Efficient Fine-tuning of Long-Context Large Language Models","date":"2023-09-21","arxiv_id":"2309.12307","repositories_listed":4,"syntology":{"n":13,"n_ran":11,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/how-far-can-camels-go-exploring-the-state-of","title":"How Far Can Camels Go? Exploring the State of Instruction Tuning on Open Resources","date":"2023-06-07","arxiv_id":"2306.04751","repositories_listed":4,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":2}},{"url":"/paper/wizardlm-empowering-large-language-models-to","title":"WizardLM: Empowering Large Language Models to Follow Complex Instructions","date":"2023-04-24","arxiv_id":"2304.12244","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/fusechat-knowledge-fusion-of-chat-models-1","title":"FuseChat: Knowledge Fusion of Chat Models","date":"2024-08-15","arxiv_id":"2408.07990","repositories_listed":3,"syntology":null},{"url":"/paper/funaudiollm-voice-understanding-and","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","date":"2024-07-04","arxiv_id":"2407.04051","repositories_listed":3,"syntology":{"n":10,"n_ran":10,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ruler-improving-llm-controllability-by-rule","title":"RuleR: Improving LLM Controllability by Rule-based Data Recycling","date":"2024-06-22","arxiv_id":"2406.15938","repositories_listed":3,"syntology":null},{"url":"/paper/mosaic-it-enhancing-instruction-tuning-with","title":"Mosaic-IT: Free Compositional Data Augmentation Improves Instruction Tuning","date":"2024-05-22","arxiv_id":"2405.13326","repositories_listed":3,"syntology":null},{"url":"/paper/shapellm-universal-3d-object-understanding","title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","date":"2024-02-27","arxiv_id":"2402.17766","repositories_listed":3,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/language-models-are-homer-simpson-safety-re","title":"Language Models are Homer Simpson! Safety Re-Alignment of Fine-tuned Language Models through Task Arithmetic","date":"2024-02-19","arxiv_id":"2402.11746","repositories_listed":3,"syntology":{"n":16,"n_ran":9,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/openfedllm-training-large-language-models-on","title":"OpenFedLLM: Training Large Language Models on Decentralized Private Data via Federated Learning","date":"2024-02-10","arxiv_id":"2402.06954","repositories_listed":3,"syntology":null},{"url":"/paper/self-rewarding-language-models","title":"Self-Rewarding Language Models","date":"2024-01-18","arxiv_id":"2401.10020","repositories_listed":3,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/waterbench-towards-holistic-evaluation-of","title":"WaterBench: Towards Holistic Evaluation of Watermarks for Large Language Models","date":"2023-11-13","arxiv_id":"2311.07138","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/language-models-are-super-mario-absorbing","title":"Language Models are Super Mario: Absorbing Abilities from Homologous Models as a Free Lunch","date":"2023-11-06","arxiv_id":"2311.03099","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/modulora-finetuning-3-bit-llms-on-consumer","title":"ModuLoRA: Finetuning 2-Bit LLMs on Consumer GPUs by Integrating with Modular Quantizers","date":"2023-09-28","arxiv_id":"2309.16119","repositories_listed":3,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":8}},{"url":"/paper/instructiongpt-4-a-200-instruction-paradigm","title":"InstructionGPT-4: A 200-Instruction Paradigm for Fine-Tuning MiniGPT-4","date":"2023-08-23","arxiv_id":"2308.12067","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/from-quantity-to-quality-boosting-llm","title":"From Quantity to Quality: Boosting LLM Performance with Self-Guided Data Selection for Instruction Tuning","date":"2023-08-23","arxiv_id":"2308.12032","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}},{"url":"/paper/l-eval-instituting-standardized-evaluation","title":"L-Eval: Instituting Standardized Evaluation for Long Context Language Models","date":"2023-07-20","arxiv_id":"2307.11088","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}}],"syntology_records":23,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}