{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/reflection-tuning-data-recycling-improves-llm","title":"Reflection-Tuning: Data Recycling Improves LLM Instruction-Tuning","arxiv_id":"2310.11716","date":"2023-10-18","proceeding":null,"authors":["Ming Li","Lichang Chen","Jiuhai Chen","Shwai He","Heng Huang","Jiuxiang Gu","Tianyi Zhou"],"abstract":"Recent advancements in Large Language Models (LLMs) have expanded the horizons of natural language understanding and generation. Notably, the output control and alignment with the input of LLMs can be refined through instruction tuning. However, as highlighted in several studies, low-quality data in the training set are usually detrimental to instruction tuning, resulting in inconsistent or even misleading LLM outputs. We propose a novel method, termed \"reflection-tuning,\" which addresses the problem by self-improvement and judging capabilities of LLMs. This approach utilizes an oracle LLM to recycle the original training data by introspecting and enhancing the quality of instructions and responses in the data. Extensive experiments on widely used evaluation benchmarks show that LLMs trained with our recycled data outperform those trained with existing datasets in various benchmarks.","url_abs":"https://arxiv.org/abs/2310.11716v1","url_pdf":"https://arxiv.org/pdf/2310.11716v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"reflection-tuning-data-recycling-improves-llm","repo_url":"https://github.com/mingliiii/reflection_tuning","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"reflection-tuning-data-recycling-improves-llm","repo_url":"https://github.com/tianyi-lab/reflection_tuning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null}],"tasks":[{"task_slug":"natural-language-understanding","task_name":"Natural Language Understanding"}],"methods":[{"method_slug":"set","method_name":"SET"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2310.11716","atlas_url":"https://app.syntology.ai/?focus=2310.11716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11716"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mingliiii/reflection_tuning","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tianyi-lab/reflection_tuning","reach":null}],"summary":{"ran_draft_wrong":1,"ran_honours":2},"by_repo_kind":{"listed":{"samples":3,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"0acb012db8f3b45d","entry":"gen_prompt_no_input","repo":"tianyi-lab/reflection_tuning","repo_kind":"listed","path":"reflection_code/reflect_instruction.py","file_url":"https://github.com/tianyi-lab/reflection_tuning/blob/HEAD/reflection_code/reflect_instruction.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0acb012db8f3b45d"}},{"code_sha256_prefix":"906dab8ee3c01e55","entry":"get_perplexity_and_embedding_part_text","repo":"tianyi-lab/reflection_tuning","repo_kind":"listed","path":"selection_code/data_analysis.py","file_url":"https://github.com/tianyi-lab/reflection_tuning/blob/HEAD/selection_code/data_analysis.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"906dab8ee3c01e55"}},{"code_sha256_prefix":"3c94e259a4b7a436","entry":"get_perplexity_and_embedding_whole_text","repo":"tianyi-lab/reflection_tuning","repo_kind":"listed","path":"selection_code/data_analysis.py","file_url":"https://github.com/tianyi-lab/reflection_tuning/blob/HEAD/selection_code/data_analysis.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3c94e259a4b7a436"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}