{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/rlcd-reinforcement-learning-from-contrast","title":"RLCD: Reinforcement Learning from Contrastive Distillation for Language Model Alignment","arxiv_id":"2307.12950","date":"2023-07-24","proceeding":null,"authors":["Kevin Yang","Dan Klein","Asli Celikyilmaz","Nanyun Peng","Yuandong Tian"],"abstract":"We propose Reinforcement Learning from Contrastive Distillation (RLCD), a method for aligning language models to follow principles expressed in natural language (e.g., to be more harmless) without using human feedback. RLCD creates preference pairs from two contrasting model outputs, one using a positive prompt designed to encourage following the given principles, and one using a negative prompt designed to encourage violating them. Using two different prompts causes model outputs to be more differentiated on average, resulting in cleaner preference labels in the absence of human annotations. We then use the preference pairs to train a preference model, which is in turn used to improve a base unaligned language model via reinforcement learning. Empirically, RLCD outperforms RLAIF (Bai et al., 2022b) and context distillation (Huang et al., 2022) baselines across three diverse alignment tasks--harmlessness, helpfulness, and story outline generation--and when using both 7B and 30B model scales for simulating preference data.","url_abs":"https://arxiv.org/abs/2307.12950v3","url_pdf":"https://arxiv.org/pdf/2307.12950v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"rlcd-reinforcement-learning-from-contrast","repo_url":"https://github.com/facebookresearch/rlcd","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"rlcd-reinforcement-learning-from-contrast","repo_url":"https://github.com/prasanns/rlhf-length-biases","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"base","method_name":"BASE"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2307.12950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12950"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/facebookresearch/rlcd","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"deterministic:regex_extraction","url":"https://github.com/anthropics/ConstitutionalHarmlessnessPaper","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/prasanns/rlhf-length-biases","reach":null}],"summary":{"ran_honours":1,"ran_draft_wrong":1,"unverified":4},"by_repo_kind":{"official":{"samples":3,"ran":2,"repositories":1},"listed":{"samples":3,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"a74b91f02512331c","entry":"parse_indices","repo":"facebookresearch/RLCD","repo_kind":"official","path":"scripts/simulate_preference_data.py","file_url":"https://github.com/facebookresearch/RLCD/blob/HEAD/scripts/simulate_preference_data.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a74b91f02512331c"}},{"code_sha256_prefix":"5683be533bddb9cf","entry":"process_filter_result","repo":"facebookresearch/RLCD","repo_kind":"official","path":"scripts/simulate_preference_data.py","file_url":"https://github.com/facebookresearch/RLCD/blob/HEAD/scripts/simulate_preference_data.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5683be533bddb9cf"}},{"code_sha256_prefix":"184cc42fa122fabd","entry":"computelike","repo":"prasanns/rlhf-length-biases","repo_kind":"listed","path":"rlhfutils/rlhfutils/rewards.py","file_url":"https://github.com/prasanns/rlhf-length-biases/blob/HEAD/rlhfutils/rlhfutils/rewards.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"184cc42fa122fabd"}},{"code_sha256_prefix":"ab43db669ce08a10","entry":"contdistill","repo":"prasanns/rlhf-length-biases","repo_kind":"listed","path":"rlhfutils/rlhfutils/rewards.py","file_url":"https://github.com/prasanns/rlhf-length-biases/blob/HEAD/rlhfutils/rlhfutils/rewards.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ab43db669ce08a10"}},{"code_sha256_prefix":"21de08a24cc455d0","entry":"generate_result","repo":"facebookresearch/RLCD","repo_kind":"official","path":"scripts/simulate_preference_data.py","file_url":"https://github.com/facebookresearch/RLCD/blob/HEAD/scripts/simulate_preference_data.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"21de08a24cc455d0"}},{"code_sha256_prefix":"1d88b65e3ddd280c","entry":"probcompute","repo":"prasanns/rlhf-length-biases","repo_kind":"listed","path":"rlhfutils/rlhfutils/rewards.py","file_url":"https://github.com/prasanns/rlhf-length-biases/blob/HEAD/rlhfutils/rlhfutils/rewards.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"1d88b65e3ddd280c"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}