{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/semi-supervised-reward-modeling-via-iterative","title":"Semi-Supervised Reward Modeling via Iterative Self-Training","arxiv_id":"2409.06903","date":"2024-09-10","proceeding":null,"authors":["Yifei He","Haoxiang Wang","Ziyan Jiang","Alexandros Papangelis","Han Zhao"],"abstract":"Reward models (RM) capture the values and preferences of humans and play a central role in Reinforcement Learning with Human Feedback (RLHF) to align pretrained large language models (LLMs). Traditionally, training these models relies on extensive human-annotated preference data, which poses significant challenges in terms of scalability and cost. To overcome these limitations, we propose Semi-Supervised Reward Modeling (SSRM), an approach that enhances RM training using unlabeled data. Given an unlabeled dataset, SSRM involves three key iterative steps: pseudo-labeling unlabeled examples, selecting high-confidence examples through a confidence threshold, and supervised finetuning on the refined dataset. Across extensive experiments on various model configurations, we demonstrate that SSRM significantly improves reward models without incurring additional labeling costs. Notably, SSRM can achieve performance comparable to models trained entirely on labeled data of equivalent volumes. Overall, SSRM substantially reduces the dependency on large volumes of human-annotated data, thereby decreasing the overall cost and time involved in training effective reward models.","url_abs":"https://arxiv.org/abs/2409.06903v1","url_pdf":"https://arxiv.org/pdf/2409.06903v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"semi-supervised-reward-modeling-via-iterative","repo_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[],"methods":[{"method_slug":"align","method_name":"ALIGN"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2409.06903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.06903"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/RLHFlow/RLHF-Reward-Modeling","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":9},"by_repo_kind":{"official":{"samples":9,"ran":9,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"af2be95726bb445c","entry":"calculate_scores_per_section","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"useful_code/eval_reward_bench_bt.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/useful_code/eval_reward_bench_bt.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"af2be95726bb445c"}},{"code_sha256_prefix":"d9c128f1ad102f62","entry":"calculate_scores_per_section","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"armo-rm/stage-2_train.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/armo-rm/stage-2_train.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d9c128f1ad102f62"}},{"code_sha256_prefix":"f8f5b369acc5168f","entry":"chat_completion_openai","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"decision_tree/collect_llm_preferences.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/decision_tree/collect_llm_preferences.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f8f5b369acc5168f"}},{"code_sha256_prefix":"b2da3df8c4cfa9a9","entry":"chat_completion_together","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"decision_tree/collect_llm_preferences.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/decision_tree/collect_llm_preferences.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b2da3df8c4cfa9a9"}},{"code_sha256_prefix":"23a0bca594cfbdcb","entry":"compute_metrics","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"bradley-terry-rm/gemma_2B_rm.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/bradley-terry-rm/gemma_2B_rm.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"23a0bca594cfbdcb"}},{"code_sha256_prefix":"3462cb5c25c31912","entry":"convert_to_chat_format","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"decision_tree/get_embeddings.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/decision_tree/get_embeddings.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"3462cb5c25c31912"}},{"code_sha256_prefix":"307a0a74a4fcdc2e","entry":"eval_reward_bench","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"armo-rm/stage-2_train.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/armo-rm/stage-2_train.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"307a0a74a4fcdc2e"}},{"code_sha256_prefix":"0a86528ef75096e8","entry":"find_proper_verbosity_penalties","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"armo-rm/stage-2_train.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/armo-rm/stage-2_train.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0a86528ef75096e8"}},{"code_sha256_prefix":"0d1671f0a8c16a44","entry":"find_token_for_gating","repo":"RLHFlow/RLHF-Reward-Modeling","repo_kind":"official","path":"armo-rm/stage-2_prepare.py","file_url":"https://github.com/RLHFlow/RLHF-Reward-Modeling/blob/HEAD/armo-rm/stage-2_prepare.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0d1671f0a8c16a44"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}