{"url":"/method/rlaif","slug":"rlaif","name":"RLAIF","full_name":"Reinforcement Learning from AI Feedback","full_name_withheld":false,"description_markdown":null,"description_state":"absent","introduced_year":null,"introduced_by":{"title":"Tuning Large Multimodal Models for Videos using Reinforcement Learning from AI Feedback","paper":"/paper/tuning-large-multimodal-models-for-videos","first_author":"Daechul Ahn","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/tuning-large-multimodal-models-for-videos"},"source":{"url":"https://arxiv.org/abs/2402.03746v3","title":"Tuning Large Multimodal Models for Videos using Reinforcement Learning from AI Feedback","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Reinforcement Learning Frameworks","url":"/methods/category/reinforcement-learning-frameworks","pwc_aliases":[]}],"n_papers_tagged":19,"archive_num_papers":19,"papers_newest_first":[{"paper":"/paper/toward-evaluative-thinking-meta-policy","title":"Toward Evaluative Thinking: Meta Policy Optimization with Evolving Reward Models","date":"2025-04-28","arxiv_id":"2504.20157","n_code_links":1,"syntology":{"ran":4,"of":15,"unverified":11,"pointer_only":15}},{"paper":"/paper/r2vul-learning-to-reason-about-software","title":"R2Vul: Learning to Reason about Software Vulnerabilities with Reinforcement Learning and Structured Reasoning Distillation","date":"2025-04-07","arxiv_id":"2504.04699","n_code_links":1,"syntology":null},{"paper":null,"title":"Synthetic Data Generation & Multi-Step RL for Reasoning & Tool Use","date":"2025-04-07","arxiv_id":"2504.04736","n_code_links":0,"syntology":null},{"paper":null,"title":"Training Dialogue Systems by AI Feedback for Improving Overall Dialogue Impression","date":"2025-01-22","arxiv_id":"2501.12698","n_code_links":0,"syntology":null},{"paper":null,"title":"PopAlign: Diversifying Contrasting Patterns for a More Comprehensive Alignment","date":"2024-10-17","arxiv_id":"2410.13785","n_code_links":0,"syntology":null},{"paper":null,"title":"Exploring LLM-based Data Annotation Strategies for Medical Dialogue Preference Alignment","date":"2024-10-05","arxiv_id":"2410.04112","n_code_links":0,"syntology":null},{"paper":null,"title":"Generative Reward Models","date":"2024-10-02","arxiv_id":"2410.12832","n_code_links":0,"syntology":null},{"paper":"/paper/maferw-query-rewriting-with-multi-aspect","title":"MaFeRw: Query Rewriting with Multi-Aspect Feedbacks for Retrieval-Augmented Large Language Models","date":"2024-08-30","arxiv_id":"2408.17072","n_code_links":1,"syntology":null},{"paper":null,"title":"Applying RLAIF for Code Generation with API-usage in Lightweight LLMs","date":"2024-06-28","arxiv_id":"2406.20060","n_code_links":0,"syntology":null},{"paper":null,"title":"Diminishing Stereotype Bias in Image Generation Model using Reinforcemenlent Learning Feedback","date":"2024-06-27","arxiv_id":"2407.09551","n_code_links":0,"syntology":null},{"paper":"/paper/multi-objective-reinforcement-learning-from","title":"Multi-objective Reinforcement learning from AI Feedback","date":"2024-06-11","arxiv_id":"2406.07295","n_code_links":1,"syntology":null},{"paper":null,"title":"Are You Sure? Rank Them Again: Repeated Ranking For Better Preference Datasets","date":"2024-05-29","arxiv_id":"2405.18952","n_code_links":0,"syntology":null},{"paper":"/paper/optimization-based-prompt-injection-attack-to","title":"Optimization-based Prompt Injection Attack to LLM-as-a-Judge","date":"2024-03-26","arxiv_id":"2403.17710","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":2}},{"paper":"/paper/codeultrafeedback-an-llm-as-a-judge-dataset","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","date":"2024-03-14","arxiv_id":"2403.09032","n_code_links":2,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":null,"title":"HRLAIF: Improvements in Helpfulness and Harmlessness in Open-domain Reinforcement Learning From AI Feedback","date":"2024-03-13","arxiv_id":"2403.08309","n_code_links":0,"syntology":null},{"paper":"/paper/a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","n_code_links":1,"syntology":{"ran":8,"of":11,"unverified":3,"pointer_only":0}},{"paper":"/paper/direct-large-language-model-alignment-through","title":"Direct Large Language Model Alignment Through Self-Rewarding Contrastive Prompt Distillation","date":"2024-02-19","arxiv_id":"2402.11907","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":null,"title":"FinTral: A Family of GPT-4 Level Multimodal Financial Large Language Models","date":"2024-02-16","arxiv_id":"2402.10986","n_code_links":0,"syntology":null},{"paper":"/paper/tuning-large-multimodal-models-for-videos","title":"Tuning Large Multimodal Models for Videos using Reinforcement Learning from AI Feedback","date":"2024-02-06","arxiv_id":"2402.03746","n_code_links":1,"syntology":null}],"papers_shown":19,"tasks":[{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":5},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":5},{"task":"/task/language-modelling","name":"Language Modelling","papers":4},{"task":"/task/large-language-model","name":"Large Language Model","papers":3},{"task":"/task/mathematical-reasoning","name":"Mathematical Reasoning","papers":3},{"task":"/task/decision-making","name":"Decision Making","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/question-answering","name":"Question Answering","papers":2},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":2},{"task":"/task/retrieval","name":"Retrieval","papers":2},{"task":"/task/code-generation","name":"Code Generation","papers":1},{"task":"/task/dataset-generation","name":"Dataset Generation","papers":1},{"task":"/task/denoising","name":"Denoising","papers":1},{"task":"/task/fairness","name":"Fairness","papers":1},{"task":"/task/financial-analysis","name":"Financial Analysis","papers":1},{"task":"/task/gsm8k","name":"GSM8K","papers":1},{"task":"/task/gender-classification","name":"Gender Classification","papers":1},{"task":"/task/hallucination","name":"Hallucination","papers":1},{"task":"/task/humaneval","name":"HumanEval","papers":1},{"task":"/task/image-generation","name":"Image Generation","papers":1}],"tasks_shown":20,"n_tasks":34,"usage_by_year":[{"year":"2024","papers":15},{"year":"2025","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/rlaif"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}