Commit 285edfde by wangchenglong

update.

parent 7e3220a0
@article{wang2026msrl,
title={MSRL: Scaling Generative Multimodal Reward Modeling via Multi-Stage Reinforcement Learning},
author={Wang, Chenglong and Huo, Yifu and Gan, Yang and He, Qiaozhi and Meng, Qi and Li, Bei and Wang, Yan and Liu, Junfu and Zhou, Tianhua and Zhu, Jingbo and others},
journal={arXiv preprint arXiv:2603.25108},
year={2026}
}
@inproceedings{wang-etal:wang2026gram,
title={Gram-r$^2$: Self-training generative foundation reward models for reward reasoning},
author={Wang, Chenglong and Mu, Yongyu and Zhou, Hang and Huo, Yifu and Zhu, Ziming and Zeng, Jiali and Yang, Murun and Li, Bei and Hao, Xiaoyang and Zhang, Chunliang and others},
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
volume={40},
number={39},
pages={33395--33403},
year={2026}
}
@article{wang-etal:wang2025gram,
title={Gram: A generative foundation reward model for reward generalization},
author={Wang, Chenglong and Gan, Yang and Huo, Yifu and Mu, Yongyu and He, Qiaozhi and Yang, Murun and Li, Bei and Xiao, Tong and Zhang, Chunliang and Liu, Tongran and others},
journal={arXiv preprint arXiv:2506.14175},
year={2025}
}
@article{zhang-etal:zhang2024generative,
title={Generative verifiers: Reward modeling as next-token prediction},
author={Zhang, Lunjun and Hosseini, Arian and Bansal, Hritik and Kazemi, Mehran and Kumar, Aviral and Agarwal, Rishabh},
journal={arXiv preprint arXiv:2408.15240},
year={2024}
}
@article{sun-etal:2023aligning, @article{sun-etal:2023aligning,
title={Aligning large multimodal models with factually augmented rlhf}, title={Aligning large multimodal models with factually augmented rlhf},
author={Sun, Zhiqing and Shen, Sheng and Cao, Shengcao and Liu, Haotian and Li, Chunyuan and Shen, Yikang and Gan, Chuang and Gui, Liang-Yan and Wang, Yu-Xiong and Yang, Yiming and others}, author={Sun, Zhiqing and Shen, Sheng and Cao, Shengcao and Liu, Haotian and Li, Chunyuan and Shen, Yikang and Gan, Chuang and Gui, Liang-Yan and Wang, Yu-Xiong and Yang, Yiming and others},
......
This is BibTeX, Version 0.99e (TeX Live 2026)
Capacity: max_strings=200000, hash_size=200000, hash_prime=170003
The top-level auxiliary file: rl-introduction.aux
The style file: rl-introduction.bst
Database file #1: rl-introduction.bib
Warning--can't use both volume and number fields in wang-etal:wang2026gram
You've used 98 entries,
2773 wiz_defined-function locations,
1106 strings with 33536 characters,
and the built_in function-call counts, 71100 in all, are:
= -- 5525
> -- 7412
< -- 34
+ -- 2481
- -- 2372
* -- 6800
:= -- 11734
add.period$ -- 380
call.type$ -- 98
change.case$ -- 1032
chr.to.int$ -- 87
cite$ -- 197
duplicate$ -- 2238
empty$ -- 3704
format.name$ -- 2482
if$ -- 14216
int.to.chr$ -- 12
int.to.str$ -- 1
missing$ -- 98
newline$ -- 562
num.names$ -- 430
pop$ -- 2052
preamble$ -- 1
purify$ -- 937
quote$ -- 0
skip$ -- 2138
stack$ -- 0
substring$ -- 926
swap$ -- 304
text.length$ -- 19
text.prefix$ -- 0
top$ -- 0
type$ -- 1063
warning$ -- 1
while$ -- 370
width$ -- 0
write$ -- 1394
(There was 1 warning)
\BOOKMARK [1][-]{section.1}{\376\377\000I\000n\000t\000r\000o\000d\000u\000c\000t\000i\000o\000n}{}% 1
\BOOKMARK [1][-]{section.2}{\376\377\000P\000r\000e\000l\000i\000m\000i\000n\000a\000r\000y}{}% 2
\BOOKMARK [2][-]{subsection.2.1}{\376\377\000F\000u\000n\000d\000a\000m\000e\000n\000t\000a\000l\000s\000\040\000o\000f\000\040\000L\000a\000r\000g\000e\000\040\000L\000a\000n\000g\000u\000a\000g\000e\000\040\000M\000o\000d\000e\000l\000s}{section.2}% 3
\BOOKMARK [2][-]{subsection.2.2}{\376\377\000F\000u\000n\000d\000a\000m\000e\000n\000t\000a\000l\000s\000\040\000o\000f\000\040\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g}{section.2}% 4
\BOOKMARK [3][-]{subsubsection.2.2.1}{\376\377\000G\000e\000n\000e\000r\000a\000l\000\040\000F\000r\000a\000m\000e\000w\000o\000r\000k\000\040\000o\000f\000\040\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g}{subsection.2.2}% 5
\BOOKMARK [3][-]{subsubsection.2.2.2}{\376\377\000K\000e\000y\000\040\000E\000l\000e\000m\000e\000n\000t\000s}{subsection.2.2}% 6
\BOOKMARK [1][-]{section.3}{\376\377\000A\000n\000\040\000E\000x\000a\000m\000p\000l\000e\000\040\000o\000f\000\040\000U\000s\000i\000n\000g\000\040\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g\000\040\000t\000o\000\040\000T\000r\000a\000i\000n\000\040\000L\000L\000M\000s}{}% 7
\BOOKMARK [2][-]{subsection.3.1}{\376\377\000P\000o\000l\000i\000c\000y\000\040\000G\000r\000a\000d\000i\000e\000n\000t}{section.3}% 8
\BOOKMARK [2][-]{subsection.3.2}{\376\377\000T\000e\000m\000p\000o\000r\000a\000l\000\040\000D\000e\000c\000o\000m\000p\000o\000s\000i\000t\000i\000o\000n}{section.3}% 9
\BOOKMARK [2][-]{subsection.3.3}{\376\377\000R\000e\000d\000u\000c\000i\000n\000g\000\040\000G\000r\000a\000d\000i\000e\000n\000t\000\040\000V\000a\000r\000i\000a\000n\000c\000e}{section.3}% 10
\BOOKMARK [2][-]{subsection.3.4}{\376\377\000I\000m\000p\000o\000r\000t\000a\000n\000c\000e\000\040\000S\000a\000m\000p\000l\000i\000n\000g}{section.3}% 11
\BOOKMARK [2][-]{subsection.3.5}{\376\377\000P\000r\000o\000x\000i\000m\000a\000l\000\040\000P\000o\000l\000i\000c\000y\000\040\000O\000p\000t\000i\000m\000i\000z\000a\000t\000i\000o\000n}{section.3}% 12
\BOOKMARK [2][-]{subsection.3.6}{\376\377\000T\000r\000a\000i\000n\000i\000n\000g\000\040\000R\000e\000w\000a\000r\000d\000\040\000M\000o\000d\000e\000l\000s}{section.3}% 13
\BOOKMARK [2][-]{subsection.3.7}{\376\377\000A\000\040\000C\000o\000m\000p\000l\000e\000t\000e\000\040\000W\000o\000r\000k\000f\000l\000o\000w}{section.3}% 14
\BOOKMARK [1][-]{section.4}{\376\377\000I\000m\000p\000r\000o\000v\000e\000d\000\040\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g\000\040\000f\000o\000r\000\040\000L\000L\000M\000s}{}% 15
\BOOKMARK [2][-]{subsection.4.1}{\376\377\000A\000d\000v\000a\000n\000c\000e\000d\000\040\000R\000e\000w\000a\000r\000d\000\040\000M\000o\000d\000e\000l\000s}{section.4}% 16
\BOOKMARK [3][-]{subsubsection.4.1.1}{\376\377\000A\000u\000t\000o\000m\000a\000t\000i\000c\000\040\000P\000r\000e\000f\000e\000r\000e\000n\000c\000e\000\040\000D\000a\000t\000a\000\040\000G\000e\000n\000e\000r\000a\000t\000i\000o\000n}{subsection.4.1}% 17
\BOOKMARK [3][-]{subsubsection.4.1.2}{\376\377\000R\000e\000w\000a\000r\000d\000\040\000S\000h\000a\000p\000i\000n\000g}{subsection.4.1}% 18
\BOOKMARK [3][-]{subsubsection.4.1.3}{\376\377\000I\000m\000p\000r\000o\000v\000e\000d\000\040\000R\000e\000w\000a\000r\000d\000\040\000G\000e\000n\000e\000r\000a\000l\000i\000z\000a\000t\000i\000o\000n}{subsection.4.1}% 19
\BOOKMARK [3][-]{subsubsection.4.1.4}{\376\377\000G\000e\000n\000e\000r\000a\000t\000i\000v\000e\000\040\000R\000e\000w\000a\000r\000d\000\040\000M\000o\000d\000e\000l\000s}{subsection.4.1}% 20
\BOOKMARK [2][-]{subsection.4.2}{\376\377\000B\000e\000t\000t\000e\000r\000\040\000A\000d\000v\000a\000n\000t\000a\000g\000e\000\040\000E\000s\000t\000i\000m\000a\000t\000i\000o\000n}{section.4}% 21
\BOOKMARK [3][-]{subsubsection.4.2.1}{\376\377\000T\000e\000m\000p\000o\000r\000a\000l\000\040\000D\000i\000f\000f\000e\000r\000e\000n\000c\000e\000-\000b\000a\000s\000e\000d\000\040\000A\000d\000v\000a\000n\000t\000a\000g\000e\000\040\000E\000s\000t\000i\000m\000a\000t\000i\000o\000n}{subsection.4.2}% 22
\BOOKMARK [3][-]{subsubsection.4.2.2}{\376\377\000G\000e\000n\000e\000r\000a\000l\000i\000z\000e\000d\000\040\000A\000d\000v\000a\000n\000t\000a\000g\000e\000\040\000E\000s\000t\000i\000m\000a\000t\000i\000o\000n}{subsection.4.2}% 23
\BOOKMARK [3][-]{subsubsection.4.2.3}{\376\377\000G\000r\000o\000u\000p\000\040\000R\000e\000l\000a\000t\000i\000v\000e\000\040\000P\000o\000l\000i\000c\000y\000\040\000O\000p\000t\000i\000m\000i\000z\000a\000t\000i\000o\000n}{subsection.4.2}% 24
\BOOKMARK [2][-]{subsection.4.3}{\376\377\000E\000f\000f\000i\000c\000i\000e\000n\000t\000\040\000R\000L\000\040\000M\000e\000t\000h\000o\000d\000s}{section.4}% 25
\BOOKMARK [3][-]{subsubsection.4.3.1}{\376\377\000D\000y\000n\000a\000m\000i\000c\000\040\000S\000a\000m\000p\000l\000i\000n\000g}{subsection.4.3}% 26
\BOOKMARK [3][-]{subsubsection.4.3.2}{\376\377\000L\000i\000g\000h\000t\000w\000e\000i\000g\000h\000t\000\040\000R\000e\000w\000a\000r\000d\000\040\000M\000e\000t\000h\000o\000d\000s}{subsection.4.3}% 27
\BOOKMARK [2][-]{subsection.4.4}{\376\377\000D\000i\000r\000e\000c\000t\000\040\000P\000r\000e\000f\000e\000r\000e\000n\000c\000e\000\040\000O\000p\000t\000i\000m\000i\000z\000a\000t\000i\000o\000n}{section.4}% 28
\BOOKMARK [1][-]{section.5}{\376\377\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g\000\040\000f\000o\000r\000\040\000L\000L\000M\000\040\000R\000e\000a\000s\000o\000n\000i\000n\000g}{}% 29
\BOOKMARK [2][-]{subsection.5.1}{\376\377\000T\000e\000s\000t\000-\000t\000i\000m\000e\000\040\000S\000c\000a\000l\000i\000n\000g}{section.5}% 30
\BOOKMARK [3][-]{subsubsection.5.1.1}{\376\377\000B\000e\000s\000t\000-\000o\000f\000-\000N\000\040\000S\000a\000m\000p\000l\000i\000n\000g}{subsection.5.1}% 31
\BOOKMARK [3][-]{subsubsection.5.1.2}{\376\377\000S\000t\000e\000p\000-\000b\000y\000-\000s\000t\000e\000p\000\040\000V\000e\000r\000i\000f\000i\000c\000a\000t\000i\000o\000n}{subsection.5.1}% 32
\BOOKMARK [3][-]{subsubsection.5.1.3}{\376\377\000M\000o\000n\000t\000e\000\040\000C\000a\000r\000l\000o\000\040\000T\000r\000e\000e\000\040\000S\000e\000a\000r\000c\000h}{subsection.5.1}% 33
\BOOKMARK [2][-]{subsection.5.2}{\376\377\000L\000a\000r\000g\000e\000-\000s\000c\000a\000l\000e\000\040\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g}{section.5}% 34
\BOOKMARK [2][-]{subsection.5.3}{\376\377\000I\000t\000e\000r\000a\000t\000i\000v\000e\000\040\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g}{section.5}% 35
\BOOKMARK [1][-]{section.6}{\376\377\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g\000\040\000f\000o\000r\000\040\000L\000L\000M\000-\000b\000a\000s\000e\000d\000\040\000A\000g\000e\000n\000t\000s}{}% 36
\BOOKMARK [1][-]{section.7}{\376\377\000R\000e\000i\000n\000f\000o\000r\000c\000e\000m\000e\000n\000t\000\040\000L\000e\000a\000r\000n\000i\000n\000g\000\040\000f\000o\000r\000\040\000M\000u\000l\000t\000i\000m\000o\000d\000a\000l\000\040\000M\000o\000d\000e\000l\000s}{}% 37
\BOOKMARK [2][-]{subsection.7.1}{\376\377\000V\000i\000s\000u\000a\000l\000\040\000L\000a\000n\000g\000u\000a\000g\000e\000\040\000M\000o\000d\000e\000l\000s}{section.7}% 38
\BOOKMARK [2][-]{subsection.7.2}{\376\377\000S\000p\000e\000e\000c\000h\000\040\000G\000e\000n\000e\000r\000a\000t\000i\000o\000n\000\040\000M\000o\000d\000e\000l\000s}{section.7}% 39
\BOOKMARK [2][-]{subsection.7.3}{\376\377\000D\000i\000f\000f\000u\000s\000i\000o\000n\000\040\000M\000o\000d\000e\000l\000s}{section.7}% 40
\BOOKMARK [1][-]{section.8}{\376\377\000S\000u\000m\000m\000a\000r\000y}{}% 41
\BOOKMARK [1][-]{section.9}{\376\377\000S\000y\000s\000t\000e\000m\000s\000\040\000a\000n\000d\000\040\000D\000a\000t\000a\000s\000e\000t\000s}{}% 42
...@@ -45,6 +45,7 @@ ...@@ -45,6 +45,7 @@
\usepackage{fontawesome5} \usepackage{fontawesome5}
\usepackage{varwidth} \usepackage{varwidth}
\usetikzlibrary{shadows} \usetikzlibrary{shadows}
\usetikzlibrary{arrows.meta,positioning,calc,fit,backgrounds}
\definecolor{ocrebase}{RGB}{243,102,25} \definecolor{ocrebase}{RGB}{243,102,25}
\definecolor{ocre}{RGB}{0,0,0} \definecolor{ocre}{RGB}{0,0,0}
\definecolor{amber}{rgb}{1.0, 0.75, 0.0} \definecolor{amber}{rgb}{1.0, 0.75, 0.0}
...@@ -167,16 +168,16 @@ However, RL was not so popular in the long history of NLP, and the field has jus ...@@ -167,16 +168,16 @@ However, RL was not so popular in the long history of NLP, and the field has jus
Of course, we'd like to learn about RL, which appears straightforward. However, common RL textbooks, such as \citet{Sutton-and-Barto:2018RL}'s book, are mostly based on classic robotics or control problems. This makes it difficult to align the concepts of RL with those of LLMs. On the other hand, most LLM literature lacks an in-depth discussion of RL techniques, often omitting related details. As a result, we find ourselves in an awkward middle ground between RL and LLMs, struggling to bridge the two fields. Of course, we'd like to learn about RL, which appears straightforward. However, common RL textbooks, such as \citet{Sutton-and-Barto:2018RL}'s book, are mostly based on classic robotics or control problems. This makes it difficult to align the concepts of RL with those of LLMs. On the other hand, most LLM literature lacks an in-depth discussion of RL techniques, often omitting related details. As a result, we find ourselves in an awkward middle ground between RL and LLMs, struggling to bridge the two fields.
The aim of this paper is to provide a comprehensive introduction to RL from the perspective of LLMs. We begin by introducing basic concepts and algorithms of RL using the LLM language. In particular, we illustrate their application through an example of training LLMs, making the concepts more accessible. We then discuss a series of refinements to the basic RL framework for LLM alignment, including advanced reward modeling, improved advantage estimation, efficient sampling, and direct LLM optimization without RL. Furthermore, we discuss how to apply RL techniques to LLM reasoning. These methods are closely related to recent LLMs, such as OpenAI o1/o3 and Deepseek R1, which adopt large-scale RL and test-time scaling to significantly advance their reasoning abilities. In addition, we discuss applications of RL to multimodal LLMs to demonstrate how RL can be adapted to various problems. Finally, we conclude by outlining promising future research directions and discussing relevant systems and datasets. The aim of this paper is to provide a comprehensive introduction to RL from the perspective of LLMs. We begin by introducing basic concepts and algorithms of RL using the LLM language. In particular, we illustrate their application through an example of training LLMs, making the concepts more accessible. We then discuss a series of refinements to the basic RL framework for LLM alignment, including advanced reward modeling, improved advantage estimation, efficient sampling, and direct LLM optimization without RL. Furthermore, we discuss how to apply RL techniques to LLM reasoning. These methods are closely related to recent LLMs, such as OpenAI o1/o3 and DeepSeek R1, which adopt large-scale RL and test-time scaling to significantly advance their reasoning abilities. In addition, we discuss applications of RL to multimodal LLMs to demonstrate how RL can be adapted to various problems. Finally, we conclude by outlining promising future research directions and discussing relevant systems and datasets.
% \clearpage % \clearpage
% New section: training LLMs with reinforcement learning % New section: training LLMs with reinforcement learning
% \input{section2/section2} \input{section2/section2}
% \clearpage % \clearpage
% New section: training LLMs with reinforcement learning % New section: training LLMs with reinforcement learning
% \input{section3/section3} \input{section3/section3}
% \clearpage % \clearpage
% New section: improved reinforcement learning for LLMs % New section: improved reinforcement learning for LLMs
...@@ -185,19 +186,23 @@ The aim of this paper is to provide a comprehensive introduction to RL from the ...@@ -185,19 +186,23 @@ The aim of this paper is to provide a comprehensive introduction to RL from the
% \clearpage % \clearpage
% New section: reinforcement learning for LLM inference % New section: reinforcement learning for LLM inference
% \input{section5/section5} \input{section5/section5}
% \clearpage % \clearpage
% New section: reinforcement learning for MLLMs % New section: reinforcement learning for MLLMs
% \input{section6/section6} \input{section6/section6}
% \clearpage
% New section: reinforcement learning for MLLMs
\input{section7/section7}
% \clearpage % \clearpage
% conclusion % conclusion
% \input{section7/section7} \input{section8/section8}
% \clearpage % \clearpage
% systems & datasets todo: ganyang % systems & datasets todo: ganyang
% \input{section8/section8} \input{section9/section9}
\clearpage \clearpage
......
\contentsline {section}{\numberline {1}Introduction}{3}{section.1}%
\contentsline {section}{\numberline {2}Preliminary}{3}{section.2}%
\contentsline {subsection}{\numberline {2.1}Fundamentals of Large Language Models}{3}{subsection.2.1}%
\contentsline {subsection}{\numberline {2.2}Fundamentals of Reinforcement Learning}{4}{subsection.2.2}%
\contentsline {subsubsection}{\numberline {2.2.1}General Framework of Reinforcement Learning}{4}{subsubsection.2.2.1}%
\contentsline {subsubsection}{\numberline {2.2.2}Key Elements}{5}{subsubsection.2.2.2}%
\contentsline {section}{\numberline {3}An Example of Using Reinforcement Learning to Train LLMs}{6}{section.3}%
\contentsline {subsection}{\numberline {3.1}Policy Gradient}{7}{subsection.3.1}%
\contentsline {subsection}{\numberline {3.2}Temporal Decomposition}{9}{subsection.3.2}%
\contentsline {subsection}{\numberline {3.3}Reducing Gradient Variance}{10}{subsection.3.3}%
\contentsline {subsection}{\numberline {3.4}Importance Sampling}{12}{subsection.3.4}%
\contentsline {subsection}{\numberline {3.5}Proximal Policy Optimization}{14}{subsection.3.5}%
\contentsline {subsection}{\numberline {3.6}Training Reward Models}{15}{subsection.3.6}%
\contentsline {subsection}{\numberline {3.7}A Complete Workflow}{18}{subsection.3.7}%
\contentsline {section}{\numberline {4}Improved Reinforcement Learning for LLMs}{20}{section.4}%
\contentsline {subsection}{\numberline {4.1}Advanced Reward Models}{20}{subsection.4.1}%
\contentsline {subsubsection}{\numberline {4.1.1}Automatic Preference Data Generation}{20}{subsubsection.4.1.1}%
\contentsline {subsubsection}{\numberline {4.1.2}Reward Shaping}{21}{subsubsection.4.1.2}%
\contentsline {subsubsection}{\numberline {4.1.3}Improved Reward Generalization}{23}{subsubsection.4.1.3}%
\contentsline {subsubsection}{\numberline {4.1.4}Generative Reward Models}{24}{subsubsection.4.1.4}%
\contentsline {subsection}{\numberline {4.2}Better Advantage Estimation}{26}{subsection.4.2}%
\contentsline {subsubsection}{\numberline {4.2.1}Temporal Difference-based Advantage Estimation}{26}{subsubsection.4.2.1}%
\contentsline {subsubsection}{\numberline {4.2.2}Generalized Advantage Estimation}{26}{subsubsection.4.2.2}%
\contentsline {subsubsection}{\numberline {4.2.3}Group Relative Policy Optimization}{27}{subsubsection.4.2.3}%
\contentsline {subsection}{\numberline {4.3}Efficient RL Methods}{29}{subsection.4.3}%
\contentsline {subsubsection}{\numberline {4.3.1}Dynamic Sampling}{29}{subsubsection.4.3.1}%
\contentsline {subsubsection}{\numberline {4.3.2}Lightweight Reward Methods}{30}{subsubsection.4.3.2}%
\contentsline {subsection}{\numberline {4.4}Direct Preference Optimization}{32}{subsection.4.4}%
\contentsline {section}{\numberline {5}Reinforcement Learning for LLM Reasoning}{34}{section.5}%
\contentsline {subsection}{\numberline {5.1}Test-time Scaling}{35}{subsection.5.1}%
\contentsline {subsubsection}{\numberline {5.1.1}Best-of-N Sampling}{35}{subsubsection.5.1.1}%
\contentsline {subsubsection}{\numberline {5.1.2}Step-by-step Verification}{37}{subsubsection.5.1.2}%
\contentsline {subsubsection}{\numberline {5.1.3}Monte Carlo Tree Search}{38}{subsubsection.5.1.3}%
\contentsline {subsection}{\numberline {5.2}Large-scale Reinforcement Learning}{40}{subsection.5.2}%
\contentsline {subsection}{\numberline {5.3}Iterative Reinforcement Learning}{41}{subsection.5.3}%
\contentsline {section}{\numberline {6}Reinforcement Learning for LLM-based Agents}{43}{section.6}%
\contentsline {section}{\numberline {7}Reinforcement Learning for Multimodal Models}{43}{section.7}%
\contentsline {subsection}{\numberline {7.1}Visual Language Models}{43}{subsection.7.1}%
\contentsline {subsection}{\numberline {7.2}Speech Generation Models}{45}{subsection.7.2}%
\contentsline {subsection}{\numberline {7.3}Diffusion Models}{45}{subsection.7.3}%
\contentsline {section}{\numberline {8}Summary}{45}{section.8}%
\contentsline {section}{\numberline {9}Systems and Datasets}{46}{section.9}%
\begin{center}
\begin{tikzpicture}[
font=\small,
% line width=0.9pt,
box/.style={
draw,
rounded corners=3pt,
minimum width=3.1cm,
minimum height=1.35cm,
align=center,
font=\normalsize
},
neuron/.style={
circle,
draw=black!55,
minimum size=7mm,
inner sep=0pt
},
conn/.style={
draw=black!65,
line width=0.5pt
},
flow/.style={
draw=black,
-Latex,
line width=0.9pt
}
]
% =========================================================
% Left part: Environment-Agent interaction
% =========================================================
\node[
anchor=center,
draw,
thick,
minimum width=3.0cm,
minimum height=1.0cm,
fill=white,
drop shadow={
fill=gray!40,
shadow xshift=0.6ex,
shadow yshift=-0.6ex
}
] (env) at (0,0) {\large Environment};
\node[anchor=center,draw, minimum width=3.0cm, minimum height=1.0cm, fill=white, drop shadow={
fill=gray!40,
shadow xshift=0.6ex,
shadow yshift=-0.6ex
}
] (agent) at (5.2,0) {\large Agent};
% top arc: state
\draw[->]
($(env.north west)+(1.6,0.05)$)
to[out=45,in=135]
node[pos=0.15, above, font=\large] {$S_t$}
($(agent.north east)+(-1.6,0.05)$);
% middle arc: reward
\draw[->]
($(env.north east)+(0.02,0.10)$)
to[out=20,in=160]
node[midway, above=2pt] {\large $r_t$}
($(agent.north west)+(-0.02,0.10)$);
% bottom arc: action
\draw[->]
($(agent.south west)+(-0.02,-0.10)$)
to[out=200,in=340]
node[midway, below=2pt] {\large $a_t$}
($(env.south east)+(0.02,-0.10)$);
% connector to policy net
\draw[dashed, line width=0.9pt, ->]
($(agent.east)+(0.10,0.00)$)
.. controls +(1.0,0.45) and +(-0.9,0.25)
.. ($(8.6,0.15)$);
% =========================================================
% Right part: Policy network
% =========================================================
% input layer
\foreach \i/\x in {1/9.3,2/10.1,3/10.9,4/11.7}
{
\node[neuron, fill=gray!25, minimum size=4mm] (I\i) at (\x,-1.15) {};
}
% hidden layer
\foreach \i/\x in {1/9.1,2/9.8,3/10.5,4/11.2,5/11.9}
{
\node[neuron, fill=blue!30, minimum size=4mm] (H\i) at (\x,0.05) {};
}
% output layer
\foreach \i/\x in {2/9.8,4/11.2}
{
\node[neuron, fill=yellow!45, minimum size=4mm] (O\i) at (\x,1.35) {};
}
% connections: input -> hidden
\foreach \i in {1,2,3,4}
{
\foreach \j in {1,2,3,4,5}
{
\draw[conn] (I\i) -- (H\j);
}
}
% connections: hidden -> output
\foreach \i in {1,2,3,4,5}
{
\foreach \j in {2,4}
{
\draw[conn] (H\i) -- (O\j);
}
}
% arrows below input layer
\foreach \i in {1,2,3,4}
{
\draw [->] ($(I\i.south)+(0,-0.36)$) -- ($(I\i.south)+(0,-0.02)$);
}
% \draw [->] ([yshift=0.2cm]\x.center) -- ([yshift=0.55cm]\x.center);
% arrows above output layer
\foreach \i in {2,4}
{
\draw[->] ($(O\i.north)+(0,0.02)$) -- ($(O\i.north)+(0,0.42)$);
}
% dashed rounded frame around policy net
\node[
draw=gray!70,
dashed,
rounded corners=5pt,
minimum width=5.0cm,
minimum height=4.0cm,
inner xsep=0.85cm,
inner ysep=1.00cm
] (policybox) at (11.2, 0.2){};
% labels
\node[anchor=west] at (10.6, -2.2) {Policy};
\node[anchor=center, font=\small] at ($(policybox.east)+(-0.9, 1.15)$) {Output};
\node[anchor=center, font=\small] at ($(policybox.east)+(-0.9, -0.1)$) {Hidden};
\node[anchor=center, font=\small] at ($(policybox.east)+(-0.9, -1.4)$) {Input};
\end{tikzpicture}
\end{center}
\ No newline at end of file
\section{Preliminary} \section{Preliminary}
In this section, we introduce the fundamentals of supervised fine-tuning for LLMs and focus on presenting the key concepts and nations necessary for the discussions in future sections.
\subsection{Fundamentals of Large Language Models}
In this subsection, we introduce the fundamentals of supervised fine-tuning for LLMs and focus on presenting the key concepts and nations necessary for the discussions in future sections.
Although pre-trained LLMs possess a vast amount of general knowledge, they are typically limited to tasks related to language modeling. Our ultimate goal is to enable them to perform various specific tasks, such as summarization, question-answer, and machine translation. Although pre-trained LLMs possess a vast amount of general knowledge, they are typically limited to tasks related to language modeling. Our ultimate goal is to enable them to perform various specific tasks, such as summarization, question-answer, and machine translation.
One straightforward method to achieve this goal is supervised fine-tuning (SFT), in which the pre-trained LLM is further trained on a dataset comprising task-specific input (i.e., instruction+user input)\footnote{In this paper, the \textit{instruction} represents the description of a specific task, e.g., ``Summarize the following article''; the \textit{user input} represents the extra content to the instruction, e.g., ``Article: In recent years, solar energy has seen a significant increase in the amount of energy available to the public. recent years, solar energy has seen unprecedented growth, becoming the fastest-growing ...''} paired with their expected outputs \citep{ouyang:2022training,wei-etal:2022finetuned}. One straightforward method to achieve this goal is supervised fine-tuning (SFT), in which the pre-trained LLM is further trained on a dataset comprising task-specific input (i.e., instruction+user input)\footnote{In this paper, the \textit{instruction} represents the description of a specific task, e.g., ``Summarize the following article''; the \textit{user input} represents the extra content to the instruction, e.g., ``Article: In recent years, solar energy has seen a significant increase in the amount of energy available to the public. recent years, solar energy has seen unprecedented growth, becoming the fastest-growing ...''} paired with their expected outputs \citep{ouyang:2022training,wei-etal:2022finetuned}.
...@@ -21,10 +23,67 @@ The objective function $\log \mathrm{Pr}_{\theta}(y_i|\mathbf{x},\mathbf{y}_{<i} ...@@ -21,10 +23,67 @@ The objective function $\log \mathrm{Pr}_{\theta}(y_i|\mathbf{x},\mathbf{y}_{<i}
This formulation is equivalent to minimizing the cross-entropy loss. In the following section, we refer to the LLM after SFT as the ``SFT LLM'' for short. Beyond SFT, the model architecture design and the pre-training process are also fundamentals of building an LLM. However, we will not discuss these topics. Interested readers can find further details in \citet{xiao-and-zhu:2025foundations}'s book. This formulation is equivalent to minimizing the cross-entropy loss. In the following section, we refer to the LLM after SFT as the ``SFT LLM'' for short. Beyond SFT, the model architecture design and the pre-training process are also fundamentals of building an LLM. However, we will not discuss these topics. Interested readers can find further details in \citet{xiao-and-zhu:2025foundations}'s book.
%%%%%%%%%% Preliminary修改建议 %%%%%%%%%% \subsection{Fundamentals of Reinforcement Learning}
% 1. 再加一些强化学习的基础知识 RL, supported by a well-established theoretical framework, has been widely applied across a broad range of domains. In particular, with the rapid advancement of LLMs, RL has become a standard approach in the post-training stage. Before exploring the interplay between RL and LLMs, we first provide a brief overview of deep reinforcement learning in this subsection. We then introduce the general formulation of RL, along with key terminologies and notations that are essential for understanding RL.
% 包含对环境、动作、policy这些术语的定义,方面后面对应起来的好定义。
% 如果修改这一章节 那么3.1这部分要对应起来,有些需要删除掉。 \subsubsection{General Framework of Reinforcement Learning}
The general framework of RL is illustrated in Figure~\ref{fig:rl_framework}. An RL system primarily consists of two components: an agent and an environment. At each timestep $t$, the agent observes a state $s_t$ from the environment and selects an action $a_t$ according to a policy $\pi$, which is typically parameterized by a neural network. After executing the action, the environment transitions to a new state $s_{t+1}$ and returns a reward $r_t$ corresponding to the taken action.
\begin{figure}[!t]
\centering
\input{section2/Figures/rl_framework}
\caption{
A general framework of deep reinforcement learning. The policy is parameterized by neural networks, and the agent learns an optimal policy through interactions with the environment to maximize cumulative rewards.
}
\label{fig:rl_framework}
\end{figure}
This interaction process is commonly modeled as a Markov Decision Process (MDP), where the agent interacts with the environment over discrete timesteps \citep{Sutton-and-Barto:2018RL}. A sequence of states and actions forms a trajectory, denoted as $\tau = (s_0, a_0, s_1, a_1, \cdots, s_{H-1}, a_{H-1})$, where $H$ represents the trajectory length. Each trajectory accumulates rewards from the environment.
The objective of RL is to learn an optimal policy $\pi^*$ that maximizes the expected cumulative reward over trajectories, which can be formulated as:
\begin{eqnarray}
\pi^{*} = \mathop{\arg\max} \limits_{\pi} \mathbb{E}\left[\sum_{t=0}^{H-1}r_{t} \mid \pi \right]
\end{eqnarray}
In practice, the agent improves its policy through repeated interactions with the environment by observing states and receiving feedback in the form of rewards. Each such interaction sequence constitutes a trajectory $\tau$. The process of generating one or more trajectories under a given policy is commonly referred to as \textit{sampling} in the RL literature.
\subsubsection{Key Elements}
To better understand the application of reinforcement learning to large language models (LLMs), we summarize the key elements of the RL framework and reinterpret them in the context of language modeling.
\begin{itemize}
\item \textbf{Agent.}
The agent is the learner or decision-maker in reinforcement learning. In the context of LLMs, the agent corresponds to the language model itself, which generates tokens sequentially and updates its behavior based on feedback signals.
\item \textbf{Environment.}
The environment comprises everything external to the agent with which it interacts. Unlike traditional RL settings that involve physical or simulated environments, the environment in LLM-based RL is typically abstract, consisting of the training framework that provides feedback (e.g., reward models, human annotations, or evaluation metrics) for generated outputs.
\item \textbf{State ($S$).}
A state represents the current situation of the environment. For language modeling, the state at timestep $t$ can be defined as the sequence of observed tokens up to that point, i.e., the context used to predict the next token. Formally, the state can be represented as $S = (x, y_{<t})$, where $x$ denotes the input prompt and $y_{<t}$ denotes the previously generated tokens.
\item \textbf{Action ($a$).}
An action corresponds to a decision made by the agent. In LLMs, actions are naturally defined as selecting the next token from the vocabulary, i.e., $a = y_t$.
\item \textbf{Reward ($r$).}
The reward provides feedback from the environment to evaluate the quality of an action. In general, the reward function can be defined as $r(s, a, s')$, representing the feedback received when the agent transitions from state $s$ to $s'$ by taking action $a$. At timestep $t$, this can be written as $r_t = r(s_t, a_t, s_{t+1})$. In deterministic settings, where the next state is uniquely determined by $(s_t, a_t)$, the reward can be simplified as $r(s_t, a_t)$.
\item \textbf{Policy ($\pi$).}
The policy defines the agent’s behavior, i.e., the probability of taking an action given a state. For LLMs, the policy corresponds to the conditional probability distribution over the next token given the context:
\begin{equation}
\pi(a \mid s) = \Pr(y_t \mid x, y_{<t})
\end{equation}
where $a = y_t$ and $s = (x, y_{<t})$. Under this formulation, an LLM can be naturally interpreted as a parameterized policy.
\item \textbf{Value Function ($V$ and $Q$).}
The value function estimates the expected cumulative reward when following a policy. The state-value function $V(s)$ measures the expected discounted return starting from state $s$:
\begin{equation}
V(s) = \mathbb{E} \left[ \sum_{t=0}^{\infty} \gamma^t r_t \middle| s_0 = s, \pi \right]
\end{equation}
where $\gamma \in [0,1]$ is the discount factor. The action-value function $Q(s,a)$ further conditions on the initial action:
\begin{equation}
Q(s,a) = \mathbb{E} \left[ \sum_{t=0}^{\infty} \gamma^t r_t \middle| s_0 = s, a_0 = a, \pi \right]
\end{equation}
\end{itemize}
By this point, the reader should have developed a foundational understanding of the core concepts and notations in RL. In contrast to conventional RL literature, which typically presents algorithms in the context of classical robotics or control tasks, \textit{we adopt an NLP perspective to introduce RL in the following section}.
...@@ -3,7 +3,7 @@ ...@@ -3,7 +3,7 @@
\section{An Example of Using Reinforcement Learning to Train LLMs} \section{An Example of Using Reinforcement Learning to Train LLMs}
\label{sec:example-using-rl-training-llms} \label{sec:example-using-rl-training-llms}
To explain how RL can be applied to train LLMs, we begin by considering a practical scenario: using an LLM as a homework assistant. In this context, we aim to improve the capability of an SFT LLM to handle education-related inputs more effectively. Suppose we have a homework assistant powered by an SFT LLM, and a student types the input ``Give me three tips to improve my accuracy in solving math problems.'' A typical SFT LLM might generate a short output, such as We begin by considering a practical scenario: using an LLM as a homework assistant. In this context, we aim to improve the capability of an SFT LLM to handle education-related inputs more effectively. Suppose we have a homework assistant powered by an SFT LLM, and a student types the input ``Give me three tips to improve my accuracy in solving math problems.'' A typical SFT LLM might generate a short output, such as
\vspace{0.1cm} \vspace{0.1cm}
\begin{tcolorbox}[frame empty] \begin{tcolorbox}[frame empty]
...@@ -23,7 +23,7 @@ There are three tips for improving accuracy in solving math problems: \\ [1mm] ...@@ -23,7 +23,7 @@ There are three tips for improving accuracy in solving math problems: \\ [1mm]
Although this output adheres to the given input, it is very short and not sufficiently detailed or comprehensive for this scenario. In practice, the ``short'' feature in the generated output stems from the training approach of the SFT LLM. During SFT, the LLM learns from a large set of labeled samples that typically contain concise and factual answers to common inputs. These samples guide the LLM to prioritize brevity and clarity, so it often generates short outputs, like the one shown above. Of course, we could annotate enough additional data to fine-tune the LLM further and adjust it to generate the desired outputs. However, this approach is limited in its ability to scale. For example, in this scenario, the input involves a variety of potential answers, and the expectations of the student can be pretty diverse. Therefore, describing the ``ideal'' output would require immense annotation effort. Consequently, collecting or annotating fine-tuning data is not as straightforward as it is with SFT, particularly when attempting to cover the breadth of possible student inputs and outputs. Instead, we can use RL to enable the model to discern outputs that better align with human preferences, such as generating more detailed and comprehensive content that not only adheres to the given input but also meets the expectations of the student in this scenario. Although this output adheres to the given input, it is very short and not sufficiently detailed or comprehensive for this scenario. In practice, the ``short'' feature in the generated output stems from the training approach of the SFT LLM. During SFT, the LLM learns from a large set of labeled samples that typically contain concise and factual answers to common inputs. These samples guide the LLM to prioritize brevity and clarity, so it often generates short outputs, like the one shown above. Of course, we could annotate enough additional data to fine-tune the LLM further and adjust it to generate the desired outputs. However, this approach is limited in its ability to scale. For example, in this scenario, the input involves a variety of potential answers, and the expectations of the student can be pretty diverse. Therefore, describing the ``ideal'' output would require immense annotation effort. Consequently, collecting or annotating fine-tuning data is not as straightforward as it is with SFT, particularly when attempting to cover the breadth of possible student inputs and outputs. Instead, we can use RL to enable the model to discern outputs that better align with human preferences, such as generating more detailed and comprehensive content that not only adheres to the given input but also meets the expectations of the student in this scenario.
In the following sections, we will demonstrate how to use RL to train LLMs through a specific example—enabling the LLM to generate a longer output in the context of a homework assistant. Note that the techniques discussed are not limited to this single case. Rather, they can be broadly applied to align LLM with any set of expectations. In the following sections, we will demonstrate how to use RL to train LLMs through a specific example—enabling the LLM to generate a longer output in the context of a homework assistant. Throughout this example, we introduce RL, including key algorithms and their improvements.
% policy gradient % policy gradient
\subsection{Policy Gradient} \subsection{Policy Gradient}
......
...@@ -151,7 +151,7 @@ where $\alpha$ is a balancing factor, we further illustrate the freezing paramet ...@@ -151,7 +151,7 @@ where $\alpha$ is a balancing factor, we further illustrate the freezing paramet
\subsubsection{Generative Reward Models} \subsubsection{Generative Reward Models}
\label{sec:generative-reward-models} \label{sec:generative-reward-models}
Reward models are typically trained as discriminative models to assign numerical rewards to outputs and classify them as preferred or dispreferred. However, this method does not leverage the text-generation capabilities for which LLMs are fundamentally designed. For example, the discriminative reward model can not perform CoT reasoning. To address this, an LLM can alternatively be employed as a reward model, thus endowing it with the ability to engage in text generation and reasoning, as depicted in Figure \ref{fig:generative-reward-model-architecture}. This model works as follows. First, we input a prompt $\mathbf{c}$, along with the tuple $(\mathbf{x},\mathbf{y}_{a},\mathbf{y}_{b})$, to the LLM. The prompt is a description of the task, as demonstrated in the example below. Reward models are typically trained as discriminative models to assign numerical rewards to outputs and classify them as preferred or dispreferred. However, this method does not leverage the text-generation capabilities for which LLMs are fundamentally designed \citep{zhang-etal:zhang2024generative,wang-etal:wang2025gram}. For example, the discriminative reward model can not perform CoT reasoning. To address this, an LLM can alternatively be employed as a reward model, thus endowing it with the ability to engage in text generation and reasoning, as depicted in Figure \ref{fig:generative-reward-model-architecture}. This model works as follows. First, we input a prompt $\mathbf{c}$, along with the tuple $(\mathbf{x},\mathbf{y}_{a},\mathbf{y}_{b})$, to the LLM. The prompt is a description of the task, as demonstrated in the example below.
\vspace{0.5em} \vspace{0.5em}
\begin{tcolorbox}[frame empty] \begin{tcolorbox}[frame empty]
...@@ -187,7 +187,7 @@ r_{\phi}(\mathbf{x}', \mathbf{y}') & = & \frac{\mathrm{Pr}_{\theta}(w=\text{A}|\ ...@@ -187,7 +187,7 @@ r_{\phi}(\mathbf{x}', \mathbf{y}') & = & \frac{\mathrm{Pr}_{\theta}(w=\text{A}|\
where the reward ranges from 0 to 1. where the reward ranges from 0 to 1.
To further improve the generative reward model, we can label the relevant explanation to enable the generative reward model to produce a CoT rationale \citep{zhang-etal:2024generative}. In this case, we use a prompt $\mathbf{c}$ with a CoT rationale generation instruction, as shown below. To further improve the generative reward model, we can label the relevant explanation to enable the generative reward model to produce a CoT rationale \citep{zhang-etal:2024generative,wang-etal:wang2026gram}. In this case, we use a prompt $\mathbf{c}$ with a CoT rationale generation instruction, as shown below.
\vspace{0.5em} \vspace{0.5em}
\begin{tcolorbox}[frame empty] \begin{tcolorbox}[frame empty]
......
\section{Reinforcement Learning for Multimodal Models} \section{Reinforcement Learning for LLM-based Agents}
Multimodal learning involves models that process and relate information from multiple data modalities (e.g. vision, text, speech), enabling more comprehensive understanding than single-modality systems. By combining modalities, models can make more robust predictions and capture complementary information that one modality alone might miss. Such multimodal approaches have benefits across various applications, such as image caption and image generation. Given these advantages, multimodal learning has become increasingly significant as AI systems aim to perceive and reason more like humans, who naturally integrate sight, sound, and language \citep{baltruvsaitis-etal:2018multimodal}. \ No newline at end of file
In the era of LLM, we can extend their capabilities into the multimodal domain by training them with diverse data types. For example, we can develop a visual language model by training an LLM with image-text pairs using SFT. Here, we return to the issue of aligning models with human preferences. Ideally, once aligned with human preferences through RL, the LLM would also generalize the target modality well by the SFT. However, the reality does not always meet expectations. Although an LLM may align well with human preferences in one aspect, it often struggles to generalize this alignment across a different modality. Thus, to align multimodal models with human preferences, we perform RL to enhance their performance further within the specific modality.
In this section, we use the visual language, speech generation, and diffusion models to discuss the application of RL in multimodal models.
\subsection{Visual Language Models}
While RL is commonly used to train LLMs, its application to other domains has been a prominent research topic. In multimodal language models\footnote{A multimodal language model is defined as a model that integrates an LLM with a multimodal encoder, such as CLIP \citep{radford-etal:2021learning}, allowing the LLM to process inputs beyond text, such as images. Recent literature has also introduced the use of LLMs for generating outputs in non-text modalities, such as images and speech \citep{xu-etal:2025qwen2,zhang-etal:2024mm}. However, in this section, we focus on the former definition of multimodal language models.}, for example, a notable trend is to perform RL training to improve their trustworthiness and helpfulness. This section considers Visual Language Models (VLMs), which connect a visual encoder to an LLM through a linear projector, facilitating general-purpose visual and language understanding. VLMs are currently the most explored extension in multimodal language research, and they form the foundation for many open-source multimodal language models, such as Qwen2.5-VL \citep{qwenTeam:2025qwen2.5-VL} and LLaMA-3.2-11B-Vision \citep{grattafiori-etal:2024llama}.
Training LLMs and VLMs with RL exhibits only minimal differences, primarily related to the input content. Unlike the textual input used for LLMs, the input for VLMs typically comprises a combination of one or multiple images and an instruction, denoted by $(\mathrm{\mathbf{I}}, \mathrm{\mathbf{x}})$, where $\mathrm{\mathbf{I}}$ represents the input images. These images are encoded into representations that are either concatenated with instruction embeddings or integrated through cross-attention mechanisms into the LLM. In practice, this subtle difference does not significantly affect the applicability of RL algorithms to VLMs. As a result, the RL training process for LLMs, as described in Section \ref{sec:example-using-rl-training-llms}, can be seamlessly adapted to train VLMs without major improvements \citep{yu-etal:2024rlhf,wang-etal:2024rovrm,zang2025internlm,ji2025safe}.
\begin{figure*}[!t]
\centering
\caption{RoVRM}
\label{fig:preference-transfer}
\end{figure*}
However, training VLMs with RL is not a low-hanging fruit in practical applications. This is because it typically encounters the challenge of training a visual reward model due to the scarcity of high-quality visual preference data. One straightforward approach to address this issue is to generate visual preference data through the automatic preference data generation method described in Section \ref{sec:automatic-preference-data-generation} \citep{yu-etal:2024rlaif}. Another alternative is a multi-stage training approach for the visual reward model, motivated by a simple idea: human preferences are well captured in text, and these preferences can be transferred across modalities \citep{wang-etal:2024rovrm}. By leveraging textual preference data, this approach reduces the dependence on visual preference data in training a visual reward model. More specifically, as illustrated in Figure \ref{fig:preference-transfer}, we can train a visual reward model in the following three stages:
\begin{itemize}
\item Stage 1: pre-training with large-scale textual preference data. Given the transferability of human preferences across different modalities, we can begin by using large-scale textual preference data to pre-train the visual reward model. Note that the projector parameters are frozen without images. This stage can be considered as providing a stronger starting point for training the visual reward model, as it enables the model to pre-learn general human preferences.
\item Stage 2: fine-tuning with image caption-based preference data. Pre-learned human preferences cannot be directly applied to vision tasks due to both \textit{task gap} and \textit{modality gap}. To bridge the task gap, we fine-tune the reward model using image caption-based preference data in this stage. The rationale behind this approach is that general textual preference data does not cover vision-specific tasks, such as ``Please describe the content in this image''. We use such data to fine-tune the model, adapting it to vision-specific tasks. The projector parameters are frozen at this stage.
\item Stage 3: fine-tuning with small-scale visual preference data. To further bridge the modality gap, we use visual preference data in the final stage to train the model. Note that in this phase, we also train the projector parameters.
\end{itemize}
In this process, not all preference data may align with the preferences used in subsequent phases, potentially leading to preference conflicts. We can enhance the visual reward model through data selection techniques, such as LESS \citep{xia-etal:2024less} to address this. In fact, preference transfer is effective across modalities and has also been shown to work across different tasks and languages \citep{cheng-etal:2023everyone,wu-etal:2024reuse}. Interested readers can refer to these papers for more detailed discussions of these topics.
Apart from preference data, another approach to improving the visual model is to integrate additional image content, such as image captions, into the reward model \citep{sun-etal:2023aligning}. This approach aims to achieve factually augmented reward prediction, addressing reward hacking. Specifically, in the original setup, the reward model predicts a reward based solely on the image, input, and output; that is, the reward model’s input is $[\mathbf{I}, \mathbf{x}, \mathbf{y}]$. In the factually augmented setup, the reward model also receives additional input in the form of the textual image caption $\mathbf{C}$, resulting in an input of $[\mathbf{I}, \mathbf{C}, \mathbf{x}, \mathbf{y}]$. The basic idea is that the backbone of the visual reward model remains a well-trained LLM with an in-context learning ability. With this ability, we can provide additional content to help the model predict rewards more accurately.
While our discussion primarily focuses on models for visual inputs in the context of visual language models, the RL techniques are adaptable across inputs of various modalities, such as video and audio.
% visual language model as reward model
% 视觉奖励模型训练大致思路
% image reward
% 融合文本数据进行提升
% 视觉奖励模型应用场景---对齐图生文模型 文生图评估模型
% 视觉语言模型的定义
% 框架
% Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback
% Aligning Large Multimodal Models with Factually Augmented RLHF
% 偏好数据
% RLAIF-V
% Silkie: Preference Distillation for Large Visual Language Models
% RoVRM
\subsection{Speech Generation Models}
% SpeechAlign: Aligning Speech Generation to Human Preferences
\subsection{Diffusion Models}
% Image Reward
% 为什么要进行强化学习
% 大体训练思路和流程
% DDPO (denoising diffusion preference optimization model)
\section{Summary}
% future work \section{Reinforcement Learning for Multimodal Models}
% 强化学习作为一种预训练方式 Multimodal learning involves models that process and relate information from multiple data modalities (e.g. vision, text, speech), enabling more comprehensive understanding than single-modality systems. By combining modalities, models can make more robust predictions and capture complementary information that one modality alone might miss. Such multimodal approaches have benefits across various applications, such as image caption and image generation. Given these advantages, multimodal learning has become increasingly significant as AI systems aim to perceive and reason more like humans, who naturally integrate sight, sound, and language \citep{baltruvsaitis-etal:2018multimodal}.
% 高效强化学习方法
% 全模态RL In the era of LLM, we can extend their capabilities into the multimodal domain by training them with diverse data types. For example, we can develop a visual language model by training an LLM with image-text pairs using SFT. Here, we return to the issue of aligning models with human preferences. Ideally, once aligned with human preferences through RL, the LLM would also generalize the target modality well by the SFT. However, the reality does not always meet expectations. Although an LLM may align well with human preferences in one aspect, it often struggles to generalize this alignment across a different modality. Thus, to align multimodal models with human preferences, we perform RL to enhance their performance further within the specific modality.
\ No newline at end of file
In this section, we use the visual language, speech generation, and diffusion models to discuss the application of RL in multimodal models.
\subsection{Visual Language Models}
While RL is commonly used to train LLMs, its application to other domains has been a prominent research topic. In multimodal language models\footnote{A multimodal language model is defined as a model that integrates an LLM with a multimodal encoder, such as CLIP \citep{radford-etal:2021learning}, allowing the LLM to process inputs beyond text, such as images. Recent literature has also introduced the use of LLMs for generating outputs in non-text modalities, such as images and speech \citep{xu-etal:2025qwen2,zhang-etal:2024mm}. However, in this section, we focus on the former definition of multimodal language models.}, for example, a notable trend is to perform RL training to improve their trustworthiness and helpfulness. This section considers Visual Language Models (VLMs), which connect a visual encoder to an LLM through a linear projector, facilitating general-purpose visual and language understanding. VLMs are currently the most explored extension in multimodal language research, and they form the foundation for many open-source multimodal language models, such as Qwen2.5-VL \citep{qwenTeam:2025qwen2.5-VL} and LLaMA-3.2-11B-Vision \citep{grattafiori-etal:2024llama}.
Training LLMs and VLMs with RL exhibits only minimal differences, primarily related to the input content. Unlike the textual input used for LLMs, the input for VLMs typically comprises a combination of one or multiple images and an instruction, denoted by $(\mathrm{\mathbf{I}}, \mathrm{\mathbf{x}})$, where $\mathrm{\mathbf{I}}$ represents the input images. These images are encoded into representations that are either concatenated with instruction embeddings or integrated through cross-attention mechanisms into the LLM. In practice, this subtle difference does not significantly affect the applicability of RL algorithms to VLMs. As a result, the RL training process for LLMs, as described in Section \ref{sec:example-using-rl-training-llms}, can be seamlessly adapted to train VLMs without major improvements \citep{yu-etal:2024rlhf,wang-etal:2024rovrm,zang2025internlm,ji2025safe}.
However, training VLMs with RL is not a low-hanging fruit in practical applications. This is because it typically encounters the challenge of training a visual reward model due to the scarcity of high-quality visual preference data. One straightforward approach to address this issue is to generate visual preference data through the automatic preference data generation method described in Section \ref{sec:automatic-preference-data-generation} \citep{yu-etal:2024rlaif}. Another alternative is a multi-stage training approach for the visual reward model, motivated by a simple idea: human preferences are well captured in text, and these preferences can be transferred across modalities \citep{wang-etal:2024rovrm}. By leveraging textual preference data, this approach reduces the dependence on visual preference data in training a visual reward model. More specifically, as illustrated in Figure \ref{fig:preference-transfer}, we can train a visual reward model in the following three stages:
\begin{itemize}
\item Stage 1: pre-training with large-scale textual preference data. Given the transferability of human preferences across different modalities, we can begin by using large-scale textual preference data to pre-train the visual reward model. Note that the projector parameters are frozen without images. This stage can be considered as providing a stronger starting point for training the visual reward model, as it enables the model to pre-learn general human preferences.
\item Stage 2: fine-tuning with image caption-based preference data. Pre-learned human preferences cannot be directly applied to vision tasks due to both \textit{task gap} and \textit{modality gap}. To bridge the task gap, we fine-tune the reward model using image caption-based preference data in this stage. The rationale behind this approach is that general textual preference data does not cover vision-specific tasks, such as ``Please describe the content in this image''. We use such data to fine-tune the model, adapting it to vision-specific tasks. The projector parameters are frozen at this stage.
\item Stage 3: fine-tuning with small-scale visual preference data. To further bridge the modality gap, we use visual preference data in the final stage to train the model. Note that in this phase, we also train the projector parameters.
\end{itemize}
In this process, not all preference data may align with the preferences used in subsequent phases, potentially leading to preference conflicts. We can enhance the visual reward model through data selection techniques, such as LESS \citep{xia-etal:2024less} to address this. In fact, preference transfer is effective across modalities and has also been shown to work across different tasks and languages \citep{cheng-etal:2023everyone,wu-etal:2024reuse}. Interested readers can refer to these papers for more detailed discussions of these topics.
As discussed in Section~\ref{sec:generative-reward-models}, reward reasoning models have demonstrated superior performance in reward prediction. A natural question then arises: \textit{can reward reasoning capabilities also be transferred from text to multimodal settings?} Recent work by \cite{wang2026msrl} provides empirical evidence supporting this hypothesis. By following a similar multi-stage training paradigm, they show that reward reasoning capabilities can indeed be effectively transferred across modalities, further enhancing the performance of visual reward models.
\begin{figure*}[!t]
\centering
\caption{RoVRM}
\label{fig:preference-transfer}
\end{figure*}
Apart from preference data, another approach to improving the visual model is to integrate additional image content, such as image captions, into the reward model \citep{sun-etal:2023aligning}. This approach aims to achieve factually augmented reward prediction, addressing reward hacking. Specifically, in the original setup, the reward model predicts a reward based solely on the image, input, and output; that is, the reward model’s input is $[\mathbf{I}, \mathbf{x}, \mathbf{y}]$. In the factually augmented setup, the reward model also receives additional input in the form of the textual image caption $\mathbf{C}$, resulting in an input of $[\mathbf{I}, \mathbf{C}, \mathbf{x}, \mathbf{y}]$. The basic idea is that the backbone of the visual reward model remains a well-trained LLM with an in-context learning ability. With this ability, we can provide additional content to help the model predict rewards more accurately.
While our discussion primarily focuses on models for visual inputs in the context of visual language models, the RL techniques are adaptable across inputs of various modalities, such as video and audio.
% visual language model as reward model
% 视觉奖励模型训练大致思路
% image reward
% 融合文本数据进行提升
% 视觉奖励模型应用场景---对齐图生文模型 文生图评估模型
% 视觉语言模型的定义
% 框架
% Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback
% Aligning Large Multimodal Models with Factually Augmented RLHF
% 偏好数据
% RLAIF-V
% Silkie: Preference Distillation for Large Visual Language Models
% RoVRM
\subsection{Speech Generation Models}
% SpeechAlign: Aligning Speech Generation to Human Preferences
\subsection{Diffusion Models}
% Image Reward
% 为什么要进行强化学习
% 大体训练思路和流程
% DDPO (denoising diffusion preference optimization model)
\clearpage \section{Summary}
\section{Systems and Datasets}
% future work
\begin{table}[h] % 强化学习作为一种预训练方式
\centering % 高效强化学习方法
\scalebox{0.88}{ % 全模态RL
\input{section8/tables/dataset}} \ No newline at end of file
\caption{datasets}
\label{tab:dataset}
\end{table}
\begin{table}[h]
\centering
\scalebox{0.88}{
\input{section8/tables/systems}}
\caption{systems}
\label{tab:system}
\end{table}
\ No newline at end of file
\clearpage
\section{Systems and Datasets}
\begin{table}[h]
\centering
\scalebox{0.88}{
\input{section9/tables/dataset}}
\caption{datasets}
\label{tab:dataset}
\end{table}
\begin{table}[h]
\centering
\scalebox{0.88}{
\input{section9/tables/systems}}
\caption{systems}
\label{tab:system}
\end{table}
\ No newline at end of file
\begin{tabular}{lccccc}
\toprule[1.1pt]
\multirow{2}{*}{Dataset Name} & \multirow{2}{*}{\begin{tabular}[c]{@{}c@{}}Sample\\Size\end{tabular}} & \multirow{2}{*}{\begin{tabular}[c]{@{}c@{}}Response\\Size\end{tabular}} & \multirow{2}{*}{Modality} & \multicolumn{2}{c}{Feedback} \\ \cmidrule(l){5-6}
& & & & Source & Category \\ \midrule
\href{https://url}{HuggingFaceH4/stack-exchange-preferences}
&10,000 &1 & \faIcon{file-alt} & Human & Score \\
\href{https://url}{Skywork/Skywork-Reward-Preference-80K-v0.2} &10,000 &2 & \faIcon{images} & AI & Ranking \\
\href{https://huggingface.co/datasets/openbmb/UltraFeedback}{openbmb/UltraFeedback} &10,000 &3 & \faIcon{video} & Rule & \\
&10,000 &4 & \faIcon{volume-up} & & \\
& & &\faIcon{file-alt} \faIcon{images} & & \\
& & & & & \\
& & & & & \\
& & & & & \\
\bottomrule[1.1pt]
\end{tabular}
\ No newline at end of file
\begin{tabular}{lccccc}
\toprule[1.1pt]
\multirow{2}{*}{System Name} & \multicolumn{4}{c}{Supported Modality} & \multirow{2}{*}{Supported Training Approaches} \\ \cmidrule(r){2-5}
& Text & Image & Video & Audio & \\ \midrule
\href{https://github.com/huggingface/trl}{TRL} & \CheckmarkBold & \CheckmarkBold &\XSolidBrush & \CheckmarkBold & Reward Modeling, PPO, GRPO, DPO, Online-DPO, etc. \\
\href{https://github.com/OpenRLHF/OpenRLHF}{OpenRLHF}& \CheckmarkBold & \XSolidBrush & \XSolidBrush & \XSolidBrush & Sft, Reject Sampling, PPO, GRPO, DPO, KTO, etc.\\
\href{https://github.com/hiyouga/EasyR1}{EasyR1}& \CheckmarkBold & \CheckmarkBold & \XSolidBrush & \XSolidBrush & GRPO, Reinforce++, Remax, RLOO, etc. \\
\href{https://github.com/volcengine/verl}{veRL}& \CheckmarkBold & \XSolidBrush & \XSolidBrush & \XSolidBrush & GRPO, PPO, Remax, RLOO, SFT, etc. \\
\href{https://github.com/OpenRLHF/OpenRLHF-M}{OpenRLHF-M}& \XSolidBrush & \CheckmarkBold & \XSolidBrush & \XSolidBrush & PPO, GRPO, RLOO, Online-RLHF, Reject-Sampling, etc. \\
\href{https://github.com/PKU-Alignment/align-anything}{Align-Anything}& \CheckmarkBold & \CheckmarkBold & \CheckmarkBold & \CheckmarkBold &PPO, GRPO, DPO,KTO, ORPO, etc. \\
\bottomrule[1.1pt]
\end{tabular}
\ No newline at end of file
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论