Commit 6de2c269 by wangchenglong

update.

parent 85256022
\begin{thebibliography}{169} \begin{thebibliography}{173}
\providecommand{\natexlab}[1]{#1} \providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}} \providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax \expandafter\ifx\csname urlstyle\endcsname\relax
...@@ -240,6 +240,10 @@ Mengkang Hu, Pu~Zhao, Can Xu, Qingfeng Sun, Jian-Guang Lou, Qingwei Lin, Ping Lu ...@@ -240,6 +240,10 @@ Mengkang Hu, Pu~Zhao, Can Xu, Qingfeng Sun, Jian-Guang Lou, Qingwei Lin, Ping Lu
\newblock Agentgen: Enhancing planning abilities for large language model based agent via environment and task generation. \newblock Agentgen: Enhancing planning abilities for large language model based agent via environment and task generation.
\newblock In \emph{Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1}, pp.\ 496--507, 2025. \newblock In \emph{Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1}, pp.\ 496--507, 2025.
\bibitem[Huo et~al.(2026)Huo, Xing, Wang, Feng, He, Ding, Ma, Gao, Liu, Xiao, and Zhu]{huo-etal:learning}
Yifu Huo, Shunjie Xing, Chenglong Wang, Peinan Feng, Qiaozhi He, Yan Ding, Anxiang Ma, Yuxin Gao, Tongran Liu, Tong Xiao, and Jingbo Zhu.
\newblock Learning from environmental feedback: Credit assignment across multiple timescales for agentic reinforcement learning, 2026.
\bibitem[Ji et~al.(2025)Ji, Chen, Pan, Zhu, Zhang, Li, Hong, Chen, Zhou, Wang, et~al.]{ji2025safe} \bibitem[Ji et~al.(2025)Ji, Chen, Pan, Zhu, Zhang, Li, Hong, Chen, Zhou, Wang, et~al.]{ji2025safe}
Jiaming Ji, Xinyu Chen, Rui Pan, Han Zhu, Conghui Zhang, Jiahao Li, Donghai Hong, Boyuan Chen, Jiayi Zhou, Kaile Wang, et~al. Jiaming Ji, Xinyu Chen, Rui Pan, Han Zhu, Conghui Zhang, Jiahao Li, Donghai Hong, Boyuan Chen, Jiayi Zhou, Kaile Wang, et~al.
\newblock Safe rlhf-v: Safe reinforcement learning from human feedback in multimodal large language models. \newblock Safe rlhf-v: Safe reinforcement learning from human feedback in multimodal large language models.
...@@ -315,6 +319,11 @@ Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Ted ...@@ -315,6 +319,11 @@ Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Ted
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}. \newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}.
\newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}. \newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}.
\bibitem[Liu et~al.(2026)Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
Xiaoqian Liu, Ke~Wang, Yuchuan Wu, Fei Huang, Yongbin Li, Jianbin Jiao, and Junge Zhang.
\newblock Agentic reinforcement learning with implicit step rewards.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval} \bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu. Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
\newblock {G}-eval: {NLG} evaluation using gpt-4 with better human alignment. \newblock {G}-eval: {NLG} evaluation using gpt-4 with better human alignment.
...@@ -573,6 +582,11 @@ Richard~S. Sutton and Andrew~G. Barto. ...@@ -573,6 +582,11 @@ Richard~S. Sutton and Andrew~G. Barto.
\newblock \emph{Reinforcement Learning: An Introduction (2nd ed.)}. \newblock \emph{Reinforcement Learning: An Introduction (2nd ed.)}.
\newblock The MIT Press, 2018. \newblock The MIT Press, 2018.
\bibitem[Sutton(1984)]{sutton-etal:temporal}
Richard~Stuart Sutton.
\newblock \emph{Temporal credit assignment in reinforcement learning}.
\newblock University of Massachusetts Amherst, 1984.
\bibitem[Szepesv{\'a}ri(2010)]{szepesvari:2010algorithms} \bibitem[Szepesv{\'a}ri(2010)]{szepesvari:2010algorithms}
Csaba Szepesv{\'a}ri. Csaba Szepesv{\'a}ri.
\newblock Algorithms for reinforcement learning. \newblock Algorithms for reinforcement learning.
...@@ -917,4 +931,9 @@ Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu. ...@@ -917,4 +931,9 @@ Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu.
\newblock Prior constraints-based reward model training for aligning large language models. \newblock Prior constraints-based reward model training for aligning large language models.
\newblock In \emph{China National Conference on Chinese Computational Linguistics}, pp.\ 555--570. Springer, 2024. \newblock In \emph{China National Conference on Chinese Computational Linguistics}, pp.\ 555--570. Springer, 2024.
\bibitem[Zhou et~al.(2020)Zhou, Liu, Sui, Li, and Chung]{zhou-etal:learning}
Meng Zhou, Ziyu Liu, Pengwei Sui, Yixuan Li, and Yuk~Ying Chung.
\newblock Learning implicit credit assignment for cooperative multi-agent reinforcement learning.
\newblock \emph{Advances in neural information processing systems}, 33:\penalty0 11853--11864, 2020.
\end{thebibliography} \end{thebibliography}
...@@ -4,9 +4,37 @@ ...@@ -4,9 +4,37 @@
@article{zhou-etal:learning,
title={Learning implicit credit assignment for cooperative multi-agent reinforcement learning},
author={Zhou, Meng and Liu, Ziyu and Sui, Pengwei and Li, Yixuan and Chung, Yuk Ying},
journal={Advances in neural information processing systems},
volume={33},
pages={11853--11864},
year={2020}
}
@book{sutton-etal:temporal,
title={Temporal credit assignment in reinforcement learning},
author={Sutton, Richard Stuart},
year={1984},
publisher={University of Massachusetts Amherst}
}
@inproceedings{liu-etal:agentic,
title={Agentic reinforcement learning with implicit step rewards},
author={Liu, Xiaoqian and Wang, Ke and Wu, Yuchuan and Huang, Fei and Li, Yongbin and Jiao, Jianbin and Zhang, Junge},
booktitle={International Conference on Learning Representations},
volume={2026},
pages={129271--129291},
year={2026}
}
@misc{huo-etal:learning,
title={Learning from Environmental Feedback: Credit Assignment across Multiple Timescales for Agentic Reinforcement Learning},
author={Yifu Huo and Shunjie Xing and Chenglong Wang and Peinan Feng and Qiaozhi He and Yan Ding and Anxiang Ma and Yuxin Gao and Tongran Liu and Tong Xiao and Jingbo Zhu},
year={2026},
journal={arXiv preprint arXiv:2608.08255}
}
@article{xue-etal:dancegrpo, @article{xue-etal:dancegrpo,
title={Dancegrpo: Unleashing grpo on visual generation}, title={Dancegrpo: Unleashing grpo on visual generation},
......
...@@ -128,4 +128,5 @@ An interesting issue arises with this design of iterative RL: why is RL aimed at ...@@ -128,4 +128,5 @@ An interesting issue arises with this design of iterative RL: why is RL aimed at
Since the optimization objective of each phase is different, the design of iterative RL can also be analyzed and understood from the perspective of multi-objective optimization \citep{wang-etal:2024hybrid}. In the context of multi-objective optimization, iterative RL for LLMs can be likened to interactive methods where the solution process is iterative, and preferences are actively defined and refined by the decision-maker during the search for the most preferred solutions \citep{miettinen-etal:2008introduction,deb-etal:2016multi}. More specifically, in this scenario, RL can be seen as a decision-maker, iteratively refining and enhancing the different capabilities of the LLM across different phases. Since the optimization objective of each phase is different, the design of iterative RL can also be analyzed and understood from the perspective of multi-objective optimization \citep{wang-etal:2024hybrid}. In the context of multi-objective optimization, iterative RL for LLMs can be likened to interactive methods where the solution process is iterative, and preferences are actively defined and refined by the decision-maker during the search for the most preferred solutions \citep{miettinen-etal:2008introduction,deb-etal:2016multi}. More specifically, in this scenario, RL can be seen as a decision-maker, iteratively refining and enhancing the different capabilities of the LLM across different phases.
\subsection{On-Policy Distillation}
% TODO
\ No newline at end of file
...@@ -78,10 +78,10 @@ ...@@ -78,10 +78,10 @@
\draw [dashed] ([xshift=0.8cm,yshift=0.1cm]b5.north east) -- ([xshift=9cm,yshift=0.1cm]b5.north east); \draw [dashed] ([xshift=0.8cm,yshift=0.1cm]b5.north east) -- ([xshift=9cm,yshift=0.1cm]b5.north east);
\node [anchor=north west, text width=7.6cm,align=left] (n5) at ([xshift=\ssep]b5.north east) {Analyze failure trajectories during training. \node [anchor=north west, text width=7.6cm,align=left] (n5) at ([xshift=\ssep]b5.north east) {Analyze failure trajectories during training.
Generate new skills or refine existing skills.}; Generate new skills or refine existing skills.};
\node [anchor=north west, text width=6cm,draw,dashed,rounded corners,minimum height=1cm,align=left] (n51) at (n5.south west) {}; \node [anchor=north west, text width=6.5cm,draw,dashed,rounded corners,minimum height=1cm,align=left] (n51) at (n5.south west) {};
\node [anchor=west,text width=2cm,align=center] (n52) at ([xshift=0.2cm]n51.west) {\textbf{Outcome}}; \node [anchor=west,text width=2cm,align=center] (n52) at ([xshift=0.2cm]n51.west) {\textbf{Outcome}};
\path (n52.east) node(n53) [anchor=west, text width=1.3cm,draw,rounded corners,minimum height=0.6cm,align=center] {new skill} \path (n52.east) node(n53) [anchor=west, text width=1.5cm,draw,rounded corners,minimum height=0.6cm,align=center] {Adding New Skills}
([xshift=0.5cm]n53.east) node(n54) [anchor=west, text width=1.3cm,draw,rounded corners,minimum height=0.6cm,align=center] {old skill}; ([xshift=0.5cm]n53.east) node(n54) [anchor=west, text width=1.5cm,draw,rounded corners,minimum height=0.6cm,align=center] {Refining Old skills};
\draw ([xshift=0.1cm]n53.south east) -- ([xshift=-0.1cm]n54.north west); \draw ([xshift=0.1cm]n53.south east) -- ([xshift=-0.1cm]n54.north west);
\end{scope} \end{scope}
\end{tikzpicture} \end{tikzpicture}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论