Commit 6de2c269 by wangchenglong

update.

parent 85256022
\begin{thebibliography}{169} \begin{thebibliography}{173}
\providecommand{\natexlab}[1]{#1} \providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}} \providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax \expandafter\ifx\csname urlstyle\endcsname\relax
...@@ -240,6 +240,10 @@ Mengkang Hu, Pu~Zhao, Can Xu, Qingfeng Sun, Jian-Guang Lou, Qingwei Lin, Ping Lu ...@@ -240,6 +240,10 @@ Mengkang Hu, Pu~Zhao, Can Xu, Qingfeng Sun, Jian-Guang Lou, Qingwei Lin, Ping Lu
\newblock Agentgen: Enhancing planning abilities for large language model based agent via environment and task generation. \newblock Agentgen: Enhancing planning abilities for large language model based agent via environment and task generation.
\newblock In \emph{Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1}, pp.\ 496--507, 2025. \newblock In \emph{Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1}, pp.\ 496--507, 2025.
\bibitem[Huo et~al.(2026)Huo, Xing, Wang, Feng, He, Ding, Ma, Gao, Liu, Xiao, and Zhu]{huo-etal:learning}
Yifu Huo, Shunjie Xing, Chenglong Wang, Peinan Feng, Qiaozhi He, Yan Ding, Anxiang Ma, Yuxin Gao, Tongran Liu, Tong Xiao, and Jingbo Zhu.
\newblock Learning from environmental feedback: Credit assignment across multiple timescales for agentic reinforcement learning, 2026.
\bibitem[Ji et~al.(2025)Ji, Chen, Pan, Zhu, Zhang, Li, Hong, Chen, Zhou, Wang, et~al.]{ji2025safe} \bibitem[Ji et~al.(2025)Ji, Chen, Pan, Zhu, Zhang, Li, Hong, Chen, Zhou, Wang, et~al.]{ji2025safe}
Jiaming Ji, Xinyu Chen, Rui Pan, Han Zhu, Conghui Zhang, Jiahao Li, Donghai Hong, Boyuan Chen, Jiayi Zhou, Kaile Wang, et~al. Jiaming Ji, Xinyu Chen, Rui Pan, Han Zhu, Conghui Zhang, Jiahao Li, Donghai Hong, Boyuan Chen, Jiayi Zhou, Kaile Wang, et~al.
\newblock Safe rlhf-v: Safe reinforcement learning from human feedback in multimodal large language models. \newblock Safe rlhf-v: Safe reinforcement learning from human feedback in multimodal large language models.
...@@ -315,6 +319,11 @@ Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Ted ...@@ -315,6 +319,11 @@ Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Ted
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}. \newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}.
\newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}. \newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}.
\bibitem[Liu et~al.(2026)Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
Xiaoqian Liu, Ke~Wang, Yuchuan Wu, Fei Huang, Yongbin Li, Jianbin Jiao, and Junge Zhang.
\newblock Agentic reinforcement learning with implicit step rewards.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval} \bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu. Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
\newblock {G}-eval: {NLG} evaluation using gpt-4 with better human alignment. \newblock {G}-eval: {NLG} evaluation using gpt-4 with better human alignment.
...@@ -573,6 +582,11 @@ Richard~S. Sutton and Andrew~G. Barto. ...@@ -573,6 +582,11 @@ Richard~S. Sutton and Andrew~G. Barto.
\newblock \emph{Reinforcement Learning: An Introduction (2nd ed.)}. \newblock \emph{Reinforcement Learning: An Introduction (2nd ed.)}.
\newblock The MIT Press, 2018. \newblock The MIT Press, 2018.
\bibitem[Sutton(1984)]{sutton-etal:temporal}
Richard~Stuart Sutton.
\newblock \emph{Temporal credit assignment in reinforcement learning}.
\newblock University of Massachusetts Amherst, 1984.
\bibitem[Szepesv{\'a}ri(2010)]{szepesvari:2010algorithms} \bibitem[Szepesv{\'a}ri(2010)]{szepesvari:2010algorithms}
Csaba Szepesv{\'a}ri. Csaba Szepesv{\'a}ri.
\newblock Algorithms for reinforcement learning. \newblock Algorithms for reinforcement learning.
...@@ -917,4 +931,9 @@ Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu. ...@@ -917,4 +931,9 @@ Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu.
\newblock Prior constraints-based reward model training for aligning large language models. \newblock Prior constraints-based reward model training for aligning large language models.
\newblock In \emph{China National Conference on Chinese Computational Linguistics}, pp.\ 555--570. Springer, 2024. \newblock In \emph{China National Conference on Chinese Computational Linguistics}, pp.\ 555--570. Springer, 2024.
\bibitem[Zhou et~al.(2020)Zhou, Liu, Sui, Li, and Chung]{zhou-etal:learning}
Meng Zhou, Ziyu Liu, Pengwei Sui, Yixuan Li, and Yuk~Ying Chung.
\newblock Learning implicit credit assignment for cooperative multi-agent reinforcement learning.
\newblock \emph{Advances in neural information processing systems}, 33:\penalty0 11853--11864, 2020.
\end{thebibliography} \end{thebibliography}
...@@ -4,9 +4,37 @@ ...@@ -4,9 +4,37 @@
@article{zhou-etal:learning,
title={Learning implicit credit assignment for cooperative multi-agent reinforcement learning},
author={Zhou, Meng and Liu, Ziyu and Sui, Pengwei and Li, Yixuan and Chung, Yuk Ying},
journal={Advances in neural information processing systems},
volume={33},
pages={11853--11864},
year={2020}
}
@book{sutton-etal:temporal,
title={Temporal credit assignment in reinforcement learning},
author={Sutton, Richard Stuart},
year={1984},
publisher={University of Massachusetts Amherst}
}
@inproceedings{liu-etal:agentic,
title={Agentic reinforcement learning with implicit step rewards},
author={Liu, Xiaoqian and Wang, Ke and Wu, Yuchuan and Huang, Fei and Li, Yongbin and Jiao, Jianbin and Zhang, Junge},
booktitle={International Conference on Learning Representations},
volume={2026},
pages={129271--129291},
year={2026}
}
@misc{huo-etal:learning,
title={Learning from Environmental Feedback: Credit Assignment across Multiple Timescales for Agentic Reinforcement Learning},
author={Yifu Huo and Shunjie Xing and Chenglong Wang and Peinan Feng and Qiaozhi He and Yan Ding and Anxiang Ma and Yuxin Gao and Tongran Liu and Tong Xiao and Jingbo Zhu},
year={2026},
journal={arXiv preprint arXiv:2608.08255}
}
@article{xue-etal:dancegrpo, @article{xue-etal:dancegrpo,
title={Dancegrpo: Unleashing grpo on visual generation}, title={Dancegrpo: Unleashing grpo on visual generation},
......
...@@ -128,4 +128,5 @@ An interesting issue arises with this design of iterative RL: why is RL aimed at ...@@ -128,4 +128,5 @@ An interesting issue arises with this design of iterative RL: why is RL aimed at
Since the optimization objective of each phase is different, the design of iterative RL can also be analyzed and understood from the perspective of multi-objective optimization \citep{wang-etal:2024hybrid}. In the context of multi-objective optimization, iterative RL for LLMs can be likened to interactive methods where the solution process is iterative, and preferences are actively defined and refined by the decision-maker during the search for the most preferred solutions \citep{miettinen-etal:2008introduction,deb-etal:2016multi}. More specifically, in this scenario, RL can be seen as a decision-maker, iteratively refining and enhancing the different capabilities of the LLM across different phases. Since the optimization objective of each phase is different, the design of iterative RL can also be analyzed and understood from the perspective of multi-objective optimization \citep{wang-etal:2024hybrid}. In the context of multi-objective optimization, iterative RL for LLMs can be likened to interactive methods where the solution process is iterative, and preferences are actively defined and refined by the decision-maker during the search for the most preferred solutions \citep{miettinen-etal:2008introduction,deb-etal:2016multi}. More specifically, in this scenario, RL can be seen as a decision-maker, iteratively refining and enhancing the different capabilities of the LLM across different phases.
\subsection{On-Policy Distillation}
% TODO
\ No newline at end of file
...@@ -78,10 +78,10 @@ ...@@ -78,10 +78,10 @@
\draw [dashed] ([xshift=0.8cm,yshift=0.1cm]b5.north east) -- ([xshift=9cm,yshift=0.1cm]b5.north east); \draw [dashed] ([xshift=0.8cm,yshift=0.1cm]b5.north east) -- ([xshift=9cm,yshift=0.1cm]b5.north east);
\node [anchor=north west, text width=7.6cm,align=left] (n5) at ([xshift=\ssep]b5.north east) {Analyze failure trajectories during training. \node [anchor=north west, text width=7.6cm,align=left] (n5) at ([xshift=\ssep]b5.north east) {Analyze failure trajectories during training.
Generate new skills or refine existing skills.}; Generate new skills or refine existing skills.};
\node [anchor=north west, text width=6cm,draw,dashed,rounded corners,minimum height=1cm,align=left] (n51) at (n5.south west) {}; \node [anchor=north west, text width=6.5cm,draw,dashed,rounded corners,minimum height=1cm,align=left] (n51) at (n5.south west) {};
\node [anchor=west,text width=2cm,align=center] (n52) at ([xshift=0.2cm]n51.west) {\textbf{Outcome}}; \node [anchor=west,text width=2cm,align=center] (n52) at ([xshift=0.2cm]n51.west) {\textbf{Outcome}};
\path (n52.east) node(n53) [anchor=west, text width=1.3cm,draw,rounded corners,minimum height=0.6cm,align=center] {new skill} \path (n52.east) node(n53) [anchor=west, text width=1.5cm,draw,rounded corners,minimum height=0.6cm,align=center] {Adding New Skills}
([xshift=0.5cm]n53.east) node(n54) [anchor=west, text width=1.3cm,draw,rounded corners,minimum height=0.6cm,align=center] {old skill}; ([xshift=0.5cm]n53.east) node(n54) [anchor=west, text width=1.5cm,draw,rounded corners,minimum height=0.6cm,align=center] {Refining Old skills};
\draw ([xshift=0.1cm]n53.south east) -- ([xshift=-0.1cm]n54.north west); \draw ([xshift=0.1cm]n53.south east) -- ([xshift=-0.1cm]n54.north west);
\end{scope} \end{scope}
\end{tikzpicture} \end{tikzpicture}
......
...@@ -5,20 +5,20 @@ ...@@ -5,20 +5,20 @@
Using RL to train agents, often referred to as \textit{agentic RL}, is not a new concept. In classical machine learning, agentic RL typically involves training a domain-specific decision model from scratch. For example, in tasks such as playing complex games like Go \citep{mnih-etal:mnih2013playing,silver-etal:silver2017mastering} or controlling robotic locomotion \citep{lee-etal:lee2024learning}, the goal is to explore a structured environment and learn an optimal policy starting from random initialization. Using RL to train agents, often referred to as \textit{agentic RL}, is not a new concept. In classical machine learning, agentic RL typically involves training a domain-specific decision model from scratch. For example, in tasks such as playing complex games like Go \citep{mnih-etal:mnih2013playing,silver-etal:silver2017mastering} or controlling robotic locomotion \citep{lee-etal:lee2024learning}, the goal is to explore a structured environment and learn an optimal policy starting from random initialization.
With the rise of LLMs as core components of autonomous systems, RL has been increasingly used to adapt these models for sequential decision-making in dynamic and open-ended environments. In this setting, agentic RL addresses a key limitation of LLMs: although they encode extensive world knowledge, their outputs are not inherently aligned with long-horizon task requirements. For example, a supervised fine-tuned LLM may correctly generate a script for calling an external API, but it cannot reliably self-correct when the environment returns unexpected errors. This creates new challenges for enabling LLM-based agents to plan autonomously, adapt to environmental feedback, and safely improve their strategies through iterative interaction. With the emergence of LLMs as core components of autonomous systems, the role of RL has shifted from learning policies from scratch to adapting pretrained models for sequential decision-making in dynamic and open-ended environments. In this context, agentic RL aims to address a key limitation of LLMs: although they contain extensive world knowledge, their outputs are not inherently optimized for long-horizon interactions. For example, an SFT-trained LLM may successfully generate code for calling an external API, but it may fail to recover from unexpected execution errors during interaction.
In this section, we focus on the emerging paradigm of agentic RL for LLM-based agents. We first discuss how RL enables agents to acquire fundamental capabilities, including long-horizon planning and tool-integrated reasoning. We then introduce the role of environment design and scaling, which provide the essential foundation for training and evaluating agentic RL systems. Finally, we explore how agents can continuously improve from their interaction experiences through mechanisms, including memory management, skill optimization, and trajectory refinement.
In this section, we focus on the emerging paradigm of agentic RL for LLM-based agents. We begin by discussing how RL builds and enhances core agentic capabilities, specifically focusing on long-horizon planning and tool-integrated reasoning. Then we consider training-free methods that apply RL principles to optimize external modules such as memory management, skill optimization, and trajectory refinement without altering the internal model weights. Finally, we introduce the crucial role of designing better environments, which serve as the foundational testbeds for training agents.
\subsection{Building Agent Capabilities} \subsection{Building Agent Capabilities}
\label{sec:building-agent-capabilities} \label{sec:building-agent-capabilities}
LLM-based agents extend LLMs from passive response generators into interactive decision-making systems. Rather than producing a single answer, an agent must repeatedly observe its environment, decide what to do next, and execute an action. This interaction loop allows the agent to solve tasks that require multiple steps, external resources, and adaptation during execution. The underlying LLM already provides several important capabilities, including instruction understanding, reasoning, code generation, and broad world knowledge. These capabilities allow the agent to interpret complex goals and propose plausible intermediate steps. However, they do not automatically guarantee reliable behavior in an interactive environment. An agent must also determine how to decompose a task, choose among alternative actions, coordinate external tools, and recover when an earlier decision leads to an unexpected result. LLM-based agents extend LLMs from passive response generators to interactive decision-making systems. Instead of producing a single response, agents continuously observe their environments, select actions, and update their behaviors through interaction. This interaction loop enables agents to solve complex tasks that require long-horizon execution. Although LLMs already possess strong capabilities, such as instruction understanding and reasoning, these capabilities are mainly developed for static text generation and may not be sufficient for dynamic interaction scenarios. When deployed as agents in real-world environments, LLMs need to further adapt their reasoning abilities to sequential decision-making, where they must decompose high-level goals into executable steps, select appropriate actions, and recover from unexpected failures during execution.
These agentic capabilities are difficult to learn from static input-output pairs alone. The quality of an individual decision often depends on its effect on later interactions and the final task outcome. A locally plausible action may lead to failure several steps later, while an initially unsuccessful attempt may provide useful information for a better strategy. RL is well suited to this setting because it optimizes complete interaction trajectories through environmental feedback and delayed rewards \citep{singh-etal:singhagentic,feng-etal:retool}. It enables the agent to explore different behaviors, compare their consequences, and gradually improve its decision-making policy.
In this subsection, we focus on two core capabilities in agentic RL: \textit{planning} and \textit{tool use}. Planning allows the agent to organize a complex goal into a sequence of executable steps. Tool use allows it to access external information, perform reliable computation, and take actions beyond the capabilities stored in the model parameters. Together, these capabilities form the basis for effective interaction in given environments. Such agentic capabilities are difficult to acquire from static input-output demonstrations alone. The effectiveness of an individual decision often depends on its long-term consequences and the final task outcome. A locally reasonable action may eventually lead to failure after multiple interactions, while an unsuccessful attempt may provide valuable experience for future decisions. To address this challenge, RL has become a standard approach for training agentic systems, enabling agents to explore different behaviors and improve their abilities through environmental feedback \citep{singh-etal:singhagentic,feng-etal:retool}.
In this subsection, we focus on two key capabilities in agentic RL: \textit{planning} and \textit{tool use}. Planning enables agents to decompose complex goals into executable steps and make effective decisions over long horizons. Tool use allows agents to interact with external resources, such as knowledge sources and computational tools, thereby extending their capabilities beyond the information encoded in model parameters.
\subsubsection{Planning} \subsubsection{Planning}
...@@ -100,51 +100,10 @@ where $\pi_{\theta}$ denotes the agent policy parameterized by the LLM, and $R(\ ...@@ -100,51 +100,10 @@ where $\pi_{\theta}$ denotes the agent policy parameterized by the LLM, and $R(\
Compared with SFT, RL provides two important advantages for agent planning. First, it enables the agent to learn from execution outcomes rather than only from static demonstrations. If a plan leads to failed tool execution or poor environmental feedback, the agent can be penalized and encouraged to explore alternative behaviors. Second, RL supports trajectory-level credit assignment, allowing the model to adjust earlier planning decisions according to later outcomes. This is crucial for agent planning, where early subgoal decomposition or tool selection can substantially affect the final task success. Compared with SFT, RL provides two important advantages for agent planning. First, it enables the agent to learn from execution outcomes rather than only from static demonstrations. If a plan leads to failed tool execution or poor environmental feedback, the agent can be penalized and encouraged to explore alternative behaviors. Second, RL supports trajectory-level credit assignment, allowing the model to adjust earlier planning decisions according to later outcomes. This is crucial for agent planning, where early subgoal decomposition or tool selection can substantially affect the final task success.
% Another challenge is training stability. In multi-turn agent RL, long-horizon decision-making and stochastic environment feedback can make trajectory-level optimization unstable, while errors in early steps may accumulate and affect later interactions \citep{wang-etal:ragen,djuhera-etal:tsr}.
\subsubsubsection{Credit Assignment}
Credit assignment is not unique to LLM-based agents. It is a long-standing challenge in traditional RL. When rewards are delayed, the agent must decide which past actions should be reinforced or suppressed. Classical methods, such as temporal-difference learning and eligibility traces, address this issue by propagating reward information backward along the trajectory. In agentic RL, credit assignment becomes more challenging. This is because that each interaction step may involve high-level planning, tool invocation, and environmental feedback, rather than a single low-level action. Therefore, credit assignment often needs to operate at both the trajectory level and the step level.
This challenge is especially important for agent planning. In many agent tasks, the environment only provides a sparse reward after the entire trajectory is completed. However, the final success or failure may come from an earlier planning decision, an incorrect tool invocation, or an ineffective response to an intermediate observation. If we directly assign the same trajectory-level reward to all interaction steps, the supervision can become noisy. The agent may fail to identify which decisions should be reinforced and which should be suppressed.
\begin{figure*}[!t]
\centering
\resizebox{\linewidth}{!}{
\input{section6/Figures/agent-credit-assignment.tex}
}
\vspace{-1.0cm}
\caption{Illustration of credit assignment in agentic RL.}
\label{fig:agent-credit-assignment}
\end{figure*}
% Consider a simple example. If the agent correctly plans to compute the babysitting payment but calls the database API before obtaining the calculation result, the final task may fail. In this case, the failure should mainly be attributed to the incorrect ordering of tool use rather than to the entire plan. Conversely, if the agent chooses the wrong formula at the beginning, then the later database operation may be executed correctly, but the final result is still wrong. These cases show that different steps in an interaction trajectory may contribute differently to the final outcome.
Formally, given a trajectory $\tau = [p,u_1,o_1,\cdots,u_T,o_T]$, a simple RL objective uses a trajectory-level reward $R(\tau)$ to optimize the whole planning process. However, such a reward only provides coarse feedback on the overall outcome. Therefore, credit assignment aims to decompose the trajectory-level reward into step-level signals:
\begin{eqnarray}
R(\tau) \rightarrow \{r_p,r_1,\cdots,r_T\}
\end{eqnarray}
where $r_p$ evaluates the quality of the generated plan, and $r_t$ evaluates the contribution of the $t$-th interaction step $(u_t,o_t)$ to the final task outcome.
The mechanism of credit assignment is illustrated in Figure \ref{fig:agent-credit-assignment}. This decomposition enables the agent to distinguish whether failure comes from an unreasonable plan, an invalid tool call, or a poor adaptation to environmental feedback \citep{li-etal:encouraging,xi-etal:agentprm,wang-etal:steppo}. For example, consider the task of computing the babysitting payment and saving the result into a database. Suppose the agent generates a reasonable plan and correctly calls the calculator, but fails to invoke the database API with the required JSON format. If we only use a final trajectory-level reward, the whole trajectory may receive a low reward, e.g., $R(\tau)=0.3$, even though the early planning and calculation steps are correct.
For illustration, we can define rule-based step-level rewards using predefined verification criteria, such as whether the plan includes all necessary steps, whether the calculation is correct, whether the JSON schema is valid, and whether the database update succeeds. Then, we can use credit assignment to decompose the final trajectory-level feedback into step-level rewards:
\begin{eqnarray}
R(\tau)=0.3
\quad \Rightarrow \quad
\{r_p=0.4,\ r_1=1.0,\ r_2=1.0,\ r_3=0.7,\ r_4=0.0\}
\end{eqnarray}
where $r_p$ indicates that the overall plan is reasonable, $r_1$ and $r_2$ indicate that the agent correctly extracts the numerical values and performs the calculation, $r_3$ indicates that the JSON formatting is partially correct, and $r_4$ indicates that the database update fails. In this way, the agent receives more precise feedback: it should preserve the correct planning.
% To address this challenge, researchers explore more informative feedback signals for agent planning. For example, tool-use rewards evaluate whether the tool invocation sequence is complete and effective, while process rewards or progress-based rewards assess whether each intermediate decision moves the agent closer to the final goal \citep{li-etal:encouraging,xi-etal:agentprm}. These methods provide denser supervision than sparse final rewards and make it possible to optimize intermediate planning and execution behaviors more directly.
\subsubsection{Tool Use} \subsubsection{Tool Use}
Tool use is another fundamental capability of LLM-based agents. Tools extend the agent beyond its internal parameters. They allow the agent to access external knowledge, perform accurate computation, and execute actions in external systems. Early studies show that LLMs can use tools through prompt-based reasoning-action patterns \citep{chang-etal:efficient,yao-etal:react}. ReAct is a representative example, where the model alternates between reasoning steps and tool-use actions \citep{yao-etal:react}. However, prompt-based tool use is often unstable. It also depends heavily on the capability of the model. Thus, SFT methods train LLMs on tool-use trajectories, so that the model can learn when and how to invoke tools from demonstrations \citep{komeili-etal:Internet,schick-etal:toolformer}. Still, SFT mainly teaches imitation. It cannot directly optimize whether the agent should use a tool, which tool it should select, or how to balance tool benefit with tool cost. RL provides a natural framework for efficient tool use. We can define rewards based on task success, tool-use correctness, or tool-use completeness \citep{qian-etal:toolrl,singh-etal:agentic}. Tool use is another fundamental capability of LLM-based agents. Tools extend the agent beyond its internal parameters. They allow the agent to access external knowledge and execute actions in external systems. Early studies show that LLMs can use tools through prompt-based reasoning-action patterns \citep{chang-etal:efficient,yao-etal:react}. ReAct is a representative example, where the model alternates between reasoning steps and tool-use actions \citep{yao-etal:react}. However, prompt-based tool use is often unstable. It depends heavily on the capability of the model. Thus, SFT methods train LLMs on tool-use trajectories, so that the model can learn when and how to invoke tools from demonstrations \citep{komeili-etal:Internet,schick-etal:toolformer}. Still, SFT mainly teaches imitation. It cannot directly optimize whether the agent should use a tool, which tool it should select, or how to balance tool benefit with tool cost. RL provides a natural framework for efficient tool use. We can define rewards based on task success, tool-use correctness, or tool-use completeness \citep{qian-etal:toolrl,singh-etal:agentic}.
% 2. Prompt-based Tool Use % 2. Prompt-based Tool Use
\subsubsubsection{Prompt-based Tool Use} \subsubsubsection{Prompt-based Tool Use}
...@@ -165,12 +124,12 @@ Prompt-based tool is a simple approach to enable LLM-based agents to interact wi ...@@ -165,12 +124,12 @@ Prompt-based tool is a simple approach to enable LLM-based agents to interact wi
\texttt{update\_record(json)}: save a JSON record into the database. \\ \texttt{update\_record(json)}: save a JSON record into the database. \\
\textcolor{gray}{Demonstration} & \textcolor{gray}{Demonstration} &
\textit{Task}: Tom earns \$15 per hour. He worked for 2 hours. How much did he earn? Please save the result into the database. \newline Task: Tom earns \$15 per hour. He worked for 2 hours. How much did he earn? Please save the result into the database. \newline
\textbf{Thought}: I need to compute the total payment. \newline \textbf{Thought}: I need to compute the total payment. \newline
{\setlength{\fboxsep}{0pt}\colorbox{green!35}{\strut\textbf{Action}: \texttt{calculator}($15 \times 2$)}} \newline {\setlength{\fboxsep}{0pt}\colorbox{green!35}{\strut\textbf{Action}: \texttt{calculator}($15 \times 2$)}} \newline
\textbf{Observation}: 30 \newline \textbf{Observation}: 30 \newline
\textbf{Thought}: The payment is 30 dollars. I need to save it into the database. \newline \textbf{Thought}: The payment is 30 dollars. I need to save it into the database. \newline
\textbf{Action}: \texttt{update\_record}(\{"earning": 30\}) \newline {\setlength{\fboxsep}{0pt}\colorbox{red!20}{\strut\textbf{Action}: \texttt{update\_record}(\{"earning": 30\})}} \newline
\textbf{Observation}: success \newline \textbf{Observation}: success \newline
\textbf{Answer}: Tom earned \$30, and the result has been saved into the database. \\ \textbf{Answer}: Tom earned \$30, and the result has been saved into the database. \\
...@@ -182,7 +141,7 @@ Weng earns \$12 per hour for babysitting. Yesterday, she babysat for 50 minutes. ...@@ -182,7 +141,7 @@ Weng earns \$12 per hour for babysitting. Yesterday, she babysat for 50 minutes.
{\setlength{\fboxsep}{0pt}\colorbox{green!35}{\strut\textbf{Action}: \texttt{calculator}($12 \times 50 / 60$)}} \newline {\setlength{\fboxsep}{0pt}\colorbox{green!35}{\strut\textbf{Action}: \texttt{calculator}($12 \times 50 / 60$)}} \newline
\textbf{Observation}: 10 \newline \textbf{Observation}: 10 \newline
\textbf{Thought}: The payment is 10 dollars. I need to save it into the database. \newline \textbf{Thought}: The payment is 10 dollars. I need to save it into the database. \newline
\textbf{Action}: \texttt{update\_record}(\{"earning": 10\}) \newline {\setlength{\fboxsep}{0pt}\colorbox{red!20}{\strut\textbf{Action}: \texttt{update\_record}(\{"earning": 10\})}} \newline
\textbf{Observation}: success \newline \textbf{Observation}: success \newline
\textbf{Answer}: Weng earned \$10, and the result has been saved into the database. \textbf{Answer}: Weng earned \$10, and the result has been saved into the database.
...@@ -196,7 +155,7 @@ This demonstration teaches the model a tool-use pattern. The model learns to rea ...@@ -196,7 +155,7 @@ This demonstration teaches the model a tool-use pattern. The model learns to rea
% 3. SFT for Tool Use % 3. SFT for Tool Use
\subsubsubsection{Supervised Tool-use Data} \subsubsubsection{Supervised Tool-use Data}
To make tool use more reliable, we can further train LLMs on tool-use trajectories with supervised fine-tuning. In this setting, each training example contains a user task, available tool descriptions, and a demonstrated tool-use process. The model learns when to invoke a tool, which tool to select, how to construct valid arguments, and how to use the returned observation. For example, we can construct a training example as follows: To make tool use more reliable, we can further train LLMs on tool-use trajectories. In this setting, each training example contains a user task, available tool descriptions, and a demonstrated tool-use process. The model learns when to invoke a tool, which tool to select, how to construct valid arguments, and how to use the returned observation. For example, we can construct a training example as follows:
\begin{center} \begin{center}
$\mathbf{x} =$ Task: Weng earns \$12 ... Tools: calculator(expression), ... , update\_record(json) \newline $\mathbf{x} =$ Task: Weng earns \$12 ... Tools: calculator(expression), ... , update\_record(json) \newline
$\mathbf{y} =$ Thought: compute the payment; Action: calculator(12 x 50 / 60); ...; Answer: Weng earned \$10. $\mathbf{y} =$ Thought: compute the payment; Action: calculator(12 x 50 / 60); ...; Answer: Weng earned \$10.
...@@ -215,7 +174,7 @@ Beyond learning to use a small set of tools, later studies further scale tool-us ...@@ -215,7 +174,7 @@ Beyond learning to use a small set of tools, later studies further scale tool-us
Although SFT can teach LLMs to imitate tool-use demonstrations, it is still limited by the coverage and quality of offline trajectories. The model mainly learns the tool-use patterns that appear in the supervised data. As a result, it may fail to generalize to unfamiliar tools, complex tool combinations, or new interaction patterns. More importantly, SFT does not directly optimize whether a tool call is necessary, whether the selected tool is appropriate, or whether the returned observation improves the final answer. This limitation becomes more serious in multi-step tool-use scenarios, \textit{where the model needs to decide when to call a tool, how to formulate the tool input, how to interpret the tool output, and when to stop using tools}. Although SFT can teach LLMs to imitate tool-use demonstrations, it is still limited by the coverage and quality of offline trajectories. The model mainly learns the tool-use patterns that appear in the supervised data. As a result, it may fail to generalize to unfamiliar tools, complex tool combinations, or new interaction patterns. More importantly, SFT does not directly optimize whether a tool call is necessary, whether the selected tool is appropriate, or whether the returned observation improves the final answer. This limitation becomes more serious in multi-step tool-use scenarios, \textit{where the model needs to decide when to call a tool, how to formulate the tool input, how to interpret the tool output, and when to stop using tools}.
To address this limitation, researchers have explored RL for tool use. Instead of only imitating fixed demonstrations, the model can interact with tools during rollout and receive feedback from task outcomes or tool execution results. Given a user task $q$ and a set of available tools $\mathcal{T}$, the model generates a tool-use trajectory: To address this limitation, researchers have explored RL for enhancing tool use. Instead of only imitating fixed demonstrations, the model can interact with tools during rollout and receive feedback from task outcomes or tool execution results. Given a user task $q$ and a set of available tools $\mathcal{T}$, the model generates a tool-use trajectory:
\begin{eqnarray} \begin{eqnarray}
\tau = [s_1,u_1,o_1,\cdots,s_T,u_T,o_T,a] \tau = [s_1,u_1,o_1,\cdots,s_T,u_T,o_T,a]
\end{eqnarray} \end{eqnarray}
...@@ -249,7 +208,7 @@ In application, this reward decomposition provides denser feedback than final-an ...@@ -249,7 +208,7 @@ In application, this reward decomposition provides denser feedback than final-an
Another direction is to use RL to improve strategic tool use. For example, ReTool first builds a cold-start model with code-augmented reasoning traces and then applies RL with real-time code execution. During rollout, the model interleaves natural language reasoning with code execution, observes execution results, and revises its subsequent reasoning. This allows the model to discover when and how to invoke a code interpreter based on outcome feedback rather than human-written rules \citep{feng-etal:retool}. Search-R1 and ReSearch follow a similar idea in retrieval-augmented reasoning. They train models to generate search queries during reasoning and use retrieved information to update later steps, rather than relying on fixed retrieval pipelines or hand-crafted search prompts \citep{jin-etal:searchr1,chen-etal:research}. Another direction is to use RL to improve strategic tool use. For example, ReTool first builds a cold-start model with code-augmented reasoning traces and then applies RL with real-time code execution. During rollout, the model interleaves natural language reasoning with code execution, observes execution results, and revises its subsequent reasoning. This allows the model to discover when and how to invoke a code interpreter based on outcome feedback rather than human-written rules \citep{feng-etal:retool}. Search-R1 and ReSearch follow a similar idea in retrieval-augmented reasoning. They train models to generate search queries during reasoning and use retrieved information to update later steps, rather than relying on fixed retrieval pipelines or hand-crafted search prompts \citep{jin-etal:searchr1,chen-etal:research}.
So far, we have introduced how RL can enhance the core capabilities of LLM-based agents, including planning and tool use. Since an LLM-based agent is inherently driven by an underlying LLM, its RL training can reuse the general LLM-based RL algorithms introduced in Section \ref{sec:example-using-rl-training-llms}. The main differences lie in the agentic setting. Compared with standard response optimization, agentic RL involves longer trajectories, environmental feedback, and intermediate decisions such as planning and tool invocation. Thus, the key challenges are not only the choice of RL algorithms, but also trajectory construction, reward design, and environment optimization. So far, we have introduced how RL can enhance the core capabilities of LLM-based agents. Since LLM-based agents are built upon underlying LLMs, their RL training can naturally leverage the general LLM-based RL algorithms introduced in Section~\ref{sec:example-using-rl-training-llms}. However, agentic RL introduces additional challenges beyond standard response optimization. A key challenge is \textit{how to obtain high-quality feedbacks from environments} and \textit{how to effectively use these signals to improve agent behaviors}. As a result, recent works on agentic RL requires not only appropriate RL algorithms, but also specialized approaches for addressing the unique challenges introduced by agentic settings \citep{huo-etal:learning,liu-etal:agentic}.
\subsection{Environment Design and Scaling} \subsection{Environment Design and Scaling}
...@@ -296,7 +255,6 @@ This process is illustrated in Figure~\ref{fig:agent-environment-scaling}. It ty ...@@ -296,7 +255,6 @@ This process is illustrated in Figure~\ref{fig:agent-environment-scaling}. It ty
\item \textbf{Agent Learning.} Finally, the verified trajectories are used to improve the agent. On the one hand, successful trajectories can serve as demonstrations for SFT, while task-level or step-level feedback can support RL training. On the other hand, failed interactions are also valuable because they reveal weaknesses in the current policy and gaps in the task distribution. This can guide the generation of targeted scenarios that exercise the agent's weak capabilities and provide more informative experience for subsequent training \citep{chen-etal:failure-attribution,sun-etal:learning-from-failure}. More broadly, this process can be viewed as learning from agentic experience, which we will discuss in Section~\ref{sec:learning_from_agentic_experience}. \item \textbf{Agent Learning.} Finally, the verified trajectories are used to improve the agent. On the one hand, successful trajectories can serve as demonstrations for SFT, while task-level or step-level feedback can support RL training. On the other hand, failed interactions are also valuable because they reveal weaknesses in the current policy and gaps in the task distribution. This can guide the generation of targeted scenarios that exercise the agent's weak capabilities and provide more informative experience for subsequent training \citep{chen-etal:failure-attribution,sun-etal:learning-from-failure}. More broadly, this process can be viewed as learning from agentic experience, which we will discuss in Section~\ref{sec:learning_from_agentic_experience}.
\end{itemize} \end{itemize}
Despite this progress, several important questions remain. First, how can we scale both the number and diversity of environments? Second, how can we ensure that generated environments are executable and verifiable? Third, how can agents generalize across heterogeneous and previously unseen environments? Despite this progress, several important questions remain. First, how can we scale both the number and diversity of environments? Second, how can we ensure that generated environments are executable and verifiable? Third, how can agents generalize across heterogeneous and previously unseen environments?
One direction focuses on scaling environment diversity. RandomWorld procedurally generates executable tools and compositional tasks for tool-use agents \citep{sullivan-etal:randomworld}. Unlike static datasets that contain only predefined trajectories, these generated environments support online execution and allow agents to explore different tool combinations during training. AgentScaler further constructs heterogeneous simulated environments and separates the learning of general function-calling behaviors from domain-specific adaptation \citep{fang-etal:agentscaler}. At a larger scale, DeepSeek-V3.2 synthesizes diverse environments and complex tasks for agentic post-training, showing that broad interaction experience is important for long-tail generalization \citep{deepseek-ai:deepseekv32}. One direction focuses on scaling environment diversity. RandomWorld procedurally generates executable tools and compositional tasks for tool-use agents \citep{sullivan-etal:randomworld}. Unlike static datasets that contain only predefined trajectories, these generated environments support online execution and allow agents to explore different tool combinations during training. AgentScaler further constructs heterogeneous simulated environments and separates the learning of general function-calling behaviors from domain-specific adaptation \citep{fang-etal:agentscaler}. At a larger scale, DeepSeek-V3.2 synthesizes diverse environments and complex tasks for agentic post-training, showing that broad interaction experience is important for long-tail generalization \citep{deepseek-ai:deepseekv32}.
...@@ -313,6 +271,47 @@ where $V_e(\cdot)$ may be implemented using executable tests, database queries, ...@@ -313,6 +271,47 @@ where $V_e(\cdot)$ may be implemented using executable tests, database queries,
Scaling environments also introduces substantial heterogeneity. Different environments may vary in task difficulty, available tools, and interaction length. These differences can make RL training unstable and cause the agent to overfit to frequently sampled or easier environments. AutoForge addresses this issue by synthesizing difficult but verifiable tasks and estimating learning signals at the environment level \citep{cai-etal:autoforge}. This design reduces the influence of unstable simulated interactions and improves training across heterogeneous environments. Scaling environments also introduces substantial heterogeneity. Different environments may vary in task difficulty, available tools, and interaction length. These differences can make RL training unstable and cause the agent to overfit to frequently sampled or easier environments. AutoForge addresses this issue by synthesizing difficult but verifiable tasks and estimating learning signals at the environment level \citep{cai-etal:autoforge}. This design reduces the influence of unstable simulated interactions and improves training across heterogeneous environments.
\subsection{Credit Assignment}
Credit assignment is not unique to LLM-based agents. It is a long-standing challenge in traditional RL \citep{sutton-etal:temporal,zhou-etal:learning}. When rewards are delayed, the agent must decide which past actions should be reinforced or suppressed. Classical methods, such as temporal-difference learning and eligibility traces, address this issue by propagating reward information backward along the trajectory. In agentic RL, credit assignment becomes more challenging. This is because that each interaction step may involve high-level planning, tool invocation, and environmental feedback, rather than a single low-level action. Therefore, credit assignment often needs to operate at both the trajectory level and the step level.
This challenge is especially important for agent planning. In many agent tasks, the environment only provides a sparse reward after the entire trajectory is completed. However, the final success or failure may come from an earlier planning decision, an incorrect tool invocation, or an ineffective response to an intermediate observation. If we directly assign the same trajectory-level reward to all interaction steps, the supervision can become noisy. The agent may fail to identify which decisions should be reinforced and which should be suppressed.
\begin{figure*}[!t]
\centering
\resizebox{\linewidth}{!}{
\input{section6/Figures/agent-credit-assignment.tex}
}
\vspace{-1.0cm}
\caption{Illustration of credit assignment in agentic RL.}
\label{fig:agent-credit-assignment}
\end{figure*}
% Consider a simple example. If the agent correctly plans to compute the babysitting payment but calls the database API before obtaining the calculation result, the final task may fail. In this case, the failure should mainly be attributed to the incorrect ordering of tool use rather than to the entire plan. Conversely, if the agent chooses the wrong formula at the beginning, then the later database operation may be executed correctly, but the final result is still wrong. These cases show that different steps in an interaction trajectory may contribute differently to the final outcome.
Formally, given a trajectory $\tau = [p,u_1,o_1,\cdots,u_T,o_T]$, a simple RL objective uses a trajectory-level reward $R(\tau)$ to optimize the whole planning process. However, such a reward only provides coarse feedback on the overall outcome. Therefore, credit assignment aims to decompose the trajectory-level reward into step-level signals:
\begin{eqnarray}
R(\tau) \rightarrow \{r_p,r_1,\cdots,r_T\}
\end{eqnarray}
where $r_p$ evaluates the quality of the generated plan, and $r_t$ evaluates the contribution of the $t$-th interaction step $(u_t,o_t)$ to the final task outcome.
The mechanism of credit assignment is illustrated in Figure \ref{fig:agent-credit-assignment}. This decomposition enables the agent to distinguish whether failure comes from an unreasonable plan, an invalid tool call, or a poor adaptation to environmental feedback \citep{li-etal:encouraging,xi-etal:agentprm,wang-etal:steppo}. For example, consider the task of computing the babysitting payment and saving the result into a database. Suppose the agent generates a reasonable plan and correctly calls the calculator, but fails to invoke the database API with the required JSON format. If we only use a final trajectory-level reward, the whole trajectory may receive a low reward, e.g., $R(\tau)=0.3$, even though the early planning and calculation steps are correct.
For illustration, we can define rule-based step-level rewards using predefined verification criteria, such as whether the plan includes all necessary steps, whether the calculation is correct, whether the JSON schema is valid, and whether the database update succeeds. Then, we can use credit assignment to decompose the final trajectory-level feedback into step-level rewards:
\begin{eqnarray}
R(\tau)=0.3
\quad \Rightarrow \quad
\{r_p=0.4,\ r_1=1.0,\ r_2=1.0,\ r_3=0.7,\ r_4=0.0\}
\end{eqnarray}
where $r_p$ indicates that the overall plan is reasonable, $r_1$ and $r_2$ indicate that the agent correctly extracts the numerical values and performs the calculation, $r_3$ indicates that the JSON formatting is partially correct, and $r_4$ indicates that the database update fails. In this way, the agent receives more precise feedback: it should preserve the correct planning.
% To address this challenge, researchers explore more informative feedback signals for agent planning. For example, tool-use rewards evaluate whether the tool invocation sequence is complete and effective, while process rewards or progress-based rewards assess whether each intermediate decision moves the agent closer to the final goal \citep{li-etal:encouraging,xi-etal:agentprm}. These methods provide denser supervision than sparse final rewards and make it possible to optimize intermediate planning and execution behaviors more directly.
\subsection{Learning from Agentic Experience} \subsection{Learning from Agentic Experience}
\label{sec:learning_from_agentic_experience} \label{sec:learning_from_agentic_experience}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论