Commit a499d184 by wangchenglong

update.

parent 448a1534
\begin{thebibliography}{128}
\begin{thebibliography}{134}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
......@@ -42,6 +42,11 @@ Jane Bromley, Isabelle Guyon, Yann LeCun, Eduard S{\"a}ckinger, and Roopak Shah.
\newblock Signature verification using a" siamese" time delay neural network.
\newblock \emph{Advances in neural information processing systems}, 6, 1993.
\bibitem[Cai et~al.(2025)Cai, Fang, Wu, Li, Wang, Jiang, Su, Zhang, Yin, Zhang, Feng, Xie, and Wang]{cai-etal:autoforge}
Shihao Cai, Runnan Fang, Jialong Wu, Baixuan Li, Xinyu Wang, Yong Jiang, Liangcai Su, Liwen Zhang, Wenbiao Yin, Zhen Zhang, Fuli Feng, Pengjun Xie, and Xiaobin Wang.
\newblock Autoforge: Automated environment synthesis for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2512.22857}, 2025.
\bibitem[Chen et~al.(2023{\natexlab{a}})Chen, Shu, Shareghi, Collier, Narasimhan, and Yao]{chen-etal:fireact}
Baian Chen, Chang Shu, Ehsan Shareghi, Nigel Collier, Karthik Narasimhan, and Shunyu Yao.
\newblock Fireact: Toward language agent fine-tuning.
......@@ -103,6 +108,11 @@ Deepseek.
\newblock Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2501.12948}, 2025.
\bibitem[DeepSeek-AI(2025)]{deepseek-ai:deepseekv32}
DeepSeek-AI.
\newblock Deepseek-v3.2: Pushing the frontier of open large language models.
\newblock \emph{arXiv preprint arXiv:2512.02556}, 2025.
\bibitem[Dubois et~al.(2023)Dubois, Li, Taori, Zhang, Gulrajani, Ba, Guestrin, Liang, and Hashimoto]{dubois-etal:2024alpacafarm}
Yann Dubois, Chen~Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy Liang, and Tatsunori~B. Hashimoto.
\newblock Alpacafarm: {A} simulation framework for methods that learn from human feedback.
......@@ -121,6 +131,11 @@ Kawin Ethayarajh, Winnie Xu, Niklas Muennighoff, Dan Jurafsky, and Douwe Kiela.
\newblock \emph{ArXiv preprint}, abs/2402.01306, 2024.
\newblock URL \url{https://arxiv.org/abs/2402.01306}.
\bibitem[Fang et~al.(2025)Fang, Cai, Li, Wu, Li, Yin, Wang, Wang, Su, Zhang, Wu, Tao, Jiang, Xie, Huang, and Zhou]{fang-etal:agentscaler}
Runnan Fang, Shihao Cai, Baixuan Li, Jialong Wu, Guangyu Li, Wenbiao Yin, Xinyu Wang, Xiaobin Wang, Liangcai Su, Zhen Zhang, Shibin Wu, Zhengwei Tao, Yong Jiang, Pengjun Xie, Fei Huang, and Jingren Zhou.
\newblock Towards general agentic intelligence via environment scaling.
\newblock \emph{arXiv preprint arXiv:2509.13311}, 2025.
\bibitem[Feng et~al.(2025)Feng, Huang, Qu, Zhang, Qin, Zhong, Jiang, Chi, and Zhong]{feng-etal:retool}
Jiazhan Feng, Shijue Huang, Xingwei Qu, Ge~Zhang, Yujia Qin, Baoquan Zhong, Chengquan Jiang, Jinxin Chi, and Wanjun Zhong.
\newblock Retool: Reinforcement learning for strategic tool use in llms.
......@@ -419,6 +434,16 @@ Prasann Singhal, Tanya Goyal, Jiacheng Xu, and Greg Durrett.
\newblock \emph{ArXiv preprint}, abs/2310.03716, 2023.
\newblock URL \url{https://arxiv.org/abs/2310.03716}.
\bibitem[Song et~al.(2026)Song, Chang, Dong, Zhu, Dou, and Wen]{song-etal:envscaler}
Xiaoshuai Song, Haofei Chang, Guanting Dong, Yutao Zhu, Zhicheng Dou, and Ji-Rong Wen.
\newblock Envscaler: Scaling tool-interactive environments for llm agent via programmatic synthesis.
\newblock \emph{arXiv preprint arXiv:2601.05808}, 2026.
\bibitem[Sullivan et~al.(2025)Sullivan, Hartmann, and Koller]{sullivan-etal:randomworld}
Michael Sullivan, Mareike Hartmann, and Alexander Koller.
\newblock Procedural environment generation for tool-use agents.
\newblock \emph{arXiv preprint arXiv:2506.11045}, 2025.
\bibitem[Sun et~al.(2024)Sun, Liu, Bair, and Kolter]{sun-etal:2023simple}
Mingjie Sun, Zhuang Liu, Anna Bair, and J.~Zico Kolter.
\newblock A simple and effective pruning approach for large language models.
......@@ -541,6 +566,11 @@ Yizhong Wang, Hamish Ivison, Pradeep Dasigi, Jack Hessel, Tushar Khot, Khyathi C
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023{\natexlab{d}}.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/ec6413875e4ab08d7bc4d8e225263398-Abstract-Datasets\_and\_Benchmarks.html}.
\bibitem[Wang et~al.(2026{\natexlab{d}})Wang, Xu, Liu, Wang, Han, Yao, Yao, and He]{wang-etal:awm}
Zhaoyang Wang, Canwen Xu, Boyi Liu, Yite Wang, Siwei Han, Zhewei Yao, Huaxiu Yao, and Yuxiong He.
\newblock Agent world model: Infinity synthetic environments for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2602.10090}, 2026{\natexlab{d}}.
\bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned}
Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le.
\newblock Finetuned language models are zero-shot learners.
......
......@@ -2,6 +2,93 @@
@article{zhang-etal:autoenv,
title = {AutoEnv: Automated Environments for Measuring
Cross-Environment Agent Learning},
author = {Zhang, Jiayi and Peng, Yiran and Kong, Fanqi and Cheng, Yang
and Wu, Yifan and Yu, Zhaoyang and Xiang, Jinyu
and Ruan, Jianhao and Wang, Jinlin and Song, Maojia
and Liu, HongZhang and Tang, Xiangru and Liu, Bang
and Wu, Chenglin and Luo, Yuyu},
journal = {arXiv preprint arXiv:2511.19304},
year = {2025},
eprint = {2511.19304},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{cai-etal:autoforge,
title = {AutoForge: Automated Environment Synthesis for
Agentic Reinforcement Learning},
author = {Cai, Shihao and Fang, Runnan and Wu, Jialong and Li, Baixuan
and Wang, Xinyu and Jiang, Yong and Su, Liangcai
and Zhang, Liwen and Yin, Wenbiao and Zhang, Zhen
and Feng, Fuli and Xie, Pengjun and Wang, Xiaobin},
journal = {arXiv preprint arXiv:2512.22857},
year = {2025},
eprint = {2512.22857},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{wang-etal:awm,
title = {Agent World Model: Infinity Synthetic Environments for
Agentic Reinforcement Learning},
author = {Wang, Zhaoyang and Xu, Canwen and Liu, Boyi and Wang, Yite
and Han, Siwei and Yao, Zhewei and Yao, Huaxiu
and He, Yuxiong},
journal = {arXiv preprint arXiv:2602.10090},
year = {2026},
eprint = {2602.10090},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{song-etal:envscaler,
title = {EnvScaler: Scaling Tool-Interactive Environments for
LLM Agent via Programmatic Synthesis},
author = {Song, Xiaoshuai and Chang, Haofei and Dong, Guanting
and Zhu, Yutao and Dou, Zhicheng and Wen, Ji-Rong},
journal = {arXiv preprint arXiv:2601.05808},
year = {2026},
eprint = {2601.05808},
archivePrefix = {arXiv},
primaryClass = {cs.CL}
}
@article{deepseek-ai:deepseekv32,
title = {DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models},
author = {DeepSeek-AI},
journal = {arXiv preprint arXiv:2512.02556},
year = {2025},
eprint = {2512.02556},
archivePrefix = {arXiv},
primaryClass = {cs.CL}
}
@article{fang-etal:agentscaler,
title = {Towards General Agentic Intelligence via Environment Scaling},
author = {Fang, Runnan and Cai, Shihao and Li, Baixuan and Wu, Jialong
and Li, Guangyu and Yin, Wenbiao and Wang, Xinyu
and Wang, Xiaobin and Su, Liangcai and Zhang, Zhen
and Wu, Shibin and Tao, Zhengwei and Jiang, Yong
and Xie, Pengjun and Huang, Fei and Zhou, Jingren},
journal = {arXiv preprint arXiv:2509.13311},
year = {2025},
eprint = {2509.13311},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{sullivan-etal:randomworld,
title = {Procedural Environment Generation for Tool-Use Agents},
author = {Sullivan, Michael and Hartmann, Mareike and Koller, Alexander},
journal = {arXiv preprint arXiv:2506.11045},
year = {2025},
eprint = {2506.11045},
archivePrefix = {arXiv},
primaryClass = {cs.CL}
}
@article{chen-etal:research,
title = {ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning},
......
No preview for this file type
% Required:
% \usetikzlibrary{positioning,shadows}
\begin{tikzpicture}[
box/.style={
draw,
fill=white,
drop shadow={shadow xshift=0.08cm, shadow yshift=-0.08cm},
text width=2.75cm,
minimum height=2.25cm,
inner sep=0.12cm,
font=\small,
align=center
},
widebox/.style={
draw,
fill=white,
drop shadow={shadow xshift=0.08cm, shadow yshift=-0.08cm},
text width=3.2cm,
minimum height=2.25cm,
inner sep=0.12cm,
font=\small,
align=center
},
arrow/.style={
->,
thick
},
dashedarrow/.style={
->,
thick,
dashed
}
]
\node[box] (spec) {
\textbf{\textit{Environment Specification}}\\[0.08cm]
State schema\\
Tools and APIs\\
Transition rules
};
\node[box, right=0.55cm of spec] (synthesis) {
\textbf{\textit{Programmatic Synthesis}}\\[0.08cm]
Executable code\\
Database states\\
Tool interfaces
};
\node[box, right=0.55cm of synthesis] (scenario) {
\textbf{\textit{Scenario Generation}}\\[0.08cm]
Initial states\\
User tasks\\
Difficulty control
};
\node[widebox, right=0.55cm of scenario] (rollout) {
\textbf{\textit{Agent--Environment Interaction}}\\[0.08cm]
$u_1 \rightarrow o_1 \rightarrow \cdots$\\
$\rightarrow u_T \rightarrow o_T$
};
\node[box, right=0.55cm of rollout] (verify) {
\textbf{\textit{Verification}}\\[0.08cm]
State checking\\
Execution success\\
Task completion
};
\node[widebox, below=1.05cm of rollout] (learning) {
\textbf{\textit{Agent Learning}}\\[0.08cm]
Trajectory collection\\
SFT or RL update
};
\draw[arrow] (spec) -- (synthesis);
\draw[arrow] (synthesis) -- (scenario);
\draw[arrow] (scenario) -- (rollout);
\draw[arrow] (rollout) -- (verify);
\draw[arrow] (verify.south) |- (learning.east);
\draw[dashedarrow]
(learning.west)
-| (scenario.south);
% \node[font=\small, align=center, below=0.42cm of learning] {
% Verified outcomes support training, while failures guide scenario generation.
% };
\end{tikzpicture}
\ No newline at end of file
......@@ -238,16 +238,65 @@ In application, this reward decomposition provides denser feedback than final-an
Another direction is to use RL to improve strategic tool use. For example, ReTool first builds a cold-start model with code-augmented reasoning traces and then applies RL with real-time code execution. During rollout, the model interleaves natural language reasoning with code execution, observes execution results, and revises its subsequent reasoning. This allows the model to discover when and how to invoke a code interpreter based on outcome feedback rather than human-written rules \citep{feng-etal:retool}. Search-R1 and ReSearch follow a similar idea in retrieval-augmented reasoning. They train models to generate search queries during reasoning and use retrieved information to update later steps, rather than relying on fixed retrieval pipelines or hand-crafted search prompts \citep{jin-etal:searchr1,chen-etal:research}.
So far, we have introduced how RL can enhance the core capabilities of LLM-based agents, including planning and tool use. Since an LLM-based agent is inherently driven by an underlying LLM, its RL training can reuse the general LLM-based RL algorithms introduced in Section \ref{sec:example-using-rl-training-llms}. The main differences lie in the agentic setting. Compared with standard response optimization, agentic RL involves longer trajectories, environmental feedback, and intermediate decisions such as planning and tool invocation. Thus, the key challenges are not only the choice of RL algorithms, but also trajectory construction, reward design, and environment optimization.
\subsubsection{Improved Environments}
\subsection{Environment Design and Scaling}
The previous subsections focus on optimizing the agent policy for planning and tool use. In this subsection, we turn to another important component of agentic RL: the environment that produces interaction experience. An environment is more than a testbed. It determines which actions the agent can perform, what observations it receives, how the underlying state changes after each action, and how task success is evaluated. More formally, we can represent an environment as
\begin{eqnarray}
e = (\mathcal{S},\mathcal{U},\mathcal{O},P_e,V_e)
\end{eqnarray}
where $\mathcal{S}$ is the state space, $\mathcal{U}$ is the set of available actions or tool invocations, and $\mathcal{O}$ is the observation space. The transition function $P_e$ determines how an action changes the environment state, while the verifier $V_e$ determines whether the resulting state satisfies the task objective. This formulation describes how the environment produces observations and feedback, rather than how the agent generates its actions.
The quality of an environment directly affects what an agent can learn. For example, an agent trained in a narrow environment may simply memorize specific APIs, task templates, or interaction patterns. An unreliable environment may return inconsistent observations or incorrect rewards, introducing noise into the learning process. Ideally, an effective environment should provide diverse tasks, executable interactions, reliable state transitions, and verifiable outcomes.
To encourage generalization, we can train the agent over a distribution of environments rather than within a single fixed environment:
\begin{eqnarray}
\theta^{*} =
\arg\max_{\theta}
\mathbb{E}_{e\sim p(\mathcal{E}), q\sim p_e(\mathcal{Q})}
\mathbb{E}_{\tau\sim\pi_{\theta}(\cdot\mid q,e)}
\left[
R_e(\tau)
\right]
\end{eqnarray}
where $p(\mathcal{E})$ denotes the environment distribution, and $p_e(\mathcal{Q})$ denotes the task distribution within environment $e$. The interaction trajectory $\tau$ follows the definition introduced in the previous subsections. This objective shows that agent performance depends not only on policy optimization, but also on the coverage and quality of the environments used for training.
However, constructing high-quality environments manually is expensive and difficult to scale. Real-world systems may be inaccessible, costly, or unsafe for large-scale exploration. Purely language-based simulations are easier to build, but they may generate inconsistent state transitions, invalid tool outputs, or unreliable evaluations. Thus, recent studies explore procedural and programmatic environment synthesis, where executable programs and structured states are used to provide a scalable and reliable interaction experience.
\begin{figure}[t!]
\centering
\resizebox{\linewidth}{!}{
\input{section6/Figures/environment-design}}
\caption{Overview of environment synthesis and interaction for agentic reinforcement learning.}
\label{fig:agent-environment-scaling}
\end{figure}
This process is illustrated in Figure~\ref{fig:agent-environment-scaling}. It typically consists of the following steps:
\begin{itemize}
\item \textbf{Environment Specification.} We first define the state schema\footnote{A state schema specifies the structured variables used to represent the current status of an environment, together with their types and possible values. For example, in a travel-booking environment, it may define fields for available flights, user preferences, and existing reservations.}, available tools and APIs, and transition rules of the environment.
\item \textbf{Programmatic Synthesis.} We then implement the environment as an executable system. This process typically converts the structured specification into program code, database tables, and callable tool interfaces.
\item \textbf{Scenario Generation.} Based on the environment specification, we generate diverse initial states and user tasks. The tasks should cover different goals, constraints, and difficulty levels. For example, in a shopping environment, a simple task may ask the agent to purchase a single item, while a more difficult task may require it to compare multiple products.
\item \textbf{Agent--Environment Interaction.} The agent interacts with the generated environment through multiple rounds of actions and observations. At each step, the agent selects a tool and provides the required arguments. The environment executes the action and returns an observation.
\item \textbf{Outcome Verification.} After the interaction is completed, a verifier checks whether the task has been successfully solved. The verifier may inspect tool execution results, database states, or other structured records.
\item \textbf{Agent Learning.} Finally, the verified trajectories are used to improve the agent. Successful trajectories can serve as demonstrations for SFT, while task-level or step-level feedback can support RL training. Note that failed interactions are also useful because they reveal weaknesses in the current policy or gaps in the task distribution. For example, repeated failures on invalid API arguments may motivate the generation of additional scenarios involving parameter constraints. In this way, learning results can further guide scenario generation and form an iterative environment--agent improvement loop.
\end{itemize}
So far, we have introduced how RL can enhance the core capabilities of LLM-based agents, including planning and tool use. Since an LLM-based agent is inherently driven by an underlying LLM, its RL training can reuse the general LLM-based RL algorithms introduced in Section \ref{sec:example-using-rl-training-llms}. The main differences lie in the agentic setting. Compared with standard response optimization, agentic RL involves longer trajectories, environmental feedback, and intermediate decisions such as planning and tool invocation. Thus, the key challenges are not only the choice of RL algorithms, but also trajectory construction, reward design, and environment optimization.
Despite this progress, several important questions remain. First, how can we scale both the number and diversity of environments? Second, how can we ensure that generated environments are executable and verifiable? Third, how can agents generalize across heterogeneous and previously unseen environments?
One direction focuses on scaling environment diversity. RandomWorld procedurally generates executable tools and compositional tasks for tool-use agents \citep{sullivan-etal:randomworld}. Unlike static datasets that contain only predefined trajectories, these generated environments support online execution and allow agents to explore different tool combinations during training. AgentScaler further constructs heterogeneous simulated environments and separates the learning of general function-calling behaviors from domain-specific adaptation \citep{fang-etal:agentscaler}. At a larger scale, DeepSeek-V3.2 synthesizes diverse environments and complex tasks for agentic post-training, showing that broad interaction experience is important for long-tail generalization \citep{deepseek-ai:deepseekv32}.
Another direction focuses on executability. EnvScaler first constructs environment skeletons that specify state schemas, available tools, and interaction rules. It then generates multiple task scenarios within each environment \citep{song-etal:envscaler}. Agent World Model follows a similar idea by implementing synthetic environments with executable code and database-backed states \citep{wang-etal:awm}. Compared with purely language-based simulations, these programmatic environments produce more consistent state transitions and generate tool outputs through actual execution.
Verifiability is equally important because RL requires reliable feedback. Instead of evaluating only the textual final response, an environment can check whether the agent has produced the desired state change. Given the final environment state $s_T$ and the task goal $g$, we can define a state-based reward as
\begin{eqnarray}
R_e(\tau)=V_e(s_T,g)
\end{eqnarray}
where $V_e(\cdot)$ may be implemented using executable tests, database queries, or rule-based validators. For example, in a database operation task, the verifier can directly check whether the required record has been inserted correctly. This provides more reliable feedback than comparing the generated answer with a reference text. Both EnvScaler and Agent World Model use such mechanisms to connect agent actions with observable state changes.
Scaling environments also introduces substantial heterogeneity. Different environments may vary in task difficulty, available tools, and interaction length. These differences can make RL training unstable and cause the agent to overfit to frequently sampled or easier environments. AutoForge addresses this issue by synthesizing difficult but verifiable tasks and estimating learning signals at the environment level \citep{cai-etal:autoforge}. This design reduces the influence of unstable simulated interactions and improves training across heterogeneous environments.
\subsection{Learning from Agentic Experience}
......@@ -256,6 +305,7 @@ So far, we have introduced how RL can enhance the core capabilities of LLM-based
\subsubsection{Skill Optimization}
\subsubsection{Trajectory Refinement}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论