Commit a499d184 by wangchenglong

update.

parent 448a1534
\begin{thebibliography}{128} \begin{thebibliography}{134}
\providecommand{\natexlab}[1]{#1} \providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}} \providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax \expandafter\ifx\csname urlstyle\endcsname\relax
...@@ -42,6 +42,11 @@ Jane Bromley, Isabelle Guyon, Yann LeCun, Eduard S{\"a}ckinger, and Roopak Shah. ...@@ -42,6 +42,11 @@ Jane Bromley, Isabelle Guyon, Yann LeCun, Eduard S{\"a}ckinger, and Roopak Shah.
\newblock Signature verification using a" siamese" time delay neural network. \newblock Signature verification using a" siamese" time delay neural network.
\newblock \emph{Advances in neural information processing systems}, 6, 1993. \newblock \emph{Advances in neural information processing systems}, 6, 1993.
\bibitem[Cai et~al.(2025)Cai, Fang, Wu, Li, Wang, Jiang, Su, Zhang, Yin, Zhang, Feng, Xie, and Wang]{cai-etal:autoforge}
Shihao Cai, Runnan Fang, Jialong Wu, Baixuan Li, Xinyu Wang, Yong Jiang, Liangcai Su, Liwen Zhang, Wenbiao Yin, Zhen Zhang, Fuli Feng, Pengjun Xie, and Xiaobin Wang.
\newblock Autoforge: Automated environment synthesis for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2512.22857}, 2025.
\bibitem[Chen et~al.(2023{\natexlab{a}})Chen, Shu, Shareghi, Collier, Narasimhan, and Yao]{chen-etal:fireact} \bibitem[Chen et~al.(2023{\natexlab{a}})Chen, Shu, Shareghi, Collier, Narasimhan, and Yao]{chen-etal:fireact}
Baian Chen, Chang Shu, Ehsan Shareghi, Nigel Collier, Karthik Narasimhan, and Shunyu Yao. Baian Chen, Chang Shu, Ehsan Shareghi, Nigel Collier, Karthik Narasimhan, and Shunyu Yao.
\newblock Fireact: Toward language agent fine-tuning. \newblock Fireact: Toward language agent fine-tuning.
...@@ -103,6 +108,11 @@ Deepseek. ...@@ -103,6 +108,11 @@ Deepseek.
\newblock Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. \newblock Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2501.12948}, 2025. \newblock \emph{arXiv preprint arXiv:2501.12948}, 2025.
\bibitem[DeepSeek-AI(2025)]{deepseek-ai:deepseekv32}
DeepSeek-AI.
\newblock Deepseek-v3.2: Pushing the frontier of open large language models.
\newblock \emph{arXiv preprint arXiv:2512.02556}, 2025.
\bibitem[Dubois et~al.(2023)Dubois, Li, Taori, Zhang, Gulrajani, Ba, Guestrin, Liang, and Hashimoto]{dubois-etal:2024alpacafarm} \bibitem[Dubois et~al.(2023)Dubois, Li, Taori, Zhang, Gulrajani, Ba, Guestrin, Liang, and Hashimoto]{dubois-etal:2024alpacafarm}
Yann Dubois, Chen~Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy Liang, and Tatsunori~B. Hashimoto. Yann Dubois, Chen~Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy Liang, and Tatsunori~B. Hashimoto.
\newblock Alpacafarm: {A} simulation framework for methods that learn from human feedback. \newblock Alpacafarm: {A} simulation framework for methods that learn from human feedback.
...@@ -121,6 +131,11 @@ Kawin Ethayarajh, Winnie Xu, Niklas Muennighoff, Dan Jurafsky, and Douwe Kiela. ...@@ -121,6 +131,11 @@ Kawin Ethayarajh, Winnie Xu, Niklas Muennighoff, Dan Jurafsky, and Douwe Kiela.
\newblock \emph{ArXiv preprint}, abs/2402.01306, 2024. \newblock \emph{ArXiv preprint}, abs/2402.01306, 2024.
\newblock URL \url{https://arxiv.org/abs/2402.01306}. \newblock URL \url{https://arxiv.org/abs/2402.01306}.
\bibitem[Fang et~al.(2025)Fang, Cai, Li, Wu, Li, Yin, Wang, Wang, Su, Zhang, Wu, Tao, Jiang, Xie, Huang, and Zhou]{fang-etal:agentscaler}
Runnan Fang, Shihao Cai, Baixuan Li, Jialong Wu, Guangyu Li, Wenbiao Yin, Xinyu Wang, Xiaobin Wang, Liangcai Su, Zhen Zhang, Shibin Wu, Zhengwei Tao, Yong Jiang, Pengjun Xie, Fei Huang, and Jingren Zhou.
\newblock Towards general agentic intelligence via environment scaling.
\newblock \emph{arXiv preprint arXiv:2509.13311}, 2025.
\bibitem[Feng et~al.(2025)Feng, Huang, Qu, Zhang, Qin, Zhong, Jiang, Chi, and Zhong]{feng-etal:retool} \bibitem[Feng et~al.(2025)Feng, Huang, Qu, Zhang, Qin, Zhong, Jiang, Chi, and Zhong]{feng-etal:retool}
Jiazhan Feng, Shijue Huang, Xingwei Qu, Ge~Zhang, Yujia Qin, Baoquan Zhong, Chengquan Jiang, Jinxin Chi, and Wanjun Zhong. Jiazhan Feng, Shijue Huang, Xingwei Qu, Ge~Zhang, Yujia Qin, Baoquan Zhong, Chengquan Jiang, Jinxin Chi, and Wanjun Zhong.
\newblock Retool: Reinforcement learning for strategic tool use in llms. \newblock Retool: Reinforcement learning for strategic tool use in llms.
...@@ -419,6 +434,16 @@ Prasann Singhal, Tanya Goyal, Jiacheng Xu, and Greg Durrett. ...@@ -419,6 +434,16 @@ Prasann Singhal, Tanya Goyal, Jiacheng Xu, and Greg Durrett.
\newblock \emph{ArXiv preprint}, abs/2310.03716, 2023. \newblock \emph{ArXiv preprint}, abs/2310.03716, 2023.
\newblock URL \url{https://arxiv.org/abs/2310.03716}. \newblock URL \url{https://arxiv.org/abs/2310.03716}.
\bibitem[Song et~al.(2026)Song, Chang, Dong, Zhu, Dou, and Wen]{song-etal:envscaler}
Xiaoshuai Song, Haofei Chang, Guanting Dong, Yutao Zhu, Zhicheng Dou, and Ji-Rong Wen.
\newblock Envscaler: Scaling tool-interactive environments for llm agent via programmatic synthesis.
\newblock \emph{arXiv preprint arXiv:2601.05808}, 2026.
\bibitem[Sullivan et~al.(2025)Sullivan, Hartmann, and Koller]{sullivan-etal:randomworld}
Michael Sullivan, Mareike Hartmann, and Alexander Koller.
\newblock Procedural environment generation for tool-use agents.
\newblock \emph{arXiv preprint arXiv:2506.11045}, 2025.
\bibitem[Sun et~al.(2024)Sun, Liu, Bair, and Kolter]{sun-etal:2023simple} \bibitem[Sun et~al.(2024)Sun, Liu, Bair, and Kolter]{sun-etal:2023simple}
Mingjie Sun, Zhuang Liu, Anna Bair, and J.~Zico Kolter. Mingjie Sun, Zhuang Liu, Anna Bair, and J.~Zico Kolter.
\newblock A simple and effective pruning approach for large language models. \newblock A simple and effective pruning approach for large language models.
...@@ -541,6 +566,11 @@ Yizhong Wang, Hamish Ivison, Pradeep Dasigi, Jack Hessel, Tushar Khot, Khyathi C ...@@ -541,6 +566,11 @@ Yizhong Wang, Hamish Ivison, Pradeep Dasigi, Jack Hessel, Tushar Khot, Khyathi C
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023{\natexlab{d}}. \newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023{\natexlab{d}}.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/ec6413875e4ab08d7bc4d8e225263398-Abstract-Datasets\_and\_Benchmarks.html}. \newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/ec6413875e4ab08d7bc4d8e225263398-Abstract-Datasets\_and\_Benchmarks.html}.
\bibitem[Wang et~al.(2026{\natexlab{d}})Wang, Xu, Liu, Wang, Han, Yao, Yao, and He]{wang-etal:awm}
Zhaoyang Wang, Canwen Xu, Boyi Liu, Yite Wang, Siwei Han, Zhewei Yao, Huaxiu Yao, and Yuxiong He.
\newblock Agent world model: Infinity synthetic environments for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2602.10090}, 2026{\natexlab{d}}.
\bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned} \bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned}
Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le. Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le.
\newblock Finetuned language models are zero-shot learners. \newblock Finetuned language models are zero-shot learners.
......
...@@ -2,6 +2,93 @@ ...@@ -2,6 +2,93 @@
@article{zhang-etal:autoenv,
title = {AutoEnv: Automated Environments for Measuring
Cross-Environment Agent Learning},
author = {Zhang, Jiayi and Peng, Yiran and Kong, Fanqi and Cheng, Yang
and Wu, Yifan and Yu, Zhaoyang and Xiang, Jinyu
and Ruan, Jianhao and Wang, Jinlin and Song, Maojia
and Liu, HongZhang and Tang, Xiangru and Liu, Bang
and Wu, Chenglin and Luo, Yuyu},
journal = {arXiv preprint arXiv:2511.19304},
year = {2025},
eprint = {2511.19304},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{cai-etal:autoforge,
title = {AutoForge: Automated Environment Synthesis for
Agentic Reinforcement Learning},
author = {Cai, Shihao and Fang, Runnan and Wu, Jialong and Li, Baixuan
and Wang, Xinyu and Jiang, Yong and Su, Liangcai
and Zhang, Liwen and Yin, Wenbiao and Zhang, Zhen
and Feng, Fuli and Xie, Pengjun and Wang, Xiaobin},
journal = {arXiv preprint arXiv:2512.22857},
year = {2025},
eprint = {2512.22857},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{wang-etal:awm,
title = {Agent World Model: Infinity Synthetic Environments for
Agentic Reinforcement Learning},
author = {Wang, Zhaoyang and Xu, Canwen and Liu, Boyi and Wang, Yite
and Han, Siwei and Yao, Zhewei and Yao, Huaxiu
and He, Yuxiong},
journal = {arXiv preprint arXiv:2602.10090},
year = {2026},
eprint = {2602.10090},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{song-etal:envscaler,
title = {EnvScaler: Scaling Tool-Interactive Environments for
LLM Agent via Programmatic Synthesis},
author = {Song, Xiaoshuai and Chang, Haofei and Dong, Guanting
and Zhu, Yutao and Dou, Zhicheng and Wen, Ji-Rong},
journal = {arXiv preprint arXiv:2601.05808},
year = {2026},
eprint = {2601.05808},
archivePrefix = {arXiv},
primaryClass = {cs.CL}
}
@article{deepseek-ai:deepseekv32,
title = {DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models},
author = {DeepSeek-AI},
journal = {arXiv preprint arXiv:2512.02556},
year = {2025},
eprint = {2512.02556},
archivePrefix = {arXiv},
primaryClass = {cs.CL}
}
@article{fang-etal:agentscaler,
title = {Towards General Agentic Intelligence via Environment Scaling},
author = {Fang, Runnan and Cai, Shihao and Li, Baixuan and Wu, Jialong
and Li, Guangyu and Yin, Wenbiao and Wang, Xinyu
and Wang, Xiaobin and Su, Liangcai and Zhang, Zhen
and Wu, Shibin and Tao, Zhengwei and Jiang, Yong
and Xie, Pengjun and Huang, Fei and Zhou, Jingren},
journal = {arXiv preprint arXiv:2509.13311},
year = {2025},
eprint = {2509.13311},
archivePrefix = {arXiv},
primaryClass = {cs.AI}
}
@article{sullivan-etal:randomworld,
title = {Procedural Environment Generation for Tool-Use Agents},
author = {Sullivan, Michael and Hartmann, Mareike and Koller, Alexander},
journal = {arXiv preprint arXiv:2506.11045},
year = {2025},
eprint = {2506.11045},
archivePrefix = {arXiv},
primaryClass = {cs.CL}
}
@article{chen-etal:research, @article{chen-etal:research,
title = {ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning}, title = {ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning},
......
No preview for this file type
% Required:
% \usetikzlibrary{positioning,shadows}
\begin{tikzpicture}[
box/.style={
draw,
fill=white,
drop shadow={shadow xshift=0.08cm, shadow yshift=-0.08cm},
text width=2.75cm,
minimum height=2.25cm,
inner sep=0.12cm,
font=\small,
align=center
},
widebox/.style={
draw,
fill=white,
drop shadow={shadow xshift=0.08cm, shadow yshift=-0.08cm},
text width=3.2cm,
minimum height=2.25cm,
inner sep=0.12cm,
font=\small,
align=center
},
arrow/.style={
->,
thick
},
dashedarrow/.style={
->,
thick,
dashed
}
]
\node[box] (spec) {
\textbf{\textit{Environment Specification}}\\[0.08cm]
State schema\\
Tools and APIs\\
Transition rules
};
\node[box, right=0.55cm of spec] (synthesis) {
\textbf{\textit{Programmatic Synthesis}}\\[0.08cm]
Executable code\\
Database states\\
Tool interfaces
};
\node[box, right=0.55cm of synthesis] (scenario) {
\textbf{\textit{Scenario Generation}}\\[0.08cm]
Initial states\\
User tasks\\
Difficulty control
};
\node[widebox, right=0.55cm of scenario] (rollout) {
\textbf{\textit{Agent--Environment Interaction}}\\[0.08cm]
$u_1 \rightarrow o_1 \rightarrow \cdots$\\
$\rightarrow u_T \rightarrow o_T$
};
\node[box, right=0.55cm of rollout] (verify) {
\textbf{\textit{Verification}}\\[0.08cm]
State checking\\
Execution success\\
Task completion
};
\node[widebox, below=1.05cm of rollout] (learning) {
\textbf{\textit{Agent Learning}}\\[0.08cm]
Trajectory collection\\
SFT or RL update
};
\draw[arrow] (spec) -- (synthesis);
\draw[arrow] (synthesis) -- (scenario);
\draw[arrow] (scenario) -- (rollout);
\draw[arrow] (rollout) -- (verify);
\draw[arrow] (verify.south) |- (learning.east);
\draw[dashedarrow]
(learning.west)
-| (scenario.south);
% \node[font=\small, align=center, below=0.42cm of learning] {
% Verified outcomes support training, while failures guide scenario generation.
% };
\end{tikzpicture}
\ No newline at end of file
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论