title = {A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications},
author = {Zhou, Yingli and Shu, Wang and Su, Yaodong and Du, Wenchuan and Fang, Yixiang and Lin, Xuemin},
journal = {arXiv preprint arXiv:2605.07358},
year = {2026}
}
@article{lu-etal:skill0,
title = {SKILL0: In-Context Agentic Reinforcement Learning for Skill Internalization},
author = {Lu, Zhengxi and Yao, Zhiyuan and Wu, Jinyang and Han, Chengcheng and Gu, Qi and Cai, Xunliang and Lu, Weiming and Xiao, Jun and Zhuang, Yueting and Shen, Yongliang},
journal = {arXiv preprint arXiv:2604.02268},
year = {2026}
}
@article{xia-etal:skillrl,
title = {SkillRL: Evolving Agents via Recursive Skill-Augmented Reinforcement Learning},
author = {Xia, Peng and Chen, Jianwen and Wang, Hanyang and Liu, Jiaqi and Zeng, Kaide and Wang, Yu and Han, Siwei and Zhou, Yiyang and Zhao, Xujiang and Chen, Haifeng and Zheng, Zeyu and Xie, Cihang and Yao, Huaxiu},
journal = {arXiv preprint arXiv:2602.08234},
year = {2026}
}
@article{zhang-etal:coevoskills,
title = {CoEvoSkills: Self-Evolving Agent Skills via Co-Evolutionary Verification},
author = {Zhang, Hanrong and Fan, Shicheng and Zou, Henry Peng and Chen, Yankai and Wang, Zhenting and Zhou, Jiayu and Li, Chengze and Huang, Wei-Chieh and Yao, Yifei and Zheng, Kening and Liu, Xue and Li, Xiaoxiao and Yu, Philip S.},
journal = {arXiv preprint arXiv:2604.01687},
year = {2026}
}
@article{zhu-etal:skill05,
title = {Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning},
author = {Zhu, Jiapeng and Yu, Jianxiang and Zhao, Yibo and Han, Chengcheng and Gu, Qi and Cai, Xunliang and Li, Xiang and Qian, Weining},
journal = {arXiv preprint arXiv:2605.28424},
year = {2026}
}
@article{shi-etal:skill1,
title = {Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning},
author = {Shi, Yaorui and Chen, Yuxin and Lu, Zhengxi and Miao, Yuchun and Liu, Shugui and Gu, Qi and Cai, Xunliang and Wang, Xiang and Zhang, An},
journal = {arXiv preprint arXiv:2605.06130},
year = {2026}
}
@article{yang-etal:skillmaster,
title = {SkillMaster: Toward Autonomous Skill Mastery in LLM Agents},
author = {Yang, Min and Piao, Jinghua and Xia, Xu and Lan, Xiaochong and Chen, Jiaju and Gong, Yongshun and Li, Yong},
journal = {arXiv preprint arXiv:2605.08693},
year = {2026}
}
@article{he-etal:reskill,
title = {ReSkill: Reconciling Skill Creation with Policy Optimization in Agentic RL},
author = {He, Zelin and Lin, Haotian and Han, Boran and Zhu, Wei and Fang, Haoyang and Wang, Bernie and Zhu, Xuan and Li, Runze and Reimherr, Matthew},
journal = {arXiv preprint arXiv:2606.01619},
year = {2026}
}
@article{li-etal:arise,
title = {ARISE: Agent Reasoning with Intrinsic Skill Evolution in Hierarchical Reinforcement Learning},
author = {Li, Yu and Miao, Rui and Qi, Zhengling and Lan, Tian},
journal = {arXiv preprint arXiv:2603.16060},
year = {2026}
}
@article{yang-etal:skillopt,
title = {SkillOpt: Executive Strategy for Self-Evolving Agent Skills},
author = {Yang, Yifan and Gong, Ziyang and Huang, Weiquan and Yang, Qihao and Zhou, Ziwei and Huang, Zisu and Li, Yan and Gao, Xuemei and Dai, Qi and Liu, Bei and Qiu, Kai and Yang, Yuqing and Chen, Dongdong and Yang, Xue and Luo, Chong},
journal = {arXiv preprint arXiv:2605.23904},
year = {2026}
}
@article{wang-etal:skillgrad,
title = {SkillGrad: Optimizing Agent Skills Like Gradient Descent},
author = {Wang, Hanyu and Lan, Yifan and Cao, Bochuan and Lin, Lu and Chen, Jinghui},
journal = {arXiv preprint arXiv:2605.27760},
year = {2026}
}
@article{liu-etal:skillrevise,
title = {SkillRevise: Improving LLM-Authored Agent Skills via Trace-Conditioned Skill Revision},
author = {Liu, Yuxuan and Su, Zhaochen and Xie, Lingyun and Zhang, Yuhao and Zong, Qing and Guo, Jiahe and Xie, Zhongwei and Ji, Yiyan and Yim, Yauwai and Luo, Hongyu and Ren, Xiyu and Chenyu, Ruan and Li, Haoran and Song, Yangqiu},
journal = {arXiv preprint arXiv:2606.01139},
year = {2026}
}
@article{alzubi-etal:evoskill,
title = {EvoSkill: Automated Skill Discovery for Multi-Agent Systems},
author = {Alzubi, Salaheddin and Provenzano, Noah and Bingham, Jaydon and Chen, Weiyuan and Vu, Tu},
journal = {arXiv preprint arXiv:2603.02766},
year = {2026}
}
@article{tanjim-etal:mocha,
title = {MOCHA: Multi-Objective Chebyshev Annealing for Agent Skill Optimization},
author = {Tanjim, Md Mehrab and Subramanian, Jayakumar and Chen, Xiang and Kveton, Branislav and Mukherjee, Subhojyoti and Zhang, Anlan and Kim, Sungchul and Sarkhel, Somdeb and Choudhury, Sunav},
journal = {arXiv preprint arXiv:2605.19330},
year = {2026}
}
@article{moll-etal:grasp,
title = {GRASP: Gated Regression-Aware Skill Proposer for Self-Improving LLM Agents},
author = {Moll, Johannes and Corbeil, Jean-Philippe and Pan, Jiazhen and Hadamitzky, Martin and Rueckert, Daniel and Adams, Lisa and Bressem, Keno},
journal = {arXiv preprint arXiv:2605.29668},
year = {2026}
}
@article{zhou-etal:mentoskills,
title = {Memento-Skills: Let Agents Design Agents},
author = {Zhou, Huichi and Guo, Siyuan and Liu, Anjie and Yu, Zhongwei and Gong, Ziqin and Zhao, Bowen and Chen, Zhixun and Zhang, Menglong and Chen, Yihang and Li, Jinsong and Yang, Runyu and Liu, Qiangbin and Yu, Xinlei and Zhou, Jianmin and Wang, Na and Sun, Chunyang and Wang, Jun},
journal = {arXiv preprint arXiv:2603.18743},
year = {2026}
}
@article{lin-etal:museautoskill,
title = {MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation},
author = {Lin, Huawei and Li, Peng and Song, Jie and Jiang, Fuxin and Zhang, Tieying},
journal = {arXiv preprint arXiv:2605.27366},
year = {2026}
}
@article{si-etal:ctx2skill,
title = {From Context to Skills: Can Language Models Learn from Context Skillfully?},
author = {Si, Shuzheng and Zhao, Haozhe and Lei, Yu and Wang, Qingyi and Chen, Dingwei and Wang, Zhitong and Wang, Zhenhailong and Luo, Kangyang and Wang, Zheng and Chen, Gang and Qi, Fanchao and Zhang, Minjia and Sun, Maosong},
journal = {arXiv preprint arXiv:2604.27660},
year = {2026}
}
@article{zhang-etal:skillcomposer,
title = {SkillComposer: Learning to Evolve Agent Skills for Specification and Generalization},
author = {Zhang, Qi and Feng, Zhaopeng and Shi, Xiaonan and Hu, Xiaomeng and Liu, Chu and Xie, Pengjun and Wang, Xiaobin and Ye, Jieping and Hooi, Bryan and Wang, Haobo and Zhao, Junbo},
journal = {arXiv preprint arXiv:2606.06079},
year = {2026}
}
@article{huang-etal:cascade,
title = {CASCADE: Cumulative Agentic Skill Creation through Autonomous Development and Evolution},
author = {Huang, Xu and Chen, Junwu and Fei, Yuxing and Li, Zhuohan and Schwaller, Philippe and Ceder, Gerbrand},
journal = {arXiv preprint arXiv:2512.23880},
year = {2025}
}
@article{wei-etal:skillsmith,
title = {SkillSmith: Co-Evolving Skills and Tools for Self-Improving Agent Systems},
author = {Wei, Yangbo and Huang, Zhen and Lu, Shaoqiang and Qian, Junhong and Wang, Qifan and Wu, Chen and He, Lei},
journal = {arXiv preprint arXiv:2606.01314},
year = {2026}
}
@article{pan-etal:skillmas,
title = {SkillMAS: Skill Co-Evolution with LLM-based Multi-Agent System},
author = {Pan, Shuai and Liu, Yixiang and Gao, Jiaye and Gao, Te and Liu, Weiwen and Lin, Jianghao and Fu, Zhihui and Wang, Jun and Zhang, Weinan and Yu, Yong},
journal = {arXiv preprint arXiv:2605.09341},
year = {2026}
}
@article{li-etal:skillhone,
title = {SkillHone: A Harness for Continual Agent Skill Evolution Through Persistent Decision History},
author = {Li, Zhiwei and Hu, Yong},
journal = {arXiv preprint arXiv:2606.08671},
year = {2026}
}
@article{li-etal:hisme,
title = {You Live More Than Once: Towards Hierarchical Skill Meta-Evolving},
author = {Li, Xujun and Zheng, Kehan and Zhao, Mingyuan and Geng, Yize and Zhou, Jinfeng and Zhu, Qi and Mi, Fei and Shang, Lifeng and Huang, Minlie and Wang, Hongning},
journal = {arXiv preprint arXiv:2605.28390},
year = {2026}
}
@article{yang-etal:federatedskill,
title = {FederatedSkill: Federated Learning for Agentic Skill Evolution},
author = {Yang, Jingbo and Yao, Guanyu and Zhang, Yang and Kompella, Ramana Rao and Liu, Gaowen and Chang, Shiyu},
journal = {arXiv preprint arXiv:2606.03143},
year = {2026}
}
@article{wu-etal:bayesianagent,
title = {Bayesian-Agent: Posterior-Guided Skill Evolution for LLM Agent Harnesses},
author = {Wu, Xiaojun and Yang, Cehao and Liu, Honghao and Lin, Xueyuan and Zhang, Wenjie and Shi, Zhichao and Jiang, Xuhui and Xu, Chengjin and Li, Jia and Guo, Jian},
@@ -250,6 +250,15 @@ So far, we have introduced how RL can enhance the core capabilities of LLM-based
...
@@ -250,6 +250,15 @@ So far, we have introduced how RL can enhance the core capabilities of LLM-based
\subsubsection{Skill Optimization}
\subsubsection{Skill Optimization}
Beyond planning and tool use at each individual step, recent work argues that agents should also learn reusable procedural building blocks, referred to as \textit{skills}. According to the survey by \citet{zhou-etal:skillsurvey}, a skill is a reusable procedural artifact that coordinates tools, memory, and runtime context under task-specific constraints. Unlike a single-step tool call, a skill encapsulates multi-step workflows, including tool invocation sequences, error handling, and intermediate decision logic. Under this view, agents and skills play complementary roles: agents handle high-level reasoning and planning, while skills form the operational layer that enables reliable, reusable, and composable execution. Research on skill optimization can be broadly divided into three paradigms: RL-based methods that update model parameters to internalize and co-evolve skills; training-free methods that treat skills as external text artifacts and refine them through iterative optimization; and frameworks that enable agents to autonomously construct and continually evolve entire skill libraries.
\textbf{RL-based skill optimization.} A central question is whether skills should be external artifacts retrieved at runtime, or internalized into model parameters so that the agent can deploy them zero-shot. \citet{lu-etal:skill0} address this with SKILL0, an in-context RL framework that trains agents to internalize skills via a curriculum: the agent starts with full skill context and the context is progressively withdrawn based on on-policy helpfulness, until the agent operates autonomously without any runtime skill retrieval. This internalization perspective naturally raises a follow-up question: once skills are internalized, how should the agent select which skill to apply, utilize it during execution, and distill new skills from experience? \citet{shi-etal:skill1} propose Skill1, which trains a single policy to unify these three operations under a shared task-outcome reward, where low-frequency reward trends credit skill selection and high-frequency variation credits skill distillation. Beyond internalization into a single model, another line of work asks whether the skill library itself can be treated as an evolving component that co-develops with the agent's policy. \citet{xia-etal:skillrl} propose SkillRL, which builds a hierarchical SkillBank distilled from interaction trajectories and allows it to co-evolve with the policy during training. \citet{yang-etal:skillmaster} push further toward autonomous skill management with SkillMaster, which introduces DualAdv-GRPO to separately estimate advantages for task-solving and skill-editing decisions, enabling the agent to review, refine, and retain skills from trajectory evidence without external teachers. \citet{he-etal:reskill} propose ReSkill, which embeds assertion-driven skill creation and Thompson Sampling with adaptive discounting directly into GRPO's rollout structure, so that skills are automatically created, tested, and pruned as the policy improves---reconciling what had previously been two decoupled processes. In the domain of mathematical reasoning, \citet{li-etal:arise} propose ARISE, a hierarchical RL framework where a Skills Manager maintains a tiered library by summarizing successful solution traces, and a Worker conditions generation on retrieved skills, with hierarchical rewards guiding both reasoning quality and library growth.
\textbf{Training-free skill refinement.} A parallel line of work asks whether skills can be optimized without touching model weights at all, by treating them as structured text parameters and applying iterative refinement driven by execution feedback. \citet{yang-etal:skillopt} propose SkillOpt, which formalizes this idea as a text-space optimizer: a separate optimizer model translates scored rollouts into bounded add/delete/replace edits on a skill document, and each edit is accepted only when it strictly improves a held-out validation score, much like a learning-rate schedule in weight-space optimization. \citet{wang-etal:skillgrad} draw a more explicit analogy to gradient descent with SkillGrad: trajectory losses serve as evidence, automatic diagnoses produce text-based gradients indicating correction directions, and a momentum mechanism accumulates recurring diagnostic patterns across iterations to stabilize optimization. While SkillOpt and SkillGrad focus on the optimization mechanism, \citet{liu-etal:skillrevise} address a practical cold-start problem in SkillRevise: given only an imperfect initial skill, how can the agent iteratively improve it without pre-collected trajectories? SkillRevise diagnoses defects from execution evidence, retrieves repair principles from a general memory, and applies execution-anchored edits within a revision budget, re-executing candidates and retaining the first verifier-passing version. A shared concern across these methods is that a new skill edit may fix one failure while silently breaking previously correct behaviors. \citet{moll-etal:grasp} address this with GRASP, which gates every candidate skill edit by a hard regression budget: a proposal is admitted only if it produces a net improvement on a balanced held-out probe set, ensuring that skill refinement is monotonic rather than cyclical.
\textbf{Autonomous skill construction and continual evolution.} The methods above assume skills already exist. A broader question is whether agents can autonomously construct skills from scratch and maintain them over extended deployment, removing the need for human-authored skill libraries entirely. \citet{zhang-etal:coevoskills} take a first step with CoEvoSkills, where a Skill Generator and a co-evolving Surrogate Verifier iteratively produce multi-file skill packages, with the verifier providing actionable feedback without ground-truth test content. \citet{si-etal:ctx2skill} propose Ctx2Skill, which adopts a multi-agent self-play loop where a Challenger generates probing tasks, a Reasoner solves them under evolving skills, and a Judge provides binary feedback---enabling skills to emerge from complex contexts without any external supervision. Once skills are constructed, they must be stored, retrieved, and improved over time. \citet{zhou-etal:mentoskills} propose Memento-Skills, a memory-based framework where skills live as structured markdown files and are continually refined through read-write reflective learning, enabling an agent to act as an agent-designing agent that improves its own toolset across interactions. \citet{zhang-etal:skillcomposer} propose SkillComposer, which decomposes skill improvement into three learnable operations---create, improve, and merge---trained via rejection sampling so that a small model can evolve skills for a much larger executor at inference time. Looking beyond a single agent, \citet{huang-etal:cascade} propose CASCADE for scientific workflows, where skills accumulate through meta-operations such as web search and code extraction and become shareable assets across agents and scientists. \citet{li-etal:skillhone} address the long-term challenge with SkillHone: as environments change, skills must evolve without losing institutional memory; SkillHone achieves this by pairing each skill revision with a persistent decision history, enabling cross-session refinement that builds on prior rationale rather than rediscovering it. Finally, \citet{yang-etal:federatedskill} extend skill evolution to distributed settings with FederatedSkill, where agents communicate semantic skill diffs rather than raw trajectories, enabling privacy-preserving collaborative improvement across clients with heterogeneous task distributions.