Commit 20632434 by wangchenglong

update.

parent 0d773174
......@@ -2,7 +2,7 @@
\clearpage
\section*{Appendix: Useful Systems and Datasets}
\phantomsection
\addcontentsline{toc}{section}{Appendix A: Useful Systems and Datasets}
\addcontentsline{toc}{section}{Appendix: Useful Systems and Datasets}
\begin{table}[h]
......
\begin{thebibliography}{179}
\begin{thebibliography}{185}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
......@@ -145,6 +145,11 @@ DeepSeek-AI.
\newblock Deepseek-v3.2: Pushing the frontier of open large language models.
\newblock \emph{arXiv preprint arXiv:2512.02556}, 2025.
\bibitem[Ding et~al.(2026)Ding, Huang, Fang, Liao, Li, Zhang, Wu, Zhao, and Wang]{ding-etal:evorubrics}
Hongxin Ding, Baixiang Huang, Yue Fang, Weibin Liao, Zheng Li, Jinyang Zhang, Zhijing Wu, Junfeng Zhao, and Yasha Wang.
\newblock Evorubrics: Dynamic rubrics as rewards via adversarial co-evolution for llm reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2606.23038}, 2026.
\bibitem[Dong et~al.(2025)Dong, Dong, Tang, Ye, Sun, Sui, and Wei]{dong-etal:rl}
Qingxiu Dong, Li~Dong, Yao Tang, Tianzhu Ye, Yutao Sun, Zhifang Sui, and Furu Wei.
\newblock Reinforcement pre-training.
......@@ -334,10 +339,15 @@ Jie Liu, Gongye Liu, Jiajun Liang, Yangguang Li, Jiaheng Liu, Xintao Wang, Pengf
\newblock Flow-grpo: Training flow matching models via online rl.
\newblock \emph{Advances in neural information processing systems}, 38:\penalty0 40783--40818, 2026{\natexlab{a}}.
\bibitem[Liu et~al.(2026{\natexlab{b}})Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
\bibitem[Liu et~al.(2026{\natexlab{b}})Liu, Xu, Yu, Hong, Yang, Zhao, and Wang]{liu-etal:openrubrics}
Tianci Liu, Ran Xu, Tony Yu, Ilgee Hong, Carl Yang, Tuo Zhao, and Haoyu Wang.
\newblock Openrubrics: Towards scalable synthetic rubric generation for reward modeling and llm alignment.
\newblock In \emph{Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics}, pp.\ 17417--17437. Association for Computational Linguistics, 2026{\natexlab{b}}.
\bibitem[Liu et~al.(2026{\natexlab{c}})Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
Xiaoqian Liu, Ke~Wang, Yuchuan Wu, Fei Huang, Yongbin Li, Jianbin Jiao, and Junge Zhang.
\newblock Agentic reinforcement learning with implicit step rewards.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026{\natexlab{b}}.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026{\natexlab{c}}.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
......@@ -468,6 +478,11 @@ Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher~D. Manning, Stefano E
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/a85b405ed65c6477a4fe8302b5e06ce7-Abstract-Conference.html}.
\bibitem[Rezaei et~al.(2025)Rezaei, Vacareanu, Wang, Wang, He, and Aky{\"u}rek]{rezaei-etal:online-rubrics}
MohammadHossein Rezaei, Robert Vacareanu, Zihao Wang, Clinton Wang, Yunzhong He, and Afra~Feyza Aky{\"u}rek.
\newblock Online rubrics elicitation from pairwise comparisons.
\newblock \emph{arXiv preprint arXiv:2510.07284}, 2025.
\bibitem[Schick et~al.(2023)Schick, Dwivedi-Yu, Dess'i, Raileanu, Lomeli, Hambro, Zettlemoyer, Cancedda, and Scialom]{schick-etal:toolformer}
Timo Schick, Jane Dwivedi-Yu, Roberto Dess'i, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom.
\newblock Toolformer: Language models can teach themselves to use tools.
......@@ -509,6 +524,11 @@ Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, Haowei Zh
\newblock \emph{ArXiv preprint}, abs/2402.03300, 2024.
\newblock URL \url{https://arxiv.org/abs/2402.03300}.
\bibitem[Shen et~al.(2026)Shen, Qiu, Whitehouse, Alazraki, Goel, Barbieri, Willi, Mathur, and Leontiadis]{shen-etal:rethinking-rubric}
William~F. Shen, Xinchi Qiu, Chenxi Whitehouse, Lisa Alazraki, Shashwat Goel, Francesco Barbieri, Timon Willi, Akhil Mathur, and Ilias Leontiadis.
\newblock Rethinking rubric generation for improving llm judge and reward modeling for open-ended tasks.
\newblock \emph{arXiv preprint arXiv:2602.05125}, 2026.
\bibitem[Shinn et~al.(2023)Shinn, Cassano, Gopinath, Narasimhan, and Yao]{shinn-etal:reflexion}
Noah Shinn, Federico Cassano, Ashwin Gopinath, Karthik Narasimhan, and Shunyu Yao.
\newblock Reflexion: Language agents with verbal reinforcement learning.
......@@ -819,10 +839,15 @@ Jin Xu, Zhifang Guo, Jinzheng He, Hangrui Hu, Ting He, Shuai Bai, Keqin Chen, Ji
\newblock Qwen2. 5-omni technical report.
\newblock \emph{arXiv preprint arXiv:2503.20215}, 2025.
\bibitem[Xu et~al.(2026)Xu, Liang, Mei, Gao, Tan, and Zhang]{xu-etal:a-mem}
\bibitem[Xu et~al.(2026{\natexlab{a}})Xu, Liu, Dong, Yu, Hong, Yang, Zhang, Zhao, and Wang]{xu-etal:rubric-arm}
Ran Xu, Tianci Liu, Zihan Dong, Tony Yu, Ilgee Hong, Carl Yang, Linjun Zhang, Tao Zhao, and Haoyu Wang.
\newblock Alternating reinforcement learning for rubric-based reward modeling in non-verifiable llm post-training.
\newblock \emph{arXiv preprint arXiv:2602.01511}, 2026{\natexlab{a}}.
\bibitem[Xu et~al.(2026{\natexlab{b}})Xu, Liang, Mei, Gao, Tan, and Zhang]{xu-etal:a-mem}
Wujiang Xu, Zujie Liang, Kai Mei, Hang Gao, Juntao Tan, and Yongfeng Zhang.
\newblock A-mem: Agentic memory for llm agents.
\newblock \emph{Advances in Neural Information Processing Systems}, 38:\penalty0 17577--17604, 2026.
\newblock \emph{Advances in Neural Information Processing Systems}, 38:\penalty0 17577--17604, 2026{\natexlab{b}}.
\bibitem[Xue et~al.(2025)Xue, Wu, Gao, Kong, Zhu, Chen, Liu, Liu, Guo, Huang, et~al.]{xue-etal:dancegrpo}
Zeyue Xue, Jie Wu, Yu~Gao, Fangyuan Kong, Lingting Zhu, Mengzhao Chen, Zhiheng Liu, Wei Liu, Qiushan Guo, Weilin Huang, et~al.
......@@ -861,10 +886,15 @@ Da~Yin, Faeze Brahman, Abhilasha Ravichander, Khyathi Chandu, Kai-Wei Chang, Yej
\newblock Agent lumos: Unified and modular training for open-source language agents.
\newblock In \emph{Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 12380--12403, 2024.
\bibitem[Yu et~al.(2026)Yu, Zhu, Lin, Cui, Ding, and Li]{yu-etal:skill}
\bibitem[Yu et~al.(2026{\natexlab{a}})Yu, Feng, Min, Lin, Xu, Xu, Yu, Liu, and Zhou]{yu-etal:audio-rubrics}
Fangxu Yu, Tao Feng, Dehai Min, Zinan Lin, Weijia Xu, Michael Xu, Philip~S. Yu, Ge~Liu, and Tianyi Zhou.
\newblock Reinforcement learning with evolving rubrics as rewards for audio reasoning.
\newblock \emph{arXiv preprint arXiv:2608.02831}, 2026{\natexlab{a}}.
\bibitem[Yu et~al.(2026{\natexlab{b}})Yu, Zhu, Lin, Cui, Ding, and Li]{yu-etal:skill}
Jianxiang Yu, Jiapeng Zhu, Bochen Lin, Qier Cui, Zichen Ding, and Xiang Li.
\newblock Skill is not one-size-fits-all: Model-aware skill alignment for llm agents.
\newblock \emph{arXiv preprint arXiv:2605.30723}, 2026.
\newblock \emph{arXiv preprint arXiv:2605.30723}, 2026{\natexlab{b}}.
\bibitem[Yu et~al.(2024{\natexlab{a}})Yu, Yao, Zhang, He, Han, Cui, Hu, Liu, Zheng, Sun, et~al.]{yu-etal:2024rlhf}
Tianyu Yu, Yuan Yao, Haoye Zhang, Taiwen He, Yifeng Han, Ganqu Cui, Jinyi Hu, Zhiyuan Liu, Hai-Tao Zheng, Maosong Sun, et~al.
......
@inproceedings{hashemi-etal:llm-rubric,
title={LLM-Rubric: A Multidimensional, Calibrated Approach to Automated Evaluation of Natural Language Texts},
author={Hashemi, Helia and Eisner, Jason and Rosset, Corby and Van Durme, Benjamin and Kedzie, Chris},
booktitle={Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics},
pages={13806--13834},
year={2024},
publisher={Association for Computational Linguistics}
}
@inproceedings{viswanathan-etal:checklists,
title={Checklists Are Better Than Reward Models For Aligning Language Models},
author={Viswanathan, Vijay and Sun, Yanchao and Kong, Xiang and Cao, Meng and Neubig, Graham and Wu, Sherry},
booktitle={Advances in Neural Information Processing Systems},
year={2025}
}
@inproceedings{gunjal-etal:rubrics-rewards,
title={Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains},
author={Gunjal, Anisha and Wang, Anthony and Lau, Elaine and Nath, Vaskar and He, Yunzhong and Liu, Bing and Hendryx, Sean},
booktitle={International Conference on Learning Representations},
year={2026}
}
@inproceedings{liu-etal:openrubrics,
title={OpenRubrics: Towards Scalable Synthetic Rubric Generation for Reward Modeling and LLM Alignment},
author={Liu, Tianci and Xu, Ran and Yu, Tony and Hong, Ilgee and Yang, Carl and Zhao, Tuo and Wang, Haoyu},
booktitle={Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics},
pages={17417--17437},
year={2026},
publisher={Association for Computational Linguistics}
}
@article{shen-etal:rethinking-rubric,
title={Rethinking Rubric Generation for Improving LLM Judge and Reward Modeling for Open-ended Tasks},
author={Shen, William F. and Qiu, Xinchi and Whitehouse, Chenxi and Alazraki, Lisa and Goel, Shashwat and Barbieri, Francesco and Willi, Timon and Mathur, Akhil and Leontiadis, Ilias},
journal={arXiv preprint arXiv:2602.05125},
year={2026}
}
@article{rezaei-etal:online-rubrics,
title={Online Rubrics Elicitation from Pairwise Comparisons},
author={Rezaei, MohammadHossein and Vacareanu, Robert and Wang, Zihao and Wang, Clinton and He, Yunzhong and Aky{\"u}rek, Afra Feyza},
journal={arXiv preprint arXiv:2510.07284},
year={2025}
}
@article{xu-etal:rubric-arm,
title={Alternating Reinforcement Learning for Rubric-Based Reward Modeling in Non-Verifiable LLM Post-Training},
author={Xu, Ran and Liu, Tianci and Dong, Zihan and Yu, Tony and Hong, Ilgee and Yang, Carl and Zhang, Linjun and Zhao, Tao and Wang, Haoyu},
journal={arXiv preprint arXiv:2602.01511},
year={2026}
}
@article{ding-etal:evorubrics,
title={EvoRubrics: Dynamic Rubrics as Rewards via Adversarial Co-Evolution for LLM Reinforcement Learning},
author={Ding, Hongxin and Huang, Baixiang and Fang, Yue and Liao, Weibin and Li, Zheng and Zhang, Jinyang and Wu, Zhijing and Zhao, Junfeng and Wang, Yasha},
journal={arXiv preprint arXiv:2606.23038},
year={2026}
}
@article{yu-etal:audio-rubrics,
title={Reinforcement Learning with Evolving Rubrics as Rewards for Audio Reasoning},
author={Yu, Fangxu and Feng, Tao and Min, Dehai and Lin, Zinan and Xu, Weijia and Xu, Michael and Yu, Philip S. and Liu, Ge and Zhou, Tianyi},
journal={arXiv preprint arXiv:2608.02831},
year={2026}
}
@article{silver-etal:reward,
title={Reward is enough},
author={Silver, David and Singh, Satinder and Precup, Doina and Sutton, Richard S},
......
......@@ -247,8 +247,8 @@ While generative reward models have shown strong performance in preference predi
\vspace{0.5em}
\item \textbf{Analytic Rubric.}
An analytic rubric evaluates each criterion separately. The criterion-level scores can then be averaged or weighted to obtain the final reward.
\item \textbf{Criterion-wise Scoring Rubric.}
A criterion-wise scoring rubric asks the reward model to assign a separate score to each evaluation criterion. These criterion-level scores are then averaged or weighted to obtain the final reward.
\vspace{0.5em}
\begin{tcolorbox}[frame empty]
......@@ -357,14 +357,29 @@ While generative reward models have shown strong performance in preference predi
\end{itemize}
\begin{figure*}[!t]
\centering
\resizebox{\linewidth}{!}{
\input{section4/Figures/figure-rubric-based-reward-modeling.tex}}
\vspace{-0.5cm}
\caption{
Two commonly used approaches for automatic rubric generation. We use checklist rubrics as an illustrative example.
}
\label{fig:rubric-based-reward-modeling}
\end{figure*}
It is worth noting that reasoning can also be incorporated into rubric-based evaluation. For example, in a checklist-style rubric, the reward model can be asked to reason about each criterion before producing the corresponding Pass/Fail judgment. Additionally, although we use pairwise evaluation, i.e., taking $(\mathbf{x}, \mathbf{y}_a, \mathbf{y}_b)$ as reward model input, as the running example, rubric-based reward modeling is equally applicable to pointwise evaluation, i.e., taking $(\mathbf{x}, \mathbf{y})$ as reward model input.
The quality of the rubric is crucial to the effectiveness of rubric-based reward modeling. As a result, recent work has devoted increasing attention to constructing high-quality rubrics for reward modeling. Similar to many other components in machine learning, rubric acquisition generally follows two approaches: manual design and automatic generation.
The quality of the rubric is crucial to the performance of rubric-based reward models. As a result, recent work has devoted increasing attention to constructing high-quality rubrics for reward modeling. Similar to many other components in machine learning, rubric acquisition generally follows two approaches: manual design and automatic generation.
For manual design, one straightforward approach is to recruit human experts to write rubrics based on task requirements and their domain knowledge. However, this way is often difficult to design a rubric that comprehensively covers the diverse cases that may arise across different inputs and tasks. This is because that evaluation criteria that are appropriate for one case may be insufficient for another. Additionally, manually designed rubrics inherit the prompt sensitivity of LLM-based evaluation: even when the underlying evaluation dimensions remain the same, small differences in wording can lead to noticeably different results.
For manual design, one straightforward approach is to recruit human experts to write rubrics based on their domain knowledge and task requirements. However, this way is often difficult to design a rubric that comprehensively covers the diverse cases that may arise across different inputs and tasks. This is because that evaluation criteria that are appropriate for one case may be insufficient for another. Additionally, manually designed rubrics inherit the prompt sensitivity of LLM-based evaluation: even when the underlying evaluation dimensions remain the same, small differences in wording can lead to noticeably different results.
One promising strategy is to generate rubrics dynamically based on the input. As shown in Figure~\ref{fig:rubric-based-reward-modeling}, existing approaches generally fall into two categories. The first uses an external LLM to generate task-specific rubrics from the input and evaluation requirements, which are then provided to the reward model for preference prediction. The second allows the generative reward model to generate its own rubrics for the current input and then evaluate candidate responses according to these self-generated criteria.
Both approaches formulate rubric generation as an explicit task. This naturally leads to another idea: can we optimize the rubric generation process itself to produce better rubrics? The answer is yes. Rubrics generated directly by LLMs may suffer from limited coverage, redundant criteria, or preference misalignment \citep{liu-etal:openrubrics,shen-etal:rethinking-rubric}. To improve rubric quality, we can explicitly refine the generated criteria. For example, OpenRubrics generates rubrics by contrasting preferred and rejected responses to identify more discriminative rules and principles \citep{liu-etal:openrubrics}. We can further learn the rubric generation process itself through RL training \citep{xu-etal:rubric-arm}. More recent studies go one step further by allowing rubrics to evolve with the policy, so that the evaluation criteria continue to capture new weaknesses as model improves \citep{rezaei-etal:online-rubrics,ding-etal:evorubrics,yu-etal:audio-rubrics}. Rubric optimization is becoming an active research direction, and interested readers can refer to the aforementioned works for further details.
\subsubsection{Reward Model Evaluation}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论