Commit 989db75b by wangchenglong

add reward model evaluation.

parent b250a4b7
\begin{thebibliography}{153} \begin{thebibliography}{160}
\providecommand{\natexlab}[1]{#1} \providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}} \providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax \expandafter\ifx\csname urlstyle\endcsname\relax
...@@ -153,6 +153,11 @@ Jiazhan Feng, Shijue Huang, Xingwei Qu, Ge~Zhang, Yujia Qin, Baoquan Zhong, Chen ...@@ -153,6 +153,11 @@ Jiazhan Feng, Shijue Huang, Xingwei Qu, Ge~Zhang, Yujia Qin, Baoquan Zhong, Chen
\newblock Retool: Reinforcement learning for strategic tool use in llms. \newblock Retool: Reinforcement learning for strategic tool use in llms.
\newblock \emph{arXiv preprint arXiv:2504.11536}, 2025. \newblock \emph{arXiv preprint arXiv:2504.11536}, 2025.
\bibitem[Frick et~al.(2025)Frick, Li, Chen, Chiang, Angelopoulos, Jiao, Zhu, Gonzalez, and Stoica]{frick-etal:ppe}
Evan Frick, Tianle Li, Connor Chen, Wei-Lin Chiang, Anastasios Angelopoulos, Jiantao Jiao, Banghua Zhu, Joseph~E Gonzalez, and Ion Stoica.
\newblock How to evaluate reward models for rlhf.
\newblock In \emph{International Conference on Learning Representations}, volume 2025, pp.\ 18128--18163, 2025.
\bibitem[Fu et~al.(2025)Fu, He, Wang, Hong, Gongque, Zeng, Wang, Wang, Cai, and Xu]{fu-etal:agentrefine} \bibitem[Fu et~al.(2025)Fu, He, Wang, Hong, Gongque, Zeng, Wang, Wang, Cai, and Xu]{fu-etal:agentrefine}
Dayuan Fu, Keqing He, Yejie Wang, Wentao Hong, Zhuoma Gongque, Weihao Zeng, Wei Wang, Jingang Wang, Xunliang Cai, and Weiran Xu. Dayuan Fu, Keqing He, Yejie Wang, Wentao Hong, Zhuoma Gongque, Weihao Zeng, Wei Wang, Jingang Wang, Xunliang Cai, and Weiran Xu.
\newblock Agentrefine: Enhancing agent generalization through refinement tuning. \newblock Agentrefine: Enhancing agent generalization through refinement tuning.
...@@ -236,6 +241,11 @@ Aviral Kumar, Vincent Zhuang, Rishabh Agarwal, Yi~Su, John~D Co-Reyes, Avi Singh ...@@ -236,6 +241,11 @@ Aviral Kumar, Vincent Zhuang, Rishabh Agarwal, Yi~Su, John~D Co-Reyes, Avi Singh
\newblock \emph{ArXiv preprint}, abs/2409.12917, 2024. \newblock \emph{ArXiv preprint}, abs/2409.12917, 2024.
\newblock URL \url{https://arxiv.org/abs/2409.12917}. \newblock URL \url{https://arxiv.org/abs/2409.12917}.
\bibitem[Lambert et~al.(2025)Lambert, Pyatkin, Morrison, Miranda, Lin, Chandu, Dziri, Kumar, Zick, Choi, et~al.]{lambert-etal:Rewardbench}
Nathan Lambert, Valentina Pyatkin, Jacob Morrison, LJ~Miranda, Bill~Yuchen Lin, Khyathi Chandu, Nouha Dziri, Sachin Kumar, Tom Zick, Yejin Choi, et~al.
\newblock Rewardbench: Evaluating reward models for language modeling.
\newblock In \emph{Findings of the Association for Computational Linguistics: NAACL 2025}, pp.\ 1755--1797, 2025.
\bibitem[Lee et~al.(2024)Lee, Bjelonic, Reske, Wellhausen, Miki, and Hutter]{lee-etal:lee2024learning} \bibitem[Lee et~al.(2024)Lee, Bjelonic, Reske, Wellhausen, Miki, and Hutter]{lee-etal:lee2024learning}
Joonho Lee, Marko Bjelonic, Alexander Reske, Lorenz Wellhausen, Takahiro Miki, and Marco Hutter. Joonho Lee, Marko Bjelonic, Alexander Reske, Lorenz Wellhausen, Takahiro Miki, and Marco Hutter.
\newblock Learning robust autonomous navigation and locomotion for wheeled-legged robots. \newblock Learning robust autonomous navigation and locomotion for wheeled-legged robots.
...@@ -287,6 +297,11 @@ Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu. ...@@ -287,6 +297,11 @@ Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
\newblock \doi{10.18653/v1/2023.emnlp-main.153}. \newblock \doi{10.18653/v1/2023.emnlp-main.153}.
\newblock URL \url{https://aclanthology.org/2023.emnlp-main.153}. \newblock URL \url{https://aclanthology.org/2023.emnlp-main.153}.
\bibitem[Liu et~al.(2025)Liu, Yao, Min, Cao, Hou, and Li]{liu-etal:rm-bench}
Yantao Liu, Zijun Yao, Rui Min, Yixin Cao, Lei Hou, and Juanzi Li.
\newblock Rm-bench: Benchmarking reward models of language models with subtlety and style.
\newblock In \emph{International Conference on Learning Representations}, volume 2025, pp.\ 44323--44355, 2025.
\bibitem[Liu et~al.(2024)Liu, Yao, Zhang, Liu, Yang, RN, Lan, Zhu, Tan, Kokane, et~al.]{liu-etal:pract} \bibitem[Liu et~al.(2024)Liu, Yao, Zhang, Liu, Yang, RN, Lan, Zhu, Tan, Kokane, et~al.]{liu-etal:pract}
Zhiwei Liu, Weiran Yao, Jianguo Zhang, Zuxin Liu, Liangwei Yang, Rithesh RN, Tian Lan, Ming Zhu, Juntao Tan, Shirley Kokane, et~al. Zhiwei Liu, Weiran Yao, Jianguo Zhang, Zuxin Liu, Liangwei Yang, Rithesh RN, Tian Lan, Ming Zhu, Juntao Tan, Shirley Kokane, et~al.
\newblock Pract: Optimizing principled reasoning and acting of llm agent. \newblock Pract: Optimizing principled reasoning and acting of llm agent.
...@@ -303,6 +318,11 @@ Dakota Mahan, Duy Van~Phung, Rafael Rafailov, Chase Blagden, Nathan Lile, Louis ...@@ -303,6 +318,11 @@ Dakota Mahan, Duy Van~Phung, Rafael Rafailov, Chase Blagden, Nathan Lile, Louis
\newblock \emph{ArXiv preprint}, abs/2410.12832, 2024. \newblock \emph{ArXiv preprint}, abs/2410.12832, 2024.
\newblock URL \url{https://arxiv.org/abs/2410.12832}. \newblock URL \url{https://arxiv.org/abs/2410.12832}.
\bibitem[Malik et~al.(2026)Malik, Pyatkin, Land, Morrison, Smith, Hajishirzi, and Lambert]{malik-etal:rewardbench}
Saumya Malik, Valentina Pyatkin, Sander Land, Jacob Morrison, Noah Smith, Hanna Hajishirzi, and Nathan Lambert.
\newblock Rewardbench 2: Advancing reward model evaluation.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 144839--144866, 2026.
\bibitem[Masoudnia \& Ebrahimpour(2014)Masoudnia and Ebrahimpour]{masoudnia-etal:2014mixture} \bibitem[Masoudnia \& Ebrahimpour(2014)Masoudnia and Ebrahimpour]{masoudnia-etal:2014mixture}
Saeed Masoudnia and Reza Ebrahimpour. Saeed Masoudnia and Reza Ebrahimpour.
\newblock Mixture of experts: a literature survey. \newblock Mixture of experts: a literature survey.
...@@ -528,6 +548,11 @@ Csaba Szepesv{\'a}ri. ...@@ -528,6 +548,11 @@ Csaba Szepesv{\'a}ri.
\newblock Algorithms for reinforcement learning. \newblock Algorithms for reinforcement learning.
\newblock \emph{Synthesis Lectures on Artificial Intelligence and Machine Learning}, 4\penalty0 (1):\penalty0 1--103, 2010. \newblock \emph{Synthesis Lectures on Artificial Intelligence and Machine Learning}, 4\penalty0 (1):\penalty0 1--103, 2010.
\bibitem[Tan et~al.(2025)Tan, Zhuang, Montgomery, Tang, Cuadron, Wang, Popa, and Stoica]{tan-etal:judgebench}
Sijun Tan, Siyuan Zhuang, Kyle Montgomery, William Tang, Alejandro Cuadron, Chenguang Wang, Raluca Popa, and Ion Stoica.
\newblock Judgebench: A benchmark for evaluating llm-based judges.
\newblock In \emph{International Conference on Learning Representations}, volume 2025, pp.\ 63277--63303, 2025.
\bibitem[Team et~al.(2025)Team, Du, Gao, Xing, Jiang, Chen, Li, Xiao, Du, Liao, et~al.]{kimi-team:2025kimi} \bibitem[Team et~al.(2025)Team, Du, Gao, Xing, Jiang, Chen, Li, Xiao, Du, Liao, et~al.]{kimi-team:2025kimi}
Kimi Team, Angang Du, Bofei Gao, Bowei Xing, Changjiu Jiang, Cheng Chen, Cheng Li, Chenjun Xiao, Chenzhuang Du, Chonghua Liao, et~al. Kimi Team, Angang Du, Bofei Gao, Bowei Xing, Changjiu Jiang, Cheng Chen, Cheng Li, Chenjun Xiao, Chenzhuang Du, Chonghua Liao, et~al.
\newblock Kimi k1. 5: Scaling reinforcement learning with llms. \newblock Kimi k1. 5: Scaling reinforcement learning with llms.
...@@ -589,10 +614,15 @@ Chenglong Wang, Yongyu Mu, Hang Zhou, Yifu Huo, Ziming Zhu, Jiali Zeng, Murun Ya ...@@ -589,10 +614,15 @@ Chenglong Wang, Yongyu Mu, Hang Zhou, Yifu Huo, Ziming Zhu, Jiali Zeng, Murun Ya
\newblock Gram-r$^2$: Self-training generative foundation reward models for reward reasoning. \newblock Gram-r$^2$: Self-training generative foundation reward models for reward reasoning.
\newblock In \emph{Proceedings of the AAAI Conference on Artificial Intelligence}, volume~40, pp.\ 33395--33403, 2026{\natexlab{b}}. \newblock In \emph{Proceedings of the AAAI Conference on Artificial Intelligence}, volume~40, pp.\ 33395--33403, 2026{\natexlab{b}}.
\bibitem[Wang et~al.(2026{\natexlab{c}})Wang, Li, Cheng, Ouyang, Yu, Liu, and Chen]{wang-etal:steppo} \bibitem[Wang et~al.(2026{\natexlab{c}})Wang, Zhu, Huo, Li, He, Ding, Hao, Gao, Zhou, Chang, Liu, and Zhu]{wang-etal:rrc}
Chenglong Wang, Ziming Zhu, Yifu Huo, Bei Li, Qiaozhi He, Yan Ding, Xiaoyang Hao, Yuxin Gao, Tianhua Zhou, Xiaojia Chang, Tongran Liu, and Jingbo Zhu.
\newblock Rrc: Unlocking generative reward models in llm reinforcement learning via ranking-based reward construction, 2026{\natexlab{c}}.
\newblock URL \url{https://arxiv.org/abs/2608.06310}.
\bibitem[Wang et~al.(2026{\natexlab{d}})Wang, Li, Cheng, Ouyang, Yu, Liu, and Chen]{wang-etal:steppo}
Daoyu Wang, Qingchuan Li, Mingyue Cheng, Jie Ouyang, Shuo Yu, Qi~Liu, and Enhong Chen. Daoyu Wang, Qingchuan Li, Mingyue Cheng, Jie Ouyang, Shuo Yu, Qi~Liu, and Enhong Chen.
\newblock Steppo: Step-aligned policy optimization for agentic reinforcement learning. \newblock Steppo: Step-aligned policy optimization for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2604.18401}, 2026{\natexlab{c}}. \newblock \emph{arXiv preprint arXiv:2604.18401}, 2026{\natexlab{d}}.
\bibitem[Wang et~al.(2021)Wang, Yan, Meng, and Zhou]{wang-etal:2021selective} \bibitem[Wang et~al.(2021)Wang, Yan, Meng, and Zhou]{wang-etal:2021selective}
Fusheng Wang, Jianhao Yan, Fandong Meng, and Jie Zhou. Fusheng Wang, Jianhao Yan, Fandong Meng, and Jie Zhou.
...@@ -606,10 +636,10 @@ Hanlin Wang, Jian Wang, Chak~Tou Leong, and Wenjie Li. ...@@ -606,10 +636,10 @@ Hanlin Wang, Jian Wang, Chak~Tou Leong, and Wenjie Li.
\newblock Steca: Step-level trajectory calibration for llm agent learning. \newblock Steca: Step-level trajectory calibration for llm agent learning.
\newblock In \emph{Findings of the Association for Computational Linguistics: ACL 2025}, pp.\ 11597--11614, 2025{\natexlab{b}}. \newblock In \emph{Findings of the Association for Computational Linguistics: ACL 2025}, pp.\ 11597--11614, 2025{\natexlab{b}}.
\bibitem[Wang et~al.(2026{\natexlab{d}})Wang, Yan, Wang, Tian, Mishra, Xu, Gandhi, Xu, and Cheong]{wang-etal:reinforcement} \bibitem[Wang et~al.(2026{\natexlab{e}})Wang, Yan, Wang, Tian, Mishra, Xu, Gandhi, Xu, and Cheong]{wang-etal:reinforcement}
Jiongxiao Wang, Qiaojing Yan, Yawei Wang, Yijun Tian, Soumya~Smruti Mishra, Zhichao Xu, Megha Gandhi, Panpan Xu, and Lin~Lee Cheong. Jiongxiao Wang, Qiaojing Yan, Yawei Wang, Yijun Tian, Soumya~Smruti Mishra, Zhichao Xu, Megha Gandhi, Panpan Xu, and Lin~Lee Cheong.
\newblock Reinforcement learning for self-improving agent with skill library. \newblock Reinforcement learning for self-improving agent with skill library.
\newblock In \emph{Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 1529--1550, 2026{\natexlab{d}}. \newblock In \emph{Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 1529--1550, 2026{\natexlab{e}}.
\bibitem[Wang et~al.(2023{\natexlab{b}})Wang, Li, Chen, Cai, Zhu, Lin, Cao, Liu, Liu, and Sui]{wang-etal:2023large} \bibitem[Wang et~al.(2023{\natexlab{b}})Wang, Li, Chen, Cai, Zhu, Lin, Cao, Liu, Liu, and Sui]{wang-etal:2023large}
Peiyi Wang, Lei Li, Liang Chen, Zefan Cai, Dawei Zhu, Binghuai Lin, Yunbo Cao, Qi~Liu, Tianyu Liu, and Zhifang Sui. Peiyi Wang, Lei Li, Liang Chen, Zefan Cai, Dawei Zhu, Binghuai Lin, Yunbo Cao, Qi~Liu, Tianyu Liu, and Zhifang Sui.
...@@ -639,10 +669,10 @@ Yu~Wang, Ryuichi Takanobu, Zhiqi Liang, Yuzhen Mao, Yuanzhe Hu, Julian McAuley, ...@@ -639,10 +669,10 @@ Yu~Wang, Ryuichi Takanobu, Zhiqi Liang, Yuzhen Mao, Yuanzhe Hu, Julian McAuley,
\newblock Mem-$\{$$\backslash$alpha$\}$: Learning memory construction via reinforcement learning. \newblock Mem-$\{$$\backslash$alpha$\}$: Learning memory construction via reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2509.25911}, 2025{\natexlab{c}}. \newblock \emph{arXiv preprint arXiv:2509.25911}, 2025{\natexlab{c}}.
\bibitem[Wang et~al.(2026{\natexlab{e}})Wang, Xu, Liu, Wang, Han, Yao, Yao, and He]{wang-etal:awm} \bibitem[Wang et~al.(2026{\natexlab{f}})Wang, Xu, Liu, Wang, Han, Yao, Yao, and He]{wang-etal:awm}
Zhaoyang Wang, Canwen Xu, Boyi Liu, Yite Wang, Siwei Han, Zhewei Yao, Huaxiu Yao, and Yuxiong He. Zhaoyang Wang, Canwen Xu, Boyi Liu, Yite Wang, Siwei Han, Zhewei Yao, Huaxiu Yao, and Yuxiong He.
\newblock Agent world model: Infinity synthetic environments for agentic reinforcement learning. \newblock Agent world model: Infinity synthetic environments for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2602.10090}, 2026{\natexlab{e}}. \newblock \emph{arXiv preprint arXiv:2602.10090}, 2026{\natexlab{f}}.
\bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned} \bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned}
Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le. Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le.
...@@ -832,6 +862,11 @@ Chunting Zhou, Pengfei Liu, Puxin Xu, Srinivasan Iyer, Jiao Sun, Yuning Mao, Xue ...@@ -832,6 +862,11 @@ Chunting Zhou, Pengfei Liu, Puxin Xu, Srinivasan Iyer, Jiao Sun, Yuning Mao, Xue
\newblock Lima: Less is more for alignment. \newblock Lima: Less is more for alignment.
\newblock \emph{Advances in Neural Information Processing Systems}, 36:\penalty0 55006--55021, 2023. \newblock \emph{Advances in Neural Information Processing Systems}, 36:\penalty0 55006--55021, 2023.
\bibitem[Zhou et~al.(2025)Zhou, Zheng, Wang, Xi, Dou, Bao, Shen, Xiong, Fan, Mou, et~al.]{zhou-etal:rmb}
Enyu Zhou, Guodong Zheng, Binghai Wang, Zhiheng Xi, Shihan Dou, Rong Bao, Wei Shen, Limao Xiong, Jessica Fan, Yurong Mou, et~al.
\newblock Rmb: Comprehensively benchmarking reward models in llm alignment.
\newblock In \emph{International Conference on Learning Representations}, volume 2025, pp.\ 26543--26589, 2025.
\bibitem[Zhou et~al.(2024)Zhou, Wang, Hu, Xiao, Zhang, and Zhu]{zhou:2024prior} \bibitem[Zhou et~al.(2024)Zhou, Wang, Hu, Xiao, Zhang, and Zhu]{zhou:2024prior}
Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu. Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu.
\newblock Prior constraints-based reward model training for aligning large language models. \newblock Prior constraints-based reward model training for aligning large language models.
......
...@@ -5,6 +5,72 @@ ...@@ -5,6 +5,72 @@
@inproceedings{frick-etal:ppe,
title={How to evaluate reward models for rlhf},
author={Frick, Evan and Li, Tianle and Chen, Connor and Chiang, Wei-Lin and Angelopoulos, Anastasios and Jiao, Jiantao and Zhu, Banghua and Gonzalez, Joseph E and Stoica, Ion},
booktitle={International Conference on Learning Representations},
volume={2025},
pages={18128--18163},
year={2025}
}
@inproceedings{tan-etal:judgebench,
title={Judgebench: A benchmark for evaluating llm-based judges},
author={Tan, Sijun and Zhuang, Siyuan and Montgomery, Kyle and Tang, William and Cuadron, Alejandro and Wang, Chenguang and Popa, Raluca and Stoica, Ion},
booktitle={International Conference on Learning Representations},
volume={2025},
pages={63277--63303},
year={2025}
}
@inproceedings{zhou-etal:rmb,
title={Rmb: Comprehensively benchmarking reward models in llm alignment},
author={Zhou, Enyu and Zheng, Guodong and Wang, Binghai and Xi, Zhiheng and Dou, Shihan and Bao, Rong and Shen, Wei and Xiong, Limao and Fan, Jessica and Mou, Yurong and others},
booktitle={International Conference on Learning Representations},
volume={2025},
pages={26543--26589},
year={2025}
}
@inproceedings{liu-etal:rm-bench,
title={Rm-bench: Benchmarking reward models of language models with subtlety and style},
author={Liu, Yantao and Yao, Zijun and Min, Rui and Cao, Yixin and Hou, Lei and Li, Juanzi},
booktitle={International Conference on Learning Representations},
volume={2025},
pages={44323--44355},
year={2025}
}
@inproceedings{malik-etal:rewardbench,
title={Rewardbench 2: Advancing reward model evaluation},
author={Malik, Saumya and Pyatkin, Valentina and Land, Sander and Morrison, Jacob and Smith, Noah and Hajishirzi, Hanna and Lambert, Nathan},
booktitle={International Conference on Learning Representations},
volume={2026},
pages={144839--144866},
year={2026}
}
@inproceedings{lambert-etal:Rewardbench,
title={Rewardbench: Evaluating reward models for language modeling},
author={Lambert, Nathan and Pyatkin, Valentina and Morrison, Jacob and Miranda, LJ and Lin, Bill Yuchen and Chandu, Khyathi and Dziri, Nouha and Kumar, Sachin and Zick, Tom and Choi, Yejin and others},
booktitle={Findings of the Association for Computational Linguistics: NAACL 2025},
pages={1755--1797},
year={2025}
}
@misc{wang-etal:rrc,
title={RRC: Unlocking Generative Reward Models in LLM Reinforcement Learning via Ranking-Based Reward Construction},
author={Chenglong Wang and Ziming Zhu and Yifu Huo and Bei Li and Qiaozhi He and Yan Ding and Xiaoyang Hao and Yuxin Gao and Tianhua Zhou and Xiaojia Chang and Tongran Liu and Jingbo Zhu},
year={2026},
eprint={2608.06310},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2608.06310},
}
@article{monea-etal:llms, @article{monea-etal:llms,
title={Llms are in-context reinforcement learners}, title={Llms are in-context reinforcement learners},
author={Monea, Giovanni and Bosselut, Antoine and Brantley, Kiant{\'e} and Artzi, Yoav}, author={Monea, Giovanni and Bosselut, Antoine and Brantley, Kiant{\'e} and Artzi, Yoav},
......
No preview for this file type
...@@ -181,7 +181,7 @@ Although we discuss methods for training a generative reward model here, an inte ...@@ -181,7 +181,7 @@ Although we discuss methods for training a generative reward model here, an inte
Additionally, to mitigate the positional bias problem \citep{wang-etal:2023large}, we can introduce an alternative input order by transposing the positions of output, i.e., presenting $\mathbf{y}_{\mathrm{ref}}$ before $\mathbf{y}'$, to construct a secondary input string $\mathbf{s}'_{T} = [\mathbf{c}, \mathbf{x}', \mathbf{y}_{\mathrm{ref}}, \mathbf{y}']$. Additionally, to mitigate the positional bias problem \citep{wang-etal:2023large}, we can introduce an alternative input order by transposing the positions of output, i.e., presenting $\mathbf{y}_{\mathrm{ref}}$ before $\mathbf{y}'$, to construct a secondary input string $\mathbf{s}'_{T} = [\mathbf{c}, \mathbf{x}', \mathbf{y}_{\mathrm{ref}}, \mathbf{y}']$.
The reward for $(\mathbf{x}', \mathbf{y}')$ is thus defined as the log-probability that $\mathbf{y}'$ is preferred over $\mathbf{y}_{\mathrm{ref}}$: The reward for $(\mathbf{x}', \mathbf{y}')$ is thus defined as the log-probability that $\mathbf{y}'$ is preferred over $\mathbf{y}_{\mathrm{ref}}$:
\begin{eqnarray} \begin{eqnarray}
r_{\phi}(\mathbf{x}', \mathbf{y}') & = & \frac{\mathrm{Pr}_{\theta}(w=\text{A}|\mathbf{s}')+\mathrm{Pr}_{\theta}(w=\text{B}|\mathbf{s}'_{T})}{2} R_{\phi}(\mathbf{x}', \mathbf{y}') & = & \frac{\mathrm{Pr}_{\theta}(w=\text{A}|\mathbf{s}')+\mathrm{Pr}_{\theta}(w=\text{B}|\mathbf{s}'_{T})}{2}
\label{eq:apply-generative-rm} \label{eq:apply-generative-rm}
\end{eqnarray} \end{eqnarray}
...@@ -210,15 +210,52 @@ We can then define a new loss function for training the generative reward model ...@@ -210,15 +210,52 @@ We can then define a new loss function for training the generative reward model
where $\mathbf{rat}$ is the labeled CoT rationale for generating a preference between $\mathbf{y}_a$ and $\mathbf{y}_b$. In practice, we can obtain the evaluation results by prompting a standard LLM (as known as LLM-as-a-judge), as demonstrated in Section \ref{sec:automatic-preference-data-generation}. However, this approach typically underperforms compared to LLMs trained with preference data \citep{zhang-etal:2024generative,mahan-etal:2024generative}. One reason is that standard LLMs are not fine-tuned specifically for the reward modeling task, and as a result, they may not fully capture the nuanced decision-making process that aligns better with human preferences. Also, LLMs trained with preference data learn to prioritize outputs based on feedback, enhancing their capability to evaluate outputs according to the desired behaviors and objectives. where $\mathbf{rat}$ is the labeled CoT rationale for generating a preference between $\mathbf{y}_a$ and $\mathbf{y}_b$. In practice, we can obtain the evaluation results by prompting a standard LLM (as known as LLM-as-a-judge), as demonstrated in Section \ref{sec:automatic-preference-data-generation}. However, this approach typically underperforms compared to LLMs trained with preference data \citep{zhang-etal:2024generative,mahan-etal:2024generative}. One reason is that standard LLMs are not fine-tuned specifically for the reward modeling task, and as a result, they may not fully capture the nuanced decision-making process that aligns better with human preferences. Also, LLMs trained with preference data learn to prioritize outputs based on feedback, enhancing their capability to evaluate outputs according to the desired behaviors and objectives.
Although incorporating CoT rationales into generative reward models improves preference learning, it also introduces a probability degeneration issue in Eq.~(\ref{eq:apply-generative-rm}). Specifically, preference token probabilities often become saturated near 0 or 1, resulting in limited reward variance. This issue is particularly challenging for RL algorithms such as GRPO, where relative reward differences are essential for policy optimization. To address this limitation, researchers investigate ranking-based reward construction approaches, which leverage the ranking capability of generative reward models to derive more effective rewards for RL \citep{wang-etal:rrc}. Interested readers can refer to this work for further details.
\subsubsection{Reward Model Evaluation} \subsubsection{Reward Model Evaluation}
After training a reward model, \textit{how we to evaluate the reward model?} a common practice for evaluating After training a reward model, an important question is \textit{how to evaluate whether the learned reward function can accurately capture human preferences}. Unlike conventional supervised models, reward models do not directly predict explicit labels, but instead learn to assign scores that reflect the relative quality of different responses. As a result, existing evaluation methods mainly focus on measuring the preference modeling ability of reward models or their effectiveness in RL training. We summarize three commonly used evaluation methods as follows.
the reward is directly assessing the performance of the aligned LLM. While this practice can respond to final metrics, it incurs significant computational costs. Additionally, this approach
\begin{itemize}
\item \textbf{RL-based Evaluation.}
A straightforward approach to evaluate a reward model is to measure its effectiveness in downstream RL optimization. Specifically, given multiple reward models $\{\mathcal{M}_r^1,\cdots,\mathcal{M}_r^k\}$, we use each reward model to provide reward signals for training a policy model while keeping the training data, RL algorithm, and hyperparameters identical. After optimization, we obtain a set of policies $\{\mathcal{P}_1,\cdots,\mathcal{P}_k\}$, which are evaluated on human preference benchmarks or downstream tasks. The performance of each optimized policy is then used as an indirect measure of the corresponding reward model quality. In this way, we consider that a high-quality reward model should provide more reliable reward signals, enabling the policy to achieve better final performance.
\item \textbf{Pairwise Ranking Evaluation.}
Although RL-based evaluation directly measures whether a reward model can improve downstream policy optimization, it suffers from limitations. First, it introduces significant evaluation costs, as RL training requires substantial computational resources and time, making it difficult to efficiently compare different reward models. Second, the evaluation results can be sensitive to other factors in the RL pipeline, such as the choice of RL algorithms, which may introduce additional variations beyond the quality of the reward model itself. To address these limitations, a more efficient approach is to directly evaluate the preference ranking ability of reward models. Given a prompt $\mathbf{x}$ and two candidate responses $\mathbf{y}^{+}$ and $\mathbf{y}^{-}$, where $\mathbf{y}^{+}$ is preferred over $\mathbf{y}^{-}$ according to human preference annotations, the reward model predicts their relative preference. Specifically, for discriminative reward models, we compare the predicted scores of the two responses. If the reward model assigns a higher score to $\mathbf{y}^{+}$, its prediction is consistent with human preference. For generative reward models, the model directly selects $\mathbf{y}^{+}$ as the preferred response based on its generated preference. Otherwise, the prediction is considered incorrect. By constructing a large number of pairwise ranking samples, we can evaluate the reward model using ranking accuracy:
\begin{equation}
\mathrm{Acc}
=
\frac{1}{N}
\sum_{i=1}^{N}
\mathbb{I}
\left[
R_\theta(\mathbf{x}_i,\mathbf{y}_i^{+})
>
R_\theta(\mathbf{x}_i,\mathbf{y}_i^{-})
\right]
\end{equation}
where $N$ denotes the number of evaluation samples. Examples of this evaluation approach include RewardBench \citep{lambert-etal:Rewardbench,malik-etal:rewardbench}, RM-Bench \citep{liu-etal:rm-bench}, RMB \citep{zhou-etal:rmb}, and JudgeBench \citep{tan-etal:judgebench}.
\item \textbf{Listwise Ranking Evaluation.}
In practical alignment scenarios, the reward model often needs to select the best output from multiple candidates, such as in best-of-$n$ sampling or reranking. Therefore, listwise ranking evaluation measures whether a reward model can produce a consistent ranking over a set of candidate outputs. Specifically, given a prompt $\mathbf{x}$ and a candidate output set:
\begin{equation}
\mathcal{Y}=\{\mathbf{y}_1,\mathbf{y}_2,\cdots,\mathbf{y}_n\}
\end{equation}
the reward model assigns scores to all candidates. Based on the predicted rewards, the reward model selects the highest-scoring response:
\begin{equation}
\hat{\mathbf{y}}=\arg\max_{\mathbf{y}_i\in\mathcal{Y}}R_\theta(\mathbf{x},\mathbf{y}_i)
\end{equation}
The predicted ranking is then compared with human preference rankings to evaluate the consistency between the reward model and human judgments. Examples of this evaluation approach include PPE \citep{frick-etal:ppe}.
\end{itemize}
1. Using RLHF to evaluate it. \\
2. Using pair ranking to evaluate it (RM-Bench, Reward-Bench, so on.). \\
3. When this model is descrminiative model, we can use a probing approach (probing preference representations). \\ 3. When this model is descrminiative model, we can use a probing approach (probing preference representations). \\
\subsection{Better Advantage Estimation} \subsection{Better Advantage Estimation}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论