Commit b75085de by wangchenglong

update.

parent 209c3781
\begin{thebibliography}{119}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
\providecommand{\doi}[1]{doi: #1}\else
\providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi
\bibitem[Amini et~al.(2024)Amini, Vieira, and Cotterell]{amini:2024direct}
Afra Amini, Tim Vieira, and Ryan Cotterell.
\newblock Direct preference optimization with an offset.
\newblock \emph{arXiv preprint arXiv:2402.10571}, 2024.
\bibitem[Azar et~al.(2024)Azar, Guo, Piot, Munos, Rowland, Valko, and Calandriello]{azar:2024general}
Mohammad~Gheshlaghi Azar, Zhaohan~Daniel Guo, Bilal Piot, R{\'{e}}mi Munos, Mark Rowland, Michal Valko, and Daniele Calandriello.
\newblock A general theoretical paradigm to understand learning from human preferences.
\newblock In Sanjoy Dasgupta, Stephan Mandt, and Yingzhen Li (eds.), \emph{International Conference on Artificial Intelligence and Statistics, 2-4 May 2024, Palau de Congressos, Valencia, Spain}, volume 238 of \emph{Proceedings of Machine Learning Research}, pp.\ 4447--4455. {PMLR}, 2024.
\newblock URL \url{https://proceedings.mlr.press/v238/gheshlaghi-azar24a.html}.
\bibitem[Bahdanau et~al.(2017)Bahdanau, Brakel, Xu, Goyal, Lowe, Pineau, Courville, and Bengio]{bahdanau-etal:2016actor}
Dzmitry Bahdanau, Philemon Brakel, Kelvin Xu, Anirudh Goyal, Ryan Lowe, Joelle Pineau, Aaron~C. Courville, and Yoshua Bengio.
\newblock An actor-critic algorithm for sequence prediction.
\newblock In \emph{5th International Conference on Learning Representations, {ICLR} 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings}. OpenReview.net, 2017.
\newblock URL \url{https://openreview.net/forum?id=SJDaqqveg}.
\bibitem[Baltru{\v{s}}aitis et~al.(2018)Baltru{\v{s}}aitis, Ahuja, and Morency]{baltruvsaitis-etal:2018multimodal}
Tadas Baltru{\v{s}}aitis, Chaitanya Ahuja, and Louis-Philippe Morency.
\newblock Multimodal machine learning: A survey and taxonomy.
\newblock \emph{IEEE transactions on pattern analysis and machine intelligence}, 41\penalty0 (2):\penalty0 423--443, 2018.
\bibitem[Bishop(2006)]{Bishop:2006}
Christopher~M. Bishop.
\newblock \emph{Pattern Recognition and Machine Learning}.
\newblock Springer, 2006.
\bibitem[Bradley \& Terry(1952)Bradley and Terry]{bradley-and-terry:rank}
Ralph~Allan Bradley and Milton~E. Terry.
\newblock Rank analysis of incomplete block designs: I. the method of paired comparisons.
\newblock \emph{Biometrika}, 39\penalty0 (3/4):\penalty0 324--345, 1952.
\bibitem[Bromley et~al.(1993)Bromley, Guyon, LeCun, S{\"a}ckinger, and Shah]{bromley-etal:1993signature}
Jane Bromley, Isabelle Guyon, Yann LeCun, Eduard S{\"a}ckinger, and Roopak Shah.
\newblock Signature verification using a" siamese" time delay neural network.
\newblock \emph{Advances in neural information processing systems}, 6, 1993.
\bibitem[Chen et~al.(2023{\natexlab{a}})Chen, Shu, Shareghi, Collier, Narasimhan, and Yao]{chen-etal:fireact}
Baian Chen, Chang Shu, Ehsan Shareghi, Nigel Collier, Karthik Narasimhan, and Shunyu Yao.
\newblock Fireact: Toward language agent fine-tuning.
\newblock \emph{arXiv preprint arXiv:2310.05915}, 2023{\natexlab{a}}.
\bibitem[Chen et~al.(2023{\natexlab{b}})Chen, Borgeaud, Irving, Lespiau, Sifre, and Jumper]{chen-etal:2023accelerating}
Charlie Chen, Sebastian Borgeaud, Geoffrey Irving, Jean-Baptiste Lespiau, Laurent Sifre, and John Jumper.
\newblock Accelerating large language model decoding with speculative sampling.
\newblock \emph{ArXiv preprint}, abs/2302.01318, 2023{\natexlab{b}}.
\newblock URL \url{https://arxiv.org/abs/2302.01318}.
\bibitem[Chen et~al.(2024)Chen, Li, Yan, Wang, Gunaratna, Yadav, Tang, Srinivasan, Zhou, Huang, and Jin]{chen-etal:2023alpagasus}
Lichang Chen, Shiyang Li, Jun Yan, Hai Wang, Kalpa Gunaratna, Vikas Yadav, Zheng Tang, Vijay Srinivasan, Tianyi Zhou, Heng Huang, and Hongxia Jin.
\newblock Alpagasus: Training a better alpaca with fewer data.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024.
\newblock URL \url{https://openreview.net/forum?id=FdVXgSJhvz}.
\bibitem[Cheng et~al.(2023)Cheng, Xie, Bai, Dai, and Du]{cheng-etal:2023everyone}
Pengyu Cheng, Jiawen Xie, Ke~Bai, Yong Dai, and Nan Du.
\newblock Everyone deserves a reward: Learning customized human preferences.
\newblock \emph{arXiv preprint arXiv:2309.03126}, 2023.
\bibitem[Christiano et~al.(2017)Christiano, Leike, Brown, Martic, Legg, and Amodei]{christiano-etal:2017deep}
Paul~F. Christiano, Jan Leike, Tom~B. Brown, Miljan Martic, Shane Legg, and Dario Amodei.
\newblock Deep reinforcement learning from human preferences.
\newblock In Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna~M. Wallach, Rob Fergus, S.~V.~N. Vishwanathan, and Roman Garnett (eds.), \emph{Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, {USA}}, pp.\ 4299--4307, 2017.
\newblock URL \url{https://proceedings.neurips.cc/paper/2017/hash/d5e2c0adad503c91f91df240d0cd4e49-Abstract.html}.
\bibitem[Coste et~al.(2024)Coste, Anwar, Kirk, and Krueger]{coste-etal:2024reward}
Thomas Coste, Usman Anwar, Robert Kirk, and David Krueger.
\newblock Reward model ensembles help mitigate overoptimization.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024.
\newblock URL \url{https://openreview.net/forum?id=dcjtMYkpXx}.
\bibitem[Cui et~al.(2024{\natexlab{a}})Cui, Yuan, Ding, Yao, He, Zhu, Ni, Xie, Xie, Lin, Liu, and Sun]{cui-etal:2023ultrafeedback}
Ganqu Cui, Lifan Yuan, Ning Ding, Guanming Yao, Bingxiang He, Wei Zhu, Yuan Ni, Guotong Xie, Ruobing Xie, Yankai Lin, Zhiyuan Liu, and Maosong Sun.
\newblock {ULTRAFEEDBACK:} boosting language models with scaled {AI} feedback.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024{\natexlab{a}}.
\newblock URL \url{https://openreview.net/forum?id=BOorDpKHiJ}.
\bibitem[Cui et~al.(2024{\natexlab{b}})Cui, Yuan, Ding, Yao, He, Zhu, Ni, Xie, Xie, Lin, Liu, and Sun]{cui-etal:2024ultra}
Ganqu Cui, Lifan Yuan, Ning Ding, Guanming Yao, Bingxiang He, Wei Zhu, Yuan Ni, Guotong Xie, Ruobing Xie, Yankai Lin, Zhiyuan Liu, and Maosong Sun.
\newblock {ULTRAFEEDBACK:} boosting language models with scaled {AI} feedback.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024{\natexlab{b}}.
\newblock URL \url{https://openreview.net/forum?id=BOorDpKHiJ}.
\bibitem[Deb et~al.(2016)Deb, Sindhya, and Hakanen]{deb-etal:2016multi}
Kalyanmoy Deb, Karthik Sindhya, and Jussi Hakanen.
\newblock Multi-objective optimization.
\newblock In \emph{Decision sciences}, pp.\ 161--200. CRC Press, 2016.
\bibitem[Deepseek(2025)]{deepseek:2025r1}
Deepseek.
\newblock Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2501.12948}, 2025.
\bibitem[Dubois et~al.(2023)Dubois, Li, Taori, Zhang, Gulrajani, Ba, Guestrin, Liang, and Hashimoto]{dubois-etal:2024alpacafarm}
Yann Dubois, Chen~Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy Liang, and Tatsunori~B. Hashimoto.
\newblock Alpacafarm: {A} simulation framework for methods that learn from human feedback.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/5fc47800ee5b30b8777fdd30abcaaf3b-Abstract-Conference.html}.
\bibitem[Eisenstein et~al.(2023)Eisenstein, Nagpal, Agarwal, Beirami, D'Amour, Dvijotham, Fisch, Heller, Pfohl, Ramachandran, et~al.]{eisenstein-etal:2023helping}
Jacob Eisenstein, Chirag Nagpal, Alekh Agarwal, Ahmad Beirami, Alex D'Amour, DJ~Dvijotham, Adam Fisch, Katherine Heller, Stephen Pfohl, Deepak Ramachandran, et~al.
\newblock Helping or herding? reward model ensembles mitigate but do not eliminate reward hacking.
\newblock \emph{ArXiv preprint}, abs/2312.09244, 2023.
\newblock URL \url{https://arxiv.org/abs/2312.09244}.
\bibitem[Ethayarajh et~al.(2024)Ethayarajh, Xu, Muennighoff, Jurafsky, and Kiela]{ethayarajh:2024kto}
Kawin Ethayarajh, Winnie Xu, Niklas Muennighoff, Dan Jurafsky, and Douwe Kiela.
\newblock Kto: Model alignment as prospect theoretic optimization.
\newblock \emph{ArXiv preprint}, abs/2402.01306, 2024.
\newblock URL \url{https://arxiv.org/abs/2402.01306}.
\bibitem[Gallego(2024)]{gallego:2024refined}
V{\'\i}ctor Gallego.
\newblock Refined direct preference optimization with synthetic data for behavioral alignment of llms.
\newblock \emph{ArXiv preprint}, abs/2402.08005, 2024.
\newblock URL \url{https://arxiv.org/abs/2402.08005}.
\bibitem[Gao et~al.(2023)Gao, Schulman, and Hilton]{gao-etal:2023scaling}
Leo Gao, John Schulman, and Jacob Hilton.
\newblock Scaling laws for reward model overoptimization.
\newblock In Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (eds.), \emph{International Conference on Machine Learning, {ICML} 2023, 23-29 July 2023, Honolulu, Hawaii, {USA}}, volume 202 of \emph{Proceedings of Machine Learning Research}, pp.\ 10835--10866. {PMLR}, 2023.
\newblock URL \url{https://proceedings.mlr.press/v202/gao23h.html}.
\bibitem[Gorbatovski et~al.(2024)Gorbatovski, Shaposhnikov, Malakhov, Surnachev, Aksenov, Maksimov, Balagansky, and Gavrilov]{gorbatovski-etal:2024learn}
Alexey Gorbatovski, Boris Shaposhnikov, Alexey Malakhov, Nikita Surnachev, Yaroslav Aksenov, Ian Maksimov, Nikita Balagansky, and Daniil Gavrilov.
\newblock Learn your reference model for real good alignment.
\newblock \emph{ArXiv preprint}, abs/2404.09656, 2024.
\newblock URL \url{https://arxiv.org/abs/2404.09656}.
\bibitem[Grattafiori et~al.(2024)Grattafiori, Dubey, Jauhri, Pandey, Kadian, Al-Dahle, Letman, Mathur, Schelten, Vaughan, et~al.]{grattafiori-etal:2024llama}
Aaron Grattafiori, Abhimanyu Dubey, Abhinav Jauhri, Abhinav Pandey, Abhishek Kadian, Ahmad Al-Dahle, Aiesha Letman, Akhil Mathur, Alan Schelten, Alex Vaughan, et~al.
\newblock The llama 3 herd of models.
\newblock \emph{arXiv preprint arXiv:2407.21783}, 2024.
\bibitem[Guo et~al.(2025)Guo, Yang, Zhang, Song, Zhang, Xu, Zhu, Ma, Wang, Bi, et~al.]{guo:2025deepseek}
Daya Guo, Dejian Yang, Haowei Zhang, Junxiao Song, Ruoyu Zhang, Runxin Xu, Qihao Zhu, Shirong Ma, Peiyi Wang, Xiao Bi, et~al.
\newblock Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.
\newblock \emph{ArXiv preprint}, abs/2501.12948, 2025.
\newblock URL \url{https://arxiv.org/abs/2501.12948}.
\bibitem[Havrilla et~al.(2024)Havrilla, Du, Raparthy, Nalmpantis, Dwivedi-Yu, Zhuravinskyi, Hambro, Sukhbaatar, and Raileanu]{havrilla-etal:2024teaching}
Alex Havrilla, Yuqing Du, Sharath~Chandra Raparthy, Christoforos Nalmpantis, Jane Dwivedi-Yu, Maksym Zhuravinskyi, Eric Hambro, Sainbayar Sukhbaatar, and Roberta Raileanu.
\newblock Teaching large language models to reason with reinforcement learning.
\newblock \emph{ArXiv preprint}, abs/2403.04642, 2024.
\newblock URL \url{https://arxiv.org/abs/2403.04642}.
\bibitem[Hong et~al.(2024)Hong, Lee, and Thorne]{hong:2024orpo}
Jiwoo Hong, Noah Lee, and James Thorne.
\newblock Orpo: Monolithic preference optimization without reference model.
\newblock \emph{ArXiv preprint}, abs/2403.07691, 2024.
\newblock URL \url{https://arxiv.org/abs/2403.07691}.
\bibitem[Hu(2025)]{hu:2025reinforce++}
Jian Hu.
\newblock Reinforce++: A simple and efficient approach for aligning large language models.
\newblock \emph{ArXiv preprint}, abs/2501.03262, 2025.
\newblock URL \url{https://arxiv.org/abs/2501.03262}.
\bibitem[Hu et~al.(2025)Hu, Zhao, Xu, Sun, Lou, Lin, Luo, and Rajmohan]{hu-etal:agentgen}
Mengkang Hu, Pu~Zhao, Can Xu, Qingfeng Sun, Jian-Guang Lou, Qingwei Lin, Ping Luo, and Saravan Rajmohan.
\newblock Agentgen: Enhancing planning abilities for large language model based agent via environment and task generation.
\newblock In \emph{Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1}, pp.\ 496--507, 2025.
\bibitem[Ji et~al.(2025)Ji, Chen, Pan, Zhu, Zhang, Li, Hong, Chen, Zhou, Wang, et~al.]{ji2025safe}
Jiaming Ji, Xinyu Chen, Rui Pan, Han Zhu, Conghui Zhang, Jiahao Li, Donghai Hong, Boyuan Chen, Jiayi Zhou, Kaile Wang, et~al.
\newblock Safe rlhf-v: Safe reinforcement learning from human feedback in multimodal large language models.
\newblock \emph{arXiv preprint arXiv:2503.17682}, 2025.
\bibitem[Kumar et~al.(2024)Kumar, Zhuang, Agarwal, Su, Co-Reyes, Singh, Baumli, Iqbal, Bishop, Roelofs, et~al.]{kumar-etal:2024training}
Aviral Kumar, Vincent Zhuang, Rishabh Agarwal, Yi~Su, John~D Co-Reyes, Avi Singh, Kate Baumli, Shariq Iqbal, Colton Bishop, Rebecca Roelofs, et~al.
\newblock Training language models to self-correct via reinforcement learning.
\newblock \emph{ArXiv preprint}, abs/2409.12917, 2024.
\newblock URL \url{https://arxiv.org/abs/2409.12917}.
\bibitem[Lee et~al.(2024)Lee, Bjelonic, Reske, Wellhausen, Miki, and Hutter]{lee-etal:lee2024learning}
Joonho Lee, Marko Bjelonic, Alexander Reske, Lorenz Wellhausen, Takahiro Miki, and Marco Hutter.
\newblock Learning robust autonomous navigation and locomotion for wheeled-legged robots.
\newblock \emph{Science Robotics}, 9\penalty0 (89):\penalty0 eadi9641, 2024.
\bibitem[Li et~al.(2025{\natexlab{a}})Li, Zou, and Liu]{li-etal:2025limr}
Xuefeng Li, Haoyang Zou, and Pengfei Liu.
\newblock Limr: Less is more for rl scaling.
\newblock \emph{ArXiv preprint}, abs/2502.11886, 2025{\natexlab{a}}.
\newblock URL \url{https://arxiv.org/abs/2502.11886}.
\bibitem[Li et~al.(2025{\natexlab{b}})Li, Hu, and Wang]{li-etal:encouraging}
Zhiwei Li, Yong Hu, and Wenqing Wang.
\newblock Encouraging good processes without the need for good answers: Reinforcement learning for llm agent planning.
\newblock In \emph{Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing: Industry Track}, pp.\ 1654--1666, 2025{\natexlab{b}}.
\bibitem[Li et~al.(2024)Li, Xu, Zhang, Lin, Yu, Sun, and Luo]{li-etal:2023remax}
Ziniu Li, Tian Xu, Yushun Zhang, Zhihang Lin, Yang Yu, Ruoyu Sun, and Zhi{-}Quan Luo.
\newblock Remax: {A} simple, effective, and efficient reinforcement learning method for aligning large language models.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024.
\newblock URL \url{https://openreview.net/forum?id=Stn8hXkpe6}.
\bibitem[Lightman et~al.(2024{\natexlab{a}})Lightman, Kosaraju, Burda, Edwards, Baker, Lee, Leike, Schulman, Sutskever, and Cobbe]{lightman-etal:2023let}
Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe.
\newblock Let's verify step by step.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{a}}.
\newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}.
\bibitem[Lightman et~al.(2024{\natexlab{b}})Lightman, Kosaraju, Burda, Edwards, Baker, Lee, Leike, Schulman, Sutskever, and Cobbe]{lightman-etal:2024lets}
Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe.
\newblock Let's verify step by step.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}.
\newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
\newblock {G}-eval: {NLG} evaluation using gpt-4 with better human alignment.
\newblock In Houda Bouamor, Juan Pino, and Kalika Bali (eds.), \emph{Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing}, pp.\ 2511--2522, Singapore, 2023. Association for Computational Linguistics.
\newblock \doi{10.18653/v1/2023.emnlp-main.153}.
\newblock URL \url{https://aclanthology.org/2023.emnlp-main.153}.
\bibitem[Longpre et~al.(2023)Longpre, Hou, Vu, Webson, Chung, Tay, Zhou, Le, Zoph, Wei, et~al.]{longpre-etal:flan}
Shayne Longpre, Le~Hou, Tu~Vu, Albert Webson, Hyung~Won Chung, Yi~Tay, Denny Zhou, Quoc~V Le, Barret Zoph, Jason Wei, et~al.
\newblock The flan collection: Designing data and methods for effective instruction tuning.
\newblock In \emph{International conference on machine learning}, pp.\ 22631--22648. PMLR, 2023.
\bibitem[Mahan et~al.(2024)Mahan, Van~Phung, Rafailov, Blagden, Lile, Castricato, Fr{\"a}nken, Finn, and Albalak]{mahan-etal:2024generative}
Dakota Mahan, Duy Van~Phung, Rafael Rafailov, Chase Blagden, Nathan Lile, Louis Castricato, Jan-Philipp Fr{\"a}nken, Chelsea Finn, and Alon Albalak.
\newblock Generative reward models.
\newblock \emph{ArXiv preprint}, abs/2410.12832, 2024.
\newblock URL \url{https://arxiv.org/abs/2410.12832}.
\bibitem[Masoudnia \& Ebrahimpour(2014)Masoudnia and Ebrahimpour]{masoudnia-etal:2014mixture}
Saeed Masoudnia and Reza Ebrahimpour.
\newblock Mixture of experts: a literature survey.
\newblock \emph{Artificial Intelligence Review}, 42:\penalty0 275--293, 2014.
\bibitem[Melnyk et~al.(2024)Melnyk, Mroueh, Belgodere, Rigotti, Nitsure, Yurochkin, Greenewald, Navratil, and Ross]{melnyk:2024distributional}
Igor Melnyk, Youssef Mroueh, Brian Belgodere, Mattia Rigotti, Apoorva Nitsure, Mikhail Yurochkin, Kristjan Greenewald, Jiri Navratil, and Jerret Ross.
\newblock Distributional preference alignment of llms via optimal transport.
\newblock In \emph{Advances in Neural Information Processing Systems}, 2024.
\bibitem[Meng et~al.(2024)Meng, Xia, and Chen]{meng:2025simpo}
Yu~Meng, Mengzhou Xia, and Danqi Chen.
\newblock Simpo: Simple preference optimization with a reference-free reward.
\newblock In Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub~M. Tomczak, and Cheng Zhang (eds.), \emph{Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024}, 2024.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2024/hash/e099c1c9699814af0be873a175361713-Abstract-Conference.html}.
\bibitem[Miettinen(1999)]{miettinen:1999nonlinear}
Kaisa Miettinen.
\newblock \emph{Nonlinear multiobjective optimization}, volume~12.
\newblock Springer Science \& Business Media, 1999.
\bibitem[Miettinen et~al.(2008)Miettinen, Ruiz, and Wierzbicki]{miettinen-etal:2008introduction}
Kaisa Miettinen, Francisco Ruiz, and Andrzej~P Wierzbicki.
\newblock Introduction to multiobjective optimization: interactive approaches.
\newblock In \emph{Multiobjective optimization: interactive and evolutionary approaches}, pp.\ 27--57. Springer, 2008.
\bibitem[Mnih et~al.(2013)Mnih, Kavukcuoglu, Silver, Graves, Antonoglou, Wierstra, and Riedmiller]{mnih-etal:mnih2013playing}
Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller.
\newblock Playing atari with deep reinforcement learning.
\newblock \emph{arXiv preprint arXiv:1312.5602}, 2013.
\bibitem[Mnih et~al.(2016)Mnih, Badia, Mirza, Graves, Lillicrap, Harley, Silver, and Kavukcuoglu]{mnih-etal:2016asynchronous}
Volodymyr Mnih, Adri{\`{a}}~Puigdom{\`{e}}nech Badia, Mehdi Mirza, Alex Graves, Timothy~P. Lillicrap, Tim Harley, David Silver, and Koray Kavukcuoglu.
\newblock Asynchronous methods for deep reinforcement learning.
\newblock In Maria{-}Florina Balcan and Kilian~Q. Weinberger (eds.), \emph{Proceedings of the 33nd International Conference on Machine Learning, {ICML} 2016, New York City, NY, USA, June 19-24, 2016}, volume~48 of \emph{{JMLR} Workshop and Conference Proceedings}, pp.\ 1928--1937. JMLR.org, 2016.
\newblock URL \url{http://proceedings.mlr.press/v48/mniha16.html}.
\bibitem[Morimura et~al.(2024)Morimura, Sakamoto, Jinnai, Abe, and Ariu]{morimura-etal:2024filtered}
Tetsuro Morimura, Mitsuki Sakamoto, Yuu Jinnai, Kenshi Abe, and Kaito Ariu.
\newblock Filtered direct preference optimization.
\newblock \emph{ArXiv preprint}, abs/2404.13846, 2024.
\newblock URL \url{https://arxiv.org/abs/2404.13846}.
\bibitem[Nakano et~al.(2021)Nakano, Hilton, Balaji, Wu, Ouyang, Kim, Hesse, Jain, Kosaraju, Saunders, Jiang, Cobbe, Eloundou, Krueger, Button, Knight, Chess, and Schulman]{nakano-etal:2021webgpt}
Reiichiro Nakano, Jacob Hilton, Suchir Balaji, Jeff Wu, Long Ouyang, Christina Kim, Christopher Hesse, Shantanu Jain, Vineet Kosaraju, William Saunders, Xu~Jiang, Karl Cobbe, Tyna Eloundou, Gretchen Krueger, Kevin Button, Matthew Knight, Benjamin Chess, and John Schulman.
\newblock Webgpt: Browser-assisted question-answering with human feedback.
\newblock \emph{ArXiv preprint}, abs/2112.09332, 2021.
\newblock URL \url{https://arxiv.org/abs/2112.09332}.
\bibitem[OpenAI(2024)]{openai:2024learning}
OpenAI.
\newblock Learning to reason with llms, September 2024.
\newblock URL \url{https://openai.com/index/learning-to-reason-with-llms/}.
\bibitem[Ouyang et~al.(2022)Ouyang, Wu, Jiang, Almeida, Wainwright, Mishkin, Zhang, Agarwal, Slama, Ray, Schulman, Hilton, Kelton, Miller, Simens, Askell, Welinder, Christiano, Leike, and Lowe]{ouyang:2022training}
Long Ouyang, Jeffrey Wu, Xu~Jiang, Diogo Almeida, Carroll~L. Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda Askell, Peter Welinder, Paul~F. Christiano, Jan Leike, and Ryan Lowe.
\newblock Training language models to follow instructions with human feedback.
\newblock In Sanmi Koyejo, S.~Mohamed, A.~Agarwal, Danielle Belgrave, K.~Cho, and A.~Oh (eds.), \emph{Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022}, 2022.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2022/hash/b1efde53be364a73914f58805a001731-Abstract-Conference.html}.
\bibitem[Pope et~al.(2023)Pope, Douglas, Chowdhery, Devlin, Bradbury, Heek, Xiao, Agrawal, and Dean]{pope-etal:2023efficiently}
Reiner Pope, Sholto Douglas, Aakanksha Chowdhery, Jacob Devlin, James Bradbury, Jonathan Heek, Kefan Xiao, Shivani Agrawal, and Jeff Dean.
\newblock Efficiently scaling transformer inference.
\newblock \emph{Proceedings of Machine Learning and Systems}, 5:\penalty0 606--624, 2023.
\bibitem[Qian et~al.(2026)Qian, Acikgoz, He, Wang, Chen, Hakkani-Tur, Tur, and Ji]{qian-etal:toolrl}
Cheng Qian, Emre~Can Acikgoz, Qi~He, Hongru Wang, Xiusi Chen, Dilek Hakkani-Tur, Gokhan Tur, and Heng Ji.
\newblock Toolrl: Reward is all tool learning needs.
\newblock \emph{Advances in Neural Information Processing Systems}, 38:\penalty0 105523--105553, 2026.
\bibitem[Radford et~al.(2021)Radford, Kim, Hallacy, Ramesh, Goh, Agarwal, Sastry, Askell, Mishkin, Clark, Krueger, and Sutskever]{radford-etal:2021learning}
Alec Radford, Jong~Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.
\newblock Learning transferable visual models from natural language supervision, 2021.
\newblock URL \url{https://arxiv.org/abs/2103.00020}.
\bibitem[Rafailov et~al.(2023)Rafailov, Sharma, Mitchell, Manning, Ermon, and Finn]{rafailov:2023direct}
Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher~D. Manning, Stefano Ermon, and Chelsea Finn.
\newblock Direct preference optimization: Your language model is secretly a reward model.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/a85b405ed65c6477a4fe8302b5e06ce7-Abstract-Conference.html}.
\bibitem[Schick et~al.(2023)Schick, Dwivedi-Yu, Dess{\`\i}, Raileanu, Lomeli, Hambro, Zettlemoyer, Cancedda, and Scialom]{schick-etal:toolformer}
Timo Schick, Jane Dwivedi-Yu, Roberto Dess{\`\i}, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom.
\newblock Toolformer: Language models can teach themselves to use tools.
\newblock \emph{Advances in neural information processing systems}, 36:\penalty0 68539--68551, 2023.
\bibitem[Schulman et~al.(2015)Schulman, Levine, Abbeel, Jordan, and Moritz]{schulman-etal:2015trust}
John Schulman, Sergey Levine, Pieter Abbeel, Michael~I. Jordan, and Philipp Moritz.
\newblock Trust region policy optimization.
\newblock In Francis~R. Bach and David~M. Blei (eds.), \emph{Proceedings of the 32nd International Conference on Machine Learning, {ICML} 2015, Lille, France, 6-11 July 2015}, volume~37 of \emph{{JMLR} Workshop and Conference Proceedings}, pp.\ 1889--1897. JMLR.org, 2015.
\newblock URL \url{http://proceedings.mlr.press/v37/schulman15.html}.
\bibitem[Schulman et~al.(2016)Schulman, Moritz, Levine, Jordan, and Abbeel]{schulman-etal:2015high}
John Schulman, Philipp Moritz, Sergey Levine, Michael~I. Jordan, and Pieter Abbeel.
\newblock High-dimensional continuous control using generalized advantage estimation.
\newblock In Yoshua Bengio and Yann LeCun (eds.), \emph{4th International Conference on Learning Representations, {ICLR} 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings}, 2016.
\newblock URL \url{http://arxiv.org/abs/1506.02438}.
\bibitem[Schulman et~al.(2017)Schulman, Wolski, Dhariwal, Radford, and Klimov]{schulman-etal:2017proximal}
John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov.
\newblock Proximal policy optimization algorithms.
\newblock \emph{ArXiv preprint}, abs/1707.06347, 2017.
\newblock URL \url{https://arxiv.org/abs/1707.06347}.
\bibitem[Setlur et~al.(2024)Setlur, Nagpal, Fisch, Geng, Eisenstein, Agarwal, Agarwal, Berant, and Kumar]{setlur-etal:2024rewarding}
Amrith Setlur, Chirag Nagpal, Adam Fisch, Xinyang Geng, Jacob Eisenstein, Rishabh Agarwal, Alekh Agarwal, Jonathan Berant, and Aviral Kumar.
\newblock Rewarding progress: Scaling automated process verifiers for llm reasoning.
\newblock \emph{ArXiv preprint}, abs/2410.08146, 2024.
\newblock URL \url{https://arxiv.org/abs/2410.08146}.
\bibitem[Shao et~al.(2025)Shao, Li, Liu, Chen, Zhou, Wang, Cai, and Li]{shao:2025earlier}
Ruichen Shao, Bei Li, Gangao Liu, Yang Chen, Xiang Zhou, Jingang Wang, Xunliang Cai, and Peng Li.
\newblock Earlier tokens contribute more: Learning direct preference optimization from temporal decay perspective.
\newblock \emph{ArXiv preprint}, abs/2502.14340, 2025.
\newblock URL \url{https://arxiv.org/abs/2502.14340}.
\bibitem[Shao et~al.(2024)Shao, Wang, Zhu, Xu, Song, Bi, Zhang, Zhang, Li, Wu, et~al.]{shao-etal:2024deepseekmath}
Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, Haowei Zhang, Mingchuan Zhang, YK~Li, Y~Wu, et~al.
\newblock Deepseekmath: Pushing the limits of mathematical reasoning in open language models.
\newblock \emph{ArXiv preprint}, abs/2402.03300, 2024.
\newblock URL \url{https://arxiv.org/abs/2402.03300}.
\bibitem[Silver et~al.(2016)Silver, Huang, Maddison, Guez, Sifre, Van Den~Driessche, Schrittwieser, Antonoglou, Panneershelvam, Lanctot, et~al.]{silver-etal:2016mastering}
David Silver, Aja Huang, Chris~J Maddison, Arthur Guez, Laurent Sifre, George Van Den~Driessche, Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et~al.
\newblock Mastering the game of go with deep neural networks and tree search.
\newblock \emph{nature}, 529\penalty0 (7587):\penalty0 484--489, 2016.
\bibitem[Silver et~al.(2017)Silver, Schrittwieser, Simonyan, Antonoglou, Huang, Guez, Hubert, Baker, Lai, Bolton, et~al.]{silver-etal:silver2017mastering}
David Silver, Julian Schrittwieser, Karen Simonyan, Ioannis Antonoglou, Aja Huang, Arthur Guez, Thomas Hubert, Lucas Baker, Matthew Lai, Adrian Bolton, et~al.
\newblock Mastering the game of go without human knowledge.
\newblock \emph{nature}, 550\penalty0 (7676):\penalty0 354--359, 2017.
\bibitem[Singh et~al.(2025)Singh, Pandya, Vajreshwari, Magazine, and Nambi]{singh-etal:singhagentic}
Joykirat Singh, Yash Pandya, Pranav Vajreshwari, Raghav Magazine, and Akshay Nambi.
\newblock Agentic reasoning and tool integration for llms via reinforcement learning.
\newblock In \emph{First Workshop on Foundations of Reasoning in Language Models}, 2025.
\bibitem[Singhal et~al.(2023)Singhal, Goyal, Xu, and Durrett]{singhal-etal:2023long}
Prasann Singhal, Tanya Goyal, Jiacheng Xu, and Greg Durrett.
\newblock A long way to go: Investigating length correlations in rlhf.
\newblock \emph{ArXiv preprint}, abs/2310.03716, 2023.
\newblock URL \url{https://arxiv.org/abs/2310.03716}.
\bibitem[Sun et~al.(2024)Sun, Liu, Bair, and Kolter]{sun-etal:2023simple}
Mingjie Sun, Zhuang Liu, Anna Bair, and J.~Zico Kolter.
\newblock A simple and effective pruning approach for large language models.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024.
\newblock URL \url{https://openreview.net/forum?id=PxoFut3dWW}.
\bibitem[Sun et~al.(2023)Sun, Shen, Cao, Liu, Li, Shen, Gan, Gui, Wang, Yang, et~al.]{sun-etal:2023aligning}
Zhiqing Sun, Sheng Shen, Shengcao Cao, Haotian Liu, Chunyuan Li, Yikang Shen, Chuang Gan, Liang-Yan Gui, Yu-Xiong Wang, Yiming Yang, et~al.
\newblock Aligning large multimodal models with factually augmented rlhf.
\newblock \emph{arXiv preprint arXiv:2309.14525}, 2023.
\bibitem[Sutton(1988)]{sutton-and-richard:1988learning}
Richard~S Sutton.
\newblock Learning to predict by the methods of temporal differences.
\newblock \emph{Machine learning}, 3:\penalty0 9--44, 1988.
\bibitem[Sutton \& Barto(2018)Sutton and Barto]{Sutton-and-Barto:2018RL}
Richard~S. Sutton and Andrew~G. Barto.
\newblock \emph{Reinforcement Learning: An Introduction (2nd ed.)}.
\newblock The MIT Press, 2018.
\bibitem[Szepesv{\'a}ri(2010)]{szepesvari:2010algorithms}
Csaba Szepesv{\'a}ri.
\newblock Algorithms for reinforcement learning.
\newblock \emph{Synthesis Lectures on Artificial Intelligence and Machine Learning}, 4\penalty0 (1):\penalty0 1--103, 2010.
\bibitem[Team et~al.(2025)Team, Du, Gao, Xing, Jiang, Chen, Li, Xiao, Du, Liao, et~al.]{kimi-team:2025kimi}
Kimi Team, Angang Du, Bofei Gao, Bowei Xing, Changjiu Jiang, Cheng Chen, Cheng Li, Chenjun Xiao, Chenzhuang Du, Chonghua Liao, et~al.
\newblock Kimi k1. 5: Scaling reinforcement learning with llms.
\newblock \emph{ArXiv preprint}, abs/2501.12599, 2025.
\newblock URL \url{https://arxiv.org/abs/2501.12599}.
\bibitem[Team(2025)]{qwenTeam:2025qwen2.5-VL}
Qwen Team.
\newblock Qwen2.5-vl, January 2025.
\newblock URL \url{https://qwenlm.github.io/blog/qwen2.5-vl/}.
\bibitem[Touvron et~al.(2023)Touvron, Martin, Stone, Albert, Almahairi, Babaei, Bashlykov, Batra, Bhargava, Bhosale, Bikel, Blecher, Ferrer, Chen, Cucurull, Esiobu, Fernandes, Fu, Fu, Fuller, Gao, Goswami, Goyal, Hartshorn, Hosseini, Hou, Inan, Kardas, Kerkez, Khabsa, Kloumann, Korenev, Koura, Lachaux, Lavril, Lee, Liskovich, Lu, Mao, Martinet, Mihaylov, Mishra, Molybog, Nie, Poulton, Reizenstein, Rungta, Saladi, Schelten, Silva, Smith, Subramanian, Tan, Tang, Taylor, Williams, Kuan, Xu, Yan, Zarov, Zhang, Fan, Kambadur, Narang, Rodriguez, Stojnic, Edunov, and Scialom]{touvron-etal:2023llama2}
Hugo Touvron, Louis Martin, Kevin Stone, Peter Albert, Amjad Almahairi, Yasmine Babaei, Nikolay Bashlykov, Soumya Batra, Prajjwal Bhargava, Shruti Bhosale, Dan Bikel, Lukas Blecher, Cristian~Canton Ferrer, Moya Chen, Guillem Cucurull, David Esiobu, Jude Fernandes, Jeremy Fu, Wenyin Fu, Brian Fuller, Cynthia Gao, Vedanuj Goswami, Naman Goyal, Anthony Hartshorn, Saghar Hosseini, Rui Hou, Hakan Inan, Marcin Kardas, Viktor Kerkez, Madian Khabsa, Isabel Kloumann, Artem Korenev, Punit~Singh Koura, Marie-Anne Lachaux, Thibaut Lavril, Jenya Lee, Diana Liskovich, Yinghai Lu, Yuning Mao, Xavier Martinet, Todor Mihaylov, Pushkar Mishra, Igor Molybog, Yixin Nie, Andrew Poulton, Jeremy Reizenstein, Rashi Rungta, Kalyan Saladi, Alan Schelten, Ruan Silva, Eric~Michael Smith, Ranjan Subramanian, Xiaoqing~Ellen Tan, Binh Tang, Ross Taylor, Adina Williams, Jian~Xiang Kuan, Puxin Xu, Zheng Yan, Iliyan Zarov, Yuchen Zhang, Angela Fan, Melanie Kambadur, Sharan Narang, Aurelien Rodriguez, Robert Stojnic, Sergey Edunov, and Thomas Scialom.
\newblock Llama 2: Open foundation and fine-tuned chat models.
\newblock \emph{ArXiv preprint}, abs/2307.09288, 2023.
\newblock URL \url{https://arxiv.org/abs/2307.09288}.
\bibitem[Vinyals et~al.(2019)Vinyals, Babuschkin, Czarnecki, Mathieu, Dudzik, Chung, Choi, Powell, Ewalds, Georgiev, et~al.]{vinyals-rtal:2019grandmaster}
Oriol Vinyals, Igor Babuschkin, Wojciech~M Czarnecki, Micha{\"e}l Mathieu, Andrew Dudzik, Junyoung Chung, David~H Choi, Richard Powell, Timo Ewalds, Petko Georgiev, et~al.
\newblock Grandmaster level in starcraft ii using multi-agent reinforcement learning.
\newblock \emph{nature}, 575\penalty0 (7782):\penalty0 350--354, 2019.
\bibitem[Wang et~al.(2023{\natexlab{a}})Wang, Zhou, Chang, Liu, Zhang, Du, Xiao, and Zhu]{wang-etal:2023learning}
Chenglong Wang, Hang Zhou, Kaiyan Chang, Tongran Liu, Chunliang Zhang, Quan Du, Tong Xiao, and Jingbo Zhu.
\newblock Learning evaluation models from large language models for sequence generation.
\newblock \emph{ArXiv preprint}, abs/2308.04386, 2023{\natexlab{a}}.
\newblock URL \url{https://arxiv.org/abs/2308.04386}.
\bibitem[Wang et~al.(2024{\natexlab{a}})Wang, Gan, Huo, Mu, Yang, He, Xiao, Zhang, Liu, Du, et~al.]{wang-etal:2024rovrm}
Chenglong Wang, Yang Gan, Yifu Huo, Yongyu Mu, Murun Yang, Qiaozhi He, Tong Xiao, Chunliang Zhang, Tongran Liu, Quan Du, et~al.
\newblock Rovrm: A robust visual reward model optimized via auxiliary textual preference data.
\newblock \emph{arXiv preprint arXiv:2408.12109}, 2024{\natexlab{a}}.
\bibitem[Wang et~al.(2024{\natexlab{b}})Wang, Zhou, Chang, Li, Mu, Xiao, Liu, and Zhu]{wang-etal:2024hybrid}
Chenglong Wang, Hang Zhou, Kaiyan Chang, Bei Li, Yongyu Mu, Tong Xiao, Tongran Liu, and Jingbo Zhu.
\newblock Hybrid alignment training for large language models.
\newblock \emph{ArXiv preprint}, abs/2406.15178, 2024{\natexlab{b}}.
\newblock URL \url{https://arxiv.org/abs/2406.15178}.
\bibitem[Wang et~al.(2024{\natexlab{c}})Wang, Zhou, Hu, Huo, Li, Liu, Xiao, and Zhu]{wang-etal:2024esrl}
Chenglong Wang, Hang Zhou, Yimin Hu, Yifu Huo, Bei Li, Tongran Liu, Tong Xiao, and Jingbo Zhu.
\newblock {ESRL:} efficient sampling-based reinforcement learning for sequence generation.
\newblock In Michael~J. Wooldridge, Jennifer~G. Dy, and Sriraam Natarajan (eds.), \emph{Thirty-Eighth {AAAI} Conference on Artificial Intelligence, {AAAI} 2024, Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence, {IAAI} 2024, Fourteenth Symposium on Educational Advances in Artificial Intelligence, {EAAI} 2014, February 20-27, 2024, Vancouver, Canada}, pp.\ 19107--19115. {AAAI} Press, 2024{\natexlab{c}}.
\newblock \doi{10.1609/AAAI.V38I17.29878}.
\newblock URL \url{https://doi.org/10.1609/aaai.v38i17.29878}.
\bibitem[Wang et~al.(2025)Wang, Gan, Huo, Mu, He, Yang, Li, Xiao, Zhang, Liu, et~al.]{wang-etal:wang2025gram}
Chenglong Wang, Yang Gan, Yifu Huo, Yongyu Mu, Qiaozhi He, Murun Yang, Bei Li, Tong Xiao, Chunliang Zhang, Tongran Liu, et~al.
\newblock Gram: A generative foundation reward model for reward generalization.
\newblock \emph{arXiv preprint arXiv:2506.14175}, 2025.
\bibitem[Wang et~al.(2026{\natexlab{a}})Wang, Huo, Gan, He, Meng, Li, Wang, Liu, Zhou, Zhu, et~al.]{wang2026msrl}
Chenglong Wang, Yifu Huo, Yang Gan, Qiaozhi He, Qi~Meng, Bei Li, Yan Wang, Junfu Liu, Tianhua Zhou, Jingbo Zhu, et~al.
\newblock Msrl: Scaling generative multimodal reward modeling via multi-stage reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2603.25108}, 2026{\natexlab{a}}.
\bibitem[Wang et~al.(2026{\natexlab{b}})Wang, Mu, Zhou, Huo, Zhu, Zeng, Yang, Li, Hao, Zhang, et~al.]{wang-etal:wang2026gramrr}
Chenglong Wang, Yongyu Mu, Hang Zhou, Yifu Huo, Ziming Zhu, Jiali Zeng, Murun Yang, Bei Li, Xiaoyang Hao, Chunliang Zhang, et~al.
\newblock Gram-r$^2$: Self-training generative foundation reward models for reward reasoning.
\newblock In \emph{Proceedings of the AAAI Conference on Artificial Intelligence}, volume~40, pp.\ 33395--33403, 2026{\natexlab{b}}.
\bibitem[Wang et~al.(2026{\natexlab{c}})Wang, Li, Cheng, Ouyang, Yu, Liu, and Chen]{wang-etal:steppo}
Daoyu Wang, Qingchuan Li, Mingyue Cheng, Jie Ouyang, Shuo Yu, Qi~Liu, and Enhong Chen.
\newblock Steppo: Step-aligned policy optimization for agentic reinforcement learning.
\newblock \emph{arXiv preprint arXiv:2604.18401}, 2026{\natexlab{c}}.
\bibitem[Wang et~al.(2021)Wang, Yan, Meng, and Zhou]{wang-etal:2021selective}
Fusheng Wang, Jianhao Yan, Fandong Meng, and Jie Zhou.
\newblock Selective knowledge distillation for neural machine translation.
\newblock In Chengqing Zong, Fei Xia, Wenjie Li, and Roberto Navigli (eds.), \emph{Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)}, pp.\ 6456--6466, Online, 2021. Association for Computational Linguistics.
\newblock \doi{10.18653/v1/2021.acl-long.504}.
\newblock URL \url{https://aclanthology.org/2021.acl-long.504}.
\bibitem[Wang et~al.(2023{\natexlab{b}})Wang, Li, Chen, Cai, Zhu, Lin, Cao, Liu, Liu, and Sui]{wang-etal:2023large}
Peiyi Wang, Lei Li, Liang Chen, Zefan Cai, Dawei Zhu, Binghuai Lin, Yunbo Cao, Qi~Liu, Tianyu Liu, and Zhifang Sui.
\newblock Large language models are not fair evaluators.
\newblock \emph{ArXiv preprint}, 2023{\natexlab{b}}.
\bibitem[Wang et~al.(2023{\natexlab{c}})Wang, Li, Shao, Xu, Dai, Li, Chen, Wu, and Sui]{wang-etal:2023math}
Peiyi Wang, Lei Li, Zhihong Shao, RX~Xu, Damai Dai, Yifei Li, Deli Chen, Yu~Wu, and Zhifang Sui.
\newblock Math-shepherd: Verify and reinforce llms step-by-step without human annotations.
\newblock \emph{ArXiv preprint}, abs/2312.08935, 2023{\natexlab{c}}.
\newblock URL \url{https://arxiv.org/abs/2312.08935}.
\bibitem[Wang \& Zhou(2024)Wang and Zhou]{wang-and-zhou:2024chain}
Xuezhi Wang and Denny Zhou.
\newblock Chain-of-thought reasoning without prompting.
\newblock In Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub~M. Tomczak, and Cheng Zhang (eds.), \emph{Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024}, 2024.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2024/hash/7a8e7fd295aa04eac4b470ae27f8785c-Abstract-Conference.html}.
\bibitem[Wang et~al.(2023{\natexlab{d}})Wang, Ivison, Dasigi, Hessel, Khot, Chandu, Wadden, MacMillan, Smith, Beltagy, and Hajishirzi]{wang-etal:2023far}
Yizhong Wang, Hamish Ivison, Pradeep Dasigi, Jack Hessel, Tushar Khot, Khyathi Chandu, David Wadden, Kelsey MacMillan, Noah~A. Smith, Iz~Beltagy, and Hannaneh Hajishirzi.
\newblock How far can camels go? exploring the state of instruction tuning on open resources.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023{\natexlab{d}}.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/ec6413875e4ab08d7bc4d8e225263398-Abstract-Datasets\_and\_Benchmarks.html}.
\bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned}
Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le.
\newblock Finetuned language models are zero-shot learners.
\newblock In \emph{The Tenth International Conference on Learning Representations, {ICLR} 2022, Virtual Event, April 25-29, 2022}. OpenReview.net, 2022.
\newblock URL \url{https://openreview.net/forum?id=gEZrGCozdqR}.
\bibitem[Wu et~al.(2023)Wu, Hu, Shi, Dziri, Suhr, Ammanabrolu, Smith, Ostendorf, and Hajishirzi]{wu-etal:2023fine}
Zeqiu Wu, Yushi Hu, Weijia Shi, Nouha Dziri, Alane Suhr, Prithviraj Ammanabrolu, Noah~A. Smith, Mari Ostendorf, and Hannaneh Hajishirzi.
\newblock Fine-grained human feedback gives better rewards for language model training.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/b8c90b65739ae8417e61eadb521f63d5-Abstract-Conference.html}.
\bibitem[Wu et~al.(2024)Wu, Balashankar, Kim, Eisenstein, and Beirami]{wu-etal:2024reuse}
Zhaofeng Wu, Ananth Balashankar, Yoon Kim, Jacob Eisenstein, and Ahmad Beirami.
\newblock Reuse your rewards: Reward model transfer for zero-shot cross-lingual alignment.
\newblock \emph{arXiv preprint arXiv:2404.12318}, 2024.
\bibitem[Xi et~al.(2026)Xi, Liao, Li, Zhang, Chen, Wang, Jin, Zhou, Guan, Wu, et~al.]{xi-etal:agentprm}
Zhiheng Xi, Chenyang Liao, Guanyu Li, Zhihao Zhang, Wenxiang Chen, Binghai Wang, Senjie Jin, Yuhao Zhou, Jian Guan, Wei Wu, et~al.
\newblock Agentprm: Process reward models for llm agents via step-wise promise and progress.
\newblock In \emph{Proceedings of the ACM Web Conference 2026}, pp.\ 4184--4195, 2026.
\bibitem[Xia et~al.(2024)Xia, Malladi, Gururangan, Arora, and Chen]{xia-etal:2024less}
Mengzhou Xia, Sadhika Malladi, Suchin Gururangan, Sanjeev Arora, and Danqi Chen.
\newblock Less: Selecting influential data for targeted instruction tuning.
\newblock \emph{arXiv preprint arXiv:2402.04333}, 2024.
\bibitem[Xiao et~al.(2024)Xiao, Yuan, Zhu, Li, and Honavar]{xiao:2024cal}
Teng Xiao, Yige Yuan, Huaisheng Zhu, Mingxiao Li, and Vasant~G. Honavar.
\newblock Cal-dpo: Calibrated direct preference optimization for language model alignment.
\newblock In \emph{The Thirty-eighth Annual Conference on Neural Information Processing Systems}, 2024.
\bibitem[Xiao \& Zhu(2023)Xiao and Zhu]{xiao-etal:2023introduction}
Tong Xiao and Jingbo Zhu.
\newblock Introduction to transformers: an nlp perspective.
\newblock \emph{ArXiv preprint}, abs/2311.17633, 2023.
\newblock URL \url{https://arxiv.org/abs/2311.17633}.
\bibitem[Xiao \& Zhu(2025)Xiao and Zhu]{xiao-and-zhu:2025foundations}
Tong Xiao and Jingbo Zhu.
\newblock Foundations of large language models.
\newblock \emph{ArXiv preprint}, abs/2501.09223, 2025.
\newblock URL \url{https://arxiv.org/abs/2501.09223}.
\bibitem[Xie et~al.(2025)Xie, Gao, Ren, Luo, Hong, Dai, Zhou, Qiu, Wu, and Luo]{xie-etal:2025logic}
Tian Xie, Zitian Gao, Qingnan Ren, Haoming Luo, Yuqian Hong, Bryan Dai, Joey Zhou, Kai Qiu, Zhirong Wu, and Chong Luo.
\newblock Logic-rl: Unleashing llm reasoning with rule-based reinforcement learning.
\newblock \emph{ArXiv preprint}, abs/2502.14768, 2025.
\newblock URL \url{https://arxiv.org/abs/2502.14768}.
\bibitem[Xu et~al.(2024)Xu, Sharaf, Chen, Tan, Shen, Durme, Murray, and Kim]{xu:2024contrastive}
Haoran Xu, Amr Sharaf, Yunmo Chen, Weiting Tan, Lingfeng Shen, Benjamin~Van Durme, Kenton Murray, and Young~Jin Kim.
\newblock Contrastive preference optimization: Pushing the boundaries of {LLM} performance in machine translation.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024.
\newblock URL \url{https://openreview.net/forum?id=51iwkioZpn}.
\bibitem[Xu et~al.(2025)Xu, Guo, He, Hu, He, Bai, Chen, Wang, Fan, Dang, et~al.]{xu-etal:2025qwen2}
Jin Xu, Zhifang Guo, Jinzheng He, Hangrui Hu, Ting He, Shuai Bai, Keqin Chen, Jialin Wang, Yang Fan, Kai Dang, et~al.
\newblock Qwen2. 5-omni technical report.
\newblock \emph{arXiv preprint arXiv:2503.20215}, 2025.
\bibitem[Yang et~al.(2024)Yang, Ding, Lin, Zhang, and Zhang]{yang-etal:2024regularizing}
Rui Yang, Ruomeng Ding, Yong Lin, Huan Zhang, and Tong Zhang.
\newblock Regularizing hidden states enables learning generalizable reward model for llms.
\newblock In Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub~M. Tomczak, and Cheng Zhang (eds.), \emph{Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024}, 2024.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2024/hash/71f7154547c748c8041505521ca433ab-Abstract-Conference.html}.
\bibitem[Yao et~al.(2022)Yao, Zhao, Yu, Du, Shafran, Narasimhan, and Cao]{yao-etal:react}
Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao.
\newblock React: Synergizing reasoning and acting in language models.
\newblock \emph{arXiv preprint arXiv:2210.03629}, 2022.
\bibitem[Yao et~al.(2023)Yao, Yu, Zhao, Shafran, Griffiths, Cao, and Narasimhan]{yao-etal:2023tree}
Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Tom Griffiths, Yuan Cao, and Karthik Narasimhan.
\newblock Tree of thoughts: Deliberate problem solving with large language models.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/271db9922b8d1f4dd7aaef84ed5ac703-Abstract-Conference.html}.
\bibitem[Yin et~al.(2024)Yin, Brahman, Ravichander, Chandu, Chang, Choi, and Lin]{yin-etal:agentlumos}
Da~Yin, Faeze Brahman, Abhilasha Ravichander, Khyathi Chandu, Kai-Wei Chang, Yejin Choi, and Bill~Yuchen Lin.
\newblock Agent lumos: Unified and modular training for open-source language agents.
\newblock In \emph{Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 12380--12403, 2024.
\bibitem[Yu et~al.(2024{\natexlab{a}})Yu, Yao, Zhang, He, Han, Cui, Hu, Liu, Zheng, Sun, et~al.]{yu-etal:2024rlhf}
Tianyu Yu, Yuan Yao, Haoye Zhang, Taiwen He, Yifeng Han, Ganqu Cui, Jinyi Hu, Zhiyuan Liu, Hai-Tao Zheng, Maosong Sun, et~al.
\newblock Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback.
\newblock In \emph{Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition}, pp.\ 13807--13816, 2024{\natexlab{a}}.
\bibitem[Yu et~al.(2024{\natexlab{b}})Yu, Zhang, Yao, Dang, Chen, Lu, Cui, He, Liu, Chua, et~al.]{yu-etal:2024rlaif}
Tianyu Yu, Haoye Zhang, Yuan Yao, Yunkai Dang, Da~Chen, Xiaoman Lu, Ganqu Cui, Taiwen He, Zhiyuan Liu, Tat-Seng Chua, et~al.
\newblock Rlaif-v: Aligning mllms through open-source ai feedback for super gpt-4v trustworthiness.
\newblock \emph{arXiv preprint arXiv:2405.17220}, 2024{\natexlab{b}}.
\bibitem[Yuan et~al.(2024)Yuan, Pang, Cho, Li, Sukhbaatar, Xu, and Weston]{yuan-etal:2024selfrewarding}
Weizhe Yuan, Richard~Yuanzhe Pang, Kyunghyun Cho, Xian Li, Sainbayar Sukhbaatar, Jing Xu, and Jason Weston.
\newblock Self-rewarding language models.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024.
\newblock URL \url{https://openreview.net/forum?id=0NphYCmgua}.
\bibitem[Yuan et~al.(2023)Yuan, Yuan, Li, Dong, Lu, Tan, Zhou, and Zhou]{yuan-etal:2023scaling}
Zheng Yuan, Hongyi Yuan, Chengpeng Li, Guanting Dong, Keming Lu, Chuanqi Tan, Chang Zhou, and Jingren Zhou.
\newblock Scaling relationship on learning mathematical reasoning with large language models.
\newblock \emph{ArXiv preprint}, abs/2308.01825, 2023.
\newblock URL \url{https://arxiv.org/abs/2308.01825}.
\bibitem[Zang et~al.(2025)Zang, Dong, Zhang, Cao, Liu, Ding, Wu, Ma, Duan, Zhang, et~al.]{zang2025internlm}
Yuhang Zang, Xiaoyi Dong, Pan Zhang, Yuhang Cao, Ziyu Liu, Shengyuan Ding, Shenxi Wu, Yubo Ma, Haodong Duan, Wenwei Zhang, et~al.
\newblock Internlm-xcomposer2. 5-reward: A simple yet effective multi-modal reward model.
\newblock \emph{arXiv preprint arXiv:2501.12368}, 2025.
\bibitem[Zeng et~al.(2024{\natexlab{a}})Zeng, Liu, Lu, Wang, Liu, Dong, and Tang]{zeng-etal:agenttuning}
Aohan Zeng, Mingdao Liu, Rui Lu, Bowen Wang, Xiao Liu, Yuxiao Dong, and Jie Tang.
\newblock Agenttuning: Enabling generalized agent abilities for llms.
\newblock In \emph{Findings of the Association for Computational Linguistics: ACL 2024}, pp.\ 3053--3077, 2024{\natexlab{a}}.
\bibitem[Zeng et~al.(2024{\natexlab{b}})Zeng, Liu, Ma, Yang, Zhang, and Wang]{zeng:2024token}
Yongcheng Zeng, Guoqing Liu, Weiyu Ma, Ning Yang, Haifeng Zhang, and Jun Wang.
\newblock Token-level direct preference optimization.
\newblock In \emph{Proceedings of the 41st International Conference on Machine Learning}, 2024{\natexlab{b}}.
\bibitem[Zeng et~al.(2025)Zeng, Cheng, Yin, Zhou, and Qiu]{zeng-etal:2025revisiting}
Zhiyuan Zeng, Qinyuan Cheng, Zhangyue Yin, Yunhua Zhou, and Xipeng Qiu.
\newblock Revisiting the test-time scaling of o1-like models: Do they truly possess test-time scaling capabilities?
\newblock \emph{ArXiv preprint}, abs/2502.12215, 2025.
\newblock URL \url{https://arxiv.org/abs/2502.12215}.
\bibitem[Zhang et~al.(2024{\natexlab{a}})Zhang, Yu, Dong, Li, Su, Chu, and Yu]{zhang-etal:2024mm}
Duzhen Zhang, Yahan Yu, Jiahua Dong, Chenxing Li, Dan Su, Chenhui Chu, and Dong Yu.
\newblock Mm-llms: Recent advances in multimodal large language models.
\newblock \emph{arXiv preprint arXiv:2401.13601}, 2024{\natexlab{a}}.
\bibitem[Zhang et~al.(2024{\natexlab{b}})Zhang, Lan, Murthy, Liu, Yao, Zhu, Tan, Hoang, Liu, Yang, et~al.]{zhang-etal:agentohana}
Jianguo Zhang, Tian Lan, Rithesh Murthy, Zhiwei Liu, Weiran Yao, Ming Zhu, Juntao Tan, Thai Hoang, Zuxin Liu, Liangwei Yang, et~al.
\newblock Agentohana: Design unified data and training pipeline for effective agent learning.
\newblock \emph{arXiv preprint arXiv:2402.15506}, 2024{\natexlab{b}}.
\bibitem[Zhang et~al.(2024{\natexlab{c}})Zhang, Hosseini, Bansal, Kazemi, Kumar, and Agarwal]{zhang-etal:2024generative}
Lunjun Zhang, Arian Hosseini, Hritik Bansal, Mehran Kazemi, Aviral Kumar, and Rishabh Agarwal.
\newblock Generative verifiers: Reward modeling as next-token prediction.
\newblock \emph{ArXiv preprint}, abs/2408.15240, 2024{\natexlab{c}}.
\newblock URL \url{https://arxiv.org/abs/2408.15240}.
\bibitem[Zhang et~al.(2024{\natexlab{d}})Zhang, Hosseini, Bansal, Kazemi, Kumar, and Agarwal]{zhang-etal:zhang2024generative}
Lunjun Zhang, Arian Hosseini, Hritik Bansal, Mehran Kazemi, Aviral Kumar, and Rishabh Agarwal.
\newblock Generative verifiers: Reward modeling as next-token prediction.
\newblock \emph{arXiv preprint arXiv:2408.15240}, 2024{\natexlab{d}}.
\bibitem[Zhao et~al.(2024)Zhao, Lin, Zhu, Ye, Chen, Zheng, Ceze, Krishnamurthy, Chen, and Kasikci]{zhao-etal:2024atom}
Yilong Zhao, Chien-Yu Lin, Kan Zhu, Zihao Ye, Lequn Chen, Size Zheng, Luis Ceze, Arvind Krishnamurthy, Tianqi Chen, and Baris Kasikci.
\newblock Atom: Low-bit quantization for efficient and accurate llm serving.
\newblock \emph{Proceedings of Machine Learning and Systems}, 6:\penalty0 196--209, 2024.
\bibitem[Zheng et~al.(2023)Zheng, Chiang, Sheng, Zhuang, Wu, Zhuang, Lin, Li, Li, Xing, Zhang, Gonzalez, and Stoica]{zheng-etal:2023judging}
Lianmin Zheng, Wei{-}Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi~Lin, Zhuohan Li, Dacheng Li, Eric~P. Xing, Hao Zhang, Joseph~E. Gonzalez, and Ion Stoica.
\newblock Judging llm-as-a-judge with mt-bench and chatbot arena.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock URL \url{http://papers.nips.cc/paper\_files/paper/2023/hash/91f18a1287b398d378ef22505bf41832-Abstract-Datasets\_and\_Benchmarks.html}.
\bibitem[Zhou et~al.(2023)Zhou, Liu, Xu, Iyer, Sun, Mao, Ma, Efrat, Yu, Yu, et~al.]{zhou-etal:lima}
Chunting Zhou, Pengfei Liu, Puxin Xu, Srinivasan Iyer, Jiao Sun, Yuning Mao, Xuezhe Ma, Avia Efrat, Ping Yu, Lili Yu, et~al.
\newblock Lima: Less is more for alignment.
\newblock \emph{Advances in Neural Information Processing Systems}, 36:\penalty0 55006--55021, 2023.
\bibitem[Zhou et~al.(2024)Zhou, Wang, Hu, Xiao, Zhang, and Zhu]{zhou:2024prior}
Hang Zhou, Chenglong Wang, Yimin Hu, Tong Xiao, Chunliang Zhang, and Jingbo Zhu.
\newblock Prior constraints-based reward model training for aligning large language models.
\newblock In \emph{China National Conference on Chinese Computational Linguistics}, pp.\ 555--570. Springer, 2024.
\end{thebibliography}
...@@ -2,6 +2,34 @@ ...@@ -2,6 +2,34 @@
@inproceedings{melnyk:2024distributional,
title={Distributional Preference Alignment of LLMs via Optimal Transport},
author={Melnyk, Igor and Mroueh, Youssef and Belgodere, Brian and Rigotti, Mattia and Nitsure, Apoorva and Yurochkin, Mikhail and Greenewald, Kristjan and Navratil, Jiri and Ross, Jerret},
booktitle={Advances in Neural Information Processing Systems},
year={2024}
}
@inproceedings{zeng:2024token,
title={Token-level Direct Preference Optimization},
author={Zeng, Yongcheng and Liu, Guoqing and Ma, Weiyu and Yang, Ning and Zhang, Haifeng and Wang, Jun},
booktitle={Proceedings of the 41st International Conference on Machine Learning},
year={2024}
}
@inproceedings{xiao:2024cal,
title={Cal-DPO: Calibrated Direct Preference Optimization for Language Model Alignment},
author={Xiao, Teng and Yuan, Yige and Zhu, Huaisheng and Li, Mingxiao and Honavar, Vasant G.},
booktitle={The Thirty-eighth Annual Conference on Neural Information Processing Systems},
year={2024}
}
@article{amini:2024direct,
title={Direct Preference Optimization with an Offset},
author={Amini, Afra and Vieira, Tim and Cotterell, Ryan},
journal={arXiv preprint arXiv:2402.10571},
year={2024}
}
@article{qian-etal:toolrl, @article{qian-etal:toolrl,
title={Toolrl: Reward is all tool learning needs}, title={Toolrl: Reward is all tool learning needs},
author={Qian, Cheng and Acikgoz, Emre Can and He, Qi and Wang, Hongru and Chen, Xiusi and Hakkani-Tur, Dilek and Tur, Gokhan and Ji, Heng}, author={Qian, Cheng and Acikgoz, Emre Can and He, Qi and Wang, Hongru and Chen, Xiusi and Hakkani-Tur, Dilek and Tur, Gokhan and Ji, Heng},
......
No preview for this file type
DPO \citep{rafailov:2023direct} \begin{tabular}{ll}
& $-\log \textrm{Sigmoid} \left( \toprule[1.1pt]
\beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})} \textbf{Method} & \textbf{Objective} \\ \midrule
- IPO \citep{azar:2024general} & $ \left( \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})} - \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_b|\mathbf{x})} - \frac{1}{2\tau} \right)^2$ \\ \midrule
\beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_b|\mathbf{x})} CPO \citep{xu:2024contrastive} & $-\log \textrm{Sigmoid} \left(\beta \log \mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x}) - \beta \log \mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x}) \right) - \lambda \log \mathrm{Pr}_\theta (\mathbf{y}_a|\mathbf{x})$ \\ \midrule
\right)$ \\ \midrule \multirow{2}{*}{KTO \citep{ethayarajh:2024kto}} & $-\lambda_a \textrm{Sigmoid} \left( \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})} - z_{\theta_\mathrm{ref}} \right) + \lambda_b \textrm{Sigmoid} \left( z_{\theta_\mathrm{ref}} - \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_b|\mathbf{x})} \right)\,$ \\
& $\text{where} \,\, z_{\theta_\mathrm{ref}} = \mathbb{E}_{(\mathbf{x}, \mathbf{y}) \sim \mathcal{S}} \left[\beta \text{KL}\left( \mathrm{Pr}_\theta(\mathbf{y}|\mathbf{x}) || \mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}|\mathbf{x}) \right) \right]$ \\ \midrule
\multirow{2}{*}{ORPO \citep{hong:2024orpo}} & $-\log p_\theta(\mathbf{y}_a|\mathbf{x}) - \lambda \log \textrm{Sigmoid} \left(\log \frac{p_\theta(\mathbf{y}_a|\mathbf{x})}{1 - p_\theta(\mathbf{y}_a|\mathbf{x})} - \log \frac{p_\theta(\mathbf{y}_b|\mathbf{x})}{1 - p_\theta(\mathbf{y}_b|\mathbf{x})} \right)\,$ \\
& $\text{where} \,\, p_\theta(\mathbf{y}|\mathbf{x}) = e^{ \frac{1}{|\mathbf{y}|} \log \mathrm{Pr}_\theta(\mathbf{y}|\mathbf{x})}$ \\ \midrule
R-DPO \citep{gallego:2024refined} & $-\log \textrm{Sigmoid} \left( \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})} - \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_b|\mathbf{x})} + \left(\alpha |\mathbf{y}_a| - \alpha |\mathbf{y}_b| \right) \right)$ \\ \midrule
\multirow{2}{*}{PCDPO \citep{zhou:2024prior}} & $-\log \textrm{Sigmoid} \left(\Delta^* - \Delta_{\mathrm{Pr}_{\theta}}\right) -\log \textrm{Sigmoid} \Delta_{\mathrm{Pr}_{\theta}} \,,$ \\
& $\text{where} \,\, \Delta_{\mathrm{Pr}_{\theta}}= \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})} - \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_b|\mathbf{x})},\,\, \Delta^* = \frac{\beta_1}{\textrm{Sim}(\textbf{y}_a,\textbf{y}_b|\textbf{x})+\beta_2}+\beta_3 $ \\ \midrule
SimPO \citep{meng:2025simpo} & $-\log \textrm{Sigmoid} \left( \frac{\beta}{|\mathbf{y}_a|} \log \mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x}) - \frac{\beta}{|\mathbf{y}_b|} \log \mathrm{Pr}_\theta(\mathbf{y}_b|\mathbf{x}) - \gamma \right)$ \\ \midrule
ODPO \citep{amini:2024direct} ODPO \citep{amini:2024direct}
& $-\log \textrm{Sigmoid} \left( & $-\log \textrm{Sigmoid} \left(
\beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})} \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a|\mathbf{x})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_a|\mathbf{x})}
...@@ -48,3 +54,6 @@ ...@@ -48,3 +54,6 @@
\beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}|\mathbf{x})} \beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}|\mathbf{x})}
{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}|\mathbf{x})}, {\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}|\mathbf{x})},
\text{ and } \mathcal{L}_{\mathrm{OT}} \text{ aligns preference distributions via optimal transport.}$ \\ \midrule \text{ and } \mathcal{L}_{\mathrm{OT}} \text{ aligns preference distributions via optimal transport.}$ \\ \midrule
D$^2$PO \citep{shao:2025earlier} & $-\log \textrm{Sigmoid} \left( \sum_{t=0}^{T}\gamma^t\beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_a^t|\mathbf{x},\mathbf{y}_{a,<t})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_{a,t}|\mathbf{x},\mathbf{y}_{a,<t})} - \sum_{t=0}^{T}\gamma^t\beta \log \frac{\mathrm{Pr}_\theta(\mathbf{y}_{b,t}|\mathbf{x},\mathbf{y}_{b,<t})}{\mathrm{Pr}_{\theta_\mathrm{ref}}(\mathbf{y}_{b,t}|\mathbf{x},\mathbf{y}_{b,<t})}\right)$ \\
\bottomrule[1.1pt]
\end{tabular}
...@@ -169,7 +169,11 @@ Tool using is another fundamental capability of LLM-based agents. Tools extend t ...@@ -169,7 +169,11 @@ Tool using is another fundamental capability of LLM-based agents. Tools extend t
\subsection{Improved Environments} \subsection{Improved Environments}
\subsubsection{Scaling Up } % large-scale environment construct
\subsection{Self-Evolving Agents}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论