\bibitem[Azar et~al.(2024)Azar, Guo, Piot, Munos, Rowland, Valko, and Calandriello]{azar:2024general}
Mohammad~Gheshlaghi Azar, Zhaohan~Daniel Guo, Bilal Piot, R{\'{e}}mi Munos, Mark Rowland, Michal Valko, and Daniele Calandriello.
\newblock A general theoretical paradigm to understand learning from human preferences.
\newblock In Sanjoy Dasgupta, Stephan Mandt, and Yingzhen Li (eds.), \emph{International Conference on Artificial Intelligence and Statistics, 2-4 May 2024, Palau de Congressos, Valencia, Spain}, volume 238 of \emph{Proceedings of Machine Learning Research}, pp.\ 4447--4455. {PMLR}, 2024.
\bibitem[Bahdanau et~al.(2017)Bahdanau, Brakel, Xu, Goyal, Lowe, Pineau, Courville, and Bengio]{bahdanau-etal:2016actor}
Dzmitry Bahdanau, Philemon Brakel, Kelvin Xu, Anirudh Goyal, Ryan Lowe, Joelle Pineau, Aaron~C. Courville, and Yoshua Bengio.
\newblock An actor-critic algorithm for sequence prediction.
\newblock In \emph{5th International Conference on Learning Representations, {ICLR} 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings}. OpenReview.net, 2017.
Lichang Chen, Shiyang Li, Jun Yan, Hai Wang, Kalpa Gunaratna, Vikas Yadav, Zheng Tang, Vijay Srinivasan, Tianyi Zhou, Heng Huang, and Hongxia Jin.
\newblock Alpagasus: Training a better alpaca with fewer data.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024.
\bibitem[Christiano et~al.(2017)Christiano, Leike, Brown, Martic, Legg, and Amodei]{christiano-etal:2017deep}
Paul~F. Christiano, Jan Leike, Tom~B. Brown, Miljan Martic, Shane Legg, and Dario Amodei.
\newblock Deep reinforcement learning from human preferences.
\newblock In Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna~M. Wallach, Rob Fergus, S.~V.~N. Vishwanathan, and Roman Garnett (eds.), \emph{Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, {USA}}, pp.\ 4299--4307, 2017.
\bibitem[Coste et~al.(2024)Coste, Anwar, Kirk, and Krueger]{coste-etal:2024reward}
Thomas Coste, Usman Anwar, Robert Kirk, and David Krueger.
\newblock Reward model ensembles help mitigate overoptimization.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024.
\newblock {ULTRAFEEDBACK:} boosting language models with scaled {AI} feedback.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024{\natexlab{a}}.
\newblock {ULTRAFEEDBACK:} boosting language models with scaled {AI} feedback.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024{\natexlab{b}}.
Yann Dubois, Chen~Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy Liang, and Tatsunori~B. Hashimoto.
\newblock Alpacafarm: {A} simulation framework for methods that learn from human feedback.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
Jacob Eisenstein, Chirag Nagpal, Alekh Agarwal, Ahmad Beirami, Alex D'Amour, DJ~Dvijotham, Adam Fisch, Katherine Heller, Stephen Pfohl, Deepak Ramachandran, et~al.
\newblock Helping or herding? reward model ensembles mitigate but do not eliminate reward hacking.
\bibitem[Gao et~al.(2023)Gao, Schulman, and Hilton]{gao-etal:2023scaling}
Leo Gao, John Schulman, and Jacob Hilton.
\newblock Scaling laws for reward model overoptimization.
\newblock In Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (eds.), \emph{International Conference on Machine Learning, {ICML} 2023, 23-29 July 2023, Honolulu, Hawaii, {USA}}, volume 202 of \emph{Proceedings of Machine Learning Research}, pp.\ 10835--10866. {PMLR}, 2023.
\bibitem[Havrilla et~al.(2024)Havrilla, Du, Raparthy, Nalmpantis, Dwivedi-Yu, Zhuravinskyi, Hambro, Sukhbaatar, and Raileanu]{havrilla-etal:2024teaching}
Alex Havrilla, Yuqing Du, Sharath~Chandra Raparthy, Christoforos Nalmpantis, Jane Dwivedi-Yu, Maksym Zhuravinskyi, Eric Hambro, Sainbayar Sukhbaatar, and Roberta Raileanu.
\newblock Teaching large language models to reason with reinforcement learning.
\bibitem[Li et~al.(2025{\natexlab{b}})Li, Hu, and Wang]{li-etal:encouraging}
Zhiwei Li, Yong Hu, and Wenqing Wang.
\newblock Encouraging good processes without the need for good answers: Reinforcement learning for llm agent planning.
\newblock In \emph{Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing: Industry Track}, pp.\ 1654--1666, 2025{\natexlab{b}}.
\bibitem[Li et~al.(2024)Li, Xu, Zhang, Lin, Yu, Sun, and Luo]{li-etal:2023remax}
Ziniu Li, Tian Xu, Yushun Zhang, Zhihang Lin, Yang Yu, Ruoyu Sun, and Zhi{-}Quan Luo.
\newblock Remax: {A} simple, effective, and efficient reinforcement learning method for aligning large language models.
\newblock In \emph{Forty-first International Conference on Machine Learning, {ICML} 2024, Vienna, Austria, July 21-27, 2024}. OpenReview.net, 2024.
Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe.
\newblock Let's verify step by step.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{a}}.
Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe.
\newblock Let's verify step by step.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
\newblock {G}-eval: {NLG} evaluation using gpt-4 with better human alignment.
\newblock In Houda Bouamor, Juan Pino, and Kalika Bali (eds.), \emph{Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing}, pp.\ 2511--2522, Singapore, 2023. Association for Computational Linguistics.
Igor Melnyk, Youssef Mroueh, Brian Belgodere, Mattia Rigotti, Apoorva Nitsure, Mikhail Yurochkin, Kristjan Greenewald, Jiri Navratil, and Jerret Ross.
\newblock Distributional preference alignment of llms via optimal transport.
\newblock In \emph{Advances in Neural Information Processing Systems}, 2024.
\bibitem[Meng et~al.(2024)Meng, Xia, and Chen]{meng:2025simpo}
Yu~Meng, Mengzhou Xia, and Danqi Chen.
\newblock Simpo: Simple preference optimization with a reference-free reward.
\newblock In Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub~M. Tomczak, and Cheng Zhang (eds.), \emph{Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024}, 2024.
\bibitem[Mnih et~al.(2016)Mnih, Badia, Mirza, Graves, Lillicrap, Harley, Silver, and Kavukcuoglu]{mnih-etal:2016asynchronous}
Volodymyr Mnih, Adri{\`{a}}~Puigdom{\`{e}}nech Badia, Mehdi Mirza, Alex Graves, Timothy~P. Lillicrap, Tim Harley, David Silver, and Koray Kavukcuoglu.
\newblock Asynchronous methods for deep reinforcement learning.
\newblock In Maria{-}Florina Balcan and Kilian~Q. Weinberger (eds.), \emph{Proceedings of the 33nd International Conference on Machine Learning, {ICML} 2016, New York City, NY, USA, June 19-24, 2016}, volume~48 of \emph{{JMLR} Workshop and Conference Proceedings}, pp.\ 1928--1937. JMLR.org, 2016.
\bibitem[Nakano et~al.(2021)Nakano, Hilton, Balaji, Wu, Ouyang, Kim, Hesse, Jain, Kosaraju, Saunders, Jiang, Cobbe, Eloundou, Krueger, Button, Knight, Chess, and Schulman]{nakano-etal:2021webgpt}
Reiichiro Nakano, Jacob Hilton, Suchir Balaji, Jeff Wu, Long Ouyang, Christina Kim, Christopher Hesse, Shantanu Jain, Vineet Kosaraju, William Saunders, Xu~Jiang, Karl Cobbe, Tyna Eloundou, Gretchen Krueger, Kevin Button, Matthew Knight, Benjamin Chess, and John Schulman.
\newblock Webgpt: Browser-assisted question-answering with human feedback.
Long Ouyang, Jeffrey Wu, Xu~Jiang, Diogo Almeida, Carroll~L. Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda Askell, Peter Welinder, Paul~F. Christiano, Jan Leike, and Ryan Lowe.
\newblock Training language models to follow instructions with human feedback.
\newblock In Sanmi Koyejo, S.~Mohamed, A.~Agarwal, Danielle Belgrave, K.~Cho, and A.~Oh (eds.), \emph{Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022}, 2022.
\newblock Toolrl: Reward is all tool learning needs.
\newblock \emph{Advances in Neural Information Processing Systems}, 38:\penalty0 105523--105553, 2026.
\bibitem[Radford et~al.(2021)Radford, Kim, Hallacy, Ramesh, Goh, Agarwal, Sastry, Askell, Mishkin, Clark, Krueger, and Sutskever]{radford-etal:2021learning}
Alec Radford, Jong~Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.
\newblock Learning transferable visual models from natural language supervision, 2021.
\bibitem[Rafailov et~al.(2023)Rafailov, Sharma, Mitchell, Manning, Ermon, and Finn]{rafailov:2023direct}
Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher~D. Manning, Stefano Ermon, and Chelsea Finn.
\newblock Direct preference optimization: Your language model is secretly a reward model.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
Timo Schick, Jane Dwivedi-Yu, Roberto Dess{\`\i}, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom.
\newblock Toolformer: Language models can teach themselves to use tools.
\newblock \emph{Advances in neural information processing systems}, 36:\penalty0 68539--68551, 2023.
\bibitem[Schulman et~al.(2015)Schulman, Levine, Abbeel, Jordan, and Moritz]{schulman-etal:2015trust}
John Schulman, Sergey Levine, Pieter Abbeel, Michael~I. Jordan, and Philipp Moritz.
\newblock Trust region policy optimization.
\newblock In Francis~R. Bach and David~M. Blei (eds.), \emph{Proceedings of the 32nd International Conference on Machine Learning, {ICML} 2015, Lille, France, 6-11 July 2015}, volume~37 of \emph{{JMLR} Workshop and Conference Proceedings}, pp.\ 1889--1897. JMLR.org, 2015.
\bibitem[Schulman et~al.(2016)Schulman, Moritz, Levine, Jordan, and Abbeel]{schulman-etal:2015high}
John Schulman, Philipp Moritz, Sergey Levine, Michael~I. Jordan, and Pieter Abbeel.
\newblock High-dimensional continuous control using generalized advantage estimation.
\newblock In Yoshua Bengio and Yann LeCun (eds.), \emph{4th International Conference on Learning Representations, {ICLR} 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings}, 2016.
David Silver, Aja Huang, Chris~J Maddison, Arthur Guez, Laurent Sifre, George Van Den~Driessche, Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et~al.
\newblock Mastering the game of go with deep neural networks and tree search.
David Silver, Julian Schrittwieser, Karen Simonyan, Ioannis Antonoglou, Aja Huang, Arthur Guez, Thomas Hubert, Lucas Baker, Matthew Lai, Adrian Bolton, et~al.
\newblock Mastering the game of go without human knowledge.
\bibitem[Sun et~al.(2024)Sun, Liu, Bair, and Kolter]{sun-etal:2023simple}
Mingjie Sun, Zhuang Liu, Anna Bair, and J.~Zico Kolter.
\newblock A simple and effective pruning approach for large language models.
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024.
\bibitem[Wang et~al.(2021)Wang, Yan, Meng, and Zhou]{wang-etal:2021selective}
Fusheng Wang, Jianhao Yan, Fandong Meng, and Jie Zhou.
\newblock Selective knowledge distillation for neural machine translation.
\newblock In Chengqing Zong, Fei Xia, Wenjie Li, and Roberto Navigli (eds.), \emph{Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)}, pp.\ 6456--6466, Online, 2021. Association for Computational Linguistics.
\bibitem[Wang \& Zhou(2024)Wang and Zhou]{wang-and-zhou:2024chain}
Xuezhi Wang and Denny Zhou.
\newblock Chain-of-thought reasoning without prompting.
\newblock In Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub~M. Tomczak, and Cheng Zhang (eds.), \emph{Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024}, 2024.
Yizhong Wang, Hamish Ivison, Pradeep Dasigi, Jack Hessel, Tushar Khot, Khyathi Chandu, David Wadden, Kelsey MacMillan, Noah~A. Smith, Iz~Beltagy, and Hannaneh Hajishirzi.
\newblock How far can camels go? exploring the state of instruction tuning on open resources.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023{\natexlab{d}}.
\bibitem[Wei et~al.(2022)Wei, Bosma, Zhao, Guu, Yu, Lester, Du, Dai, and Le]{wei-etal:2022finetuned}
Jason Wei, Maarten Bosma, Vincent~Y. Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester, Nan Du, Andrew~M. Dai, and Quoc~V. Le.
\newblock Finetuned language models are zero-shot learners.
\newblock In \emph{The Tenth International Conference on Learning Representations, {ICLR} 2022, Virtual Event, April 25-29, 2022}. OpenReview.net, 2022.
Zeqiu Wu, Yushi Hu, Weijia Shi, Nouha Dziri, Alane Suhr, Prithviraj Ammanabrolu, Noah~A. Smith, Mari Ostendorf, and Hannaneh Hajishirzi.
\newblock Fine-grained human feedback gives better rewards for language model training.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock Regularizing hidden states enables learning generalizable reward model for llms.
\newblock In Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub~M. Tomczak, and Cheng Zhang (eds.), \emph{Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024}, 2024.
\bibitem[Yao et~al.(2023)Yao, Yu, Zhao, Shafran, Griffiths, Cao, and Narasimhan]{yao-etal:2023tree}
Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Tom Griffiths, Yuan Cao, and Karthik Narasimhan.
\newblock Tree of thoughts: Deliberate problem solving with large language models.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock Agent lumos: Unified and modular training for open-source language agents.
\newblock In \emph{Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 12380--12403, 2024.
\newblock Judging llm-as-a-judge with mt-bench and chatbot arena.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
title={Distributional Preference Alignment of LLMs via Optimal Transport},
author={Melnyk, Igor and Mroueh, Youssef and Belgodere, Brian and Rigotti, Mattia and Nitsure, Apoorva and Yurochkin, Mikhail and Greenewald, Kristjan and Navratil, Jiri and Ross, Jerret},
booktitle={Advances in Neural Information Processing Systems},
year={2024}
}
@inproceedings{zeng:2024token,
title={Token-level Direct Preference Optimization},
author={Zeng, Yongcheng and Liu, Guoqing and Ma, Weiyu and Yang, Ning and Zhang, Haifeng and Wang, Jun},
booktitle={Proceedings of the 41st International Conference on Machine Learning},
year={2024}
}
@inproceedings{xiao:2024cal,
title={Cal-DPO: Calibrated Direct Preference Optimization for Language Model Alignment},
author={Xiao, Teng and Yuan, Yige and Zhu, Huaisheng and Li, Mingxiao and Honavar, Vasant G.},
booktitle={The Thirty-eighth Annual Conference on Neural Information Processing Systems},
year={2024}
}
@article{amini:2024direct,
title={Direct Preference Optimization with an Offset},
author={Amini, Afra and Vieira, Tim and Cotterell, Ryan},
journal={arXiv preprint arXiv:2402.10571},
year={2024}
}
@article{qian-etal:toolrl,
title={Toolrl: Reward is all tool learning needs},
author={Qian, Cheng and Acikgoz, Emre Can and He, Qi and Wang, Hongru and Chen, Xiusi and Hakkani-Tur, Dilek and Tur, Gokhan and Ji, Heng},