title={Distributional Preference Alignment of LLMs via Optimal Transport},
author={Melnyk, Igor and Mroueh, Youssef and Belgodere, Brian and Rigotti, Mattia and Nitsure, Apoorva and Yurochkin, Mikhail and Greenewald, Kristjan and Navratil, Jiri and Ross, Jerret},
booktitle={Advances in Neural Information Processing Systems},
year={2024}
}
@inproceedings{zeng:2024token,
title={Token-level Direct Preference Optimization},
author={Zeng, Yongcheng and Liu, Guoqing and Ma, Weiyu and Yang, Ning and Zhang, Haifeng and Wang, Jun},
booktitle={Proceedings of the 41st International Conference on Machine Learning},
year={2024}
}
@inproceedings{xiao:2024cal,
title={Cal-DPO: Calibrated Direct Preference Optimization for Language Model Alignment},
author={Xiao, Teng and Yuan, Yige and Zhu, Huaisheng and Li, Mingxiao and Honavar, Vasant G.},
booktitle={The Thirty-eighth Annual Conference on Neural Information Processing Systems},
year={2024}
}
@article{amini:2024direct,
title={Direct Preference Optimization with an Offset},
author={Amini, Afra and Vieira, Tim and Cotterell, Ryan},
journal={arXiv preprint arXiv:2402.10571},
year={2024}
}
@article{qian-etal:toolrl,
@article{qian-etal:toolrl,
title={Toolrl: Reward is all tool learning needs},
title={Toolrl: Reward is all tool learning needs},
author={Qian, Cheng and Acikgoz, Emre Can and He, Qi and Wang, Hongru and Chen, Xiusi and Hakkani-Tur, Dilek and Tur, Gokhan and Ji, Heng},
author={Qian, Cheng and Acikgoz, Emre Can and He, Qi and Wang, Hongru and Chen, Xiusi and Hakkani-Tur, Dilek and Tur, Gokhan and Ji, Heng},