Commit 031428fa by wangchenglong

update.

parent 679b367c
\begin{thebibliography}{120}
\begin{thebibliography}{121}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
......@@ -173,6 +173,11 @@ Jiaming Ji, Xinyu Chen, Rui Pan, Han Zhu, Conghui Zhang, Jiahao Li, Donghai Hong
\newblock Safe rlhf-v: Safe reinforcement learning from human feedback in multimodal large language models.
\newblock \emph{arXiv preprint arXiv:2503.17682}, 2025.
\bibitem[Komeili et~al.(2022)Komeili, Shuster, and Weston]{komeili-etal:Internet}
Mojtaba Komeili, Kurt Shuster, and Jason Weston.
\newblock Internet-augmented dialogue generation.
\newblock In \emph{Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 8460--8478, 2022.
\bibitem[Kumar et~al.(2024)Kumar, Zhuang, Agarwal, Su, Co-Reyes, Singh, Baumli, Iqbal, Bishop, Roelofs, et~al.]{kumar-etal:2024training}
Aviral Kumar, Vincent Zhuang, Rishabh Agarwal, Yi~Su, John~D Co-Reyes, Avi Singh, Kate Baumli, Shariq Iqbal, Colton Bishop, Rebecca Roelofs, et~al.
\newblock Training language models to self-correct via reinforcement learning.
......
......@@ -4,7 +4,13 @@
@inproceedings{komeili-etal:Internet,
title={Internet-augmented dialogue generation},
author={Komeili, Mojtaba and Shuster, Kurt and Weston, Jason},
booktitle={Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)},
pages={8460--8478},
year={2022}
}
@inproceedings{singh-etal:agentic,
title={Agentic Reasoning and Tool Integration for LLMs via Reinforcement Learning},
......
No preview for this file type
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论