\newblock Flow-grpo: Training flow matching models via online rl.
\newblock Flow-grpo: Training flow matching models via online rl.
\newblock \emph{Advances in neural information processing systems}, 38:\penalty0 40783--40818, 2026{\natexlab{a}}.
\newblock \emph{Advances in neural information processing systems}, 38:\penalty0 40783--40818, 2026{\natexlab{a}}.
\bibitem[Liu et~al.(2026{\natexlab{b}})Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
\bibitem[Liu et~al.(2026{\natexlab{b}})Liu, Xu, Yu, Hong, Yang, Zhao, and Wang]{liu-etal:openrubrics}
Tianci Liu, Ran Xu, Tony Yu, Ilgee Hong, Carl Yang, Tuo Zhao, and Haoyu Wang.
\newblock Openrubrics: Towards scalable synthetic rubric generation for reward modeling and llm alignment.
\newblock In \emph{Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics}, pp.\ 17417--17437. Association for Computational Linguistics, 2026{\natexlab{b}}.
\bibitem[Liu et~al.(2026{\natexlab{c}})Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
\newblock Agentic reinforcement learning with implicit step rewards.
\newblock Agentic reinforcement learning with implicit step rewards.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026{\natexlab{b}}.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026{\natexlab{c}}.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
...
@@ -468,6 +478,11 @@ Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher~D. Manning, Stefano E
...
@@ -468,6 +478,11 @@ Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher~D. Manning, Stefano E
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock In Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (eds.), \emph{Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023}, 2023.
\newblock Agent lumos: Unified and modular training for open-source language agents.
\newblock Agent lumos: Unified and modular training for open-source language agents.
\newblock In \emph{Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 12380--12403, 2024.
\newblock In \emph{Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pp.\ 12380--12403, 2024.
\bibitem[Yu et~al.(2026)Yu, Zhu, Lin, Cui, Ding, and Li]{yu-etal:skill}
title={LLM-Rubric: A Multidimensional, Calibrated Approach to Automated Evaluation of Natural Language Texts},
author={Hashemi, Helia and Eisner, Jason and Rosset, Corby and Van Durme, Benjamin and Kedzie, Chris},
booktitle={Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics},
pages={13806--13834},
year={2024},
publisher={Association for Computational Linguistics}
}
@inproceedings{viswanathan-etal:checklists,
title={Checklists Are Better Than Reward Models For Aligning Language Models},
author={Viswanathan, Vijay and Sun, Yanchao and Kong, Xiang and Cao, Meng and Neubig, Graham and Wu, Sherry},
booktitle={Advances in Neural Information Processing Systems},
year={2025}
}
@inproceedings{gunjal-etal:rubrics-rewards,
title={Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains},
author={Gunjal, Anisha and Wang, Anthony and Lau, Elaine and Nath, Vaskar and He, Yunzhong and Liu, Bing and Hendryx, Sean},
booktitle={International Conference on Learning Representations},
year={2026}
}
@inproceedings{liu-etal:openrubrics,
title={OpenRubrics: Towards Scalable Synthetic Rubric Generation for Reward Modeling and LLM Alignment},
author={Liu, Tianci and Xu, Ran and Yu, Tony and Hong, Ilgee and Yang, Carl and Zhao, Tuo and Wang, Haoyu},
booktitle={Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics},
pages={17417--17437},
year={2026},
publisher={Association for Computational Linguistics}
}
@article{shen-etal:rethinking-rubric,
title={Rethinking Rubric Generation for Improving LLM Judge and Reward Modeling for Open-ended Tasks},
author={Shen, William F. and Qiu, Xinchi and Whitehouse, Chenxi and Alazraki, Lisa and Goel, Shashwat and Barbieri, Francesco and Willi, Timon and Mathur, Akhil and Leontiadis, Ilias},
journal={arXiv preprint arXiv:2602.05125},
year={2026}
}
@article{rezaei-etal:online-rubrics,
title={Online Rubrics Elicitation from Pairwise Comparisons},
author={Rezaei, MohammadHossein and Vacareanu, Robert and Wang, Zihao and Wang, Clinton and He, Yunzhong and Aky{\"u}rek, Afra Feyza},
journal={arXiv preprint arXiv:2510.07284},
year={2025}
}
@article{xu-etal:rubric-arm,
title={Alternating Reinforcement Learning for Rubric-Based Reward Modeling in Non-Verifiable LLM Post-Training},
author={Xu, Ran and Liu, Tianci and Dong, Zihan and Yu, Tony and Hong, Ilgee and Yang, Carl and Zhang, Linjun and Zhao, Tao and Wang, Haoyu},
journal={arXiv preprint arXiv:2602.01511},
year={2026}
}
@article{ding-etal:evorubrics,
title={EvoRubrics: Dynamic Rubrics as Rewards via Adversarial Co-Evolution for LLM Reinforcement Learning},
author={Ding, Hongxin and Huang, Baixiang and Fang, Yue and Liao, Weibin and Li, Zheng and Zhang, Jinyang and Wu, Zhijing and Zhao, Junfeng and Wang, Yasha},
journal={arXiv preprint arXiv:2606.23038},
year={2026}
}
@article{yu-etal:audio-rubrics,
title={Reinforcement Learning with Evolving Rubrics as Rewards for Audio Reasoning},
author={Yu, Fangxu and Feng, Tao and Min, Dehai and Lin, Zinan and Xu, Weijia and Xu, Michael and Yu, Philip S. and Liu, Ge and Zhou, Tianyi},
journal={arXiv preprint arXiv:2608.02831},
year={2026}
}
@article{silver-etal:reward,
@article{silver-etal:reward,
title={Reward is enough},
title={Reward is enough},
author={Silver, David and Singh, Satinder and Precup, Doina and Sutton, Richard S},
author={Silver, David and Singh, Satinder and Precup, Doina and Sutton, Richard S},
@@ -247,8 +247,8 @@ While generative reward models have shown strong performance in preference predi
...
@@ -247,8 +247,8 @@ While generative reward models have shown strong performance in preference predi
\vspace{0.5em}
\vspace{0.5em}
\item\textbf{Analytic Rubric.}
\item\textbf{Criterion-wise Scoring Rubric.}
An analytic rubric evaluates each criterion separately. The criterion-level scores can then be averaged or weighted to obtain the final reward.
A criterion-wise scoring rubric asks the reward model to assign a separate score to each evaluation criterion. These criterion-level scores are then averaged or weighted to obtain the final reward.
\vspace{0.5em}
\vspace{0.5em}
\begin{tcolorbox}[frame empty]
\begin{tcolorbox}[frame empty]
...
@@ -357,14 +357,29 @@ While generative reward models have shown strong performance in preference predi
...
@@ -357,14 +357,29 @@ While generative reward models have shown strong performance in preference predi
Two commonly used approaches for automatic rubric generation. We use checklist rubrics as an illustrative example.
}
\label{fig:rubric-based-reward-modeling}
\end{figure*}
It is worth noting that reasoning can also be incorporated into rubric-based evaluation. For example, in a checklist-style rubric, the reward model can be asked to reason about each criterion before producing the corresponding Pass/Fail judgment. Additionally, although we use pairwise evaluation, i.e., taking $(\mathbf{x}, \mathbf{y}_a, \mathbf{y}_b)$ as reward model input, as the running example, rubric-based reward modeling is equally applicable to pointwise evaluation, i.e., taking $(\mathbf{x}, \mathbf{y})$ as reward model input.
It is worth noting that reasoning can also be incorporated into rubric-based evaluation. For example, in a checklist-style rubric, the reward model can be asked to reason about each criterion before producing the corresponding Pass/Fail judgment. Additionally, although we use pairwise evaluation, i.e., taking $(\mathbf{x}, \mathbf{y}_a, \mathbf{y}_b)$ as reward model input, as the running example, rubric-based reward modeling is equally applicable to pointwise evaluation, i.e., taking $(\mathbf{x}, \mathbf{y})$ as reward model input.
The quality of the rubric is crucial to the effectiveness of rubric-based reward modeling. As a result, recent work has devoted increasing attention to constructing high-quality rubrics for reward modeling. Similar to many other components in machine learning, rubric acquisition generally follows two approaches: manual design and automatic generation.
The quality of the rubric is crucial to the performance of rubric-based reward models. As a result, recent work has devoted increasing attention to constructing high-quality rubrics for reward modeling. Similar to many other components in machine learning, rubric acquisition generally follows two approaches: manual design and automatic generation.
For manual design, one straightforward approach is to recruit human experts to write rubrics based on task requirements and their domain knowledge. However, this way is often difficult to design a rubric that comprehensively covers the diverse cases that may arise across different inputs and tasks. This is because that evaluation criteria that are appropriate for one case may be insufficient for another. Additionally, manually designed rubrics inherit the prompt sensitivity of LLM-based evaluation: even when the underlying evaluation dimensions remain the same, small differences in wording can lead to noticeably different results.
For manual design, one straightforward approach is to recruit human experts to write rubrics based on their domain knowledge and task requirements. However, this way is often difficult to design a rubric that comprehensively covers the diverse cases that may arise across different inputs and tasks. This is because that evaluation criteria that are appropriate for one case may be insufficient for another. Additionally, manually designed rubrics inherit the prompt sensitivity of LLM-based evaluation: even when the underlying evaluation dimensions remain the same, small differences in wording can lead to noticeably different results.
One promising strategy is to generate rubrics dynamically based on the input. As shown in Figure~\ref{fig:rubric-based-reward-modeling}, existing approaches generally fall into two categories. The first uses an external LLM to generate task-specific rubrics from the input and evaluation requirements, which are then provided to the reward model for preference prediction. The second allows the generative reward model to generate its own rubrics for the current input and then evaluate candidate responses according to these self-generated criteria.
Both approaches formulate rubric generation as an explicit task. This naturally leads to another idea: can we optimize the rubric generation process itself to produce better rubrics? The answer is yes. Rubrics generated directly by LLMs may suffer from limited coverage, redundant criteria, or preference misalignment \citep{liu-etal:openrubrics,shen-etal:rethinking-rubric}. To improve rubric quality, we can explicitly refine the generated criteria. For example, OpenRubrics generates rubrics by contrasting preferred and rejected responses to identify more discriminative rules and principles \citep{liu-etal:openrubrics}. We can further learn the rubric generation process itself through RL training \citep{xu-etal:rubric-arm}. More recent studies go one step further by allowing rubrics to evolve with the policy, so that the evaluation criteria continue to capture new weaknesses as model improves \citep{rezaei-etal:online-rubrics,ding-etal:evorubrics,yu-etal:audio-rubrics}. Rubric optimization is becoming an active research direction, and interested readers can refer to the aforementioned works for further details.