\makebox[\linewidth][c]{\textbf{Reward Model Input}}\\[2pt]
\textit{Input:} Give me three tips to improve my accuracy in solving math problems.\\[2pt]
\textit{Output A:} Improving math accuracy requires careful attention to detail and consistent practice. Here are three tips: practice regularly, understand the concepts, and \ldots\\[2pt]
\textit{Output B:} Practice more, read carefully, and check your answers. \ldots
};
\node[rubric generator, right=13mm of inputA.north east, anchor=north west] (llm) {
Rubric Generator\\
(an LLM)
};
\node[box, text width=4.3cm, below=8mm of llm] (rubricA) {
\node[box, text width=5.2cm, right=60mm of inputA] (inputB) {
\makebox[\linewidth][c]{\textbf{Reward Model Input}}\\[2pt]
\textit{Input:} Give me three tips to improve my accuracy in solving math problems.\\[2pt]
\textit{Output A:} Improving math accuracy requires careful attention to detail and consistent practice. Here are three tips: practice regularly, understand the concepts, and \ldots\\[2pt]
\textit{Output B:} Practice more, read carefully, and check your answers. \ldots
};
\node[reward model, below=12mm of inputB] (grmB) {
Generative\\
Reward Model
};
\node[box, text width=5.6cm, below=8mm of grmB] (outputB) {
\makebox[\linewidth][c]{\textbf{Reward Model Output}}\\[2pt]