Commit 759c7625 by wangchenglong

update.

parent 6de2c269
\begin{thebibliography}{173} \begin{thebibliography}{175}
\providecommand{\natexlab}[1]{#1} \providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}} \providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax \expandafter\ifx\csname urlstyle\endcsname\relax
...@@ -319,10 +319,15 @@ Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Ted ...@@ -319,10 +319,15 @@ Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Ted
\newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}. \newblock In \emph{The Twelfth International Conference on Learning Representations, {ICLR} 2024, Vienna, Austria, May 7-11, 2024}. OpenReview.net, 2024{\natexlab{b}}.
\newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}. \newblock URL \url{https://openreview.net/forum?id=v8L0pN6EOi}.
\bibitem[Liu et~al.(2026)Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic} \bibitem[Liu et~al.(2026{\natexlab{a}})Liu, Liu, Liang, Li, Liu, Wang, Wan, Zhang, and Ouyang]{liu-etal:flow}
Jie Liu, Gongye Liu, Jiajun Liang, Yangguang Li, Jiaheng Liu, Xintao Wang, Pengfei Wan, Di~Zhang, and Wanli Ouyang.
\newblock Flow-grpo: Training flow matching models via online rl.
\newblock \emph{Advances in neural information processing systems}, 38:\penalty0 40783--40818, 2026{\natexlab{a}}.
\bibitem[Liu et~al.(2026{\natexlab{b}})Liu, Wang, Wu, Huang, Li, Jiao, and Zhang]{liu-etal:agentic}
Xiaoqian Liu, Ke~Wang, Yuchuan Wu, Fei Huang, Yongbin Li, Jianbin Jiao, and Junge Zhang. Xiaoqian Liu, Ke~Wang, Yuchuan Wu, Fei Huang, Yongbin Li, Jianbin Jiao, and Junge Zhang.
\newblock Agentic reinforcement learning with implicit step rewards. \newblock Agentic reinforcement learning with implicit step rewards.
\newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026. \newblock In \emph{International Conference on Learning Representations}, volume 2026, pp.\ 129271--129291, 2026{\natexlab{b}}.
\bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval} \bibitem[Liu et~al.(2023)Liu, Iter, Xu, Wang, Xu, and Zhu]{liu-etal:2023GEval}
Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu. Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu.
...@@ -772,6 +777,11 @@ Tong Xiao and Jingbo Zhu. ...@@ -772,6 +777,11 @@ Tong Xiao and Jingbo Zhu.
\newblock \emph{ArXiv preprint}, abs/2501.09223, 2025. \newblock \emph{ArXiv preprint}, abs/2501.09223, 2025.
\newblock URL \url{https://arxiv.org/abs/2501.09223}. \newblock URL \url{https://arxiv.org/abs/2501.09223}.
\bibitem[Xiao et~al.(2026)Xiao, Ruan, Li, Yu, Zhang, and Zhu]{xiao-etal:ordinary}
Tong Xiao, Junhao Ruan, Bei Li, Zhengtao Yu, Min Zhang, and Jingbo Zhu.
\newblock Ordinary differential equations in vision and language.
\newblock 2026.
\bibitem[Xie et~al.(2025)Xie, Gao, Ren, Luo, Hong, Dai, Zhou, Qiu, Wu, and Luo]{xie-etal:2025logic} \bibitem[Xie et~al.(2025)Xie, Gao, Ren, Luo, Hong, Dai, Zhou, Qiu, Wu, and Luo]{xie-etal:2025logic}
Tian Xie, Zitian Gao, Qingnan Ren, Haoming Luo, Yuqian Hong, Bryan Dai, Joey Zhou, Kai Qiu, Zhirong Wu, and Chong Luo. Tian Xie, Zitian Gao, Qingnan Ren, Haoming Luo, Yuqian Hong, Bryan Dai, Joey Zhou, Kai Qiu, Zhirong Wu, and Chong Luo.
\newblock Logic-rl: Unleashing llm reasoning with rule-based reinforcement learning. \newblock Logic-rl: Unleashing llm reasoning with rule-based reinforcement learning.
......
...@@ -4,6 +4,23 @@ ...@@ -4,6 +4,23 @@
@article{liu-etal:flow,
title={Flow-grpo: Training flow matching models via online rl},
author={Liu, Jie and Liu, Gongye and Liang, Jiajun and Li, Yangguang and Liu, Jiaheng and Wang, Xintao and Wan, Pengfei and Zhang, Di and Ouyang, Wanli},
journal={Advances in neural information processing systems},
volume={38},
pages={40783--40818},
year={2026}
}
@article{xiao-etal:ordinary,
title={Ordinary Differential Equations in Vision and Language},
author={Xiao, Tong and Ruan, Junhao and Li, Bei and Yu, Zhengtao and Zhang, Min and Zhu, Jingbo},
year={2026},
publisher={TechRxiv}
}
@article{zhou-etal:learning, @article{zhou-etal:learning,
title={Learning implicit credit assignment for cooperative multi-agent reinforcement learning}, title={Learning implicit credit assignment for cooperative multi-agent reinforcement learning},
author={Zhou, Meng and Liu, Ziyu and Sui, Pengwei and Li, Yixuan and Chung, Yuk Ying}, author={Zhou, Meng and Liu, Ziyu and Sui, Pengwei and Li, Yixuan and Chung, Yuk Ying},
......
...@@ -45,35 +45,28 @@ While our discussion primarily focuses on models for visual inputs in the contex ...@@ -45,35 +45,28 @@ While our discussion primarily focuses on models for visual inputs in the contex
Unlike multimodal understanding models that typically produce textual responses, multimodal generation models aim to synthesize new content, such as images and videos. Recent advances in generative models, including diffusion models and flow matching models, have enabled high-quality content generation by learning complex data distributions. However, researchers have found that optimizing these models remains challenging because conventional maximum likelihood or reconstruction objectives cannot fully capture high-level human preferences, such as satisfying specific requirements (e.g., object quantities, colors, and visual layouts). To address this challenge, RL has been introduced to directly optimize multimodal generation models based on flexible reward signals \citep{lee-etal:aligning,fan-etal:dpok}. In this way, instead of relying solely on token-level or pixel-level reconstruction objectives, RL allows models to optimize task-specific criteria provided by humans or automatic evaluators. This property makes RL particularly suitable for multimodal generation, where generation quality is often difficult to describe through explicit supervision. Unlike multimodal understanding models that typically produce textual responses, multimodal generation models aim to synthesize new content, such as images and videos. Recent advances in generative models, including diffusion models and flow matching models, have enabled high-quality content generation by learning complex data distributions. However, researchers have found that optimizing these models remains challenging because conventional maximum likelihood or reconstruction objectives cannot fully capture high-level human preferences, such as satisfying specific requirements (e.g., object quantities, colors, and visual layouts). To address this challenge, RL has been introduced to directly optimize multimodal generation models based on flexible reward signals \citep{lee-etal:aligning,fan-etal:dpok}. In this way, instead of relying solely on token-level or pixel-level reconstruction objectives, RL allows models to optimize task-specific criteria provided by humans or automatic evaluators. This property makes RL particularly suitable for multimodal generation, where generation quality is often difficult to describe through explicit supervision.
In this subsection, we describe the application of RL in two representative classes of multimodal generation models: diffusion-based models and flow matching-based models. In this subsection, we describe the application of RL in two representative classes of multimodal generation models: diffusion models and flow matching models.
\subsubsection{Diffusion-based Models} \subsubsection{Diffusion Models}
We first briefly introduce the basic generation process of diffusion-based models to provide the necessary notations. There are many aspects of diffusion models, such as image representations and mathematical formulations, that cannot be fully covered in this paper. Interested readers can refer to existing surveys on diffusion models for further details \citep{yang-etal:diffusion,croitoru-etal:diffusion}. We first briefly introduce the basic generation process of diffusion-based models to provide the necessary notations. There are many aspects of diffusion models, such as image representations and mathematical formulations, that cannot be fully covered in this paper. Interested readers can refer to existing surveys on diffusion models for further details \citep{yang-etal:diffusion,croitoru-etal:diffusion}.
Specifically, diffusion models generate samples through a gradual denoising process \citep{sohl-etal:deep}. Given a clean sample $\mathbf{x}_0$, the forward diffusion process progressively adds Gaussian noise to the sample. At each timestep $t$, the transition distribution is defined as: Specifically, diffusion models generate samples through a gradual denoising process \citep{sohl-etal:deep}. Given a clean sample $\mathbf{x}_0$, the forward diffusion process progressively adds Gaussian noise to the sample. At each timestep $t$, the transition distribution is defined as:
\begin{eqnarray} \begin{eqnarray}
q(\mathbf{x}_t|\mathbf{x}_{t-1}) Q(\mathbf{x}_t|\mathbf{x}_{t-1}) = \mathcal{N} (\mathbf{x}_t;\sqrt{1-\beta_t}\mathbf{x}_{t-1},\beta_t\mathbf{I})
=
\mathcal{N}
(\mathbf{x}_t;
\sqrt{1-\beta_t}\mathbf{x}_{t-1},
\beta_t\mathbf{I})
\end{eqnarray} \end{eqnarray}
where $q(\cdot)$ denotes the predefined forward diffusion process, $\beta_t$ controls the noise magnitude at timestep $t$, and $\mathcal{N}(\mu,\sigma^2)$ denotes a Gaussian distribution with mean $\mu$ and variance $\sigma^2$. Through this forward process, the original data distribution is gradually transformed into a simple Gaussian noise distribution.
where $Q(\cdot)$ denotes the predefined forward diffusion process, $\beta_t$ controls the noise magnitude at timestep $t$, and $\mathcal{N}(\mu,\sigma^2)$ denotes a Gaussian distribution with mean $\mu$ and variance $\sigma^2$. Through this forward process, the original data distribution is gradually transformed into a simple Gaussian noise distribution.
The generation process performs the reverse denoising procedure, where a neural network learns to recover clean samples from noisy inputs: The generation process performs the reverse denoising procedure, where a neural network learns to recover clean samples from noisy inputs:
\begin{eqnarray} \begin{eqnarray}
p_\theta(\mathbf{x}_{t-1}|\mathbf{x}_{t}) U_\theta(\mathbf{x}_{t-1}|\mathbf{x}_{t}) = \mathcal{N} (\mathbf{x}_{t-1}; \mu_\theta(\mathbf{x}_t,t),
=
\mathcal{N}
(\mathbf{x}_{t-1};
\mu_\theta(\mathbf{x}_t,t),
\Sigma_\theta(\mathbf{x}_t,t)) \Sigma_\theta(\mathbf{x}_t,t))
\end{eqnarray} \end{eqnarray}
where $p_\theta(\cdot)$ denotes the learned reverse denoising process used for generation, and $\mu_\theta(\cdot)$ and $\Sigma_\theta(\cdot)$ denote the learned mean and variance of the reverse transition. By iteratively applying the denoising process from timestep $T$ to $0$, diffusion models can generate high-quality samples from random noise.
During training, diffusion models mainly focus on recovering the original data distribution, which can be regarded as a supervised learning objective. Similar to SFT in LLMs, such training paradigms optimize models toward the provided training data but do not explicitly consider human preferences. As a result, diffusion models may generate high-quality samples while still failing to satisfy fine-grained user requirements. For example, as described in \citep{lee-etal:aligning}, although text-to-image diffusion models can generate visually realistic images, they may struggle with specific attributes, such as generating a desired number of objects. Therefore, how to incorporate additional learning objectives to better align diffusion models with human preferences has become an important research topic. where $U_\theta(\cdot)$ denotes the learned reverse denoising process used for generation, and $\mu_\theta(\cdot)$ and $\Sigma_\theta(\cdot)$ denote the learned mean and variance of the reverse transition. By iteratively applying the denoising process from timestep $T$ to $0$, diffusion models can generate high-quality samples from random noise.
During training, diffusion models mainly focus on recovering the original data distribution, which can be regarded as a supervised learning objective. Similar to SFT in LLMs, such training paradigms optimize models toward the provided training data but do not explicitly consider human preferences. As a result, diffusion models may generate high-quality samples while still failing to satisfy fine-grained user requirements. For example, as described in \citet{lee-etal:aligning}, although text-to-image diffusion models can generate visually realistic images, they may struggle with specific attributes, such as generating a desired number of objects. Therefore, how to incorporate additional learning objectives to better align diffusion models with human preferences has become an important research topic.
To address this challenge, researchers have explored RL as an effective approach for aligning diffusion models with human preferences. A straightforward approach is to treat generated samples as optimization targets and use external reward signals to guide the generation process. Specifically, given a text prompt $\mathbf{z}$, a reward function $R_\mathrm{dm}(\mathbf{x}_0,\mathbf{z})$ evaluates the generated image $\mathbf{x}_0$, and the diffusion model is optimized to maximize the expected reward: To address this challenge, researchers have explored RL as an effective approach for aligning diffusion models with human preferences. A straightforward approach is to treat generated samples as optimization targets and use external reward signals to guide the generation process. Specifically, given a text prompt $\mathbf{z}$, a reward function $R_\mathrm{dm}(\mathbf{x}_0,\mathbf{z})$ evaluates the generated image $\mathbf{x}_0$, and the diffusion model is optimized to maximize the expected reward:
\begin{eqnarray} \begin{eqnarray}
...@@ -82,6 +75,7 @@ To address this challenge, researchers have explored RL as an effective approach ...@@ -82,6 +75,7 @@ To address this challenge, researchers have explored RL as an effective approach
[R_\mathrm{dm}(\mathbf{x}_0,\mathbf{z})] [R_\mathrm{dm}(\mathbf{x}_0,\mathbf{z})]
\label{eq:optimization_objective} \label{eq:optimization_objective}
\end{eqnarray} \end{eqnarray}
Here, the reward function can be instantiated by different types of evaluators depending on the optimization objective. For example, it can measure semantic alignment between the generated image and the text prompt using a vision-language model or measure human preferences through a learned reward model. For example, given a prompt requiring ``three red apples on a table'', a reward function can assess whether the generated image contains the correct object number and color: Here, the reward function can be instantiated by different types of evaluators depending on the optimization objective. For example, it can measure semantic alignment between the generated image and the text prompt using a vision-language model or measure human preferences through a learned reward model. For example, given a prompt requiring ``three red apples on a table'', a reward function can assess whether the generated image contains the correct object number and color:
However, directly optimizing this objective with the RL formulation introduced in Section~\ref{sec:policy-gradient} is not straightforward. This is because that diffusion models generate samples through an iterative denoising process rather than an autoregressive generation process. Therefore, we need to redefine the RL formulation according to the characteristics of diffusion generation. Taking DPOK \citep{fan-etal:dpok} as an example, recent studies observe that the reverse diffusion process naturally forms a multi-step trajectory, where each denoising step can be viewed as an action conditioned on the current noisy state. Based on this observation, the diffusion generation process can be formulated as a MDP. Specifically, we can consider the denoising procedure as a multi-step MDP and applies a policy gradient-based RL algorithm to optimize the reward obtained from generated images. The state corresponds to the current noisy latent representation, while the action represents the next denoising step: However, directly optimizing this objective with the RL formulation introduced in Section~\ref{sec:policy-gradient} is not straightforward. This is because that diffusion models generate samples through an iterative denoising process rather than an autoregressive generation process. Therefore, we need to redefine the RL formulation according to the characteristics of diffusion generation. Taking DPOK \citep{fan-etal:dpok} as an example, recent studies observe that the reverse diffusion process naturally forms a multi-step trajectory, where each denoising step can be viewed as an action conditioned on the current noisy state. Based on this observation, the diffusion generation process can be formulated as a MDP. Specifically, we can consider the denoising procedure as a multi-step MDP and applies a policy gradient-based RL algorithm to optimize the reward obtained from generated images. The state corresponds to the current noisy latent representation, while the action represents the next denoising step:
...@@ -93,21 +87,113 @@ a_t=\mathbf{x}_{T-t-1} ...@@ -93,21 +87,113 @@ a_t=\mathbf{x}_{T-t-1}
The policy is defined as the reverse diffusion transition: The policy is defined as the reverse diffusion transition:
\begin{eqnarray} \begin{eqnarray}
\mathrm{Pr}_\theta(a_t|s_t) \mathrm{Pr}_\theta(a_t|s_t) = U_\theta(\mathbf{x}_{T-t-1}|\mathbf{x}_{T-t},\mathbf{z})
=
p_\theta(\mathbf{x}_{T-t-1}|\mathbf{x}_{T-t},\mathbf{z})
\end{eqnarray} \end{eqnarray}
where the final generated image receives the reward signal from the reward model. Based on this formulation, the optimization objective in Eq.~(\ref{eq:optimization_objective}) can be optimized using a simple policy gradient loss function: where the final generated image receives the reward signal from the reward model. Based on this formulation, the optimization objective in Eq.~(\ref{eq:optimization_objective}) can be optimized using a simple policy gradient loss function:
\begin{eqnarray} \begin{eqnarray}
\mathcal{L}_{\mathrm{dmrl}}(\theta) \mathcal{L}_{\mathrm{dmrl}}(\theta) =
= -\mathbb{E}_{\tau\sim U_\theta(\cdot)}
-\mathbb{E}_{\tau\sim p_\theta(\cdot)}
\left[ \left[
\sum_{t=0}^{T-1} \sum_{t=0}^{T-1}
\log p_\theta(\mathbf{x}_{T-t-1}|\mathbf{x}_{T-t},\mathbf{z}) R(\mathbf{x}_0,\mathbf{z}) \log U_\theta(\mathbf{x}_{T-t-1}|\mathbf{x}_{T-t},\mathbf{z}) R_{\mathrm{dm}}(\mathbf{x}_0,\mathbf{z})
\right] \right]
\end{eqnarray} \end{eqnarray}
where $\tau$ denotes the denoising trajectory $\{\mathbf{x}_{T},\mathbf{x}_{T-1},\cdots,\mathbf{x}_{0}\}$. Here, many RL techniques developed for LLM alignment can be naturally extended to diffusion model optimization, such as incorporating KL regularization \citep{fan-etal:dpok}, introducing a reference diffusion model for importance sampling \citep{black-etal:training}, and applying GRPO-based optimization \citep{xue-etal:dancegrpo}. where $\tau$ denotes the denoising trajectory $\{\mathbf{x}_{T},\mathbf{x}_{T-1},\cdots,\mathbf{x}_{0}\}$. Here, many RL techniques developed for LLM alignment can be naturally extended to diffusion model optimization, such as incorporating KL regularization \citep{fan-etal:dpok}, introducing a reference diffusion model for importance sampling \citep{black-etal:training}, and applying GRPO-based optimization \citep{xue-etal:dancegrpo}.
\subsubsection{Flow Matching-based Models} \subsubsection{Flow Matching Models}
Different from diffusion models, flow matching models achieve the multimodal generation process based on ordinary differential equations (ODEs), where data transformation is modeled as a continuous-time flow. Instead of gradually adding and removing noise through a stochastic diffusion process, flow matching learns a continuous vector field that transports samples from a simple prior distribution to the target data distribution. In this subsection, we briefly introduce the training and generation processes of flow matching models, and then discuss how RL can be applied to optimize them. Note that we do not provide a detailed introduction to the underlying principles of flow matching models in this subsection. Interested readers can refer to existing tutorials for further details \citep{xiao-etal:ordinary}.
Let $\mathbf{x}_0 \sim X_0$ denote a data sample from the target distribution and $\mathbf{x}_1 \sim X_1$ denote a noise sample from the prior distribution. Flow matching constructs an intermediate state by interpolating between the data and noise distributions:
\begin{eqnarray}
\mathbf{x}_t=(1-t)\mathbf{x}_0+t\mathbf{x}_1
\end{eqnarray}
where $t\in[0,1]$ denotes the continuous time variable. Based on this interpolation process, flow matching trains a neural network to predict the velocity field that describes how samples move along the trajectory:
\begin{eqnarray}
\mathcal{L}_{\mathrm{fm}}(\theta)
= \mathbb{E}_{t,\mathbf{x}_0 \sim X_0 ,\mathbf{x}_1\sim X_1}
\left[
\left\|
\mathbf{v}_\theta(\mathbf{x}_t,t)-\mathbf{v}
\right\|^2
\right]
\end{eqnarray}
where $\mathbf{v}=\mathbf{x}_1-\mathbf{x}_0$ denotes the target velocity field and $\mathbf{v}_\theta(\mathbf{x}_t,t)$ denotes the learned velocity function. After training, the generation process starts from a noise sample $\mathbf{x}_1$ and follows the learned continuous flow by solving the ordinary differential equation:
\begin{eqnarray}
\frac{d\mathbf{x}_t}{dt} = \mathbf{v}_\theta(\mathbf{x}_t,t)
\end{eqnarray}
Similar to diffusion models, the continuous generation process of flow matching models can also be formulated as an MDP to enable RL optimization. Specifically, by discretizing the continuous flow trajectory, we can consider the generation process as a sequence of decision steps, where the model determines how to transform the current state toward the target data distribution \citep{liu-etal:flow}. Under this formulation, the state, action, transition, initial state distribution, and reward function are defined as follows. The state at timestep $t$ is defined as:
\begin{eqnarray}
s_t=(\mathbf{z},t,\mathbf{x}_t)
\end{eqnarray}
The action corresponds to the next state predicted by the flow model:
\begin{eqnarray}
a_t=\mathbf{x}_{t-\Delta t}
\end{eqnarray}
Since the flow model deterministically predicts the velocity field, the policy can be represented as:
\begin{eqnarray}
\mathrm{Pr}_{\theta}(a_t|s_t) = \delta(a_t-\mathbf{v}_{\theta}(\mathbf{x}_t,t,\mathbf{z}))
\end{eqnarray}
where $\delta(\cdot)$ denotes the Dirac delta function. Given the predicted velocity, the next state is obtained by discretizing the ODE:
\begin{eqnarray}
\mathbf{x}_{t-\Delta t} = \mathbf{x}_t-\Delta t a_t
\end{eqnarray}
The initial state is sampled from the noise distribution:
\begin{eqnarray}
\rho_0(s_1) = p(\mathbf{z})\delta(t-1)\mathcal{N}(0,I)
\end{eqnarray}
where $\mathcal{N}(0,I)$ denotes the standard Gaussian distribution and $I$ represents the identity covariance matrix.
% Similar to diffusion models, the reward is usually provided at the end of the generation process:
% \begin{eqnarray}
% R(s_t,a_t)=
% \begin{cases}
% r(\mathbf{x}_0,\mathbf{z}),&t=0,\\
% 0,&\text{otherwise}.
% \end{cases}
% \end{eqnarray}
Based on this MDP formulation, we can optimize flow matching models with policy gradient-based RL algorithms. Taking GRPO as an example, we sample multiple generation trajectories from the current flow model and compute their relative rewards to construct the advantage signals for policy optimization:
\begin{eqnarray}
\mathcal{L}_{\mathrm{GRPO}}
=
-\mathbb{E}
\left[
\frac{1}{G}
\sum_{i=1}^{G}
\min
\left(
r_i(\theta)A_i,
\mathrm{clip}(r_i(\theta),1-\epsilon,1+\epsilon)A_i
\right)
\right],
\end{eqnarray}
where $A_i$ denotes the normalized advantage computed from the final rewards and $r_i(\theta)$ represents the policy ratio between the updated and reference flow models.
However, directly applying policy optimization to flow matching models introduces additional challenges. Unlike autoregressive models, flow matching models do not explicitly define the likelihood of each generation step, making the policy ratio difficult to compute. To address this issue, recent studies propose to construct surrogate objectives based on the flow matching loss. For example, Flow Policy Optimization (FPO) replaces the exact likelihood ratio with a flow-matching-based approximation:
\begin{eqnarray}
\hat{r}_{\mathrm{FPO}}(\theta)
=
\exp
(
\mathcal{L}_{\mathrm{CFM},\theta_{\mathrm{old}}}
-
\mathcal{L}_{\mathrm{CFM},\theta}
),
\end{eqnarray}
where $\mathcal{L}_{\mathrm{CFM}}$ denotes the conditional flow matching loss. This formulation avoids expensive likelihood estimation while maintaining compatibility with PPO-style optimization. Moreover, compared with methods that treat every integration step as an independent decision, FPO regards the sampling process as a whole and supports different numerical solvers and sampling strategies \citep{mcallister-etal:fpo}.
% Similar to diffusion-based RL, the reward is typically assigned at the terminal step based on the quality of the final generated sample:
\ No newline at end of file
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论