@online{huangndrl,
  author       = {Huang, Xuanqiang Angelo},
  title        = {{RL} Losses},
  organization = {Xuanqiang Angelo Huang's Blog},
  url          = {https://flecart.github.io/notes/rl-losses/},
  langid       = {english},
  abstract     = {SDPO \# See (Hübotter et al. 2026) GRPO \# https://hlfshell.ai/posts/grpo/ GRPO (Group Relative Policy Optimization) comes from the DeepSeekMath paper. Its whole reason for existing is to get rid of the value/critic network that PPO needs. Instead of learning a separate model to es}
}
