← Back
@inproceedings{xu2026beyond,
  title={Beyond GRPO and On-Policy Distillation: An Empirical Sparse-to-Dense Reward Principle for Language-Model Post-Training},
  author={Xu, Yuanda and Sang, Hejian and Zhou, Zhengze and He, Ran and Wang, Zhipeng and Geramifard, Alborz},
  booktitle={Findings of the Association for Computational Linguistics: EMNLP 2026},
  year={2026}
}