@inproceedings{xu2026beyond,
title={Beyond GRPO and On-Policy Distillation: An Empirical Sparse-to-Dense Reward Principle for Language-Model Post-Training},
author={Xu, Yuanda and Sang, Hejian and Zhou, Zhengze and He, Ran and Wang, Zhipeng and Geramifard, Alborz},
booktitle={Findings of the Association for Computational Linguistics: EMNLP 2026},
year={2026}
}