@misc{indiciae38e0acb0d108, title = {ScalePRM: Training Process Reward Models by Scaling Verification Compute Without Ground Truth}, author = {Salman Rahman and Sruthi Gorantla and Arpit Gupta and Swastik Roy and Nanyun Peng and Yang Liu}, year = {2026}, url = {https://arxiv.org/abs/2512.03244}, note = {Source identifier: 2512.03244} }