@inproceedings{arxiv250503318,
  title     = {{Unified Multimodal Chain-of-Thought Reward Model through Reinforcement Fine-Tuning}},
  author    = {Yibin Wang and Zhimin Li and Yuhang Zang and Chunyu Wang and Qinglin Lu and Cheng Jin and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2025},
  volume    = {38},
  publisher = {Curran Associates, Inc.},
  pages     = {159130--159157},
  doi       = {10.52202/085713-5315},
  url       = {https://proceedings.neurips.cc/paper_files/paper/2025/hash/e95e9f0c127aa1cfa2628adb2f3cb107-Abstract-Conference.html}
}
