% Publications of Yuhang Zang — https://yuhangzang.github.io/
% 66 entries generated from https://yuhangzang.github.io/data/publications.json
% Each entry is also available at https://yuhangzang.github.io/papers/<id>.bib next to its paper page.

@article{arxiv260619338,
  title     = {{Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games}},
  author    = {Shengyuan Ding and Xilin Wei and Xinyu Fang and Haodong Duan and Dahua Lin and Jiaqi Wang and Yuhang Zang},
  journal   = {arXiv preprint arXiv:2606.19338},
  year      = {2026},
  url       = {https://arxiv.org/abs/2606.19338}
}

@article{arxiv260603890,
  title     = {{OVO-S-Bench: A Hierarchical Benchmark for Streaming Spatial Intelligence in Multimodal LLMs}},
  author    = {Yifei Li and Pengyiang Liu and Yuhang Zang and Zhongyue Shi and Qi Fu and Hongye Hao and Jiwen Lu},
  journal   = {arXiv preprint arXiv:2606.03890},
  year      = {2026},
  url       = {https://arxiv.org/abs/2606.03890}
}

@article{arxiv260208439,
  title     = {{Demo-ICL: In-Context Learning for Procedural Video Knowledge Acquisition}},
  author    = {Yuhao Dong and Shulin Tian and Shuai Liu and Shuangrui Ding and Yuhang Zang and Xiaoyi Dong and Yuhang Cao and Jiaqi Wang and Ziwei Liu},
  journal   = {arXiv preprint arXiv:2602.08439},
  year      = {2026},
  url       = {https://arxiv.org/abs/2602.08439}
}

@article{arxiv260527955,
  title     = {{Skill-as-Pseudocode: Refactoring Skill Libraries to Pseudocode for LLM Agents}},
  author    = {Xinze Li and Yuhang Zang and Yixin Cao and Aixin Sun},
  journal   = {arXiv preprint arXiv:2605.27955},
  year      = {2026},
  url       = {https://arxiv.org/abs/2605.27955}
}

@article{arxiv260116690,
  title     = {{EMemBench: Interactive Benchmarking of Episodic Memory for VLM Agents}},
  author    = {Xinze Li and Ziyue Zhu and Siyuan Liu and Yubo Ma and Yuhang Zang and Yixin Cao and Aixin Sun},
  journal   = {arXiv preprint arXiv:2601.16690},
  year      = {2026},
  url       = {https://arxiv.org/abs/2601.16690}
}

@article{arxiv250514677,
  title     = {{Visionary-R1: Mitigating Shortcuts in Visual Reasoning with Reinforcement Learning}},
  author    = {Jiaer Xia and Yuhang Zang and Peng Gao and Sharon Li and Kaiyang Zhou},
  journal   = {Transactions on Machine Learning Research},
  year      = {2026},
  url       = {https://jmlr.org/tmlr/papers/bib/JWkZXBgh5a.bib}
}

@article{dai2026endocot,
  title     = {{EndoCoT: Scaling Endogenous Chain-of-Thought Reasoning in Diffusion Models}},
  author    = {Xuanlang Dai and Yujie Zhou and Long Xing and Jiazi Bu and Xilin Wei and Yuhong Liu and Beichen Zhang and Kai Chen and Yuhang Zang},
  journal   = {arXiv preprint arXiv:2603.12252},
  year      = {2026},
  url       = {https://arxiv.org/abs/2603.12252}
}

@article{arxiv260312648,
  title     = {{From Sparse to Dense: Multi-View GRPO for Flow Models via Augmented Condition Space}},
  author    = {Jiazi Bu and Pengyang Ling and Yujie Zhou and Yibin Wang and Yuhang Zang and Tianyi Wei and Xiaohang Zhan and Jiaqi Wang and Tong Wu and Xingang Pan and Dahua Lin},
  journal   = {arXiv preprint arXiv:2603.12648},
  year      = {2026},
  url       = {https://arxiv.org/abs/2603.12648}
}

@inproceedings{arxiv250922186,
  title     = {{MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing}},
  author    = {Junbo Niu and Zheng Liu and Zhuangcheng Gu and Bin Wang and Linke Ouyang and Zhiyuan Zhao and Tao Chu and Tianyao He and Fan Wu and Qintong Zhang and Zhenjiang Jin and Guang Liang and Rui Zhang and Wenzheng Zhang and Yuan Qu and Zhifei Ren and Yuefeng Sun and Zirui Tang and Boyu Niu and Yuanhong Zheng and Dongsheng Ma and Ziyang Miao and Hejun Dong and Siyi Qian and Junyuan Zhang and Fangdong Wang and Jingzhou Chen and Xiaomeng Zhao and Liqun Wei and Wei Li and Shasha Wang and Ruiliang Xu and Yuanyuan Cao and Lu Chen and Qianqian Wu and Huaiyu Gu and Lindong Lu and Dechen Lin and Guanlin Shen and Xuanhe Zhou and Linfeng Zhang and Yuhang Zang and Xiaoyi Dong and Jiaqi Wang and Bo Zhang and Lei Bai and Pei Chu and Weijia Li and Jiang Wu and Lijun Wu and Zhenxiang Li and Guangyu Wang and Zhongying Tu and Chao Xu and Kai Chen and Bowen Zhou and Dahua Lin and Wentao Zhang and Conghui He},
  booktitle = {Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 6: Industry Track)},
  month     = {July},
  year      = {2026},
  publisher = {Association for Computational Linguistics},
  pages     = {13--42},
  doi       = {10.18653/v1/2026.acl-industry.3},
  url       = {https://aclanthology.org/2026.acl-industry.3/}
}

@inproceedings{arxiv250804700,
  title     = {{SEAgent: Self-Evolving Computer Use Agent with Autonomous Learning from Experience}},
  author    = {Zeyi Sun and Ziyu Liu and Yuhang Zang and Yuhang Cao and Xiaoyi Dong and Tong Wu and Dahua Lin and Jiaqi Wang},
  booktitle = {International Conference on Machine Learning (ICML)},
  year      = {2026},
  url       = {https://icml.cc/virtual/2026/poster/65711}
}

@inproceedings{arxiv251205111,
  title     = {{ARM-Thinker: Reinforcing Multimodal Generative Reward Models with Agentic Tool Use and Visual Reasoning}},
  author    = {Shengyuan Ding and Xinyu Fang and Ziyu Liu and Yuhang Zang and Yuhang Cao and Xiangyu Zhao and Haodong Duan and Xiaoyi Dong and Jianze Liang and Bin Wang and Conghui He and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2026},
  pages     = {22195--22205},
  url       = {https://openaccess.thecvf.com/content/CVPR2026/html/Ding_ARM-Thinker_Reinforcing_Multimodal_Generative_Reward_Models_with_Agentic_Tool_Use_CVPR_2026_paper.html}
}

@inproceedings{arxiv251115703,
  title     = {{Think Visually, Reason Textually: Vision-Language Synergy in Abstract Reasoning}},
  author    = {Beichen Zhang and Yuhang Zang and Xiaoyi Dong and Yuhang Cao and Haodong Duan and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2026},
  pages     = {41203--41212},
  url       = {https://openaccess.thecvf.com/content/CVPR2026/html/Zhang_Think_Visually_Reason_Textually_Vision-Language_Synergy_in_Abstract_Reasoning_CVPR_2026_paper.html}
}

@inproceedings{arxiv251001982,
  title     = {{Fine-Grained GRPO for Precise Preference Alignment in Flow Models}},
  author    = {Yujie Zhou and Pengyang Ling and Jiazi Bu and Yibin Wang and Yuhang Zang and Jiaqi Wang and Li Niu and Guangtao Zhai},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2026},
  pages     = {20045--20054},
  url       = {https://openaccess.thecvf.com/content/CVPR2026/html/Zhou_Fine-Grained_GRPO_for_Precise_Preference_Alignment_in_Flow_Models_CVPR_2026_paper.html}
}

@inproceedings{arxiv251027606,
  title     = {{Spatial-SSRL: Enhancing Spatial Understanding via Self-Supervised Reinforcement Learning}},
  author    = {Yuhong Liu and Beichen Zhang and Yuhang Zang and Yuhang Cao and Long Xing and Xiaoyi Dong and Haodong Duan and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2026},
  pages     = {9570--9581},
  url       = {https://openaccess.thecvf.com/content/CVPR2026/html/Liu_Spatial-SSRL_Enhancing_Spatial_Understanding_via_Self-Supervised_Reinforcement_Learning_CVPR_2026_paper.html}
}

@inproceedings{arxiv251201248,
  title     = {{TRivia: Self-supervised Fine-tuning of Vision-Language Models for Table Recognition}},
  author    = {Junyuan Zhang and Bin Wang and Qintong Zhang and Fan Wu and Zichen Wen and Jialin Lu and Junjie Shan and Ziqi Zhao and Shuya Yang and Ziling Wang and Ziyang Miao and Huaping Zhong and Yuhang Zang and Xiaoyi Dong and Ka-Ho Chow and Conghui He},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2026},
  pages     = {33196--33206},
  url       = {https://openaccess.thecvf.com/content/CVPR2026/html/Zhang_TRivia_Self-supervised_Fine-tuning_of_Vision-Language_Models_for_Table_Recognition_CVPR_2026_paper.html}
}

@inproceedings{arxiv250800819,
  title     = {{Beyond Fixed: Training-Free Variable-Length Denoising for Diffusion Large Language Models}},
  author    = {Jinsong Li and Xiaoyi Dong and Yuhang Zang and Yuhang Cao and Jiaqi Wang and Dahua Lin},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {91715--91731},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/940380c12da75e64351cceff4c880557-Abstract-Conference.html}
}

@inproceedings{arxiv250817356,
  title     = {{DiCache: Let Diffusion Model Determine Its Own Cache}},
  author    = {Jiazi Bu and Pengyang Ling and Yujie Zhou and Yibin Wang and Yuhang Zang and Dahua Lin and Jiaqi Wang},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {73778--73803},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/78288ef33b18a351c3cd679dc9a15c8d-Abstract-Conference.html}
}

@inproceedings{arxiv250715852,
  title     = {{Advancing Complex Video Object Segmentation via Progressive Concept Construction}},
  author    = {Zhixiong Zhang and Shuangrui Ding and Xiaoyi Dong and Songxin He and Jianfan Lin and Junsong Tang and Yuhang Zang and Yuhang Cao and Dahua Lin and Jiaqi Wang},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {130411--130434},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/d3f48777432f64bc96b1203713f20352-Abstract-Conference.html}
}

@inproceedings{arxiv250920317,
  title     = {{SIM-CoT: Supervised Implicit Chain-of-Thought}},
  author    = {Xilin Wei and Xiaoran Liu and Yuhang Zang and Xiaoyi Dong and Yuhang Cao and Jiaqi Wang and Xipeng Qiu and Dahua Lin},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {56721--56742},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/5d087955ee13fe9a7402eedec879b9c3-Abstract-Conference.html}
}

@inproceedings{arxiv250619848,
  title     = {{ScaleCap: Scalable Image Captioning via Dual-Modality Debiasing}},
  author    = {Long Xing and Qidong Huang and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Yuhang Cao and Jinsong Li and Shuangrui Ding and Weiming Zhang and Nenghai Yu and Jiaqi Wang and Feng Wu and Dahua Lin},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {32266--32285},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/37050ebbbd7096719ab96cec19a4c69f-Abstract-Conference.html}
}

@inproceedings{arxiv251024693,
  title     = {{STAR-Bench: Probing Deep Spatio-Temporal Reasoning as Audio 4D Intelligence}},
  author    = {Zihan Liu and Zhikang Niu and Qiuyang Xiao and Zhisheng Zheng and Ruoqi Yuan and Yuhang Zang and Yuhang Cao and Xiaoyi Dong and Jianze Liang and Xie Chen and Leilei Sun and Dahua Lin and Jiaqi Wang},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {134703--134731},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/d9f8b5abc8e0926539ecbb492af7b2f1-Abstract-Conference.html}
}

@inproceedings{arxiv260216455,
  title     = {{Visual Self-Refine: A Pixel-Guided Paradigm for Accurate Chart Parsing}},
  author    = {Jinsong Li and Xiaoyi Dong and Yuhang Zang and Yuhang Cao and Jiaqi Wang and Dahua Lin},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {85499--85522},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/89b0e466b46292ce0bfe53618aadd3de-Abstract-Conference.html}
}

@inproceedings{arxiv250922647,
  title     = {{CapRL: Stimulating Dense Image Caption Capabilities via Reinforcement Learning}},
  author    = {Long Xing and Xiaoyi Dong and Yuhang Zang and Yuhang Cao and Jianze Liang and Qidong Huang and Jiaqi Wang and Feng Wu and Dahua Lin},
  booktitle = {International Conference on Learning Representations},
  year      = {2026},
  volume    = {2026},
  pages     = {13066--13093},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2026/hash/15f1dbc086bfd94d8c32557b573cbe18-Abstract-Conference.html}
}

@article{arxiv240313805,
  title     = {{RAR: Retrieving and Ranking Augmented MLLMs for Visual Recognition}},
  author    = {Ziyu Liu and Zeyi Sun and Yuhang Zang and Wei Li and Pan Zhang and Xiaoyi Dong and Yuanjun Xiong and Dahua Lin and Jiaqi Wang},
  journal   = {IEEE Transactions on Image Processing},
  year      = {2026},
  volume    = {35},
  publisher = {Institute of Electrical and Electronics Engineers (IEEE)},
  pages     = {388--401},
  doi       = {10.1109/tip.2025.3644175},
  url       = {https://doi.org/10.1109/tip.2025.3644175}
}

@inproceedings{arxiv250503318,
  title     = {{Unified Multimodal Chain-of-Thought Reward Model through Reinforcement Fine-Tuning}},
  author    = {Yibin Wang and Zhimin Li and Yuhang Zang and Chunyu Wang and Qinglin Lu and Cheng Jin and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2025},
  volume    = {38},
  publisher = {Curran Associates, Inc.},
  pages     = {159130--159157},
  doi       = {10.52202/085713-5315},
  url       = {https://proceedings.neurips.cc/paper_files/paper/2025/hash/e95e9f0c127aa1cfa2628adb2f3cb107-Abstract-Conference.html}
}

@inproceedings{arxiv250406232,
  title     = {{HiFlow: Training-free High-Resolution Image Generation with Flow-Aligned Guidance}},
  author    = {Jiazi Bu and Pengyang Ling and Yujie Zhou and Pan Zhang and Tong Wu and Xiaoyi Dong and Yuhang Zang and Yuhang Cao and Dahua Lin and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2025},
  volume    = {38},
  publisher = {Curran Associates, Inc.},
  pages     = {144317--144351},
  doi       = {10.52202/085713-4831},
  url       = {https://proceedings.neurips.cc/paper_files/paper/2025/hash/d4ecde5d18d36150c529d1b9c5f0d727-Abstract-Conference.html}
}

@inproceedings{Liu_2025_ICCV,
  title     = {{Visual-RFT: Visual Reinforcement Fine-Tuning}},
  author    = {Ziyu Liu and Zeyi Sun and Yuhang Zang and Xiaoyi Dong and Yuhang Cao and Haodong Duan and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {2034--2044},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.html}
}

@inproceedings{arxiv250407957,
  title     = {{MM-IFEngine: Towards Multimodal Instruction Following}},
  author    = {Shengyuan Ding and Shenxi Wu and Xiangyu Zhao and Yuhang Zang and Haodong Duan and Xiaoyi Dong and Pan Zhang and Yuhang Cao and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {1099--1109},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Ding_MM-IFEngine_Towards_Multimodal_Instruction_Following_ICCV_2025_paper.html}
}

@inproceedings{arxiv241201824,
  title     = {{X-Prompt: Generalizable Auto-Regressive Visual Learning with In-Context Prompting}},
  author    = {Zeyi Sun and Ziyang Chu and Pan Zhang and Tong Wu and Yuhang Zang and Xiaoyi Dong and Yuanjun Xiong and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {17268--17280},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Sun_X-Prompt_Generalizable_Auto-Regressive_Visual_Learning_with_In-Context_Prompting_ICCV_2025_paper.html}
}

@inproceedings{arxiv240600093,
  title     = {{Bootstrap3D: Improving Multi-view Diffusion Model with Synthetic Data}},
  author    = {Zeyi Sun and Tong Wu and Pan Zhang and Yuhang Zang and Xiaoyi Dong and Yuanjun Xiong and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {15714--15726},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Sun_Bootstrap3D_Improving_Multi-view_Diffusion_Model_with_Synthetic_Data_ICCV_2025_paper.html}
}

@inproceedings{arxiv250702859,
  title     = {{Bootstrapping Grounded Chain-of-Thought in Multimodal LLMs for Data-Efficient Model Adaptation}},
  author    = {Jiaer Xia and Bingkui Tong and Yuhang Zang and Rui Shao and Kaiyang Zhou},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {208--217},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Xia_Bootstrapping_Grounded_Chain-of-Thought_in_Multimodal_LLMs_for_Data-Efficient_Model_Adaptation_ICCV_2025_paper.html}
}

@inproceedings{arxiv250208590,
  title     = {{Light-A-Video: Training-free Video Relighting via Progressive Light Fusion}},
  author    = {Yujie Zhou and Jiazi Bu and Pengyang Ling and Pan Zhang and Tong Wu and Qidong Huang and Jinsong Li and Xiaoyi Dong and Yuhang Zang and Yuhang Cao and Anyi Rao and Jiaqi Wang and Li Niu},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {13315--13325},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Zhou_Light-A-Video_Training-free_Video_Relighting_via_Progressive_Light_Fusion_ICCV_2025_paper.html}
}

@inproceedings{arxiv241007167,
  title     = {{Deciphering Cross-Modal Alignment in Large Vision-Language Models via Modality Integration Rate}},
  author    = {Qidong Huang and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Yuhang Cao and Jiaqi Wang and Weiming Zhang and Nenghai Yu},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {218--227},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Huang_Deciphering_Cross-Modal_Alignment_in_Large_Vision-Language_Models_via_Modality_Integration_ICCV_2025_paper.html}
}

@inproceedings{arxiv241016268,
  title     = {{SAM2Long: Enhancing SAM 2 for Long Video Segmentation with a Training-Free Memory Tree}},
  author    = {Shuangrui Ding and Rui Qian and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Yuhang Cao and Yuwei Guo and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2025},
  pages     = {13614--13624},
  url       = {https://openaccess.thecvf.com/content/ICCV2025/html/Ding_SAM2Long_Enhancing_SAM_2_for_Long_Video_Segmentation_with_a_ICCV_2025_paper.html}
}

@inproceedings{arxiv250112368,
  title     = {{InternLM-XComposer2.5-Reward: A Simple Yet Effective Multi-Modal Reward Model}},
  author    = {Yuhang Zang and Xiaoyi Dong and Pan Zhang and Yuhang Cao and Ziyu Liu and Shengyuan Ding and Shenxi Wu and Yubo Ma and Haodong Duan and Wenwei Zhang and Kai Chen and Dahua Lin and Jiaqi Wang},
  booktitle = {Findings of the Association for Computational Linguistics: ACL 2025},
  month     = {July},
  year      = {2025},
  publisher = {Association for Computational Linguistics},
  pages     = {6547--6563},
  doi       = {10.18653/v1/2025.findings-acl.340},
  url       = {https://aclanthology.org/2025.findings-acl.340/}
}

@inproceedings{arxiv250604997,
  title     = {{Towards Storage-Efficient Visual Document Retrieval: An Empirical Study on Reducing Patch-Level Embeddings}},
  author    = {Yubo Ma and Jinsong Li and Yuhang Zang and Xiaobao Wu and Xiaoyi Dong and Pan Zhang and Yuhang Cao and Haodong Duan and Jiaqi Wang and Yixin Cao and Aixin Sun},
  booktitle = {Findings of the Association for Computational Linguistics: ACL 2025},
  month     = {July},
  year      = {2025},
  publisher = {Association for Computational Linguistics},
  pages     = {19568--19580},
  doi       = {10.18653/v1/2025.findings-acl.1003},
  url       = {https://aclanthology.org/2025.findings-acl.1003/}
}

@inproceedings{arxiv250205173,
  title     = {{VideoRoPE: What Makes for Good Video Rotary Position Embedding?}},
  author    = {Xilin Wei and Xiaoran Liu and Yuhang Zang and Xiaoyi Dong and Pan Zhang and Yuhang Cao and Jian Tong and Haodong Duan and Qipeng Guo and Jiaqi Wang and Xipeng Qiu and Dahua Lin},
  booktitle = {Proceedings of the 42nd International Conference on Machine Learning},
  year      = {2025},
  volume    = {267},
  publisher = {PMLR},
  pages     = {66118--66136},
  url       = {https://proceedings.mlr.press/v267/wei25h.html}
}

@inproceedings{arxiv250213128,
  title     = {{SongGen: A Single Stage Auto-regressive Transformer for Text-to-Song Generation}},
  author    = {Zihan Liu and Shuangrui Ding and Zhixiong Zhang and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Yuhang Cao and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the 42nd International Conference on Machine Learning},
  year      = {2025},
  volume    = {267},
  publisher = {PMLR},
  pages     = {38351--38364},
  url       = {https://proceedings.mlr.press/v267/liu25m.html}
}

@inproceedings{arxiv241006241,
  title     = {{ByTheWay: Boost Your Text-to-Video Generation Model to Higher Quality in a Training-free Way}},
  author    = {Jiazi Bu and Pengyang Ling and Pan Zhang and Tong Wu and Xiaoyi Dong and Yuhang Zang and Yuhang Cao and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2025},
  pages     = {12999--13008},
  url       = {https://openaccess.thecvf.com/content/CVPR2025/html/Bu_ByTheWay_Boost_Your_Text-to-Video_Generation_Model_to_Higher_Quality_in_CVPR_2025_paper.html}
}

@inproceedings{arxiv250105510,
  title     = {{OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?}},
  author    = {Junbo Niu and Yifei Li and Ziyang Miao and Chunjiang Ge and Yuanhang Zhou and Qihao He and Xiaoyi Dong and Haodong Duan and Shuangrui Ding and Rui Qian and Pan Zhang and Yuhang Zang and Yuhang Cao and Conghui He and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2025},
  pages     = {18902--18913},
  url       = {https://openaccess.thecvf.com/content/CVPR2025/html/Niu_OVO-Bench_How_Far_is_Your_Video-LLMs_from_Real-World_Online_Video_CVPR_2025_paper.html}
}

@inproceedings{arxiv250103218,
  title     = {{Dispider: Enabling Video LLMs with Active Real-Time Interaction via Disentangled Perception, Decision, and Reaction}},
  author    = {Rui Qian and Shuangrui Ding and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Yuhang Cao and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2025},
  pages     = {24045--24055},
  url       = {https://openaccess.thecvf.com/content/CVPR2025/html/Qian_Dispider_Enabling_Video_LLMs_with_Active_Real-Time_Interaction_via_Disentangled_CVPR_2025_paper.html}
}

@inproceedings{arxiv241017247,
  title     = {{Conical Visual Concentration for Efficient Large Vision-Language Models}},
  author    = {Long Xing and Qidong Huang and Xiaoyi Dong and Jiajie Lu and Pan Zhang and Yuhang Zang and Yuhang Cao and Conghui He and Jiaqi Wang and Feng Wu and Dahua Lin},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2025},
  pages     = {14593--14603},
  url       = {https://openaccess.thecvf.com/content/CVPR2025/html/Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper.html}
}

@inproceedings{arxiv240702165,
  title     = {{WildAvatar: Learning In-the-wild 3D Avatars from the Web}},
  author    = {Zihao Huang and Shoukang Hu and Guangcong Wang and Tianqi Liu and Yuhang Zang and Zhiguo Cao and Wei Li and Ziwei Liu},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2025},
  pages     = {15963--15975},
  url       = {https://openaccess.thecvf.com/content/CVPR2025/html/Huang_WildAvatar_Learning_In-the-wild_3D_Avatars_from_the_Web_CVPR_2025_paper.html}
}

@inproceedings{arxiv241017637,
  title     = {{MIA-DPO: Multi-Image Augmented Direct Preference Optimization For Large Vision-Language Models}},
  author    = {Ziyu Liu and Yuhang Zang and Xiaoyi Dong and Pan Zhang and Yuhang Cao and Haodong Duan and Conghui He and Yuanjun Xiong and Dahua Lin and Jiaqi Wang},
  booktitle = {International Conference on Learning Representations},
  year      = {2025},
  volume    = {2025},
  pages     = {34583--34610},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2025/hash/557a20663907ed637c2807f608d5bec2-Abstract-Conference.html}
}

@inproceedings{arxiv240605338,
  title     = {{MotionClone: Training-Free Motion Cloning for Controllable Video Generation}},
  author    = {Pengyang Ling and Jiazi Bu and Pan Zhang and Xiaoyi Dong and Yuhang Zang and Tong Wu and Huaian Chen and Jiaqi Wang and Yi Jin},
  booktitle = {International Conference on Learning Representations},
  year      = {2025},
  volume    = {2025},
  pages     = {75579--75601},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2025/hash/bc82dbfbfa43232be85b8d9838f49c3e-Abstract-Conference.html}
}

@inproceedings{arxiv240711691,
  title     = {{VLMEvalKit: An Open-Source ToolKit for Evaluating Large Multi-Modality Models}},
  author    = {Haodong Duan and Junming Yang and Yuxuan Qiao and Xinyu Fang and Lin Chen and Yuan Liu and Xiaoyi Dong and Yuhang Zang and Pan Zhang and Jiaqi Wang and Dahua Lin and Kai Chen},
  booktitle = {Proceedings of the 32nd ACM International Conference on Multimedia},
  month     = {October},
  year      = {2024},
  publisher = {ACM},
  pages     = {11198--11201},
  doi       = {10.1145/3664647.3685520},
  url       = {https://doi.org/10.1145/3664647.3685520}
}

@inproceedings{arxiv240701523,
  title     = {{MMLONGBENCH-DOC: Benchmarking Long-context Document Understanding with Visualizations}},
  author    = {Yubo Ma and Yuhang Zang and Liangyu Chen and Meiqi Chen and Yizhu Jiao and Xinze Li and Xinyuan Lu and Ziyu Liu and Yan Ma and Xiaoyi Dong and Pan Zhang and Liangming Pan and Yu-Gang Jiang and Jiaqi Wang and Yixin Cao and Aixin Sun},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  volume    = {37},
  publisher = {Curran Associates, Inc.},
  pages     = {95963--96010},
  doi       = {10.52202/079017-3041},
  url       = {https://papers.nips.cc/paper_files/paper/2024/hash/ae0e43289bffea0c1fa34633fc608e92-Abstract-Datasets_and_Benchmarks_Track.html}
}

@inproceedings{arxiv240611833,
  title     = {{MMDU: A Multi-Turn Multi-Image Dialog Understanding Benchmark and Instruction-Tuning Dataset for LVLMs}},
  author    = {Ziyu Liu and Tao Chu and Yuhang Zang and Xilin Wei and Xiaoyi Dong and Pan Zhang and Zijian Liang and Yuanjun Xiong and Yu Qiao and Dahua Lin and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  volume    = {37},
  publisher = {Curran Associates, Inc.},
  pages     = {8698--8733},
  doi       = {10.52202/079017-0278},
  url       = {https://papers.nips.cc/paper_files/paper/2024/hash/1057053100de064a44286239724f7865-Abstract-Datasets_and_Benchmarks_Track.html}
}

@inproceedings{arxiv240604325,
  title     = {{ShareGPT4Video: Improving Video Understanding and Generation with Better Captions}},
  author    = {Lin Chen and Xilin Wei and Jinsong Li and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Zehui Chen and Haodong Duan and Bin Lin and Zhenyu Tang and Li Yuan and Yu Qiao and Dahua Lin and Feng Zhao and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  volume    = {37},
  publisher = {Curran Associates, Inc.},
  pages     = {19472--19495},
  doi       = {10.52202/079017-0614},
  url       = {https://papers.nips.cc/paper_files/paper/2024/hash/22a7476e4fd36818777c47e666f61a41-Abstract-Datasets_and_Benchmarks_Track.html}
}

@inproceedings{arxiv240516009,
  title     = {{Streaming Long Video Understanding with Large Language Models}},
  author    = {Rui Qian and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Shuangrui Ding and Dahua Lin and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  volume    = {37},
  publisher = {Curran Associates, Inc.},
  pages     = {119336--119360},
  doi       = {10.52202/079017-3792},
  url       = {https://papers.nips.cc/paper_files/paper/2024/hash/d7ce06e9293c3d8e6cb3f80b4157f875-Abstract-Conference.html}
}

@inproceedings{arxiv240320330,
  title     = {{Are We on the Right Way for Evaluating Large Vision-Language Models?}},
  author    = {Lin Chen and Jinsong Li and Xiaoyi Dong and Pan Zhang and Yuhang Zang and Zehui Chen and Haodong Duan and Jiaqi Wang and Yu Qiao and Dahua Lin and Feng Zhao},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  volume    = {37},
  publisher = {Curran Associates, Inc.},
  pages     = {27056--27087},
  doi       = {10.52202/079017-0850},
  url       = {https://papers.nips.cc/paper_files/paper/2024/hash/2f8ee6a3d766b426d2618e555b5aeb39-Abstract-Conference.html}
}

@inproceedings{arxiv240406512,
  title     = {{InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD}},
  author    = {Xiaoyi Dong and Pan Zhang and Yuhang Zang and Yuhang Cao and Bin Wang and Linke Ouyang and Songyang Zhang and Haodong Duan and Wenwei Zhang and Yining Li and Hang Yan and Yang Gao and Zhe Chen and Xinyue Zhang and Wei Li and Jingwen Li and Wenhai Wang and Kai Chen and Conghui He and Xingcheng Zhang and Jifeng Dai and Yu Qiao and Dahua Lin and Jiaqi Wang},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  volume    = {37},
  publisher = {Curran Associates, Inc.},
  pages     = {42566--42592},
  doi       = {10.52202/079017-1348},
  url       = {https://papers.nips.cc/paper_files/paper/2024/hash/4b06cdddb1cde6624c0be1465c7b800f-Abstract-Conference.html}
}

@inproceedings{arxiv240512218,
  title     = {{MVSGaussian: Fast Generalizable Gaussian Splatting Reconstruction from Multi-View Stereo}},
  author    = {Tianqi Liu and Guangcong Wang and Shoukang Hu and Liao Shen and Xinyi Ye and Yuhang Zang and Zhiguo Cao and Wei Li and Ziwei Liu},
  booktitle = {Computer Vision – ECCV 2024},
  year      = {2025},
  publisher = {Springer Nature Switzerland},
  pages     = {37--53},
  doi       = {10.1007/978-3-031-72649-1\_3},
  url       = {https://doi.org/10.1007/978-3-031-72649-1_3}
}

@inproceedings{arxiv240315378,
  title     = {{Long-CLIP: Unlocking the Long-Text Capability of CLIP}},
  author    = {Beichen Zhang and Pan Zhang and Xiaoyi Dong and Yuhang Zang and Jiaqi Wang},
  booktitle = {Computer Vision – ECCV 2024},
  year      = {2025},
  publisher = {Springer Nature Switzerland},
  pages     = {310--325},
  doi       = {10.1007/978-3-031-72983-6\_18},
  url       = {https://doi.org/10.1007/978-3-031-72983-6_18}
}

@inproceedings{arxiv231203818,
  title     = {{Alpha-CLIP: A CLIP Model Focusing on Wherever You Want}},
  author    = {Zeyi Sun and Ye Fang and Tong Wu and Pan Zhang and Yuhang Zang and Shu Kong and Yuanjun Xiong and Dahua Lin and Jiaqi Wang},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2024},
  pages     = {13019--13029},
  url       = {https://openaccess.thecvf.com/content/CVPR2024/html/Sun_Alpha-CLIP_A_CLIP_Model_Focusing_on_Wherever_You_Want_CVPR_2024_paper.html}
}

@inproceedings{arxiv240115914,
  title     = {{Overcoming the Pitfalls of Vision-Language Model Finetuning for OOD Generalization}},
  author    = {Yuhang Zang and Hanlin Goh and Joshua Susskind and Chen Huang},
  booktitle = {International Conference on Learning Representations},
  year      = {2024},
  volume    = {2024},
  pages     = {30327--30342},
  url       = {https://proceedings.iclr.cc/paper_files/paper/2024/hash/8140f43b06c9c7e14fb96953caed2665-Abstract-Conference.html}
}

@article{arxiv230518279,
  title     = {{Contextual Object Detection with Multimodal Large Language Models}},
  author    = {Yuhang Zang and Wei Li and Jun Han and Kaiyang Zhou and Chen Change Loy},
  journal   = {International Journal of Computer Vision},
  month     = {February},
  year      = {2025},
  volume    = {133},
  number    = {2},
  publisher = {Springer Science and Business Media LLC},
  pages     = {825--843},
  doi       = {10.1007/s11263-024-02214-4},
  url       = {https://doi.org/10.1007/s11263-024-02214-4}
}

@phdthesis{zang2023realworld,
  title     = {{Real-World Object Detection}},
  author    = {Yuhang Zang},
  school    = {Nanyang Technological University},
  year      = {2023},
  doi       = {10.32657/10356/171489},
  url       = {https://doi.org/10.32657/10356/171489}
}

@article{arxiv221007225,
  title     = {{Unified Vision and Language Prompt Learning}},
  author    = {Yuhang Zang and Wei Li and Kaiyang Zhou and Chen Huang and Chen Change Loy},
  journal   = {arXiv preprint arXiv:2210.07225},
  year      = {2022},
  url       = {https://arxiv.org/abs/2210.07225}
}

@article{arxiv230514813,
  title     = {{Semi-Supervised and Long-Tailed Object Detection with CascadeMatch}},
  author    = {Yuhang Zang and Kaiyang Zhou and Chen Huang and Chen Change Loy},
  journal   = {International Journal of Computer Vision},
  month     = {April},
  year      = {2023},
  volume    = {131},
  number    = {4},
  publisher = {Springer Science and Business Media LLC},
  pages     = {987--1001},
  doi       = {10.1007/s11263-022-01738-x},
  url       = {https://doi.org/10.1007/s11263-022-01738-x}
}

@inproceedings{arxiv220311876,
  title     = {{Open-Vocabulary DETR with Conditional Matching}},
  author    = {Yuhang Zang and Wei Li and Kaiyang Zhou and Chen Huang and Chen Change Loy},
  booktitle = {Computer Vision – ECCV 2022},
  year      = {2022},
  publisher = {Springer Nature Switzerland},
  pages     = {106--122},
  doi       = {10.1007/978-3-031-20077-9\_7},
  url       = {https://doi.org/10.1007/978-3-031-20077-9_7}
}

@inproceedings{arxiv210212867,
  title     = {{FASA: Feature Augmentation and Sampling Adaptation for Long-Tailed Instance Segmentation}},
  author    = {Yuhang Zang and Chen Huang and Chen Change Loy},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2021},
  pages     = {3457--3466},
  url       = {https://openaccess.thecvf.com/content/ICCV2021/html/Zang_FASA_Feature_Augmentation_and_Sampling_Adaptation_for_Long-Tailed_Instance_Segmentation_ICCV_2021_paper.html}
}

@inproceedings{arxiv200810032,
  title     = {{Seesaw Loss for Long-Tailed Instance Segmentation}},
  author    = {Jiaqi Wang and Wenwei Zhang and Yuhang Zang and Yuhang Cao and Jiangmiao Pang and Tao Gong and Kai Chen and Ziwei Liu and Chen Change Loy and Dahua Lin},
  booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
  month     = {June},
  year      = {2021},
  pages     = {9695--9704},
  url       = {https://openaccess.thecvf.com/content/CVPR2021/html/Wang_Seesaw_Loss_for_Long-Tailed_Instance_Segmentation_CVPR_2021_paper.html}
}

@article{arxiv200307543,
  title     = {{KPNet: Towards Minimal Face Detector}},
  author    = {Guanglu Song and Yu Liu and Yuhang Zang and Xiaogang Wang and Biao Leng and Qingsheng Yuan},
  journal   = {Proceedings of the AAAI Conference on Artificial Intelligence},
  month     = {April},
  year      = {2020},
  volume    = {34},
  number    = {07},
  publisher = {Association for the Advancement of Artificial Intelligence (AAAI)},
  pages     = {12015--12022},
  doi       = {10.1609/aaai.v34i07.6878},
  url       = {https://doi.org/10.1609/aaai.v34i07.6878}
}

@inproceedings{arxiv190805900,
  title     = {{Efficient and Accurate Arbitrary-Shaped Text Detection With Pixel Aggregation Network}},
  author    = {Wenhai Wang and Enze Xie and Xiaoge Song and Yuhang Zang and Wenjia Wang and Tong Lu and Gang Yu and Chunhua Shen},
  booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
  month     = {October},
  year      = {2019},
  pages     = {8440--8449},
  url       = {https://openaccess.thecvf.com/content_ICCV_2019/html/Wang_Efficient_and_Accurate_Arbitrary-Shaped_Text_Detection_With_Pixel_Aggregation_Network_ICCV_2019_paper.html}
}

@article{arxiv181108605,
  title     = {{Scene Text Detection with Supervised Pyramid Context Network}},
  author    = {Enze Xie and Yuhang Zang and Shuai Shao and Gang Yu and Cong Yao and Guangyao Li},
  journal   = {Proceedings of the AAAI Conference on Artificial Intelligence},
  month     = {July},
  year      = {2019},
  volume    = {33},
  number    = {01},
  publisher = {Association for the Advancement of Artificial Intelligence (AAAI)},
  pages     = {9038--9045},
  doi       = {10.1609/aaai.v33i01.33019038},
  url       = {https://doi.org/10.1609/aaai.v33i01.33019038}
}
