@inproceedings{NL2VIS, author = {Luo, Yuyu and Tang, Nan and Li, Guoliang and Chai, Chengliang and Li, Wenbo and Qin, Xuedi}, title = {Synthesizing Natural Language to Visualization ({NL2VIS}) Benchmarks from {NL2SQL} Benchmarks}, booktitle = {{SIGMOD} '21: International Conference on Management of Data}, pages = {1235--1247}, publisher = {{ACM}}, year = {2021}, } @inproceedings{DataVisT5, author = {Wan, Zhuoyue and Song, Yuanfeng and Li, Shuaimin and Zhang, Chen Jason and Wong, Raymond Chi-Wing}, title = {{DataVisT5}: A Pre-Trained Language Model for Jointly Understanding Text and Data Visualization}, booktitle = {41st {IEEE} International Conference on Data Engineering, {ICDE} 2025}, pages = {1704--1717}, publisher = {{IEEE}}, year = {2025}, } @inproceedings{zhang2017stackgan, title={Stackgan: Text to photo-realistic image synthesis with stacked generative adversarial networks}, author={Zhang, Han and Xu, Tao and Li, Hongsheng and Zhang, Shaoting and Wang, Xiaogang and Huang, Xiaolei and Metaxas, Dimitris N}, booktitle={Proceedings of the IEEE international conference on computer vision}, pages={5907--5915}, year={2017} } @inproceedings{ramesh2021zero, title={Zero-shot text-to-image generation}, author={Ramesh, Aditya and Pavlov, Mikhail and Goh, Gabriel and Gray, Scott and Voss, Chelsea and Radford, Alec and Chen, Mark and Sutskever, Ilya}, booktitle={International conference on machine learning}, pages={8821--8831}, year={2021}, organization={Pmlr} } @article{betker2023improving, title={Improving image generation with better captions}, author={Betker, James and Goh, Gabriel and Jing, Li and Brooks, Tim and Wang, Jianfeng and Li, Linjie and Ouyang, Long and Zhuang, Juntang and Lee, Joyce and Guo, Yufei and others}, journal={Computer Science. https://cdn. openai. com/papers/dall-e-3. pdf}, volume={2}, number={3}, pages={8}, year={2023} } @article{li2024if, title={What if we recaption billions of web images with llama-3?}, author={Li, Xianhang and Tu, Haoqin and Hui, Mude and Wang, Zeyu and Zhao, Bingchen and Xiao, Junfei and Ren, Sucheng and Mei, Jieru and Liu, Qing and Zheng, Huangjie and others}, journal={arXiv preprint arXiv:2406.08478}, year={2024} } @article{FDABench, author = {Wang, Ziting and Zhang, Shize and Yuan, Haitao and Zhu, Jinwei and Li, Shifu and Dong, Wei and Cong, Gao}, title = {{FDABench}: A Benchmark for Data Agents on Analytical Queries over Heterogeneous Data}, journal = {arXiv preprint arXiv:2509.02473}, year = {2025}, } @inproceedings{STRaptor, author = {Tang, Zirui and Niu, Boyu and Zhou, Xuanhe and Li, Boxiu and Zhou, Wei and Wang, Jiannan and Li, Guoliang and Zhang, Xinyi}, title = {{ST-Raptor}: {LLM}-Powered Semi-Structured Table Question Answering}, booktitle = {Proceedings of the ACM SIGMOD International Conference on Management of Data}, year = {2026}, } @article{peng2024dreambench++, title={Dreambench++: A human-aligned benchmark for personalized image generation}, author={Peng, Yuang and Cui, Yuxin and Tang, Haomiao and Qi, Zekun and Dong, Runpei and Bai, Jing and Han, Chunrui and Ge, Zheng and Zhang, Xiangyu and Xia, Shu-Tao}, journal={arXiv preprint arXiv:2406.16855}, year={2024} } @inproceedings{mou2025dreamo, title={Dreamo: A unified framework for image customization}, author={Mou, Chong and Wu, Yanze and Wu, Wenxu and Guo, Zinan and Zhang, Pengze and Cheng, Yufeng and Luo, Yiming and Ding, Fei and Zhang, Shiwen and Li, Xinghui and others}, booktitle={Proceedings of the SIGGRAPH Asia 2025 Conference Papers}, pages={1--12}, year={2025} } @inproceedings{MoDora, author = {Xu, Bangrui and Yao, Qihang and Tang, Zirui and Zhou, Xuanhe and He, Yeye and Yu, Shihan and Xu, Qianqian and Wang, Bin and Li, Guoliang}, title = {{MoDora}: Tree-Based Semi-Structured Document Analysis System}, booktitle = {Proceedings of the ACM SIGMOD International Conference on Management of Data}, year = {2026}, } @InProceedings{OminiControl, author = {Tan, Zhenxiong and Liu, Songhua and Yang, Xingyi and Xue, Qiaochu and Wang, Xinchao}, title = {OminiControl: Minimal and Universal Control for Diffusion Transformer}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {14940-14950} } @misc{Chimera, title={Chimera: Compositional Image Generation using Part-based Concepting}, author={Shivam Singh and Yiming Chen and Agneet Chatterjee and Amit Raj and James Hays and Yezhou Yang and Chitta Baral}, year={2025}, eprint={2510.18083}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2510.18083}, } @InProceedings{FROSS, author = {Hou, Hao-Yu and Lee, Chun-Yi and Sonogashira, Motoharu and Kawanishi, Yasutomo}, title = {FROSS: Faster-Than-Real-Time Online 3D Semantic Scene Graph Generation from RGB-D Images}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {28818-28827} } @InProceedings{AIComposer, author = {Li, Haowen and Fan, Zhenfeng and Wen, Zhang and Zhu, Zhengzhou and Li, Yunjin}, title = {AIComposer: Any Style and Content Image Composition via Feature Integration}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {16840-16850} } @inproceedings{Easycontrol, title={Easycontrol: Adding efficient and flexible control for diffusion transformer}, author={Zhang, Yuxuan and Yuan, Yirui and Song, Yiren and Wang, Haofan and Liu, Jiaming}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={19513--19524}, year={2025} } @inproceedings{Mv-adapter, title={Mv-adapter: Multi-view consistent image generation made easy}, author={Huang, Zehuan and Guo, Yuan-Chen and Wang, Haoran and Yi, Ran and Ma, Lizhuang and Cao, Yan-Pei and Sheng, Lu}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={16377--16387}, year={2025} } @misc{PRISM, title={PRISM: A Unified Framework for Photorealistic Reconstruction and Intrinsic Scene Modeling}, author={Alara Dirik and Tuanfeng Wang and Duygu Ceylan and Stefanos Zafeiriou and Anna Frühstück}, year={2025}, eprint={2504.14219}, archivePrefix={arXiv}, primaryClass={cs.GR}, url={https://arxiv.org/abs/2504.14219} } @misc{MIGLoRA, title={Efficient Multi-Instance Generation with Janus-Pro-Dirven Prompt Parsing}, author={Fan Qi and Yu Duan and Changsheng Xu}, year={2025}, eprint={2503.21069}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2503.21069}, } @inproceedings{controlnet++, author = {Ming Li and Taojiannan Yang and Huafeng Kuang and Jie Wu and Zhaoning Wang and Xuefeng Xiao and Chen Chen}, title = {ControlNet++: Improving Conditional Controls with Efficient Consistency Feedback}, booktitle = {European Conference on Computer Vision (ECCV)}, year = {2024}, } @InProceedings{FlexGen, author = {Xu, Xinli and Ge, Wenhang and Lin, Jiantao and Feng, Jiawei and Xu, Lie and Zhao, Hanfeng and Zhang, Shunsi and Chen, Ying-Cong}, title = {FlexGen: Flexible Multi-View Generation from Text and Image Inputs}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {18714-18724} } @InProceedings{SpinMeRound, author = {Galanakis, Stathis and Lattas, Alexandros and Moschoglou, Stylianos and Kainz, Bernhard and Zafeiriou, Stefanos}, title = {SpinMeRound: Consistent Multi-View Identity Generation Using Diffusion Models}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {14346-14356} } @InProceedings{MOSAIC, author = {Liu, Zhixuan and Zhu, Haokun and Chen, Rui and Francis, Jonathan and Hwang, Soonmin and Zhang, Ji and Oh, Jean}, title = {MOSAIC: Generating Consistent, Privacy-Preserving Scenes from Multiple Depth Views in Multi-Room Environments}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {27456-27465} } @InProceedings{UniPortrait, author = {He, Junjie and Geng, Yifeng and Bo, Liefeng}, title = {UniPortrait: A Unified Framework for Identity-Preserving Single- and Multi-Human Image Personalization}, booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, month = {October}, year = {2025}, pages = {14399-14408} } @inproceedings{ruiz2023dreambooth, title={Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation}, author={Ruiz, Nataniel and Li, Yuanzhen and Jampani, Varun and Pritch, Yael and Rubinstein, Michael and Aberman, Kfir}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={22500--22510}, year={2023} } %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% %%% %%% %%% Section 7.2 Custom Domain Adaptation %%% %%% %%% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% % Luo2025BeyondFT, Liu2024GlyphByT5v2AS, Lu2025EasyTextCD, Chen2025POSTAAG, Jiang2025ControlTextUC, Wang2025UniGlyphUS, Ma2024CharGenHA, Ma2026UMTextAU, Chen2025PosterCraftRH, Shi2025WordConWT, Zhang2025PosterGenAP, Liu2026PosterVerseAF, Chen2025RethinkingLG, Chen2026PosterOmniGA, Li2025ICCustomDI, @article{gal2022image, title={An image is worth one word: Personalizing text-to-image generation using textual inversion}, author={Gal, Rinon and Alaluf, Yuval and Atzmon, Yuval and Patashnik, Or and Bermano, Amit H and Chechik, Gal and Cohen-Or, Daniel}, journal={arXiv preprint arXiv:2208.01618}, year={2022} } @article{hu2022lora, title={Lora: Low-rank adaptation of large language models}, author={Hu, Edward J and Shen, Yelong and Wallis, Phillip and Allen-Zhu, Zeyuan and Li, Yuanzhi and Wang, Shean and Wang, Liang and Chen, Weizhu and others}, journal={Iclr}, volume={1}, number={2}, pages={3}, year={2022} } @article{gal2023encoder, title={Encoder-based domain tuning for fast personalization of text-to-image models}, author={Gal, Rinon and Arar, Moab and Atzmon, Yuval and Bermano, Amit H and Chechik, Gal and Cohen-Or, Daniel}, journal={ACM Transactions on Graphics (TOG)}, volume={42}, number={4}, pages={1--13}, year={2023}, publisher={ACM New York, NY, USA} } @article{ye2023ip, title={Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models}, author={Ye, Hu and Zhang, Jun and Liu, Sibo and Han, Xiao and Yang, Wei}, journal={arXiv preprint arXiv:2308.06721}, year={2023} } @article{wang2024instantid, title={Instantid: Zero-shot identity-preserving generation in seconds}, author={Wang, Qixun and Bai, Xu and Wang, Haofan and Qin, Zekui and Chen, Anthony}, journal={arXiv preprint arXiv:2401.07519}, year={2024} } @article{guo2024pulid, title={Pulid: Pure and lightning id customization via contrastive alignment}, author={Guo, Zinan and Wu, Yanze and Zhuowei, Chen and Zhang, Peng and He, Qian and others}, journal={Advances in neural information processing systems}, volume={37}, pages={36777--36804}, year={2024} } @inproceedings{shi2024instantbooth, title={Instantbooth: Personalized text-to-image generation without test-time finetuning}, author={Shi, Jing and Xiong, Wei and Lin, Zhe and Jung, Hyun Joon}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={8543--8552}, year={2024} } @inproceedings{li2024photomaker, title={Photomaker: Customizing realistic human photos via stacked id embedding}, author={Li, Zhen and Cao, Mingdeng and Wang, Xintao and Qi, Zhongang and Cheng, Ming-Ming and Shan, Ying}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={8640--8650}, year={2024} } @inproceedings{peng2024portraitbooth, title={Portraitbooth: A versatile portrait model for fast identity-preserved personalization}, author={Peng, Xu and Zhu, Junwei and Jiang, Boyuan and Tai, Ying and Luo, Donghao and Zhang, Jiangning and Lin, Wei and Jin, Taisong and Wang, Chengjie and Ji, Rongrong}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition}, pages={27080--27090}, year={2024} } @article{zhou2024storymaker, title={Storymaker: Towards holistic consistent characters in text-to-image generation}, author={Zhou, Zhengguang and Li, Jing and Li, Huaxia and Chen, Nemo and Tang, Xu}, journal={arXiv preprint arXiv:2409.12576}, year={2024} } @article{xiao2025fastcomposer, title={Fastcomposer: Tuning-free multi-subject image generation with localized attention}, author={Xiao, Guangxuan and Yin, Tianwei and Freeman, William T and Durand, Fr{\'e}do and Han, Song}, journal={International Journal of Computer Vision}, volume={133}, number={3}, pages={1175--1194}, year={2025}, publisher={Springer} } @inproceedings{nam2025visual, title={Visual persona: Foundation model for full-body human customization}, author={Nam, Jisu and Son, Soowon and Xu, Zhan and Shi, Jing and Liu, Difan and Liu, Feng and Kim, Seungryong and Zhou, Yang}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={18630--18641}, year={2025} } @article{Li2025ICCustomDI, title={IC-Custom: Diverse Image Customization via In-Context Learning}, author={Yaowei Li and Xiaoyu Li and Zhaoyang Zhang and Yuxuan Bian and Gan Liu and Xinyuan Li and Jiale Xu and Wenbo Hu and Yating Liu and Lingen Li and Jing Cai and Yuexian Zou and Yancheng He and Ying Shan}, journal={ArXiv}, year={2025}, volume={abs/2507.01926}, url={https://api.semanticscholar.org/CorpusID:280150169} } @article{Chen2026PosterOmniGA, title={PosterOmni: Generalized Artistic Poster Creation via Task Distillation and Unified Reward Feedback}, author={Sixiang Chen and Jianyu Lai and Jialin Gao and Hengyu Shi and Zhongying Liu and Tian Ye and Junfeng Luo and Xiaoming Wei and Lei Zhu}, journal={arXiv preprint}, year={2026} } @article{Liu2026PosterVerseAF, title={PosterVerse: A Full-Workflow Framework for Commercial-Grade Poster Generation with HTML-Based Scalable Typography}, author={Junle Liu and Peirong Zhang and Yuyi Zhang and Pengyu Yan and Hui Zhou and Xinyue Zhou and Fengjun Guo and Lianwen Jin}, journal={ArXiv}, year={2026}, volume={abs/2601.03993}, url={https://api.semanticscholar.org/CorpusID:284532482} } @article{Zhang2025PosterGenAP, title={PosterGen: Aesthetic-Aware Paper-to-Poster Generation via Multi-Agent LLMs}, author={Zhilin Zhang and Xiang Zhang and Jiaqi Wei and Yiwei Xu and Chenyu You}, journal={ArXiv}, year={2025}, volume={abs/2508.17188}, url={https://api.semanticscholar.org/CorpusID:280710654} } @article{zhang2025creatidesign, title={Creatidesign: A unified multi-conditional diffusion transformer for creative graphic design}, author={Zhang, Hui and Hong, Dexiang and Yang, Maoke and Cheng, Yutao and Zhang, Zhao and Shao, Jie and Wu, Xinglong and Wu, Zuxuan and Jiang, Yu-Gang}, journal={arXiv preprint arXiv:2505.19114}, year={2025} } @article{Shi2025WordConWT, title={WordCon: Word-level Typography Control in Scene Text Rendering}, author={Wenda Shi and Yiren Song and Zihan Rao and Dengming Zhang and Jiaming Liu and Xingxing Zou}, journal={ArXiv}, year={2025}, volume={abs/2506.21276}, url={https://api.semanticscholar.org/CorpusID:280010723} } @article{Chen2025PosterCraftRH, title={PosterCraft: Rethinking High-Quality Aesthetic Poster Generation in a Unified Framework}, author={Sixiang Chen and Jianyu Lai and Jialin Gao and Tian Ye and Haoyu Chen and Hengyu Shi and Shitong Shao and Yunlong Lin and Song Fei and Zhaohu Xing and Yeying Jin and Junfeng Luo and Xiaoming Wei and Lei Zhu}, journal={ArXiv}, year={2025}, volume={abs/2506.10741}, url={https://api.semanticscholar.org/CorpusID:279318662} } @article{Ma2026UMTextAU, title={UM-Text: A Unified Multimodal Model for Image Understanding and Visual Text Editing}, author={Lichen Ma and Xiaolong Fu and Gaojing Zhou and Zipeng Guo and Ting Zhu and Yichun Liu and Yu Shi and Jason Li and Junshi Huang}, journal={arXiv preprint}, year={2026} } @article{Ma2024CharGenHA, title={CharGen: High Accurate Character-Level Visual Text Generation Model with MultiModal Encoder}, author={Lichen Ma and Tiezhu Yue and Pei Fu and Yujie Zhong and Kai Zhou and Xiaoming Wei and Jie Hu}, journal={ArXiv}, year={2024}, volume={abs/2412.17225}, url={https://api.semanticscholar.org/CorpusID:274981660} } @article{Wang2025UniGlyphUS, title={UniGlyph: Unified Segmentation-Conditioned Diffusion for Precise Visual Text Synthesis}, author={Yuanrui Wang and Cong Han and Yafei Li and Zhipeng Jin and Xiawei Li and Sinan Du and Wen Tao and Yi Yang and Shuanglong Li and Chun Yuan and Liu Lin}, journal={ArXiv}, year={2025}, volume={abs/2507.00992}, url={https://api.semanticscholar.org/CorpusID:280069916} } @article{Jiang2025ControlTextUC, title={ControlText: Unlocking Controllable Fonts in Multilingual Text Rendering without Font Annotations}, author={Bowen Jiang and Yuan Yuan and Xinyi Bai and Zhuoqun Hao and Alyson Yin and Yaojie Hu and Wenyu Liao and Lyle Ungar and Camillo Jose Taylor}, journal={ArXiv}, year={2025}, volume={abs/2502.10999}, url={https://api.semanticscholar.org/CorpusID:276408159} } @article{Chen2025POSTAAG, title={POSTA: A Go-to Framework for Customized Artistic Poster Generation}, author={Haoyu Chen and Xiaojie Xu and Wenbo Li and Jingjing Ren and Tian Ye and Songhua Liu and Ying-Cong Chen and Lei Zhu and Xinchao Wang}, journal={2025 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, year={2025}, pages={28694-28704}, url={https://api.semanticscholar.org/CorpusID:277113604} } @article{Lu2025EasyTextCD, title={EasyText: Controllable Diffusion Transformer for Multilingual Text Rendering}, author={Runnan Lu and Yuxuan Zhang and Jailing Liu and Haifa Wang and Yiren Song}, journal={ArXiv}, year={2025}, volume={abs/2505.24417}, url={https://api.semanticscholar.org/CorpusID:279070661} } @misc{PhysicEdit, title={From Statics to Dynamics: Physics-Aware Image Editing with Latent Transition Priors}, author={Liangbing Zhao and Le Zhuo and Sayak Paul and Hongsheng Li and Mohamed Elhoseiny}, year={2026}, eprint={2602.21778}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2602.21778}, } @article{Liu2024GlyphByT5v2AS, title={Glyph-ByT5-v2: A Strong Aesthetic Baseline for Accurate Multilingual Visual Text Rendering}, author={Zeyu Liu and Weicong Liang and Yiming Zhao and Bohan Chen and Ji Li and Yuhui Yuan}, journal={ArXiv}, year={2024}, volume={abs/2406.10208}, url={https://api.semanticscholar.org/CorpusID:270521692} } @article{Luo2025BeyondFT, title={Beyond Flat Text: Dual Self-inherited Guidance for Visual Text Generation}, author={Minxing Luo and Zixun Xia and Liaojun Chen and Zhenhang Li and Weichao Zeng and Jianye Wang and Wentao Cheng and Yaxing Wang and Yu ZHOU and Jian Yang}, journal={ArXiv}, year={2025}, volume={abs/2501.05892}, url={https://api.semanticscholar.org/CorpusID:275458598} } @misc{CreatiLayout, title={CreatiLayout: Siamese Multimodal Diffusion Transformer for Creative Layout-to-Image Generation}, author={Hui Zhang and Dexiang Hong and Yitong Wang and Jie Shao and Xinglong Wu and Zuxuan Wu and Yu-Gang Jiang}, year={2025}, eprint={2412.03859}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2412.03859}, } @misc{ReCon, title={ReCon: Region-Controllable Data Augmentation with Rectification and Alignment for Object Detection}, author={Haowei Zhu and Tianxiang Pan and Rui Qin and Jun-Hai Yong and Bin Wang}, year={2025}, eprint={2510.15783}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2510.15783}, } @ARTICLE{StyleShot, author={Gao, Junyao and Sun, Yanan and Liu, Yanchen and Tang, Yinhao and Zeng, Yanhong and Qi, Ding and Chen, Kai and Zhao, Cairong}, journal={IEEE Transactions on Pattern Analysis and Machine Intelligence}, title={StyleShot: A Snapshot on Any Style}, year={2026}, volume={48}, number={2}, pages={1215-1228}, keywords={Feature extraction;Training;Diffusion models;Image synthesis;Benchmark testing;Tuning;Semantics;Noise reduction;Data mining;Text to image;Style transfer;diffusion model;open-domain;text-to-image}, doi={10.1109/TPAMI.2025.3610614}} @article{InstantCharacter, title={InstantCharacter: Personalize Any Characters with a Scalable Diffusion Transformer Framework}, author={Jiale Tao and Yanbing Zhang and Qixun Wang and Yiji Cheng and Haofan Wang and Xu Bai and Zhengguang Zhou and Ruihuang Li and Linqing Wang and Chunyu Wang and Qin Lin and Qinglin Lu}, journal={ArXiv}, year={2025}, volume={abs/2504.12395}, url={https://api.semanticscholar.org/CorpusID:277856764} } @article{RichControl, title={RichControl: Structure- and Appearance-Rich Training-Free Spatial Control for Text-to-Image Generation}, author={Liheng Zhang and Lexi Pang and Hang Ye and Xiaoxuan Ma and Yizhou Wang}, journal={ArXiv}, year={2025}, volume={abs/2507.02792}, url={https://api.semanticscholar.org/CorpusID:280047690} } @article{OmniRefiner, title={OmniRefiner: Reinforcement-Guided Local Diffusion Refinement}, author={Yaoli Liu and Zi-Juan Ouyang and Shengtao Lou and Yiren Song}, journal={ArXiv}, year={2025}, volume={abs/2511.19990}, url={https://api.semanticscholar.org/CorpusID:283250744} } %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% %%% %%% %%% Section 7.3 Conditional Image Editing %%% %%% %%% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% @article{Zhang2025InContextEE, title={In-Context Edit: Enabling Instructional Image Editing with In-Context Generation in Large Scale Diffusion Transformer}, author={Zechuan Zhang and Ji Xie and Yu Lu and Zongxin Yang and Yi Yang}, journal={ArXiv}, year={2025}, volume={abs/2504.20690}, url={https://api.semanticscholar.org/CorpusID:278171476} } @article{Yin2025ReasonEditTR, title={ReasonEdit: Towards Reasoning-Enhanced Image Editing Models}, author={Fukun Yin and Shiyu Liu and Yucheng Han and Zhibo Wang and Peng Xing and Rui Wang and Wei Cheng and Yingming Wang and Aojie Li and Zixi Yin and Pengtao Chen and Xiangyu Zhang and Daxin Jiang and Xianfang Zeng and Gang Yu}, journal={ArXiv}, year={2025}, volume={abs/2511.22625}, url={https://api.semanticscholar.org/CorpusID:283439275} } @article{Ghazanfari2025SpotEditEV, title={SpotEdit: Evaluating Visually-Guided Image Editing Methods}, author={Sara Ghazanfari and Wei-An Lin and Haitong Tian and Ersin Yumer}, journal={ArXiv}, year={2025}, volume={abs/2508.18159}, url={https://api.semanticscholar.org/CorpusID:280710822} } @article{Jia2025LegoEditAG, title={Lego-Edit: A General Image Editing Framework with Model-Level Bricks and MLLM Builder}, author={Qifei Jia and Yu Liu and Yajie Chai and Xintong Yao and Qiming Lu and Yasen Zhang and Runyu Shi and Ying Huang and Guoquan Zhang}, journal={ArXiv}, year={2025}, volume={abs/2509.12883}, url={https://api.semanticscholar.org/CorpusID:281325583} } @misc{PaCaNet, title={PaCaNet: A Study on CycleGAN with Transfer Learning for Diversifying Fused Chinese Painting and Calligraphy}, author={Zuhao Yang and Huajun Bai and Zhang Luo and Yang Xu and Wei Pang and Yue Wang and Yisheng Yuan and Yingfang Yuan}, year={2023}, eprint={2301.13082}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2301.13082}, } @inproceedings{Rechar, author = {Yang, Zhongyu and Song, Junhao and Luo, Zhang and Yang, Zuhao and Xu, Yang and Lan, Jingfen and Zhang, Yonghan and Pang, Wei and Song, Siyang and Yuan, Yingfang}, title = {ReChar: Revitalising Characters with Structure Preserved and User-Specified Aesthetic Enhancements}, year = {2025}, isbn = {9798400721366}, publisher = {Association for Computing Machinery}, address = {New York, NY, USA}, url = {https://doi.org/10.1145/3757376.3771409}, doi = {10.1145/3757376.3771409}, abstract = {Despite recent advances in generative models, artistic character generation remains an open problem. The key challenge is to balance the preservation of character structures to ensure integrity while incorporating aesthetic enhancements, which can be broadly categorized into visual styles and user-specified decorative elements. To address this, we propose ReChar, a plug-and-play framework composed of three complementary modules that preserve structure, extract style, and generate decorative elements. These modules are integrated via a fusion model to enable precise and coherent artistic character generation. To systematically evaluate artistic character generation, we introduce ImageNet-ReChar, the first large-scale benchmark for this task, covering multiple writing systems, diverse visual styles, and over 1,000 semantically grounded decorative prompts. Extensive experiments show that ReChar outperforms state-of-the-art baselines in structural integrity, stylistic fidelity, and prompt adherence, achieving an SSIM of 0.8690 and over 93\% human preference across all criteria.}, booktitle = {Proceedings of the SIGGRAPH Asia 2025 Technical Communications}, articleno = {30}, numpages = {5}, location = { }, series = {SA Technical Communications '25} } @article{Yeh2025BeyondSE, title={Beyond Simple Edits: X-Planner for Complex Instruction-Based Image Editing}, author={Chun-Hsiao Yeh and Yilin Wang and Nanxuan Zhao and Richard Zhang and Yuheng Li and Yi Ma and Krishna Kumar Singh}, journal={ArXiv}, year={2025}, volume={abs/2507.05259}, url={https://api.semanticscholar.org/CorpusID:280150428} } @article{Zeng2025MIRAMI, title={MIRA: Multimodal Iterative Reasoning Agent for Image Editing}, author={Ziyun Zeng and Hang Hua and Jiebo Luo}, journal={ArXiv}, year={2025}, volume={abs/2511.21087}, url={https://api.semanticscholar.org/CorpusID:283262006} } @article{Shen2025IMAGHarmonyCI, title={IMAGHarmony: Controllable Image Editing with Consistent Object Quantity and Layout}, author={Fei Shen and Xiaoyu Du and Yutong Gao and Jian Yu and Yushe Cao and Xing Lei and Jinhui Tang}, journal={ArXiv}, year={2025}, volume={abs/2506.01949}, url={https://api.semanticscholar.org/CorpusID:279119734} } @article{Yang2026ControllableLI, title={Controllable Layered Image Generation for Real-World Editing}, author={Jinrui Yang and Qing Liu and Yijun Li and Mengwei Ren and Letian Zhang and Zhe Lin and Cihang Xie and Yuyin Zhou}, journal={arXiv preprint}, year={2026} } @article{Mao2025VisualAM, title={Visual Autoregressive Modeling for Instruction-Guided Image Editing}, author={Qingyang Mao and Qi Cai and Yehao Li and Yingwei Pan and Mingyue Cheng and Ting Yao and Qi Liu and Tao Mei}, journal={ArXiv}, year={2025}, volume={abs/2508.15772}, url={https://api.semanticscholar.org/CorpusID:280700028} } @article{Liu2025Step1XEditAP, title={Step1X-Edit: A Practical Framework for General Image Editing}, author={Shiyu Liu and Yucheng Han and Peng Xing and Fukun Yin and Rui Wang and Wei Cheng and Jiaqi Liao and Yingming Wang and Honghao Fu and Chunrui Han and Guopeng Li and Yuang Peng and Quan Sun and Jingwei Wu and Yan Cai and Zheng Ge and Ranchen Ming and Lei Xia and Xianfang Zeng and Yibo Zhu and Binxing Jiao and Xiangyu Zhang and Gang Yu and Daxin Jiang}, journal={ArXiv}, year={2025}, volume={abs/2504.17761}, url={https://api.semanticscholar.org/CorpusID:278033726} } @article{Ma2025X2EditRA, title={X2Edit: Revisiting Arbitrary-Instruction Image Editing through Self-Constructed Data and Task-Aware Representation Learning}, author={Jiancang Ma and Xujie Zhu and Zihao Pan and Qirong Peng and Xu Guo and Chen Chen and H. Lu}, journal={ArXiv}, year={2025}, volume={abs/2508.07607}, url={https://api.semanticscholar.org/CorpusID:280567028} } @article{Hu2025ImageEA, title={Image Editing As Programs with Diffusion Models}, author={Yujia Hu and Songhua Liu and Zhenxiong Tan and Xingyi Yang and Xinchao Wang}, journal={ArXiv}, year={2025}, volume={abs/2506.04158}, url={https://api.semanticscholar.org/CorpusID:279154460} } @article{Yao2025ImplementationFF, title={Implementation Framework for Instruction-Driven Image Editing}, author={Chaosheng Yao and Lei Cui and Jinbo Zhang and Changcai Lu and Ziyang Zhang and Zhengyan Fan}, journal={2025 2nd International Symposium on AI and Cybersecurity (ISAICS)}, year={2025}, pages={1-5}, url={https://api.semanticscholar.org/CorpusID:285106947} } @article{Fang2025TBStarEditFI, title={TBStar-Edit: From Image Editing Pattern Shifting to Consistency Enhancement}, author={Hao Fang and Zechao Zhan and Weixin Feng and Ziwei Huang and Xubin Li and Tiezheng Ge}, journal={ArXiv}, year={2025}, volume={abs/2510.04483}, url={https://api.semanticscholar.org/CorpusID:281842655} } %%% Post-Training @article{ho2020denoising, title={Denoising diffusion probabilistic models}, author={Ho, Jonathan and Jain, Ajay and Abbeel, Pieter}, journal={Advances in neural information processing systems}, volume={33}, pages={6840--6851}, year={2020} } @article{black2023training, title={Training diffusion models with reinforcement learning}, author={Black, Kevin and Janner, Michael and Du, Yilun and Kostrikov, Ilya and Levine, Sergey}, journal={arXiv preprint arXiv:2305.13301}, year={2023} } @article{fan2023dpok, title={DPOK: Reinforcement learning for fine-tuning text-to-image diffusion models}, author={Fan, Ying and Watkins, Olivia and Du, Yuqing and Liu, Hao and Ryu, Moonkyung and Boutilier, Craig and Abbeel, Pieter and Ghavamzadeh, Mohammad and Lee, Kangwook and Lee, Kimin}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={79858--79885}, year={2023} } @article{prabhudesai2023alignprop, title={Aligning text-to-image diffusion models with reward backpropagation}, author={Prabhudesai, Mihir and Goyal, Anirudh and Pathak, Deepak and Fragkiadaki, Katerina}, journal={arXiv preprint arXiv:2310.03739}, year={2023} } @inproceedings{wallace2024diffusion, title={Diffusion model alignment using direct preference optimization}, author={Wallace, Bram and Dang, Meihua and Rafailov, Rafael and Zhou, Linqi and Lou, Aaron and Purushwalkam, Senthil and Ermon, Stefano and Xiong, Caiming and Joty, Shafiq and Naik, Nikhil}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition}, pages={8228--8238}, year={2024} } @inproceedings{yang2024dense, title={A dense reward view on aligning text-to-image diffusion with preference}, author={Yang, Shentao and Chen, Tianqi and Zhou, Mingyuan}, booktitle={Proceedings of the 41st International Conference on Machine Learning}, year={2024} } @inproceedings{wu2025hybrid, title={Hybrid layout control for diffusion transformer: Fewer annotations, superior aesthetics}, author={Wu, Keming and Chen, Junwen and Liang, Zhanhao and Wang, Yinuo and Li, Ji and Zhang, Chao and Wang, Bin and Yuan, Yuhui}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={17930--17940}, year={2025} } @inproceedings{liu2025videodpo, title={Videodpo: Omni-preference alignment for video diffusion generation}, author={Liu, Runtao and Wu, Haoyu and Zheng, Ziqiang and Wei, Chen and He, Yingqing and Pi, Renjie and Chen, Qifeng}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={8009--8019}, year={2025} } @article{liu2022flow, title={Flow straight and fast: Learning to generate and transfer data with rectified flow}, author={Liu, Xingchao and Gong, Chengyue and Liu, Qiang}, journal={arXiv preprint arXiv:2209.03003}, year={2022} } @article{xue2025dancegrpo, title={Dancegrpo: Unleashing grpo on visual generation}, author={Xue, Zeyue and Wu, Jie and Gao, Yu and Kong, Fangyuan and Zhu, Lingting and Chen, Mengzhao and Liu, Zhiheng and Liu, Wei and Guo, Qiushan and Huang, Weilin and others}, journal={arXiv preprint arXiv:2505.07818}, year={2025} } @article{liu2025flow, title={Flow-grpo: Training flow matching models via online rl}, author={Liu, Jie and Liu, Gongye and Liang, Jiajun and Li, Yangguang and Liu, Jiaheng and Wang, Xintao and Wan, Pengfei and Zhang, Di and Ouyang, Wanli}, journal={arXiv preprint arXiv:2505.05470}, year={2025} } @inproceedings{hessel2021clipscore, title={Clipscore: A reference-free evaluation metric for image captioning}, author={Hessel, Jack and Holtzman, Ari and Forbes, Maxwell and Le Bras, Ronan and Choi, Yejin}, booktitle={Proceedings of the 2021 conference on empirical methods in natural language processing}, pages={7514--7528}, year={2021} } @article{kirstain2023pick, title={Pick-a-pic: An open dataset of user preferences for text-to-image generation}, author={Kirstain, Yuval and Polyak, Adam and Singer, Uriel and Matiana, Shahbuland and Penna, Joe and Levy, Omer}, journal={Advances in neural information processing systems}, volume={36}, pages={36652--36663}, year={2023} } @inproceedings{ma2025hpsv3, title={Hpsv3: Towards wide-spectrum human preference score}, author={Ma, Yuhang and Wu, Xiaoshi and Sun, Keqiang and Li, Hongsheng}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={15086--15095}, year={2025} } @inproceedings{zhang2024learning, title={Learning multi-dimensional human preference for text-to-image generation}, author={Zhang, Sixian and Wang, Bohan and Wu, Junqiang and Li, Yan and Gao, Tingting and Zhang, Di and Wang, Zhongyuan}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition}, pages={8018--8027}, year={2024} } @article{xu2024visionreward, title={Visionreward: Fine-grained multi-dimensional human preference learning for image and video generation}, author={Xu, Jiazheng and Huang, Yu and Cheng, Jiale and Yang, Yuanming and Xu, Jiajun and Wang, Yuan and Duan, Wenbo and Yang, Shen and Jin, Qunlin and Li, Shurun and others}, journal={arXiv preprint arXiv:2412.21059}, year={2024} } @inproceedings{ku2024viescore, title={Viescore: Towards explainable metrics for conditional image synthesis evaluation}, author={Ku, Max and Jiang, Dongfu and Wei, Cong and Yue, Xiang and Chen, Wenhu}, booktitle={Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pages={12268--12290}, year={2024} } @article{wei2025skywork, title={Skywork unipic 2.0: Building kontext model with online rl for unified multimodal model}, author={Wei, Hongyang and Xu, Baixin and Liu, Hongbo and Wu, Size and Liu, Jie and Peng, Yi and Wang, Peiyu and Liu, Zexiang and He, Jingwen and Xietian, Yidan and others}, journal={arXiv preprint arXiv:2509.04548}, year={2025} } @article{wu2025rewarddance, title={Rewarddance: Reward scaling in visual generation}, author={Wu, Jie and Gao, Yu and Ye, Zilyu and Li, Ming and Li, Liang and Guo, Hanzhong and Liu, Jie and Xue, Zeyue and Hou, Xiaoxia and Liu, Wei and others}, journal={arXiv preprint arXiv:2509.08826}, year={2025} } @article{gong2025onereward, title={Onereward: Unified mask-guided image generation via multi-task human preference learning}, author={Gong, Yuan and Wang, Xionghui and Wu, Jie and Wang, Shiyin and Wang, Yitong and Wu, Xinglong}, journal={arXiv preprint arXiv:2508.21066}, year={2025} } @article{wu2025editreward, title={Editreward: A human-aligned reward model for instruction-guided image editing}, author={Wu, Keming and Jiang, Sicong and Ku, Max and Nie, Ping and Liu, Minghao and Chen, Wenhu}, journal={arXiv preprint arXiv:2509.26346}, year={2025} } @article{luo2025editscore, title={Editscore: Unlocking online rl for image editing via high-fidelity reward modeling}, author={Luo, Xin and Wang, Jiahao and Wu, Chenyuan and Xiao, Shitao and Jiang, Xiyan and Lian, Defu and Zhang, Jiajun and Liu, Dong and others}, journal={arXiv preprint arXiv:2509.23909}, year={2025} } @article{zhang2025reasongen, title={ReasonGen-R1: CoT for Autoregressive Image generation models through SFT and RL}, author={Zhang, Yu and Li, Yunqi and Yang, Yifan and Wang, Rui and Yang, Yuqing and Qi, Dai and Bao, Jianmin and Chen, Dongdong and Luo, Chong and Qiu, Lili}, journal={arXiv preprint arXiv:2505.24875}, year={2025} } @article{jiang2025t2i, title={T2i-r1: Reinforcing image generation with collaborative semantic-level and token-level cot}, author={Jiang, Dongzhi and Guo, Ziyu and Zhang, Renrui and Zong, Zhuofan and Li, Hao and Zhuo, Le and Yan, Shilin and Heng, Pheng-Ann and Li, Hongsheng}, journal={arXiv preprint arXiv:2505.00703}, year={2025} } @inproceedings{sohl2015deep, title={Deep unsupervised learning using nonequilibrium thermodynamics}, author={Sohl-Dickstein, Jascha and Weiss, Eric and Maheswaranathan, Niru and Ganguli, Surya}, booktitle={International conference on machine learning}, pages={2256--2265}, year={2015}, organization={PMLR} } %%%%%%%%%%%%%%%%%% 1. Efficient Training %%%%%%%%%%%%%%%%%% %%%%%%%%%%%%%%%%%% 1.1 Rectified Flow %%%%%%%%%%%%%%%%%% @article{guo2024gaussian, title={Gaussian mixture solvers for diffusion models}, author={Guo, Hanzhong and Lu, Cheng and Bao, Fan and Pang, Tianyu and Yan, Shuicheng and Du, Chao and Li, Chongxuan}, journal={Advances in Neural Information Processing Systems}, volume={36}, year={2024} } @inproceedings{xue2024sa, title={SA-Solver: Stochastic Adams Solver for Fast Sampling of Diffusion Models}, author={Xue, Shuchen and Yi, Mingyang and Luo, Weijian and Zhang, Shifeng and Sun, Jiacheng and Li, Zhenguo and Ma, Zhi-Ming}, booktitle={Advances in Neural Information Processing Systems}, volume={36}, year={2024} } @inproceedings{lu2022dpm, title={DPM-Solver: A Fast ODE Solver for Diffusion Probabilistic Model Sampling in Around 10 Steps}, author={Lu, Cheng and Zhou, Yuhao and Bao, Fan and Chen, Jianfei and Li, Chongxuan and Zhu, Jun}, booktitle={Advances in Neural Information Processing Systems}, volume={35}, year={2022} } @inproceedings{zhang2022fast, title={Fast Sampling of Diffusion Models with Exponential Integrator}, author={Zhang, Qinsheng and Chen, Yongxin}, booktitle={International Conference on Learning Representations}, year={2023} } @article{jolicoeur2021gotta, title={Gotta Go Fast When Generating Data with Score-Based Models}, author={Jolicoeur-Martineau, Alexia and Li, Ke and Pich\'{e}-Taillefer, R\'{e}mi and Kachman, Tal and Mitliagkas, Ioannis}, journal={arXiv preprint arXiv:2105.14080}, year={2021} } @inproceedings{karras2022elucidating, title={Elucidating the Design Space of Diffusion-Based Generative Models}, author={Karras, Tero and Aittala, Miika and Aila, Timo and Laine, Samuli}, booktitle={Advances in Neural Information Processing Systems}, volume={35}, year={2022} } @inproceedings{meng2022on, title={On fast sampling of diffusion models}, author={Meng, Chenlin and Song, Jiaming and Ermon, Stefano}, booktitle={International Conference on Machine Learning}, year={2022} } @inproceedings{liu2022pseudo, title={Pseudo Numerical Methods for Diffusion Models on Manifolds}, author={Liu, Luping and Ren, Yi and Lin, Zhijie and Zhao, Zhou}, booktitle={International Conference on Learning Representations}, year={2022} } @article{cai2025z, title={Z-image: An efficient image generation foundation model with single-stream diffusion transformer}, author={Cai, Huanqia and Cao, Sihan and Du, Ruoyi and Gao, Peng and Hoi, Steven and Hou, Zhaohui and Huang, Shijie and Jiang, Dengyang and Jin, Xin and Li, Liangchen and others}, journal={arXiv preprint arXiv:2511.22699}, year={2025} } @article{wu2025qwen, title={Qwen-image technical report}, author={Wu, Chenfei and Li, Jiahao and Zhou, Jingren and Lin, Junyang and Gao, Kaiyuan and Yan, Kun and Yin, Sheng-ming and Bai, Shuai and Xu, Xiao and Chen, Yilei and others}, journal={arXiv preprint arXiv:2508.02324}, year={2025} } @article{gao2025seedream, title={Seedream 3.0 technical report}, author={Gao, Yu and Gong, Lixue and Guo, Qiushan and Hou, Xiaoxia and Lai, Zhichao and Li, Fanshi and Li, Liang and Lian, Xiaochen and Liao, Chao and Liu, Liyang and others}, journal={arXiv preprint arXiv:2504.11346}, year={2025} } @article{seedream2025seedream, title={Seedream 4.0: Toward next-generation multimodal image generation}, author={Seedream, Team and Chen, Yunpeng and Gao, Yu and Gong, Lixue and Guo, Meng and Guo, Qiushan and Guo, Zhiyao and Hou, Xiaoxia and Huang, Weilin and Huang, Yixuan and others}, journal={arXiv preprint arXiv:2509.20427}, year={2025} } @article{cao2025hunyuanimage, title={Hunyuanimage 3.0 technical report}, author={Cao, Siyu and Chen, Hangting and Chen, Peng and Cheng, Yiji and Cui, Yutao and Deng, Xinchi and Dong, Ying and Gong, Kipper and Gu, Tianpeng and Gu, Xiusen and others}, journal={arXiv preprint arXiv:2509.23951}, year={2025} } @article{wang2025seededit, title={Seededit 3.0: Fast and high-quality generative image editing}, author={Wang, Peng and Shi, Yichun and Lian, Xiaochen and Zhai, Zhonghua and Xia, Xin and Xiao, Xuefeng and Huang, Weilin and Yang, Jianchao}, journal={arXiv preprint arXiv:2506.05083}, year={2025} } @article{team2025longcat, title={Longcat-image technical report}, author={Team, Meituan LongCat and Ma, Hanghang and Tan, Haoxian and Huang, Jiale and Wu, Junqiang and He, Jun-Yan and Gao, Lishuai and Xiao, Songlin and Wei, Xiaoming and Ma, Xiaoqi and others}, journal={arXiv preprint arXiv:2512.07584}, year={2025} } @article{team2026firered, title={FireRed-Image-Edit-1.0 Technical Report}, author={Team, Super Intelligence and Qiao, Changhao and Hui, Chao and Li, Chen and Wang, Cunzheng and Song, Dejia and Zhang, Jiale and Li, Jing and Xiang, Qiang and Wang, Runqi and others}, journal={arXiv preprint arXiv:2602.13344}, year={2026} } @misc{jdjoyaiimage, title={JoyAI-Image: Awakening Spatial Intelligence in Unified Multimodal Understanding and Generation}, author={{Joy Future Academy, JD}}, year={2026}, howpublished={Technical report}, note={Available at \url{https://joyai-image.s3.cn-north-1.jdcloud-oss.com/JoyAI-Image.pdf}, accessed 2026-04-22} } @inproceedings{xu2023restart, title={Restart sampling for improving generative processes}, author={Xu, Yilun and Deng, Mingyang and Cheng, Xiang and Tian, Yonglong and Liu, Ziming and Jaakkola, Tommi}, booktitle={Advances in Neural Information Processing Systems}, year={2023} } @inproceedings{fang2023structural, title={Structural Pruning for Diffusion Models}, author={Fang, Gongfan and Ma, Xinyin and Wang, Xinchao}, booktitle={Advances in Neural Information Processing Systems}, volume={36}, year={2023} } @inproceedings{castells2024ld, title={LD-Pruner: Efficient Pruning of Latent Diffusion Models using Task-Agnostic Insights}, author={Castells, Thibault and Song, Hyoung-Kyu and Kim, Bo-Kyeong and Choi, Shinkook}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops}, pages={821--830}, year={2024} } @inproceedings{kim2024layermerge, title={LayerMerge: Neural Network Depth Compression through Layer Pruning and Merging}, author={Kim, Jinuk and El Halabi, Marwa and Ji, Mingi and Song, Hyun Oh}, booktitle={International Conference on Machine Learning}, year={2024} } @article{zhang2024laptop, title={Laptop-diff: Layer pruning and normalized distillation for compressing diffusion models}, author={Zhang, Dingkun and Li, Sijia and Chen, Chen and Xie, Qingsong and Lu, Haonan}, journal={arXiv preprint arXiv:2404.11098}, year={2024} } @inproceedings{shang2023post, title={Post-training quantization on diffusion models}, author={Shang, Yuzhang and Yuan, Zhihang and Xie, Bin and Wu, Bingzhe and Yan, Yan}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={1972--1981}, year={2023} } @inproceedings{songDDIM, title = {Denoising Diffusion Implicit Models}, author = {Song, Jiaming and Meng, Chenlin and Ermon, Stefano}, year = 2021, booktitle = {International Conference on Learning Representations} } @article{lu2022dpm++, title={DPM-Solver++: Fast Solver for Guided Sampling of Diffusion Probabilistic Models}, author={Lu, Cheng and Zhou, Yuhao and Bao, Fan and Chen, Jianfei and Li, Chongxuan and Zhu, Jun}, journal={arXiv preprint arXiv:2211.01095}, year={2022} } @inproceedings{zheng2023dpmsolvervF, title={DPM-Solver-v3: Improved Diffusion ODE Solver with Empirical Model Statistics}, author={Zheng, Kaiwen and Lu, Cheng and Chen, Jianfei and Zhu, Jun}, booktitle={Advances in Neural Information Processing Systems}, volume={36}, year={2023} } @inproceedings{10377259, title={Q-Diffusion: Quantizing Diffusion Models}, author={Li, Xiuyu and Liu, Yijiang and Lian, Long and Yang, Huanrui and Dong, Zhen and Kang, Daniel and Zhang, Shanghang and Keutzer, Kurt}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={17535--17545}, year={2023} } @inproceedings{kim2025ditto, title={Ditto: Accelerating Diffusion Model via Temporal Value Similarity}, author={Kim, Sungbin and Lee, Hyunwuk and Cho, Wonho and Park, Mincheol and Ro, Won Woo}, booktitle={IEEE International Symposium on High-Performance Computer Architecture}, year={2025} } @inproceedings{bolya2023tomesd, title={Token Merging for Fast Stable Diffusion}, author={Bolya, Daniel and Hoffman, Judy}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops}, pages={4599--4603}, year={2023} } @inproceedings{kim2024tofu, title={Token Fusion: Bridging the Gap between Token Pruning and Token Merging}, author={Kim, Minchul and Gao, Shangqian and Hsu, Yen-Chang and Shen, Yilin and Jin, Hongxia}, booktitle={Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision}, year={2024} } @inproceedings{ma2024deepcache, title = {Deepcache: Accelerating diffusion models for free}, author = {Ma, Xinyin and Fang, Gongfan and Wang, Xinchao}, year = 2024, booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition}, pages = {15762--15772} } @article{chen2024delta-dit, title = {$\Delta$-DiT: A Training-Free Acceleration Method Tailored for Diffusion Transformers}, author = {Chen, Pengtao and Shen, Mingzhu and Ye, Peng and Cao, Jianjian and Tu, Chongjun and Bouganis, Christos-Savvas and Zhao, Yiren and Chen, Tao}, year = 2024, journal = {arXiv preprint arXiv:2406.01125} } @misc{liu2024timestep, title={Timestep Embedding Tells: It's Time to Cache for Video Diffusion Model}, author={Feng Liu and Shiwei Zhang and Xiaofeng Wang and Yujie Wei and Haonan Qiu and Yuzhong Zhao and Yingya Zhang and Qixiang Ye and Fang Wan}, year={2024}, eprint={2411.19108}, archivePrefix={arXiv}, primaryClass={cs.CV} } @article{zou2024accelerating, title = {Accelerating Diffusion Transformers with Token-wise Feature Caching}, author = {Zou, Chang and Liu, Xuyang and Liu, Ting and Huang, Siteng and Zhang, Linfeng}, year = 2024, journal = {arXiv preprint arXiv:2410.05317} } @article{liuReusingForecastingAccelerating2025, title = {From Reusing to Forecasting: Accelerating Diffusion Models with TaylorSeers}, author = {Liu, Jiacheng and Zou, Chang and Lyu, Yuanhuiyi and Chen, Junjie and Zhang, Linfeng}, journal = {arXiv preprint arXiv:2503.06923}, year = {2025} } @article{liu2025freqca, title={Freqca: Accelerating diffusion models via frequency-aware caching}, author={Liu, Jiacheng and Cai, Peiliang and Zhou, Qinming and Lin, Yuqi and Kong, Deyang and Huang, Benhao and Pan, Yupei and Xu, Haowen and Zou, Chang and Tang, Junshu and others}, journal={arXiv preprint arXiv:2510.08669}, year={2025} } @inproceedings{liu2025speca, title={Speca: Accelerating diffusion transformers with speculative feature caching}, author={Liu, Jiacheng and Zou, Chang and Lyu, Yuanhuiyi and Ren, Fei and Wang, Shaobo and Li, Kaixin and Zhang, Linfeng}, booktitle={Proceedings of the 33rd ACM International Conference on Multimedia}, pages={10024--10033}, year={2025} } @article{zhu2026tap, title={TAP: A Token-Adaptive Predictor Framework for Training-Free Diffusion Acceleration}, author={Zhu, Haowei and Huang, Tingxuan and Wang, Xing and Zhao, Tianyu and Wang, Jiexi and Chen, Weifeng and Peng, Xurui and Chen, Fangmin and Yong, Junhai and Wang, Bin}, journal={arXiv preprint arXiv:2603.03792}, year={2026} } @article{zhu2026diffsparse, title={DiffSparse: Accelerating Diffusion Transformers with Learned Token Sparsity}, author={Zhu, Haowei and Liu, Ji and Liu, Ziqiong and Li, Dong and Yong, Junhai and Wang, Bin and Barsoum, Emad}, journal={arXiv preprint arXiv:2604.03674}, year={2026} } @article{ma2024learning, title={Learning-to-cache: Accelerating diffusion transformer via layer caching}, author={Ma, Xinyin and Fang, Gongfan and Bi Mi, Michael and Wang, Xinchao}, journal={Advances in Neural Information Processing Systems}, volume={37}, pages={133282--133304}, year={2024} } @inproceedings{fang2025tinyfusion, title={Tinyfusion: Diffusion transformers learned shallow}, author={Fang, Gongfan and Li, Kunjun and Ma, Xinyin and Wang, Xinchao}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={18144--18154}, year={2025} } @article{yuan2024ditfastattn, title={Ditfastattn: Attention compression for diffusion transformer models}, author={Yuan, Zhihang and Zhang, Hanling and Lu, Pu and Ning, Xuefei and Zhang, Linfeng and Zhao, Tianchen and Yan, Shengen and Dai, Guohao and Wang, Yu}, journal={Advances in Neural Information Processing Systems}, volume={37}, pages={1196--1219}, year={2024} } @article{zhu2024dip, title={Dip-go: A diffusion pruner via few-step gradient optimization}, author={Zhu, Haowei and Tang, Dehua and Liu, Ji and Lu, Mingjie and Zheng, Jintu and Peng, Jinzhang and Li, Dong and Wang, Yu and Jiang, Fan and Tian, Lu and others}, journal={Advances in Neural Information Processing Systems}, volume={37}, pages={92581--92604}, year={2024} } @misc{goodfellow2014generativeadversarialnetworks, title={Generative Adversarial Networks}, author={Ian J. Goodfellow and Jean Pouget-Abadie and Mehdi Mirza and Bing Xu and David Warde-Farley and Sherjil Ozair and Aaron Courville and Yoshua Bengio}, year={2014}, eprint={1406.2661}, archivePrefix={arXiv}, primaryClass={stat.ML}, url={https://arxiv.org/abs/1406.2661}, } @misc{arjovsky2017wassersteingan, title={Wasserstein GAN}, author={Martin Arjovsky and Soumith Chintala and Léon Bottou}, year={2017}, eprint={1701.07875}, archivePrefix={arXiv}, primaryClass={stat.ML}, url={https://arxiv.org/abs/1701.07875}, } @misc{gulrajani2017improvedtrainingwassersteingans, title={Improved Training of Wasserstein GANs}, author={Ishaan Gulrajani and Faruk Ahmed and Martin Arjovsky and Vincent Dumoulin and Aaron Courville}, year={2017}, eprint={1704.00028}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/1704.00028}, } @misc{mao2017squaresgenerativeadversarialnetworks, title={Least Squares Generative Adversarial Networks}, author={Xudong Mao and Qing Li and Haoran Xie and Raymond Y. K. Lau and Zhen Wang and Stephen Paul Smolley}, year={2017}, eprint={1611.04076}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/1611.04076}, } @misc{miyato2018spectralnormalizationgenerativeadversarial, title={Spectral Normalization for Generative Adversarial Networks}, author={Takeru Miyato and Toshiki Kataoka and Masanori Koyama and Yuichi Yoshida}, year={2018}, eprint={1802.05957}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/1802.05957}, } @misc{radford2016unsupervisedrepresentationlearningdeep, title={Unsupervised Representation Learning with Deep Convolutional Generative Adversarial Networks}, author={Alec Radford and Luke Metz and Soumith Chintala}, year={2016}, eprint={1511.06434}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/1511.06434}, } @misc{brock2019largescalegantraining, title={Large Scale GAN Training for High Fidelity Natural Image Synthesis}, author={Andrew Brock and Jeff Donahue and Karen Simonyan}, year={2019}, eprint={1809.11096}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/1809.11096}, } @misc{karras2019stylebasedgeneratorarchitecturegenerative, title={A Style-Based Generator Architecture for Generative Adversarial Networks}, author={Tero Karras and Samuli Laine and Timo Aila}, year={2019}, eprint={1812.04948}, archivePrefix={arXiv}, primaryClass={cs.NE}, url={https://arxiv.org/abs/1812.04948}, } @misc{chen2016infoganinterpretablerepresentationlearning, title={InfoGAN: Interpretable Representation Learning by Information Maximizing Generative Adversarial Nets}, author={Xi Chen and Yan Duan and Rein Houthooft and John Schulman and Ilya Sutskever and Pieter Abbeel}, year={2016}, eprint={1606.03657}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/1606.03657}, } @misc{zhu2020unpairedimagetoimagetranslationusing, title={Unpaired Image-to-Image Translation using Cycle-Consistent Adversarial Networks}, author={Jun-Yan Zhu and Taesung Park and Phillip Isola and Alexei A. Efros}, year={2020}, eprint={1703.10593}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/1703.10593}, } @misc{song2021scorebasedgenerativemodelingstochastic, title={Score-Based Generative Modeling through Stochastic Differential Equations}, author={Yang Song and Jascha Sohl-Dickstein and Diederik P. Kingma and Abhishek Kumar and Stefano Ermon and Ben Poole}, year={2021}, eprint={2011.13456}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/2011.13456}, } @misc{nichol2022glidephotorealisticimagegeneration, title={GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models}, author={Alex Nichol and Prafulla Dhariwal and Aditya Ramesh and Pranav Shyam and Pamela Mishkin and Bob McGrew and Ilya Sutskever and Mark Chen}, year={2022}, eprint={2112.10741}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2112.10741}, } @misc{rombach2022stablediffusion, title={High-Resolution Image Synthesis with Latent Diffusion Models}, author={Robin Rombach and Andreas Blattmann and Dominik Lorenz and Patrick Esser and Björn Ommer}, year={2022}, eprint={2112.10752}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2112.10752}, } @misc{esser2024stablediffusion3, title={Scaling Rectified Flow Transformers for High-Resolution Image Synthesis}, author={Patrick Esser and Sumith Kulal and Andreas Blattmann and Rahim Entezari and Jonas Müller and Harry Saini and Yam Levi and Dominik Lorenz and Axel Sauer and Frederic Boesel and Dustin Podell and Tim Dockhorn and Zion English and Kyle Lacey and Alex Goodwin and Yannik Marek and Robin Rombach}, year={2024}, eprint={2403.03206}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2403.03206}, } @misc{saharia2022imagegen, title={Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding}, author={Chitwan Saharia and William Chan and Saurabh Saxena and Lala Li and Jay Whang and Emily Denton and Seyed Kamyar Seyed Ghasemipour and Burcu Karagol Ayan and S. Sara Mahdavi and Rapha Gontijo Lopes and Tim Salimans and Jonathan Ho and David J Fleet and Mohammad Norouzi}, year={2022}, eprint={2205.11487}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2205.11487}, } @misc{ramesh2022dalle2, title={Hierarchical Text-Conditional Image Generation with CLIP Latents}, author={Aditya Ramesh and Prafulla Dhariwal and Alex Nichol and Casey Chu and Mark Chen}, year={2022}, eprint={2204.06125}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2204.06125}, } @misc{deng2025bagelemerging, title={Emerging Properties in Unified Multimodal Pretraining}, author={Chaorui Deng and Deyao Zhu and Kunchang Li and Chenhui Gou and Feng Li and Zeyu Wang and Shu Zhong and Weihao Yu and Xiaonan Nie and Ziang Song and Guang Shi and Haoqi Fan}, year={2025}, eprint={2505.14683}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2505.14683}, } @misc{deng2025emergingpropertiesunifiedmultimodal, title={Emerging Properties in Unified Multimodal Pretraining}, author={Chaorui Deng and Deyao Zhu and Kunchang Li and Chenhui Gou and Feng Li and Zeyu Wang and Shu Zhong and Weihao Yu and Xiaonan Nie and Ziang Song and Guang Shi and Haoqi Fan}, year={2025}, eprint={2505.14683}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2505.14683}, } @misc{chen2025janusprounifiedmultimodalunderstanding, title={Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling}, author={Xiaokang Chen and Zhiyu Wu and Xingchao Liu and Zizheng Pan and Wen Liu and Zhenda Xie and Xingkai Yu and Chong Ruan}, year={2025}, eprint={2501.17811}, archivePrefix={arXiv}, primaryClass={cs.AI}, url={https://arxiv.org/abs/2501.17811}, } @misc{lipman2023flowmatchinggenerativemodeling, title={Flow Matching for Generative Modeling}, author={Yaron Lipman and Ricky T. Q. Chen and Heli Ben-Hamu and Maximilian Nickel and Matt Le}, year={2023}, eprint={2210.02747}, archivePrefix={arXiv}, primaryClass={cs.LG}, url={https://arxiv.org/abs/2210.02747}, } @misc{schusterbauer2025diff2flowtrainingflowmatching, title={Diff2Flow: Training Flow Matching Models via Diffusion Model Alignment}, author={Johannes Schusterbauer and Ming Gui and Frank Fundel and Björn Ommer}, year={2025}, eprint={2506.02221}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2506.02221}, } %% ----- zuhao start ----- @inproceedings{peebles2023scalable, title={Scalable Diffusion Models with Transformers}, author={Peebles, William and Xie, Saining}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, year={2023}, url={https://arxiv.org/abs/2212.09748}, } @inproceedings{tian2024visual, title={Visual Autoregressive Modeling: Scalable Image Generation via Next-Scale Prediction}, author={Tian, Keyu and Jiang, Yi and Yuan, Zehuan and Peng, Bingyue and Wang, Liwei}, booktitle={Advances in Neural Information Processing Systems (NeurIPS)}, year={2024}, url={https://arxiv.org/abs/2404.02905}, } @article{sun2024autoregressive, title={Autoregressive Model Beats Diffusion: {Llama} for Scalable Image Generation}, author={Sun, Peize and Jiang, Yi and Chen, Shoufa and Zhang, Shilong and Peng, Bingyue and Luo, Ping and Yuan, Zehuan}, journal={arXiv preprint arXiv:2406.06525}, year={2024}, url={https://arxiv.org/abs/2406.06525}, } @misc{sun2024autoregressivemodelbeatsdiffusion, title={Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation}, author={Peize Sun and Yi Jiang and Shoufa Chen and Shilong Zhang and Bingyue Peng and Ping Luo and Zehuan Yuan}, year={2024}, eprint={2406.06525}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2406.06525}, } @article{zou2025omnimamba, title={{OmniMamba}: Efficient and Unified Multimodal Understanding and Generation via State Space Models}, author={Zou, Jialv and Liao, Bencheng and Zhang, Qian and Liu, Wenyu and Wang, Xinggang}, journal={arXiv preprint arXiv:2503.08686}, year={2025}, url={https://arxiv.org/abs/2503.08686}, } @article{wang2025bridge, title={Growing Visual Generative Capacity for Pre-Trained {MLLMs}}, author={Wang, Hanyu and Han, Jiaming and Yang, Ziyan and Zhao, Qi and Lin, Shanchuan and Yue, Xiangyu and Shrivastava, Abhinav and Yang, Zhenheng and Chen, Hao}, journal={arXiv preprint arXiv:2510.01546}, year={2025}, url={https://arxiv.org/abs/2510.01546}, } @article{chen2025blip3onext, title={{BLIP3o-NEXT}: Next Frontier of Native Image Generation}, author={Chen, Jiuhai and Xue, Le and Xu, Zhiyang and Pan, Xichen and Yang, Shusheng and Qin, Can and Yan, An and Zhou, Honglu and Chen, Zeyuan and Huang, Lifu and Zhou, Tianyi and Li, Junnan and Savarese, Silvio and Xiong, Caiming and Xu, Ran}, journal={arXiv preprint arXiv:2510.15857}, year={2025}, url={https://arxiv.org/abs/2510.15857}, } @article{li2025back, title={Back to Basics: Let Denoising Generative Models Denoise}, author={Li, Tianhong and He, Kaiming}, journal={arXiv preprint arXiv:2511.13720}, year={2025}, url={https://arxiv.org/abs/2511.13720}, } @article{huang2025r3gan, title={The {GAN} is Dead; Long Live the {GAN}! {A} Modern {GAN} Baseline}, author={Huang, Yiwen and Gokaslan, Aaron and Kuleshov, Volodymyr and Tompkin, James}, journal={arXiv preprint arXiv:2501.05441}, year={2025}, url={https://arxiv.org/abs/2501.05441}, } @article{zhang2025markov, title={Markovian Scale Prediction: {A} New Era of Visual Autoregressive Generation}, author={Zhang, Yu and Liu, Jingyi and Shi, Yiwei and Zhang, Qi and Miao, Duoqian and Wang, Changwei and Cao, Longbing}, journal={arXiv preprint arXiv:2511.23334}, year={2025}, url={https://arxiv.org/abs/2511.23334}, } %% ===== Encoder / Tokenizer ===== @article{zheng2025rae, title={Diffusion Transformers with Representation Autoencoders}, author={Zheng, Boyang and Ma, Nanye and Tong, Shengbang and Xie, Saining}, journal={arXiv preprint arXiv:2510.11690}, year={2025}, url={https://arxiv.org/abs/2510.11690}, } @article{shi2025svg, title={Latent Diffusion Model without Variational Autoencoder}, author={Shi, Minglei and Wang, Haolin and Zheng, Wenzhao and Yuan, Ziyang and Wu, Xiaoshi and Wang, Xintao and Wan, Pengfei and Zhou, Jie and Lu, Jiwen}, journal={arXiv preprint arXiv:2510.15301}, year={2025}, url={https://arxiv.org/abs/2510.15301}, } @article{yu2024representation, title={Representation Alignment for Generation: Training Diffusion Transformers Is Easier Than You Think}, author={Yu, Sihyun and Kwak, Sangkyung and Jang, Huiwon and Jeong, Jongheon and Huang, Jonathan and Shin, Jinwoo and Xie, Saining}, journal={arXiv preprint arXiv:2410.06940}, year={2024}, url={https://arxiv.org/abs/2410.06940}, } @article{han2025vision, title={Vision as a Dialect: Unifying Visual Understanding and Generation via Text-Aligned Representations}, author={Han, Jiaming and Chen, Hao and Zhao, Yang and Wang, Hanyu and Zhao, Qi and Yang, Ziyan and He, Hao and Yue, Xiangyu and Jiang, Lu}, journal={arXiv preprint arXiv:2506.18898}, year={2025}, url={https://arxiv.org/abs/2506.18898}, } @article{fan2025prism, title={The Prism Hypothesis: Harmonizing Semantic and Pixel Representations via Unified Autoencoding}, author={Fan, Weichen and Diao, Haiwen and Wang, Quan and Lin, Dahua and Liu, Ziwei}, journal={arXiv preprint arXiv:2512.19693}, year={2025}, url={https://arxiv.org/abs/2512.19693}, } @article{yan2025uae, title={Unified Multimodal Model as Auto-Encoder}, author={Yan, Zhiyuan and Lin, Kaiqing and Li, Zongjian and Ye, Junyan and Han, Hui and Wang, Zhendong and Liu, Hao and Lin, Bin and Li, Hao and Xu, Xue and Xiao, Xinyan and Wang, Jingdong and Wang, Haifeng and Yuan, Li}, journal={arXiv preprint arXiv:2509.09666}, year={2025}, url={https://arxiv.org/abs/2509.09666}, } %% ===== Layout-to-Image ===== @inproceedings{li2023gligen, title={Gligen: Open-set grounded text-to-image generation}, author={Li, Yuheng and Liu, Haotian and Wu, Qingyang and Mu, Fangzhou and Yang, Jianwei and Gao, Jianfeng and Li, Chunyuan and Lee, Yong Jae}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={22511--22521}, year={2023} } @inproceedings{wang2024instancediffusion, title={Instancediffusion: Instance-level control for image generation}, author={Wang, Xudong and Darrell, Trevor and Rambhatla, Sai Saketh and Girdhar, Rohit and Misra, Ishan}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={6232--6242}, year={2024} } @inproceedings{zhou2024migc, title={Migc: Multi-instance generation controller for text-to-image synthesis}, author={Zhou, Dewei and Li, You and Ma, Fan and Zhang, Xiaoting and Yang, Yi}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={6818--6828}, year={2024} } @inproceedings{zhou2025dreamrenderer, title={Dreamrenderer: Taming multi-instance attribute control in large-scale text-to-image models}, author={Zhou, Dewei and Li, Mingwei and Yang, Zongxin and Yang, Yi}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={16712--16722}, year={2025} } %% ===== Condition Module ===== @inproceedings{zhang2023controlnet, title={Adding Conditional Control to Text-to-Image Diffusion Models}, author={Zhang, Lvmin and Rao, Anyi and Agrawala, Maneesh}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, year={2023}, url={https://arxiv.org/abs/2302.05543}, } @article{ho2022classifier, title={Classifier-Free Diffusion Guidance}, author={Ho, Jonathan and Salimans, Tim}, journal={arXiv preprint arXiv:2207.12598}, year={2022}, url={https://arxiv.org/abs/2207.12598}, } @article{yang2025mmada, title={{MMaDA}: Multimodal Large Diffusion Language Models}, author={Yang, Ling and Tian, Ye and Li, Bowen and Zhang, Xinchen and Shen, Ke and Tong, Yunhai and Wang, Mengdi}, journal={arXiv preprint arXiv:2505.15809}, year={2025}, url={https://arxiv.org/abs/2505.15809}, } %% ===== Multimodal Fusion ===== @article{xie2024showo, title={Show-o: One Single Transformer to Unify Multimodal Understanding and Generation}, author={Xie, Jinheng and Mao, Weijia and Bai, Zechen and Zhang, David Junhao and Wang, Weihao and Lin, Kevin Qinghong and Gu, Yuchao and Chen, Zhijie and Yang, Zhenheng and Shou, Mike Zheng}, journal={arXiv preprint arXiv:2408.12528}, year={2024}, url={https://arxiv.org/abs/2408.12528}, } @article{zhou2024transfusion, title={Transfusion: Predict the Next Token and Diffuse Images with One Multi-Modal Model}, author={Zhou, Chunting and Yu, Lili and Babu, Arun and Tirumala, Kushal and Yasunaga, Michihiro and Shamis, Leonid and Kahn, Jacob and Ma, Xuezhe and Zettlemoyer, Luke and Levy, Omer}, journal={arXiv preprint arXiv:2408.11039}, year={2024}, url={https://arxiv.org/abs/2408.11039}, } @article{geng2025xomni, title={{X-Omni}: Reinforcement Learning Makes Discrete Autoregressive Image Generative Models Great Again}, author={Geng, Zigang and Wang, Yibing and Ma, Yeyao and Li, Chen and Rao, Yongming and Gu, Shuyang and Zhong, Zhao and Lu, Qinglin and Hu, Han and Zhang, Xiaosong and Wang, Di and Jiang, Jie}, journal={arXiv preprint arXiv:2507.22058}, year={2025}, url={https://arxiv.org/abs/2507.22058}, } %% ===== 3.3(2) Unified Understanding & Generation ===== @article{chameleon2024, title={Chameleon: Mixed-Modal Early-Fusion Foundation Models}, author={{Chameleon Team}}, journal={arXiv preprint arXiv:2405.09818}, year={2024}, url={https://arxiv.org/abs/2405.09818}, } @misc{chameleonteam2025chameleonmixedmodalearlyfusionfoundation, title={Chameleon: Mixed-Modal Early-Fusion Foundation Models}, author={Chameleon Team}, year={2025}, eprint={2405.09818}, archivePrefix={arXiv}, primaryClass={cs.CL}, url={https://arxiv.org/abs/2405.09818}, } @article{wang2024emu3, title={Emu3: Next-Token Prediction is All You Need}, author={Wang, Xinlong and Zhang, Xiaosong and Luo, Zhengxiong and Sun, Quan and Cui, Yufeng and Wang, Jinsheng and Zhang, Fan and Wang, Yueze and Li, Zhen and Yu, Qiying and Zhao, Yingli and Ao, Yulong and Min, Xuebin and Li, Tao and Wu, Boya and Zhao, Bo and Zhang, Bowen and Wang, Liangdong and Liu, Guang and He, Zheqi and Yang, Xi and Liu, Jingjing and Lin, Yonghua and Huang, Tiejun and Wang, Zhongyuan}, journal={arXiv preprint arXiv:2409.18869}, year={2024}, url={https://arxiv.org/abs/2409.18869}, } @article{ma2024janusflow, title={{JanusFlow}: Harmonizing Autoregression and Rectified Flow for Unified Multimodal Understanding and Generation}, author={Ma, Yiyang and Liu, Xingchao and Chen, Xiaokang and Liu, Wen and Wu, Chengyue and Wu, Zhiyu and Pan, Zizheng and Xie, Zhenda and Zhang, Haowei and Yu, Xingkai and Zhao, Liang and Wang, Yisong and Liu, Jiaying and Ruan, Chong}, journal={arXiv preprint arXiv:2411.07975}, year={2024}, url={https://arxiv.org/abs/2411.07975}, } @article{xie2025showo2, title={Show-o2: Improved Native Unified Multimodal Models}, author={Xie, Jinheng and Yang, Zhenheng and Shou, Mike Zheng}, journal={arXiv preprint arXiv:2506.15564}, year={2025}, url={https://arxiv.org/abs/2506.15564}, } @article{liu2023llava, title={Visual Instruction Tuning}, author={Liu, Haotian and Li, Chunyuan and Wu, Qingyang and Lee, Yong Jae}, journal={arXiv preprint arXiv:2304.08485}, year={2023}, url={https://arxiv.org/abs/2304.08485}, } @article{pan2025metaqueries, title={Transfer between Modalities with {MetaQueries}}, author={Pan, Xichen and Shukla, Satya Narayan and Singh, Aashu and Zhao, Zhuokai and Mishra, Shlok Kumar and Wang, Jialiang and Xu, Zhiyang and Chen, Jiuhai and Li, Kunpeng and Juefei-Xu, Felix and Hou, Ji and Xie, Saining}, journal={arXiv preprint arXiv:2504.06256}, year={2025}, url={https://arxiv.org/abs/2504.06256}, } @article{qu2024tokenflow, title={{TokenFlow}: Unified Image Tokenizer for Multimodal Understanding and Generation}, author={Qu, Liao and Zhang, Huichao and Liu, Yiheng and Wang, Xu and Jiang, Yi and Gao, Yiming and Ye, Hu and Du, Daniel K. and Yuan, Zehuan and Wu, Xinglong}, journal={arXiv preprint arXiv:2412.03069}, year={2024}, url={https://arxiv.org/abs/2412.03069}, } @article{yang2025hermesflow, title={{HermesFlow}: Seamlessly Closing the Gap in Multimodal Understanding and Generation}, author={Yang, Ling and Zhang, Xinchen and Tian, Ye and Shang, Chenming and Xu, Minghao and Zhang, Wentao and Cui, Bin}, journal={arXiv preprint arXiv:2502.12148}, year={2025}, url={https://arxiv.org/abs/2502.12148}, } @techreport{cayton2008algorithms, title = {Algorithms for manifold learning}, author = {Cayton, Lawrence}, institution = {University of California, San Diego, Department of Computer Science and Engineering}, year = {2008}, number = {CS2008-0923}, type = {Technical Report}, url = {https://escholarship.org/uc/item/8969r8tc} } @misc{kingma2022autoencodingvariationalbayes, title={Auto-Encoding Variational Bayes}, author={Diederik P Kingma and Max Welling}, year={2022}, eprint={1312.6114}, archivePrefix={arXiv}, primaryClass={stat.ML}, url={https://arxiv.org/abs/1312.6114}, } %% ----- zuhao end ----- %% ----- Shizun start ----- @inproceedings{yu2025anyedit, title={Anyedit: Mastering unified high-quality image editing for any idea}, author={Yu, Qifan and Chow, Wei and Yue, Zhongqi and Pan, Kaihang and Wu, Yang and Wan, Xiaoyang and Li, Juncheng and Tang, Siliang and Zhang, Hanwang and Zhuang, Yueting}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={26125--26135}, year={2025} } @inproceedings{pu2025art, title={Art: Anonymous region transformer for variable multi-layer transparent image generation}, author={Pu, Yifan and Zhao, Yiming and Tang, Zhicong and Yin, Ruihong and Ye, Haoxing and Yuan, Yuhui and Chen, Dong and Bao, Jianmin and Zhang, Sirui and Wang, Yanbin and others}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={7952--7962}, year={2025} } @article{chen2025blip3o, title={Blip3o-next: Next frontier of native image generation}, author={Chen, Jiuhai and Xue, Le and Xu, Zhiyang and Pan, Xichen and Yang, Shusheng and Qin, Can and Yan, An and Zhou, Honglu and Chen, Zeyuan and Huang, Lifu and others}, journal={arXiv preprint arXiv:2510.15857}, year={2025} } @article{chang2025bytemorph, title={Bytemorph: Benchmarking instruction-guided image editing with non-rigid motions}, author={Chang, Di and Cao, Mingdeng and Shi, Yichun and Liu, Bo and Cai, Shengqu and Zhou, Shijie and Huang, Weilin and Wetzstein, Gordon and Soleymani, Mohammad and Wang, Peng}, journal={arXiv preprint arXiv:2506.03107}, year={2025} } @inproceedings{zhang2025diffusion, title={Diffusion-4k: Ultra-high-resolution image synthesis with latent diffusion models}, author={Zhang, Jinjin and Huang, Qiuyu and Liu, Junjie and Guo, Xiefan and Huang, Di}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={23464--23473}, year={2025} } @article{bertazzini2025dragon, title={DRAGON: A Large-Scale Dataset of Realistic Images Generated by Diffusion Models}, author={Bertazzini, Giulia and Baracchi, Daniele and Shullani, Dasara and Echizen, Isao and Piva, Alessandro}, journal={arXiv preprint arXiv:2505.11257}, year={2025} } @article{jang2025dreamgen, title={Dreamgen: Unlocking generalization in robot learning through video world models}, author={Jang, Joel and Ye, Seonghyeon and Lin, Zongyu and Xiang, Jiannan and Bjorck, Johan and Fang, Yu and Hu, Fengyuan and Huang, Spencer and Kundalia, Kaushil and Lin, Yen-Chen and others}, journal={arXiv preprint arXiv:2505.12705}, year={2025} } @inproceedings{zeng2025editworld, title={Editworld: Simulating world dynamics for instruction-following image editing}, author={Zeng, Bohan and Yang, Ling and Liu, Jiaming and Xu, Minghao and Zhang, Yuanxing and Wan, Pengfei and Zhang, Wentao and Yan, Shuicheng}, booktitle={Proceedings of the 33rd ACM International Conference on Multimedia}, pages={12674--12681}, year={2025} } @article{hu2024ella, title={Ella: Equip diffusion models with llm for enhanced semantic alignment}, author={Hu, Xiwei and Wang, Rui and Fang, Yixiao and Fu, Bin and Cheng, Pei and Yu, Gang}, journal={arXiv preprint arXiv:2403.05135}, year={2024} } @article{ma2025veomni, title={Veomni: Scaling any modality model training with model-centric distributed recipe zoo}, author={Ma, Qianli and Zheng, Yaowei and Shi, Zhelun and Zhao, Zhongkai and Jia, Bin and Huang, Ziyue and Lin, Zhiqi and Li, Youjie and Yang, Jiacheng and Peng, Yanghua and others}, journal={arXiv preprint arXiv:2508.02317}, year={2025} } @article{wang2025unirl, title={UniRL-Zero: Reinforcement Learning on Unified Models with Joint Language Model and Diffusion Model Experts}, author={Wang, Fu-Yun and Zhang, Han and Gharbi, Michael and Li, Hongsheng and Park, Taesung}, journal={arXiv preprint arXiv:2510.17937}, year={2025} } @article{wang2026promptrl, title={PromptRL: Prompt Matters in RL for Flow-Based Image Generation}, author={Wang, Fu-Yun and Zhang, Han and Gharbi, Michael and Li, Hongsheng and Park, Taesung}, journal={arXiv preprint arXiv:2602.01382}, year={2026} } @article{zhang2025vsa, title={Vsa: Faster video diffusion with trainable sparse attention}, author={Zhang, Peiyuan and Chen, Yongqi and Huang, Haofeng and Lin, Will and Liu, Zhengzhong and Stoica, Ion and Xing, Eric and Zhang, Hao}, journal={arXiv preprint arXiv:2505.13389}, year={2025} } @article{zheng2024open, title={Open-sora: Democratizing efficient video production for all}, author={Zheng, Zangwei and Peng, Xiangyu and Yang, Tianji and Shen, Chenhui and Li, Shenggui and Liu, Hongxin and Zhou, Yukun and Li, Tianyi and You, Yang}, journal={arXiv preprint arXiv:2412.20404}, year={2024} } @misc{von-platen-etal-2022-diffusers, author = {Patrick von Platen and Suraj Patil and Anton Lozhkov and Pedro Cuenca and Nathan Lambert and Kashif Rasul and Mishig Davaadorj and Dhruv Nair and Sayak Paul and William Berman and Yiyi Xu and Steven Liu and Thomas Wolf}, title = {Diffusers: State-of-the-art diffusion models}, year = {2022}, publisher = {GitHub}, journal = {GitHub repository}, howpublished = {\url{https://github.com/huggingface/diffusers}} } @misc{mmagic2023, title = {{MMagic}: {OpenMMLab} Multimodal Advanced, Generative, and Intelligent Creation Toolbox}, author = {{MMagic Contributors}}, howpublished = {\url{https://github.com/open-mmlab/mmagic}}, year = {2023} } @article{zhuo2025factuality, title={Factuality Matters: When Image Generation and Editing Meet Structured Visuals}, author={Zhuo, Le and Han, Songhao and Pu, Yuandong and Qiu, Boxiang and Paul, Sayak and Liao, Yue and Liu, Yihao and Shao, Jie and Chen, Xi and Liu, Si and others}, journal={arXiv preprint arXiv:2510.05091}, year={2025} } @article{fang2025flux, title={Flux-reason-6m \& prism-bench: A million-scale text-to-image reasoning dataset and comprehensive benchmark}, author={Fang, Rongyao and Yu, Aldrich and Duan, Chengqi and Huang, Linjiang and Bai, Shuai and Cai, Yuxuan and Wang, Kun and Liu, Si and Liu, Xihui and Li, Hongsheng}, journal={arXiv preprint arXiv:2509.09680}, year={2025} } @inproceedings{katara2024gen2sim, title={Gen2sim: Scaling up robot learning in simulation with generative models}, author={Katara, Pushkal and Xian, Zhou and Fragkiadaki, Katerina}, booktitle={2024 IEEE International Conference on Robotics and Automation (ICRA)}, pages={6672--6679}, year={2024}, organization={IEEE} } @article{chen2023genaug, title={Genaug: Retargeting behaviors to unseen situations via generative augmentation}, author={Chen, Zoey and Kiami, Sho and Gupta, Abhishek and Kumar, Vikash}, journal={arXiv preprint arXiv:2302.06671}, year={2023} } @inproceedings{kumari2025generating, title={Generating multi-image synthetic data for text-to-image customization}, author={Kumari, Nupur and Yin, Xi and Zhu, Jun-Yan and Misra, Ishan and Azadi, Samaneh}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={16524--16534}, year={2025} } @article{ghosh2023geneval, title={Geneval: An object-focused framework for evaluating text-to-image alignment}, author={Ghosh, Dhruba and Hajishirzi, Hannaneh and Schmidt, Ludwig}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={52132--52152}, year={2023} } @article{wang2025genexam, title={GenExam: A Multidisciplinary Text-to-Image Exam}, author={Wang, Zhaokai and Yin, Penghao and Zhao, Xiangyu and Tian, Changyao and Qiao, Yu and Wang, Wenhai and Dai, Jifeng and Luo, Gen}, journal={arXiv preprint arXiv:2509.14232}, year={2025} } @article{wang2025gpt, title={Gpt-image-edit-1.5 m: A million-scale, gpt-generated image dataset}, author={Wang, Yuhan and Yang, Siwei and Zhao, Bingchen and Zhang, Letian and Liu, Qing and Zhou, Yuyin and Xie, Cihang}, journal={arXiv preprint arXiv:2507.21033}, year={2025} } @article{hui2024hq, title={Hq-edit: A high-quality dataset for instruction-based image editing}, author={Hui, Mude and Yang, Siwei and Zhao, Bingchen and Shi, Yichun and Wang, Heng and Wang, Peng and Zhou, Yuyin and Xie, Cihang}, journal={arXiv preprint arXiv:2404.09990}, year={2024} } @inproceedings{sani2026imagenworld, title={ImagenWorld: Stress-Testing Image Generation Models with Explainable Human Evaluation on Open-ended Real-World Tasks}, author={Sani, Samin Mahdizadeh and Ku, Max and Jamali, Nima and Sani, Matina Mahdizadeh and Khoshtab, Paria and Sun, Wei-Chieh and Fazel, Parnian and Tam, Zhi Rui and Chong, Thomas and Chan, Edisy Kin Wai and others}, booktitle={The Fourteenth International Conference on Learning Representations}, year={2026} } @article{ye2025imgedit, title={Imgedit: A unified image editing dataset and benchmark}, author={Ye, Yang and He, Xianyi and Li, Zongjian and Lin, Bin and Yuan, Shenghai and Yan, Zhiyuan and Hou, Bohan and Yuan, Li}, journal={arXiv preprint arXiv:2505.20275}, year={2025} } @article{lepert2025masquerade, title={Masquerade: Learning from in-the-wild human videos using data-editing}, author={Lepert, Marion and Fang, Jiaying and Bohg, Jeannette}, journal={arXiv preprint arXiv:2508.09976}, year={2025} } @inproceedings{chen2025multiref, title={MultiRef: Controllable Image Generation with Multiple Visual References}, author={Chen, Ruoxi and Chen, Dongping and Wu, Siyuan and Wang, Sinan and Lang, Shiyun and Sushko, Peter and Jiang, Gaoyang and Wan, Yao and Krishna, Ranjay}, booktitle={Proceedings of the 33rd ACM International Conference on Multimedia}, pages={13325--13331}, year={2025} } @article{pakdamansavoji2025improving, title={Improving Robotic Manipulation Robustness via NICE Scene Surgery}, author={Pakdamansavoji, Sajjad and Pourkeshavarz, Mozhgan and Sigal, Adam and Li, Zhiyuan and Yang, Rui Heng and Rasouli, Amir}, journal={arXiv preprint arXiv:2511.22777}, year={2025} } @inproceedings{wei2024omniedit, title={Omniedit: Building image editing generalist models through specialist supervision}, author={Wei, Cong and Xiong, Zheyang and Ren, Weiming and Du, Xeron and Zhang, Ge and Chen, Wenhu}, booktitle={The Thirteenth International Conference on Learning Representations}, year={2024} } @article{chang2025oneig, title={Oneig-bench: Omni-dimensional nuanced evaluation for image generation}, author={Chang, Jingjing and Fang, Yixiao and Xing, Peng and Wu, Shuhan and Cheng, Wei and Wang, Rui and Zeng, Xianfang and Yu, Gang and Chen, Hai-Bao}, journal={arXiv preprint arXiv:2506.07977}, year={2025} } @article{chen2025opengpt, title={Opengpt-4o-image: A comprehensive dataset for advanced image generation and editing}, author={Chen, Zhihong and Bai, Xuehai and Shi, Yang and Fu, Chaoyou and Zhang, Huanyu and Wang, Haotian and Sun, Xiaoyan and Zhang, Zhang and Wang, Liang and Zhang, Yuanxing and others}, journal={arXiv preprint arXiv:2509.24900}, year={2025} } @article{qian2025pico, title={Pico-banana-400k: A large-scale dataset for text-guided image editing}, author={Qian, Yusu and Bocek-Rivele, Eli and Song, Liangchen and Tong, Jialing and Yang, Yinfei and Lu, Jiasen and Hu, Wenze and Gan, Zhe}, journal={arXiv preprint arXiv:2510.19808}, year={2025} } @inproceedings{sushko2025realedit, title={Realedit: Reddit edits as a large-scale empirical dataset for image transformations}, author={Sushko, Peter and Bharadwaj, Ayana and Lim, Zhi Yang and Ilin, Vasily and Caffee, Ben and Chen, Dongping and Salehi, Mohammadreza and Hsieh, Cheng-Yu and Krishna, Ranjay}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={13403--13413}, year={2025} } @inproceedings{yuan2025roboengine, title={Roboengine: Plug-and-play robot data augmentation with semantic robot segmentation and background generation}, author={Yuan, Chengbo and Joshi, Suraj and Zhu, Shaoting and Su, Hang and Zhao, Hang and Gao, Yang}, booktitle={2025 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)}, pages={7622--7629}, year={2025}, organization={IEEE} } @inproceedings{tao2025robopearls, title={RoboPearls: editable video simulation for robot manipulation}, author={Tao, Tang and Zhang, Likui and Wen, Youpeng and Zhang, Kaidong and Bian, Jia-Wang and Zhou, Xia and Yan, Tianyi and Zhan, Kun and Jia, Peng and Wu, Hefeng and others}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={10118--10129}, year={2025} } @article{yu2023scaling, title={Scaling robot learning with semantically imagined experience}, author={Yu, Tianhe and Xiao, Ted and Stone, Austin and Tompson, Jonathan and Brohan, Anthony and Wang, Su and Singh, Jaspiar and Tan, Clayton and Peralta, Jodilyn and Ichter, Brian and others}, journal={arXiv preprint arXiv:2302.11550}, year={2023} } @article{ge2024seed, title={Seed-data-edit technical report: A hybrid dataset for instructional image editing}, author={Ge, Yuying and Zhao, Sijie and Li, Chen and Ge, Yixiao and Shan, Ying}, journal={arXiv preprint arXiv:2405.04007}, year={2024} } @article{chen2025sharegpt, title={Sharegpt-4o-image: Aligning multimodal models with gpt-4o-level image generation}, author={Chen, Junying and Cai, Zhenyang and Chen, Pengcheng and Chen, Shunian and Ji, Ke and Wang, Xidong and Yang, Yunjin and Wang, Benyou}, journal={arXiv preprint arXiv:2506.18095}, year={2025} } @article{liu2025step1x, title={Step1x-edit: A practical framework for general image editing}, author={Liu, Shiyu and Han, Yucheng and Xing, Peng and Yin, Fukun and Wang, Rui and Cheng, Wei and Liao, Jiaqi and Wang, Yingming and Fu, Honghao and Han, Chunrui and others}, journal={arXiv preprint arXiv:2504.17761}, year={2025} } @article{wang2025textatlas5m, title={Textatlas5m: A large-scale dataset for dense text image generation}, author={Wang, Alex Jinpeng and Mao, Dongxing and Zhang, Jiawei and Han, Weiming and Dong, Zhuobai and Li, Linjie and Lin, Yiqi and Yang, Zhengyuan and Qin, Libo and Zhang, Fuwei and others}, journal={arXiv preprint arXiv:2502.07870}, year={2025} } @article{wei2025tiif, title={TIIF-Bench: How Does Your T2I Model Follow Your Instructions?}, author={Wei, Xinyu and Zhang, Jinrui and Wang, Zeqing and Wei, Hongyang and Guo, Zhen and Zhang, Lei}, journal={arXiv preprint arXiv:2506.02161}, year={2025} } @article{zhao2024ultraedit, title={Ultraedit: Instruction-based fine-grained image editing at scale}, author={Zhao, Haozhe and Ma, Xiaojian and Chen, Liang and Si, Shuzheng and Wu, Rujie and An, Kaikai and Yu, Peiyu and Zhang, Minjia and Li, Qing and Chang, Baobao}, journal={Advances in Neural Information Processing Systems}, volume={37}, pages={3058--3093}, year={2024} } @article{taesiri2025understanding, title={Understanding Generative AI Capabilities in Everyday Image Editing Tasks}, author={Taesiri, Mohammad Reza and Collins, Brandon and Bolton, Logan and Lai, Viet Dac and Dernoncourt, Franck and Bui, Trung and Nguyen, Anh Totti}, journal={arXiv preprint arXiv:2505.16181}, year={2025} } @article{geng2025x, title={X-omni: Reinforcement learning makes discrete autoregressive image generative models great again}, author={Geng, Zigang and Wang, Yibing and Ma, Yeyao and Li, Chen and Rao, Yongming and Gu, Shuyang and Zhong, Zhao and Lu, Qinglin and Hu, Han and Zhang, Xiaosong and others}, journal={arXiv preprint arXiv:2507.22058}, year={2025} } @article{huang2023t2i, title={T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation}, author={Huang, Kaiyi and Sun, Kaiyue and Xie, Enze and Li, Zhenguo and Liu, Xihui}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={78723--78747}, year={2023} } @inproceedings{kumari2023multi, title={Multi-concept customization of text-to-image diffusion}, author={Kumari, Nupur and Zhang, Bingliang and Zhang, Richard and Shechtman, Eli and Zhu, Jun-Yan}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={1931--1941}, year={2023} } @article{zhang2023magicbrush, title={Magicbrush: A manually annotated dataset for instruction-guided image editing}, author={Zhang, Kai and Mo, Lingbo and Chen, Wenhu and Sun, Huan and Su, Yu}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={31428--31449}, year={2023} } @inproceedings{hu2023tifa, title={Tifa: Accurate and interpretable text-to-image faithfulness evaluation with question answering}, author={Hu, Yushi and Liu, Benlin and Kasai, Jungo and Wang, Yizhong and Ostendorf, Mari and Krishna, Ranjay and Smith, Noah A}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={20406--20417}, year={2023} } @article{yang2025complexedit, title={Complexedit: Cot-like instruction generation for complexity-controllable image editing benchmark}, author={Yang, Siwei and Hui, Mude and Zhao, Bingchen and Zhou, Yuyin and Ruiz, Nataniel and Xie, Cihang}, journal={arXiv preprint arXiv:2504.13143}, volume={5}, number={1}, year={2025} } @article{wu2025omnigen2, title={Omnigen2: Exploration to advanced multimodal generation}, author={Wu, Chenyuan and Zheng, Pengfei and Yan, Ruiran and Xiao, Shitao and Luo, Xin and Wang, Yueze and Li, Wanli and Jiang, Xiyan and Liu, Yexin and Zhou, Junjie and others}, journal={arXiv preprint arXiv:2506.18871}, year={2025} } @inproceedings{AnyEdit, author = {Qifan Yu and Wei Chow and Zhongqi Yue and Kaihang Pan and Yang Wu and Xiaoyang Wan and Juncheng Li and Siliang Tang and Hanwang Zhang and Yueting Zhuang}, title = {AnyEdit: Mastering Unified High-Quality Image Editing for Any Idea}, booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, month = {June}, year = {2025} } @article{ART_plus_multi-layer, title = {ART: Anonymous Region Transformer for Variable Multi-Layer Transparent Image Generation}, author = {Yifan Pu and Yiming Zhao and Zhicong Tang and Ruihong Yin and Haoxing Ye and Yuhui Yuan and Dong Chen and Jianmin Bao and Sirui Zhang and Yanbin Wang and Lin Liang and Lijuan Wang and Ji Li and Xiu Li and Zhouhui Lian and Gao Huang and Baining Guo}, journal = {arXiv preprint arXiv:2502.18364}, year = {2025} } @article{BLIP3o-NEXT, title = {BLIP3o-NEXT: Next Frontier of Native Image Generation}, author = {Jiuhai Chen and Le Xue and Zhiyang Xu and Xichen Pan and Shusheng Yang and Can Qin and An Yan and Honglu Zhou and Zeyuan Chen and Lifu Huang and Tianyi Zhou and Junnan Li and Silvio Savarese and Caiming Xiong and Ran Xu}, journal = {arXiv preprint arXiv:2510.15857}, year = {2025} } @article{ByteMorph, title = {ByteMorph: Benchmarking Instruction-Guided Image Editing with Non-Rigid Motions}, author = {Di Chang and Mingdeng Cao and Yichun Shi and Bo Liu and Shengqu Cai and Shijie Zhou and Weilin Huang and Gordon Wetzstein and Mohammad Soleymani and Peng Wang}, journal = {arXiv preprint arXiv:2506.03107}, year = {2025} } @inproceedings{Diffusion-4K, title = {Diffusion-4K: Ultra-High-Resolution Image Synthesis with Latent Diffusion Models}, author = {Jinjin Zhang and Qiuyu Huang and Junjie Liu and Xiefan Guo and Di Huang}, booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, year = {2025} } @article{DRAGON, title = {DRAGON: A Large-Scale Dataset of Realistic Images Generated by Diffusion Models}, author = {Giulia Bertazzini and Daniele Baracchi and Dasara Shullani and Isao Echizen and Alessandro Piva}, journal = {arXiv preprint arXiv:2505.11257}, year = {2025} } @article{DreamGen, title={DreamGen: Unlocking Generalization in Robot Learning through Video World Models}, author={Jang, Joel and Ye, Seonghyeon and Lin, Zongyu and Xiang, Jiannan and Bjorck, Johan and Fang, Yu and Hu, Fengyuan and Huang, Spencer and Kundalia, Kaushil and Lin, Yen-Chen and Magne, Loic and Mandlekar, Ajay and Narayan, Avnish and Tan, You Liang and Wang, Guanzhi and Wang, Jing and Wang, Qi and Xu, Yinzhen and Zeng, Xiaohui and Zheng, Kaiyuan and Zheng, Ruijie and Liu, Ming-Yu and Zettlemoyer, Luke and Fox, Dieter and Kautz, Jan and Reed, Scott and Zhu, Yuke and Fan, Linxi}, journal={arXiv preprint arXiv:2505.12705}, year={2025} } @article{EditWorld, title={EditWorld: Simulating World Dynamics for Instruction-Following Image Editing}, author={Yang, Ling and Zeng, Bohan and Liu, Jiaming and Li, Hong and Xu, Minghao and Zhang, Wentao and Yan, Shuicheng}, journal={arXiv preprint arXiv:2405.14785}, year={2024} } @article{Factuality_Matters, title={Factuality Matters: When Image Generation and Editing Meet Structured Visuals}, author={Zhuo, Le and Han, Songhao and Pu, Yuandong and Qiu, Boxiang and Paul, Sayak and Liao, Yue and Liu, Yihao and Shao, Jie and Chen, Xi and Liu, Si and Li, Hongsheng}, journal={arXiv preprint arXiv:2510.05091}, year={2025} } @article{FLUX-Reason-6M_PRISM-Bench, title={FLUX-Reason-6M \& PRISM-Bench: A Million-Scale Text-to-Image Reasoning Dataset and Comprehensive Benchmark}, author={Fang, Rongyao and Yu, Aldrich and Duan, Chengqi and Huang, Linjiang and Bai, Shuai and Cai, Yuxuan and Wang, Kun and Liu, Si and Liu, Xihui and Li, Hongsheng}, journal={arXiv preprint arXiv:2509.09680}, year={2025} } @article{Gen2Sim, title={Gen2Sim: Scaling up Robot Learning in Simulation with Generative Models}, author={Katara, Pushkal and Xian, Zhou and Fragkiadaki, Katerina}, journal={arXiv preprint arXiv:2310.18308}, year={2023} } @inproceedings{Generating_Multi-Image_Synthetic_Data, title={Generating Multi-Image Synthetic Data for Text-to-Image Customization}, author={Kumari, Nupur and Yin, Xi and Zhu, Jun-Yan and Misra, Ishan and Azadi, Samaneh}, booktitle={International Conference on Computer Vision (ICCV)}, year={2025} } @article{GenEval, title={GenEval: An Object-Focused Framework for Evaluating Text-to-Image Alignment}, author={Ghosh, Dhruba and Hajishirzi, Hannaneh and Schmidt, Ludwig}, journal={arXiv preprint arXiv:2310.11513}, year={2023} } @article{GenExam, title={GenExam: A Multidisciplinary Text-to-Image Exam}, author={Wang, Zhaokai and Yin, Penghao and Zhao, Xiangyu and Tian, Changyao and Qiao, Yu and Wang, Wenhai and Dai, Jifeng and Luo, Gen}, journal={arXiv preprint arXiv:2509.14232}, year={2025} } @article{GPT-IMAGE-EDIT-1.5M, title={GPT-IMAGE-EDIT-1.5M: A Million-Scale, GPT-Generated Image Dataset}, author={Wang, Yuhan and Yang, Siwei and Zhao, Bingchen and Zhang, Letian and Liu, Qing and Zhou, Yuyin and Xie, Cihang}, journal={arXiv preprint arXiv:2507.21033}, year={2025} } @article{HQ-Edit, title={HQ-Edit: A High-Quality Dataset for Instruction-based Image Editing}, author={Hui, Mude and Yang, Siwei and Zhao, Bingchen and Shi, Yichun and Wang, Heng and Wang, Peng and Zhou, Yuyin and Xie, Cihang}, journal={arXiv preprint arXiv:2404.09990}, year={2024} } @misc{ImagenWorld, title={{ImagenWorld}: Stress-Testing Image Generation Models with Explainable Human Evaluation on Open-ended Real-World Tasks}, author={Samin Mahdizadeh Sani and Max Ku and Nima Jamali and Matina Mahdizadeh Sani and Paria Khoshtab and Wei-Chieh Sun and Parnian Fazel and Zhi Rui Tam and Thomas Chong and Edisy Kin Wai Chan and Donald Wai Tong Tsang and Chiao-Wei Hsu and Ting Wai Lam and Ho Yin Sam Ng and Chiafeng Chu and Chak-Wing Mak and Keming Wu and Hiu Tung Wong and Yik Chun Ho and Chi Ruan and Zhuofeng Li and I-Sheng Fang and Shih-Ying Yeh and Ho Kei Cheng and Ping Nie and Wenhu Chen}, year={2025}, doi={10.5281/zenodo.17344183}, url={https://zenodo.org/records/17344183} } @inproceedings{peng2025bizgen, title={Bizgen: Advancing article-level visual text rendering for infographics generation}, author={Peng, Yuyang and Xiao, Shishi and Wu, Keming and Liao, Qisheng and Chen, Bohan and Lin, Kevin and Huang, Danqing and Li, Ji and Yuan, Yuhui}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={23615--23624}, year={2025} } @inproceedings{tsai2025completeme, title={CompleteMe: Reference-based Human Image Completion}, author={Tsai, Yu-Ju and Price, Brian and Liu, Qing and Figueroa, Luis and Pakhomov, Daniil and Ding, Zhihong and Cohen, Scott and Yang, Ming-Hsuan}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={18252--18261}, year={2025} } @article{ImgEdit, title={{ImgEdit}: A Unified Image Editing Dataset and Benchmark}, author={Ye, Yang and He, Xianyi and Li, Zongjian and Lin, Bin and Yuan, Shenghai and Yan, Zhiyuan and Hou, Bohan and Yuan, Li}, journal={arXiv preprint arXiv:2505.20275}, year={2025} } @inproceedings{MultiRef, title={{MultiRef}: Controllable Image Generation with Multiple Visual References}, author={Chen, Ruoxi and Chen, Dongping and Wu, Siyuan and Wang, Sinan and Lang, Shiyun and Sushko, Petr and Jiang, Gaoyang and Wan, Yao and Krishna, Ranjay}, booktitle={Proceedings of the 33rd ACM International Conference on Multimedia}, year={2025}, url={https://dl.acm.org/doi/10.1145/3746027.3758292} } @article{NICE, title={Improving Robotic Manipulation Robustness via {NICE} Scene Surgery}, author={Pakdamansavoji, Sajjad and Pourkeshavarz, Mozhgan and Sigal, Adam and Li, Zhiyuan and Yang, Rui Heng and Rasouli, Amir}, journal={arXiv preprint arXiv:2511.22777}, year={2025} } @article{OmniEdit, title={{OmniEdit}: Building Image Editing Generalist Models Through Specialist Supervision}, author={Wei, Cong and Xiong, Zheyang and Ren, Weiming and Du, Xinrun and Zhang, Ge and Chen, Wenhu}, journal={arXiv preprint arXiv:2411.07199}, year={2024} } @article{OneIG-Bench, title={{OneIG-Bench}: Omni-dimensional Nuanced Evaluation for Image Generation}, author={Chang, Jingjing and Fang, Yixiao and Xing, Peng and Wu, Shuhan and Cheng, Wei and Wang, Rui and Zeng, Xianfang and Yu, Gang and Chen, Hai-Bao}, journal={arXiv preprint arXiv:2506.07977}, year={2025} } @article{OpenGPT-4o-Image, title={OpenGPT-4o-Image: A Comprehensive Dataset for Advanced Image Generation and Editing}, author={Chen, Zhihong and Bai, Xuehai and Shi, Yang and Fu, Chaoyou and Zhang, Huanyu and Wang, Haotian and Sun, Xiaoyan and Zhang, Zhang and Wang, Liang and Zhang, Yuanxing and Wan, Pengfei and Zhang, Yi-Fan}, journal={arXiv preprint arXiv:2509.24900}, year={2025} } @article{Pico-Banana-400K, title={Pico-Banana-400K: A Large-Scale Dataset for Text-Guided Image Editing}, author={Qian, Yusu and Bocek-Rivele, Eli and Song, Liangchen and Tong, Jialing and Yang, Yinfei and Lu, Jiasen and Hu, Wenze and Gan, Zhe}, journal={arXiv preprint arXiv:2510.19808}, year={2025} } @inproceedings{REALEDIT, title={RealEdit: Reddit Edits As a Large-scale Empirical Dataset for Image Transformations}, author={Sushko, Peter and Bharadwaj, Ayana and Lim, Zhi Yang and Ilin, Vasily and Caffee, Ben and Chen, Dongping and Salehi, Mohammadreza and Hsieh, Cheng-Yu and Krishna, Ranjay}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, pages={13403--13413}, year={2025} } @article{RoboEngine, title={RoboEngine: Plug-and-Play Robot Data Augmentation with Semantic Robot Segmentation and Background Generation}, author={Yuan, Chengbo and Joshi, Suraj and Zhu, Shaoting and Su, Hang and Zhao, Hang and Gao, Yang}, journal={arXiv preprint arXiv:2503.18738}, year={2025} } @inproceedings{RoboPearls, title={RoboPearls: Editable Video Simulation for Robot Manipulation}, author={Tang, Tao and Zhang, Likui and Wen, Youpeng and Zhang, Kaidong and Bian, Jia-Wang and Zhou, Xia and Yan, Tianyi and Zhan, Kun and Jia, Peng and Wu, Hefeng and Lin, Liang and Liang, Xiaodan}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)}, year={2025} } @inproceedings{ROSIE, title={Scaling Robot Learning with Semantically Imagined Experience}, author={Yu, Tianhe and Xiao, Ted and Tompson, Jonathan and Stone, Austin and Wang, Su and Brohan, Anthony and Singh, Jaspiar and Tan, Clayton and M, Dee and Peralta, Jodilyn and Hausman, Karol and Ichter, Brian and Xia, Fei}, booktitle={Robotics: Science and Systems (RSS)}, year={2023} } @article{SEED-Data-Edit, title={SEED-Data-Edit Technical Report: A Hybrid Dataset for Instructional Image Editing}, author={Ge, Yuying and Zhao, Sijie and Li, Chen and Ge, Yixiao and Shan, Ying}, journal={arXiv preprint arXiv:2405.04007}, year={2024} } @article{ShareGPT-4o-Image, title={ShareGPT-4o-Image: Aligning Multimodal Models with GPT-4o-Level Image Generation}, author={Junying Chen and Zhenyang Cai and Pengcheng Chen and Shunian Chen and Ke Ji and Xidong Wang and Yunjin Yang and Benyou Wang}, journal={arXiv preprint arXiv:2506.18095}, year={2025} } @article{TextAtlas5M, title={TextAtlas5M: A Large-scale Dataset for Dense Text Image Generation}, author={Alex Jinpeng Wang and Dongxing Mao and Jiawei Zhang and Weiming Han and Zhuobai Dong and Linjie Li and Yiqi Lin and Zhengyuan Yang and Libo Qin and Fuwei Zhang and Lijuan Wang and Min Li}, journal={arXiv preprint arXiv:2502.07870}, year={2025} } @article{TIIF-Bench, title={TIIF-Bench: How Does Your T2I Model Follow Your Instructions?}, author={Xinyu Wei and Jinrui Zhang and Zeqing Wang and Hongyang Wei and Zhenzhao Guo and Lei Zhang}, journal={arXiv preprint arXiv:2506.02161}, year={2025} } @article{UltraEdit, title={UltraEdit: Instruction-based Fine-Grained Image Editing at Scale}, author={Haozhe Zhao and Xiaojian Ma and Liang Chen and Shuzheng Si and Rujie Wu and Kaikai An and Peiyu Yu and Minjia Zhang and Qing Li and Baobao Chang}, journal={arXiv preprint arXiv:2407.05282}, year={2024} } @article{Understanding_GenAI_Image_Editing, title={Understanding Generative AI Capabilities in Everyday Image Editing Tasks}, author={Mohammad Reza Taesiri and Brandon Collins and Logan Bolton and Viet Dac Lai and Franck Dernoncourt and Trung Bui and Anh Totti Nguyen}, journal={arXiv preprint arXiv:2505.16181}, year={2025} } @article{X-Omni, title={X-Omni: Reinforcement Learning Makes Discrete Autoregressive Image Generative Models Great Again}, author={Zigang Geng and Yibing Wang and Yeyao Ma and Chen Li and Yongming Rao and Shuyang Gu and Zhao Zhong and Qinglin Lu and Han Hu and Xiaosong Zhang and Linus and Di Wang and Jie Jiang}, journal={arXiv preprint arXiv:2507.22058}, year={2025} } @inproceedings{lin2014microsoft, title={Microsoft coco: Common objects in context}, author={Lin, Tsung-Yi and Maire, Michael and Belongie, Serge and Hays, James and Perona, Pietro and Ramanan, Deva and Doll{\'a}r, Piotr and Zitnick, C Lawrence}, booktitle={European conference on computer vision}, pages={740--755}, year={2014}, organization={Springer} } @inproceedings{agrawal2019nocaps, title={Nocaps: Novel object captioning at scale}, author={Agrawal, Harsh and Desai, Karan and Wang, Yufei and Chen, Xinlei and Jain, Rishabh and Johnson, Mark and Batra, Dhruv and Parikh, Devi and Lee, Stefan and Anderson, Peter}, booktitle={Proceedings of the IEEE/CVF international conference on computer vision}, pages={8948--8957}, year={2019} } @inproceedings{gurari2018vizwiz, title={Vizwiz grand challenge: Answering visual questions from blind people}, author={Gurari, Danna and Li, Qing and Stangl, Abigale J and Guo, Anhong and Lin, Chi and Grauman, Kristen and Luo, Jiebo and Bigham, Jeffrey P}, booktitle={Proceedings of the IEEE conference on computer vision and pattern recognition}, pages={3608--3617}, year={2018} } @inproceedings{sidorov2020textcaps, title={Textcaps: a dataset for image captioning with reading comprehension}, author={Sidorov, Oleksii and Hu, Ronghang and Rohrbach, Marcus and Singh, Amanpreet}, booktitle={European conference on computer vision}, pages={742--758}, year={2020}, organization={Springer} } @article{schuhmann2022laion, title={Laion-5b: An open large-scale dataset for training next generation image-text models}, author={Schuhmann, Christoph and Beaumont, Romain and Vencu, Richard and Gordon, Cade and Wightman, Ross and Cherti, Mehdi and Coombes, Theo and Katta, Aarush and Mullis, Clayton and Wortsman, Mitchell and others}, journal={Advances in neural information processing systems}, volume={35}, pages={25278--25294}, year={2022} } @inproceedings{deng2009imagenet, title={Imagenet: A large-scale hierarchical image database}, author={Deng, Jia and Dong, Wei and Socher, Richard and Li, Li-Jia and Li, Kai and Fei-Fei, Li}, booktitle={2009 IEEE conference on computer vision and pattern recognition}, pages={248--255}, year={2009}, organization={Ieee} } @misc{kakaobrain2022coyo-700m, title = {COYO-700M: Image-Text Pair Dataset}, author = {Minwoo Byeon and Beomhee Park and Haecheon Kim and Sungjun Lee and Woonhyuk Baek and Saehoon Kim}, year = {2022}, howpublished = {\url{https://github.com/kakaobrain/coyo-dataset}}, } @article{gadre2023datacomp, title={Datacomp: In search of the next generation of multimodal datasets}, author={Gadre, Samir Yitzhak and Ilharco, Gabriel and Fang, Alex and Hayase, Jonathan and Smyrnis, Georgios and Nguyen, Thao and Marten, Ryan and Wortsman, Mitchell and Ghosh, Dhruba and Zhang, Jieyu and others}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={27092--27112}, year={2023} } @inproceedings{yu2023mvimgnet, title={Mvimgnet: A large-scale dataset of multi-view images}, author={Yu, Xianggang and Xu, Mutian and Zhang, Yidan and Liu, Haolin and Ye, Chongjie and Wu, Yushuang and Yan, Zizheng and Zhu, Chenming and Xiong, Zhangyang and Liang, Tianyou and others}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={9150--9161}, year={2023} } @inproceedings{sharma2018conceptual, title={Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning}, author={Sharma, Piyush and Ding, Nan and Goodman, Sebastian and Soricut, Radu}, booktitle={Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)}, pages={2556--2565}, year={2018} } @article{wang2023internvid, title={Internvid: A large-scale video-text dataset for multimodal understanding and generation}, author={Wang, Yi and He, Yinan and Li, Yizhuo and Li, Kunchang and Yu, Jiashuo and Ma, Xin and Li, Xinhao and Chen, Guo and Chen, Xinyuan and Wang, Yaohui and others}, journal={arXiv preprint arXiv:2307.06942}, year={2023} } @inproceedings{damen2018scaling, title={Scaling egocentric vision: The epic-kitchens dataset}, author={Damen, Dima and Doughty, Hazel and Farinella, Giovanni Maria and Fidler, Sanja and Furnari, Antonino and Kazakos, Evangelos and Moltisanti, Davide and Munro, Jonathan and Perrett, Toby and Price, Will and others}, booktitle={Proceedings of the European conference on computer vision (ECCV)}, pages={720--736}, year={2018} } @inproceedings{deitke2023objaverse, title={Objaverse: A universe of annotated 3d objects}, author={Deitke, Matt and Schwenk, Dustin and Salvador, Jordi and Weihs, Luca and Michel, Oscar and VanderBilt, Eli and Schmidt, Ludwig and Ehsani, Kiana and Kembhavi, Aniruddha and Farhadi, Ali}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={13142--13153}, year={2023} } @inproceedings{liu2023zero, title={Zero-1-to-3: Zero-shot one image to 3d object}, author={Liu, Ruoshi and Wu, Rundi and Van Hoorick, Basile and Tokmakov, Pavel and Zakharov, Sergey and Vondrick, Carl}, booktitle={Proceedings of the IEEE/CVF international conference on computer vision}, pages={9298--9309}, year={2023} } @article{ma2024shapesplat, title={Shapesplat: A large-scale dataset of gaussian splats and their self-supervised pretraining}, author={Ma, Qi and Li, Yue and Ren, Bin and Sebe, Nicu and Konukoglu, Ender and Gevers, Theo and Van Gool, Luc and Paudel, Danda Pani}, journal={arXiv preprint arXiv:2408.10906}, year={2024} } @inproceedings{liu24uco3d, Author = {Liu, Xingchen and Tayal, Piyush and Wang, Jianyuan and Zarzar, Jesus and Monnier, Tom and Tertikas, Konstantinos and Duan, Jiali and Toisoul, Antoine and Zhang, Jason Y. and Neverova, Natalia and Vedaldi, Andrea and Shapovalov, Roman and Novotny, David}, Booktitle = {arXiv}, Title = {UnCommon Objects in 3D}, Year = {2025}, } @article{sun2023journeydb, title={Journeydb: A benchmark for generative image understanding}, author={Sun, Keqiang and Pan, Junting and Ge, Yuying and Li, Hao and Duan, Haodong and Wu, Xiaoshi and Zhang, Renrui and Zhou, Aojun and Qin, Zipeng and Wang, Yi and others}, journal={Advances in neural information processing systems}, volume={36}, pages={49659--49678}, year={2023} } @article{young2014image, title={From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions}, author={Young, Peter and Lai, Alice and Hodosh, Micah and Hockenmaier, Julia}, journal={Transactions of the Association for Computational Linguistics}, volume={2}, pages={67--78}, year={2014}, publisher={MIT Press} } @article{GenAI-Arena, title={Genai arena: An open evaluation platform for generative models}, author={Jiang, Dongfu and Ku, Max and Li, Tianle and Ni, Yuansheng and Sun, Shizhuo and Fan, Rongqi and Chen, Wenhu}, journal={Advances in Neural Information Processing Systems}, volume={37}, pages={79889--79908}, year={2024} } @inproceedings{ChatbotArena, title={Chatbot arena: An open platform for evaluating llms by human preference}, author={Chiang, Wei-Lin and Zheng, Lianmin and Sheng, Ying and Angelopoulos, Anastasios Nikolas and Li, Tianle and Li, Dacheng and Zhu, Banghua and Zhang, Hao and Jordan, Michael and Gonzalez, Joseph E and others}, booktitle={Forty-first International Conference on Machine Learning}, year={2024} } %% ----- Shizun end ----- %%% ---- haowei distillation ---- % Core Distillation @inproceedings{salimans2022progressive, title={Progressive distillation for fast sampling of diffusion models}, author={Salimans, Tim and Ho, Jonathan and Dhariwal, Prafulla and others}, booktitle={International Conference on Machine Learning}, year={2022} } % Trajectory Matching @article{geng2025mean, title={Mean flows for one-step generative modeling}, author={Geng, Zhengyang and Deng, Mingyang and Bai, Xingjian and Kolter, J Zico and He, Kaiming}, journal={arXiv preprint arXiv:2505.13447}, year={2025} } @article{frans2024one, title={One step diffusion via shortcut models}, author={Frans, Kevin and Hafner, Danijar and Levine, Sergey and Abbeel, Pieter}, journal={arXiv preprint arXiv:2410.12557}, year={2024} } % Consistency Models @inproceedings{song2023consistency, title={Consistency Models}, author={Song, Yang and Dhariwal, Prafulla and Chen, Mark and Sutskever, Ilya}, booktitle={International Conference on Machine Learning}, pages={32211--32252}, year={2023}, organization={PMLR} } @article{lu2024simplifying, title={Simplifying, stabilizing and scaling continuous-time consistency models}, author={Lu, Cheng and Song, Yang}, journal={arXiv preprint arXiv:2410.11081}, year={2024} } @article{zheng2025rcm, title={Large Scale Diffusion Distillation via Score-Regularized Continuous-Time Consistency}, author={Zheng, Kaiwen and Wang, Yuji and Ma, Qianli and Chen, Huayu and Zhang, Jintao and Balaji, Yogesh and Chen, Jianfei and Liu, Ming-Yu and Zhu, Jun and Zhang, Qinsheng}, journal={arXiv preprint arXiv:2510.08431}, year={2025} } % Distribution Matching @inproceedings{sauer2024adversarial, title={Adversarial diffusion distillation}, author={Sauer, Axel and Lorenz, Dominik and Blattmann, Andreas and Rombach, Robin}, booktitle={European Conference on Computer Vision}, pages={87--103}, year={2024}, organization={Springer} } @inproceedings{yin2024one, title={One-step diffusion with distribution matching distillation}, author={Yin, Tianwei and Gharbi, Micha{\"e}l and Zhang, Richard and Shechtman, Eli and Durand, Fredo and Freeman, William T and Park, Taesung}, booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pages={6613--6623}, year={2024} } @article{yin2024improved, title={Improved distribution matching distillation for fast image synthesis}, author={Yin, Tianwei and Gharbi, Micha{\"e}l and Park, Taesung and Zhang, Richard and Shechtman, Eli and Durand, Fredo and Freeman, Bill}, journal={Advances in neural information processing systems}, volume={37}, pages={47455--47487}, year={2024} } % Video Distillation @article{nie2026tmd, title={Transition Matching Distillation for Fast Video Generation}, author={Nie, Weili and Berner, Julius and Ma, Nanye and Liu, Chao and Xie, Saining and Vahdat, Arash}, journal={arXiv preprint arXiv:2601.09881}, year={2026} } %%$ --- haowei distillation end ----- %%% ---- qijie start ---- @article{li2024mar, title={Autoregressive image generation without vector quantization}, author={Li, Tianhong and Tian, Yonglong and Li, He and Deng, Mingyang and He, Kaiming}, journal={Advances in Neural Information Processing Systems}, volume={37}, pages={56424--56445}, year={2024} } @article{team2025nextstep1, title={Nextstep-1: Toward autoregressive image generation with continuous tokens at scale}, author={Team, NextStep and Han, Chunrui and Li, Guopeng and Wu, Jingwei and Sun, Quan and Cai, Yan and Peng, Yuang and Ge, Zheng and Zhou, Deyu and Tang, Haomiao and others}, journal={arXiv preprint arXiv:2508.10711}, year={2025} } @inproceedings{bevilacqua2012set5, title={Low-complexity single-image super-resolution based on nonnegative neighbor embedding}, author={Bevilacqua, Marco and Roumy, Aline and Guillemot, Christine and Alberi-Morel, Marie Line}, booktitle={Proceedings of the British Machine Vision Conference (BMVC)}, year={2012} } @article{wei2018lol, title={Deep retinex decomposition for low-light enhancement}, author={Wei, Chen and Wang, Wenjing and Yang, Wenhan and Liu, Jiaying}, journal={arXiv preprint arXiv:1808.04560}, year={2018} } @inproceedings{martin2001bsd68, title={A database of human segmented natural images and its application to evaluating segmentation algorithms and measuring ecological statistics}, author={Martin, David and Fowlkes, Charless and Tal, Doron and Malik, Jitendra}, booktitle={Proceedings eighth IEEE international conference on computer vision. ICCV 2001}, volume={2}, pages={416--423}, year={2001}, organization={Ieee} } @inproceedings{yang2017rain100h, title={Deep joint rain detection and removal from a single image}, author={Yang, Wenhan and Tan, Robby T and Feng, Jiashi and Liu, Jiaying and Guo, Zongming and Yan, Shuicheng}, booktitle={Proceedings of the IEEE conference on computer vision and pattern recognition}, pages={1357--1366}, year={2017} } @inproceedings{nah2017gopro, title={Deep multi-scale convolutional neural network for dynamic scene deblurring}, author={Nah, Seungjun and Hyun Kim, Tae and Mu Lee, Kyoung}, booktitle={Proceedings of the IEEE conference on computer vision and pattern recognition}, pages={3883--3891}, year={2017} } @article{niu2025wise, title={Wise: A world knowledge-informed semantic evaluation for text-to-image generation}, author={Niu, Yuwei and Ning, Munan and Zheng, Mengren and Jin, Weiyang and Lin, Bin and Jin, Peng and Liao, Jiaqi and Feng, Chaoran and Ning, Kunpeng and Zhu, Bin and others}, journal={arXiv preprint arXiv:2503.07265}, year={2025} } @inproceedings{chen2025r2ibench, title={R2i-bench: Benchmarking reasoning-driven text-to-image generation}, author={Chen, Kaijie and Lin, Zihao and Xu, Zhiyang and Shen, Ying and Yao, Yuguang and Rimchala, Joy and Zhang, Jiaxin and Huang, Lifu}, booktitle={Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing}, pages={12606--12641}, year={2025} } @article{meng2024phybench, title={Phybench: A physical commonsense benchmark for evaluating text-to-image models}, author={Meng, Fanqing and Shao, Wenqi and Luo, Lixin and Wang, Yahong and Chen, Yiran and Lu, Quanfeng and Yang, Yue and Yang, Tianshuo and Zhang, Kaipeng and Qiao, Yu and others}, journal={arXiv preprint arXiv:2406.11802}, year={2024} } %%% ---- qijie end ---- %%% ---- sicong start ---- @article{khazatsky2024droid, title={Droid: A large-scale in-the-wild robot manipulation dataset}, author={Khazatsky, Alexander and Pertsch, Karl and Nair, Suraj and Balakrishna, Ashwin and Dasari, Sudeep and Karamcheti, Siddharth and Nasiriany, Soroush and Srirama, Mohan Kumar and Chen, Lawrence Yunliang and Ellis, Kirsty and others}, journal={arXiv preprint arXiv:2403.12945}, year={2024} } @inproceedings{openx2024rtx, title={Open x-embodiment: Robotic learning datasets and rt-x models: Open x-embodiment collaboration 0}, author={O’Neill, Abby and Rehman, Abdul and Maddukuri, Abhiram and Gupta, Abhishek and Padalkar, Abhishek and Lee, Abraham and Pooley, Acorn and Gupta, Agrim and Mandlekar, Ajay and Jain, Ajinkya and others}, booktitle={2024 IEEE International Conference on Robotics and Automation (ICRA)}, pages={6892--6903}, year={2024}, organization={IEEE} } @inproceedings{walke2023bridgedatav2, title={Bridgedata v2: A dataset for robot learning at scale}, author={Walke, Homer Rich and Black, Kevin and Zhao, Tony Z and Vuong, Quan and Zheng, Chongyi and Hansen-Estruch, Philippe and He, Andre Wang and Myers, Vivek and Kim, Moo Jin and Du, Max and others}, booktitle={Conference on Robot Learning}, pages={1723--1736}, year={2023}, organization={PMLR} } @article{chen2024roviaug, title={Rovi-aug: Robot and viewpoint augmentation for cross-embodiment robot learning}, author={Chen, Lawrence Yunliang and Xu, Chenfeng and Dharmarajan, Karthik and Irshad, Muhammad Zubair and Cheng, Richard and Keutzer, Kurt and Tomizuka, Masayoshi and Vuong, Quan and Goldberg, Ken}, journal={arXiv preprint arXiv:2409.03403}, year={2024} } @article{lepert2025phantom, title={Phantom: Training robots without robots using only human videos}, author={Lepert, Marion and Fang, Jiaying and Bohg, Jeannette}, journal={arXiv preprint arXiv:2503.00779}, year={2025} } @article{li2025h2r, title={H2r: A human-to-robot data augmentation for robot pre-training from videos}, author={Li, Guangrun and Lyu, Yaoxu and Liu, Zhuoyang and Hou, Chengkai and Zhang, Jieyu and Zhang, Shanghang}, journal={arXiv preprint arXiv:2505.11920}, year={2025} } @article{fan2026robopaint, title={RoboPaint: From Human Demonstration to Any Robot and Any View}, author={Fan, Jiacheng and Zhao, Zhiyue and Zhang, Yiqian and Chen, Chao and Wang, Peide and Zhang, Hengdi and Cheng, Zhengxue}, journal={arXiv preprint arXiv:2602.05325}, year={2026} } @article{song2025mitty, title={Mitty: Diffusion-based Human-to-Robot Video Generation}, author={Song, Yiren and Liu, Cheng and Mao, Weijia and Shou, Mike Zheng}, journal={arXiv preprint arXiv:2512.17253}, year={2025} } @article{zhu2024irasim, title={Irasim: Learning interactive real-robot action simulators}, author={Zhu, Fangqi and Wu, Hongtao and Guo, Song and Liu, Yuxiao and Cheang, Chilam and Kong, Tao}, journal={arXiv preprint arXiv:2406.14540}, volume={1}, number={2}, pages={3}, year={2024} } @article{fu2025robomaster, title={Learning video generation for robotic manipulation with collaborative trajectory control}, author={Fu, Xiao and Wang, Xintao and Liu, Xian and Bai, Jianhong and Xu, Runsen and Wan, Pengfei and Zhang, Di and Lin, Dahua}, journal={arXiv preprint arXiv:2506.01943}, year={2025} } @article{zhou2024robodreamer, title={Robodreamer: Learning compositional world models for robot imagination}, author={Zhou, Siyuan and Du, Yilun and Chen, Jiaben and Li, Yandong and Yeung, Dit-Yan and Gan, Chuang}, journal={arXiv preprint arXiv:2404.12377}, year={2024} } @article{barcellona2024dreamtomanipulate, title={Dream to manipulate: Compositional world models empowering robot imitation learning with imagination}, author={Barcellona, Leonardo and Zadaianchuk, Andrii and Allegro, Davide and Papa, Samuele and Ghidoni, Stefano and Gavves, Efstratios}, journal={arXiv preprint arXiv:2412.14957}, year={2024} } @inproceedings{zhao2025cotvla, title={Cot-vla: Visual chain-of-thought reasoning for vision-language-action models}, author={Zhao, Qingqing and Lu, Yao and Kim, Moo Jin and Fu, Zipeng and Zhang, Zhuoyang and Wu, Yecheng and Li, Zhaoshuo and Ma, Qianli and Han, Song and Finn, Chelsea and others}, booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference}, pages={1702--1713}, year={2025} } @article{tian2024predictive, title={Predictive inverse dynamics models are scalable learners for robotic manipulation}, author={Tian, Yang and Yang, Sizhe and Zeng, Jia and Wang, Ping and Lin, Dahua and Dong, Hao and Pang, Jiangmiao}, journal={arXiv preprint arXiv:2412.15109}, year={2024} } @article{ko2024actionlessvideos, title={Learning to act from actionless videos through dense correspondences}, author={Ko, Po-Chen and Mao, Jiayuan and Du, Yilun and Sun, Shao-Hua and Tenenbaum, Joshua B}, journal={arXiv preprint arXiv:2310.08576}, year={2023} } @article{du2023unipi, title={Learning universal policies via text-guided video generation}, author={Du, Yilun and Yang, Sherry and Dai, Bo and Dai, Hanjun and Nachum, Ofir and Tenenbaum, Josh and Schuurmans, Dale and Abbeel, Pieter}, journal={Advances in neural information processing systems}, volume={36}, pages={9156--9172}, year={2023} } @article{hu2024vpp, title={Video prediction policy: A generalist robot policy with predictive visual representations}, author={Hu, Yucheng and Guo, Yanjiang and Wang, Pengchao and Chen, Xiaoyu and Wang, Yen-Jen and Zhang, Jianke and Sreenath, Koushil and Lu, Chaochao and Chen, Jianyu}, journal={arXiv preprint arXiv:2412.14803}, year={2024} } @article{li2025uva, title={Unified video action model}, author={Li, Shuang and Gao, Yihuai and Sadigh, Dorsa and Song, Shuran}, journal={arXiv preprint arXiv:2503.00200}, year={2025} } @article{zhu2025uwm, title={Unified world models: Coupling video and action diffusion for pretraining on large robotic datasets}, author={Zhu, Chuning and Yu, Raymond and Feng, Siyuan and Burchfiel, Benjamin and Shah, Paarth and Gupta, Abhishek}, journal={arXiv preprint arXiv:2504.02792}, year={2025} } @article{chen2025largevideoplanner, title={Large video planner enables generalizable robot control}, author={Chen, Boyuan and Zhang, Tianyuan and Geng, Haoran and Song, Kiwhan and Zhang, Caiyi and Li, Peihao and Freeman, William T and Malik, Jitendra and Abbeel, Pieter and Tedrake, Russ and others}, journal={arXiv preprint arXiv:2512.15840}, year={2025} } @article{liang2025videopolicy, title={Video generators are robot policies}, author={Liang, Junbang and Tokmakov, Pavel and Liu, Ruoshi and Sudhakar, Sruthi and Shah, Paarth and Ambrus, Rares and Vondrick, Carl}, journal={arXiv preprint arXiv:2508.00795}, year={2025} } @article{shen2025videovla, title={Videovla: Video generators can be generalizable robot manipulators}, author={Shen, Yichao and Wei, Fangyun and Du, Zhiying and Liang, Yaobo and Lu, Yan and Yang, Jiaolong and Zheng, Nanning and Guo, Baining}, journal={arXiv preprint arXiv:2512.06963}, year={2025} } @article{meietal2026videoroboticsurvey, title={Video Generation Models in Robotics: Applications, Research Challenges, Future Directions}, author={Mei, Zhiting and Yin, Tenny and Shorinwa, Ola and Badithela, Apurva and Zheng, Zhonghe and Bruno, Joseph and Bland, Madison and Zha, Lihan and Hancock, Asher and Fisac, Jaime Fern{\'a}ndez and Dames, Philip and Majumdar, Anirudha}, journal={arXiv preprint arXiv:2601.07823}, year={2026} } @article{ping2026flowfactory, title={Flow-Factory: A Unified Framework for Reinforcement Learning in Flow-Matching Models}, author={Bowen Ping and Chengyou Jia and Minnan Luo and Hangwei Qian and Ivor Tsang}, journal={arXiv preprint arXiv:2602.12529}, year={2026} } @article{zhang2025fast, title={Fast video generation with sliding tile attention}, author={Zhang, Peiyuan and Chen, Yongqi and Su, Runlong and Ding, Hangliang and Stoica, Ion and Liu, Zhengzhong and Zhang, Hao}, journal={arXiv preprint arXiv:2502.04507}, year={2025} } % ------ world model / interactive video ------ @article{ha2018worldmodels, title={Recurrent world models facilitate policy evolution}, author={Ha, David and Schmidhuber, J{\"u}rgen}, journal={Advances in Neural Information Processing Systems}, volume={31}, year={2018} } @inproceedings{bruce2024genie, title={Genie: Generative Interactive Environments}, author={Bruce, Jake and Dennis, Michael and Edwards, Ashley and Parker-Holder, Jack and Shi, Yuge and Hughes, Edward and Lai, Matthew and Mavalankar, Aditi and Steigerwald, Richie and Apps, Chris and Aytar, Yusuf and Bechtle, Sarah and Behbahani, Feryal and Chan, Stephanie and Heess, Nicolas and Gonzalez, Lucy and Osindero, Simon and Ozair, Sherjil and Reed, Scott and Zhang, Jingwei and Zolna, Konrad and Clune, Jeff and de Freitas, Nando and Singh, Satinder and Rockt{\"a}schel, Tim}, booktitle={Proceedings of the 41st International Conference on Machine Learning}, year={2024} } @article{parker2024genie2, title={Genie 2: A Large-Scale Foundation World Model}, author={Parker-Holder, Jack and Ball, Philip and Bruce, Jake and Dasagi, Vibhavari and Holsheimer, Kristian and Kaplanis, Christos and Moufarek, Alexandre and Scully, Guy and Shar, Jeremy and Shi, Jimmy and Spencer, Stephen and Yung, Jessica and Dennis, Michael and Kenjeyev, Sultan and Long, Shangbang and Mnih, Vlad and Chan, Harris and Gazeau, Maxime and Li, Bonnie and Pardo, Fabio and Wang, Luyu and Zhang, Lei and Besse, Frederic and Harley, Tim and Mitenkova, Anna and Wang, Jane and Clune, Jeff and Hassabis, Demis and Hadsell, Raia and Bolton, Adrian and Singh, Satinder and Rockt{\"a}schel, Tim}, journal={Google DeepMind Blog Post}, year={2024}, note={Available at \url{https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/}} } @article{valevski2024gamenngen, title={Diffusion models are real-time game engines}, author={Valevski, Dani and Leviathan, Yaniv and Arar, Moab and Fruchter, Shlomi}, journal={arXiv preprint arXiv:2408.14837}, year={2024} } @article{alonso2024diamond, title={Diffusion for world modeling: Visual details matter in {A}tari}, author={Alonso, Eloi and Jelley, Adam and Micheli, Vincent and Kanervisto, Anssi and Storkey, Amos and Pearce, Tim and Fleuret, Fran{\c{c}}ois}, journal={Advances in Neural Information Processing Systems}, volume={37}, year={2024} } @article{oasis2024, title={Oasis: A universe in a transformer}, author={Quevedo, Julian and McIntyre, Quinn and Campbell, Spruce and Chen, Xinlei and Wachen, Robert}, journal={Technical Report}, year={2024} } @inproceedings{yang2024unisim, title={Learning interactive real-world simulators}, author={Yang, Sherry and Du, Yilun and Ghasemipour, Kamyar and Tompson, Jonathan and Kaelbling, Leslie Pack and Schuurmans, Dale and Abbeel, Pieter}, booktitle={International Conference on Learning Representations}, year={2024} } @inproceedings{che2025gamegenx, title={{GameGen-X}: Interactive open-world game video generation}, author={Che, Haoxuan and He, Xuanhua and Liu, Quande and Jin, Cheng and Chen, Hao}, booktitle={International Conference on Learning Representations}, year={2025} } @article{hafner2025dreamerv3, title={Mastering diverse control tasks through world models}, author={Hafner, Danijar and Pasukonis, Jurgis and Ba, Jimmy and Lillicrap, Timothy}, journal={Nature}, volume={640}, pages={647--653}, year={2025} } @article{lecun2022jepa, title={A path towards autonomous machine intelligence}, author={LeCun, Yann}, journal={Open Review}, year={2022} } @article{hu2024gaia1, title={{GAIA-1}: A generative world model for autonomous driving}, author={Hu, Anthony and Russell, Lloyd and Yeo, Hudson and Murez, Zak and Fedoseev, George and Kendall, Alex and Shotton, Jamie and Corrado, Gianluca}, journal={arXiv preprint arXiv:2309.17080}, year={2023} } @article{wang2024worlddreamer, title={WorldDreamer: Towards general world models for video generation via predicting masked tokens}, author={Wang, Xiaofeng and Zhu, Zheng and Huang, Guan and Wang, Boyuan and Chen, Xinze and Lu, Jiwen}, journal={arXiv preprint arXiv:2401.09985}, year={2024} } @article{xiang2024pandora, title={Pandora: Towards general world model with natural language actions and video states}, author={Xiang, Jiannan and Liu, Guangyi and Gu, Yi and Gao, Qiyue and Ning, Yuting and Zha, Yuheng and Feng, Zeyu and Tao, Tianhua and Hao, Shibo and Shi, Yemin and Liu, Zhengzhong and Xing, Eric P. and Hu, Zhiting}, journal={arXiv preprint arXiv:2406.09455}, year={2024} } @article{CosmosPredict, title={World simulation with video foundation models for physical {AI}}, author={Ali, Arslan and Bai, Junjie and Bala, Maciej and Balaji, Yogesh and Blakeman, Aaron and Cai, Tiffany and Cao, Jiaxin and Cao, Tianshi and Cha, Elizabeth and Chao, Yu-Wei and others}, journal={arXiv preprint arXiv:2511.00062}, year={2025} } % ------ world model end ------ %%% ---- sicong end ---- % ------ layer start ------ @article{zhang2023text2layer, title={Text2layer: Layered image generation using latent diffusion model}, author={Zhang, Xinyang and Zhao, Wentian and Lu, Xin and Chien, Jeff}, journal={arXiv preprint arXiv:2307.09781}, year={2023} } @inproceedings{huang2024layerdiff, title={{LayerDiff}: Exploring Text-guided Multi-layered Composable Image Synthesis via Layer-Collaborative Diffusion Model}, author={Huang, Runhui and Cai, Kaixin and Han, Jianhua and Liang, Xiaodan and Pei, Renjing and Lu, Guansong and Xu, Songcen and Zhang, Wei and Xu, Hang}, booktitle={ECCV}, year={2024} } @article{chen2025prismlayers, title={PrismLayers: Open Data for High-Quality Multi-Layer Transparent Image Generative Models}, author={Chen, Junwen and Jiang, Heyang and Wang, Yanbin and Wu, Keming and Li, Ji and Zhang, Chao and Yanai, Keiji and Chen, Dong and Yuan, Yuhui}, journal={arXiv preprint arXiv:2505.22523}, year={2025} } @article{yin2025qwenimagelayered, title={Qwen-Image-Layered: Towards Inherent Editability via Layer Decomposition}, author={Shengming Yin and Zekai Zhang and Zecheng Tang and Kaiyuan Gao and Xiao Xu and Kun Yan and Jiahao Li and Yilei Chen and Yuxiang Chen and Heung-Yeung Shum and Lionel M. Ni and Jingren Zhou and Junyang Lin and Chenfei Wu}, journal={arXiv preprint arXiv:2512.15603}, year={2025}, } @article{zhang2024transparent, author = {Zhang, Lvmin and Agrawala, Maneesh}, title = {Transparent Image Layer Diffusion using Latent Transparency}, year = {2024}, issue_date = {July 2024}, publisher = {Association for Computing Machinery}, volume = {43}, number = {4}, doi = {10.1145/3658150}, journal = {ACM Transactions on Graphics}, pages={1--15}, } @article{jia2023cole, title={{COLE}: A hierarchical generation framework for graphic design}, author={Jia, Peidong and Li, Chenxuan and Liu, Zeyu and Shen, Yichao and Chen, Xingru and Yuan, Yuhui and Zheng, Yinglin and Chen, Dong and Li, Ji and Xie, Xiaodong and others}, journal={arXiv preprint arXiv:2311.16974}, year={2023} } @inproceedings{inoue2024opencole, title={{OpenCOLE}: Towards Reproducible Automatic Graphic Design Generation}, author={Inoue, Naoto and Masui, Kento and Shimoda, Wataru and Yamaguchi, Kota}, booktitle={CVPR Workshops}, year={2024} } @inproceedings{suzuki2025layerd, title={LayerD: Decomposing Raster Graphic Designs into Layers}, author={Suzuki, Tomoyuki and Liu, Kang-Jun and Inoue, Naoto and Yamaguchi, Kota}, booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision}, pages={17783--17792}, year={2025} } @inproceedings{jiang2024scedit, title={SCEdit: Efficient and Controllable Image Diffusion Generation via Skip Connection Editing}, author={Jiang, Zeyu and Mao, Huaiyu and others}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, year={2024} } @article{liu2025step1xedit, title={Step-1X-Edit: A Practical Framework for General Image Editing}, author={Liu, Shiyu and others}, journal={arXiv preprint arXiv:2504.17761}, year={2025} } @article{he2026gems, title={GEMS: Agent-Native Multimodal Generation with Memory and Skills}, author={He, Zefeng and Huang, Siyuan and Qu, Xiaoye and Li, Yafu and Zhu, Tong and Cheng, Yu and Yang, Yang}, journal={arXiv preprint arXiv:2603.28088}, year={2026} } @article{feng2026gen, title={Gen-Searcher: Reinforcing Agentic Search for Image Generation}, author={Feng, Kaituo and Zhang, Manyuan and Chen, Shuang and Lin, Yunlong and Fan, Kaixuan and Jiang, Yilei and Li, Hongyu and Zheng, Dian and Wang, Chenyang and Yue, Xiangyu}, journal={arXiv preprint arXiv:2603.28767}, year={2026} } @article{lin2025jarvisart, title={Jarvisart: Liberating human artistic creativity via an intelligent photo retouching agent}, author={Lin, Yunlong and Lin, Zixu and Lin, Kunjie and Bai, Jinbin and Pan, Panwang and Li, Chenxin and Chen, Haoyu and Wang, Zhongdao and Ding, Xinghao and Li, Wenbo and others}, journal={arXiv preprint arXiv:2506.17612}, year={2025} } @misc{sglang-diffusion-2026, title = {SGLang Diffusion: Unified Inference for AR and Diffusion Visual Generation}, author = {{LMSYS Org}}, year = {2026}, month = jan, howpublished = {\url{https://www.lmsys.org/blog/2026-01-16-sglang-diffusion/}}, note = {Blog post} } @misc{diffsynth-studio, title = {DiffSynth-Studio: A Unified Framework for Training and Inference of Diffusion Models}, author = {{ModelScope Team}}, year = {2024}, howpublished = {\url{https://github.com/modelscope/DiffSynth-Studio}}, note = {Open-source repository} } @article{zhao2024monoformer, title={Monoformer: One transformer for both diffusion and autoregression}, author={Zhao, Chuyang and Song, Yuxing and Wang, Wenhao and Feng, Haocheng and Ding, Errui and Sun, Yifan and Xiao, Xinyan and Wang, Jingdong}, journal={arXiv preprint arXiv:2409.16280}, year={2024} } @article{song2025query, title={Query-kontext: An unified multimodal model for image generation and editing}, author={Song, Yuxin and Dong, Wenkai and Wang, Shizun and Zhang, Qi and Xue, Song and Yuan, Tao and Yang, Hu and Feng, Haocheng and Zhou, Hang and Xiao, Xinyan and others}, journal={arXiv preprint arXiv:2509.26641}, year={2025} } @article{song2026cologen, title={CoLoGen: Progressive Learning of Concept-Localization Duality for Unified Image Generation}, author={Song, YuXin and Lu, Yu and Sun, Haoyuan and Yao, Huanjin and Liu, Fanglong and Sun, Yifan and Feng, Haocheng and Zhou, Hang and Wang, Jingdong}, journal={arXiv preprint arXiv:2602.22150}, year={2026} } @article{liu2024luminamgpt, title={Lumina-mgpt: Illuminate flexible photorealistic text-to-image generation with multimodal generative pretraining}, author={Liu, Dongyang and Zhao, Shitian and Zhuo, Le and Lin, Weifeng and Xin, Yi and Li, Xinyue and Qin, Qi and Qiao, Yu and Li, Hongsheng and Gao, Peng}, journal={arXiv preprint arXiv:2408.02657}, year={2024} } @article{shen2024explanatory, title={Explanatory instructions: Towards unified vision tasks understanding and zero-shot generalization}, author={Shen, Yang and Wei, Xiu-Shen and Sun, Yifan and Song, Yuxin and Yuan, Tao and Jin, Jian and Xu, Heyang and Yao, Yazhou and Ding, Errui}, journal={arXiv preprint arXiv:2412.18525}, year={2024} } % ------ layer end ------ % ------ S10: sec5.3 Infrastructure citations ------ @article{jacobs2023deepspeed, title={Deepspeed ulysses: System optimizations for enabling training of extreme long sequence transformer models}, author={Jacobs, Sam Ade and Tanaka, Masahiro and Zhang, Chengming and Zhang, Minjia and Song, Shuaiwen Leon and Rajbhandari, Samyam and He, Yuxiong}, journal={arXiv preprint arXiv:2309.14509}, year={2023} } @article{liu2023ring, title={Ring attention with blockwise transformers for near-infinite context}, author={Liu, Hao and Zaharia, Matei and Abbeel, Pieter}, journal={arXiv preprint arXiv:2310.01889}, year={2023} } @article{dehghani2023patch, title={Patch n'pack: Navit, a vision transformer for any aspect ratio and resolution}, author={Dehghani, Mostafa and Mustafa, Basil and Djolonga, Josip and Heek, Jonathan and Minderer, Matthias and Caron, Mathilde and Steiner, Andreas and Puigcerver, Joan and Geirhos, Robert and Alabdulmohsin, Ibrahim M and others}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={2252--2274}, year={2023} } @article{schulman2017proximal, title={Proximal policy optimization algorithms}, author={Schulman, John and Wolski, Filip and Dhariwal, Prafulla and Radford, Alec and Klimov, Oleg}, journal={arXiv preprint arXiv:1707.06347}, year={2017} } @article{shao2024deepseekmath, title={Deepseekmath: Pushing the limits of mathematical reasoning in open language models}, author={Shao, Zhihong and Wang, Peiyi and Zhu, Qihao and Xu, Runxin and Song, Junxiao and Bi, Xiao and Zhang, Haowei and Zhang, Mingchuan and Li, YK and Wu, Yang and others}, journal={arXiv preprint arXiv:2402.03300}, year={2024} } @inproceedings{kwon2023efficient, title={Efficient memory management for large language model serving with pagedattention}, author={Kwon, Woosuk and Li, Zhuohan and Zhuang, Siyuan and Sheng, Ying and Zheng, Lianmin and Yu, Cody Hao and Gonzalez, Joseph and Zhang, Hao and Stoica, Ion}, booktitle={Proceedings of the 29th symposium on operating systems principles}, pages={611--626}, year={2023} } @article{zheng2024sglang, title={Sglang: Efficient execution of structured language model programs}, author={Zheng, Lianmin and Yin, Liangsheng and Xie, Zhiqiang and Sun, Chuyue and Huang, Jeff and Yu, Cody H and Cao, Shiyi and Kozyrakis, Christos and Stoica, Ion and Gonzalez, Joseph E and others}, journal={Advances in neural information processing systems}, volume={37}, pages={62557--62583}, year={2024} } @inproceedings{yu2022orca, title={Orca: A distributed serving system for $\{$Transformer-Based$\}$ generative models}, author={Yu, Gyeong-In and Jeong, Joo Seong and Kim, Geon-Woo and Kim, Soojeong and Chun, Byung-Gon}, booktitle={16th USENIX symposium on operating systems design and implementation (OSDI 22)}, pages={521--538}, year={2022} } @inproceedings{tillet2019triton, title={Triton: an intermediate language and compiler for tiled neural network computations}, author={Tillet, Philippe and Kung, Hsiang-Tsung and Cox, David}, booktitle={Proceedings of the 3rd ACM SIGPLAN International Workshop on Machine Learning and Programming Languages}, pages={10--19}, year={2019} } @article{xu2023imagereward, title={Imagereward: Learning and evaluating human preferences for text-to-image generation}, author={Xu, Jiazheng and Liu, Xiao and Wu, Yuchen and Tong, Yuxuan and Li, Qinkai and Ding, Ming and Tang, Jie and Dong, Yuxiao}, journal={Advances in Neural Information Processing Systems}, volume={36}, pages={15903--15935}, year={2023} } @article{wu2023human, title={Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis}, author={Wu, Xiaoshi and Hao, Yiming and Sun, Keqiang and Chen, Yixiong and Zhu, Feng and Zhao, Rui and Li, Hongsheng}, journal={arXiv preprint arXiv:2306.09341}, year={2023} } @misc{ma2025hpsv3widespectrumhumanpreference, title={HPSv3: Towards Wide-Spectrum Human Preference Score}, author={Yuhang Ma and Xiaoshi Wu and Keqiang Sun and Hongsheng Li}, year={2025}, eprint={2508.03789}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2508.03789} } @misc{cui2025paddleocr30technicalreport, title={PaddleOCR 3.0 Technical Report}, author={Cui, Cheng and Sun, Ting and Lin, Manhui and Gao, Tingquan and Zhang, Yubo and Liu, Jiaxuan and Wang, Xueqing and Zhang, Zelun and Zhou, Changda and Liu, Hongen and Zhang, Yue and Lv, Wenyu and Huang, Kui and Zhang, Yichao and Zhang, Jing and Zhang, Jun and Liu, Yi and Yu, Dianhai and Ma, Yanjun}, year={2025}, eprint={2507.05595}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2507.05595} } @misc{labs2025flux1kontextflowmatching, title={FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space}, author={Black Forest Labs and Batifol, Stephen and Blattmann, Andreas and Boesel, Frederic and Consul, Saksham and Diagne, Cyril and Dockhorn, Tim and English, Jack and English, Zion and Esser, Patrick and Kulal, Sumith and Lacey, Kyle and Levi, Yam and Li, Cheng and Lorenz, Dominik and M{\"u}ller, Jonas and Podell, Dustin and Rombach, Robin and Saini, Harry and Sauer, Axel and Smith, Luke}, year={2025}, eprint={2506.15742}, archivePrefix={arXiv}, primaryClass={cs.GR}, url={https://arxiv.org/abs/2506.15742} } % ------ S10 end ------ % verified via arXiv:2211.09800, accepted at CVPR 2023 @inproceedings{brooks2023instructpix2pix, title={InstructPix2Pix: Learning to Follow Image Editing Instructions}, author={Brooks, Tim and Holynski, Aleksander and Efros, Alexei A.}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, pages={18392--18402}, year={2023} } % verified via arXiv:2310.01830, arXiv-only positioning paper on synthetic data @misc{yang2023aigs, title={AI-Generated Images as Data Source: The Dawn of Synthetic Era}, author={Yang, Zuhao and Zhan, Fangneng and Liu, Kunhao and Xu, Muyu and Lu, Shijian}, year={2023}, eprint={2310.01830}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2310.01830} } % verified via arXiv:2508.01698, arXiv-only preprint @misc{yang2025vtg, title={Versatile Transition Generation with Image-to-Video Diffusion}, author={Yang, Zuhao and Zhang, Jiahui and Yu, Yingchen and Lu, Shijian and Bai, Song}, year={2025}, eprint={2508.01698}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2508.01698} } % ------ Added 2026-04-28 ------ @article{chen2023photoverse, title={Photoverse: Tuning-free image customization with text-to-image diffusion models}, author={Chen, Li and Zhao, Mengyi and Liu, Yiheng and Ding, Mingxu and Song, Yangyang and Wang, Shizun and Wang, Xu and Yang, Hao and Liu, Jing and Du, Kang and others}, journal={arXiv preprint arXiv:2309.05793}, year={2023} } @inproceedings{wang2024mindbridge, title={Mindbridge: A cross-subject brain decoding framework}, author={Wang, Shizun and Liu, Songhua and Tan, Zhenxiong and Wang, Xinchao}, booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition}, pages={11333--11342}, year={2024} } @article{zhang2026make, title={Make Geometry Matter for Spatial Reasoning}, author={Zhang, Shihua and Shen, Qiuhong and Wang, Shizun and Pan, Tianbo and Wang, Xinchao}, journal={arXiv preprint arXiv:2603.26639}, year={2026} }