{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T14:56:42Z","timestamp":1784041002989,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62325206"],"award-info":[{"award-number":["62325206"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Research and Development Program of Jiangsu Province","award":["BE2023016-4"],"award-info":[{"award-number":["BE2023016-4"]}]},{"name":"Postgraduate Research & Practice Innovation Program of Jiangsu Province","award":["KYCX23_1033"],"award-info":[{"award-number":["KYCX23_1033"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754727","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:55Z","timestamp":1761377215000},"page":"9287-9295","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Chain-of-Cooking: Cooking Process Visualization via Bidirectional Chain-of-Thought Guidance"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-4809-9129","authenticated-orcid":false,"given":"Mengling","family":"Xu","sequence":"first","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4662-7170","authenticated-orcid":false,"given":"Ming","family":"Tao","sequence":"additional","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China and Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5956-831X","authenticated-orcid":false,"given":"Bing-Kun","family":"Bao","sequence":"additional","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China and Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"e_1_3_2_1_4_1","first-page":"50742","article-title":"DreamSim: Learning New Dimensions of Human Visual Similarity using Synthetic Data","volume":"36","author":"Fu Stephanie","year":"2023","unstructured":"Stephanie Fu, Netanel Tamir, Shobhita Sundaram, Lucy Chai, Richard Zhang, Tali Dekel, and Phillip Isola. 2023. DreamSim: Learning New Dimensions of Human Visual Similarity using Synthetic Data. Advances in Neural Information Processing Systems, Vol. 36 (2023), 50742-50768.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","volume-title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step. arXiv preprint arXiv:2501.13926","author":"Guo Ziyu","year":"2025","unstructured":"Ziyu Guo, Renrui Zhang, Chengzhuo Tong, Zhizheng Zhao, Peng Gao, Hongsheng Li, and Pheng-Ann Heng. 2025. Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step. arXiv preprint arXiv:2501.13926 (2025)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3607828.3617796"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080686"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i16.33882"},{"key":"e_1_3_2_1_9_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium. In Advances in Neural Information Processing Systems, Vol. 30."},{"key":"e_1_3_2_1_10_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations.","author":"Hu Edward J.","year":"2022","unstructured":"Edward J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00592"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3554738"},{"key":"e_1_3_2_1_14_1","first-page":"10931","article-title":"Multimodal Procedural Planning via Dual Text-Image Prompting","author":"Lu Yujie","year":"2024","unstructured":"Yujie Lu, Pan Lu, Zhiyu Chen, Wanrong Zhu, Xin Wang, and William Yang Wang. 2024. Multimodal Procedural Planning via Dual Text-Image Prompting. In Findings of the Association for Computational Linguistics. 10931-10954.","journal-title":"Findings of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_15_1","volume-title":"Chi-en Amy Tai, and Alexander Wong","author":"Markham Olivia","year":"2023","unstructured":"Olivia Markham, Yuhao Chen, Chi-en Amy Tai, and Alexander Wong. 2023. FoodFusion: a latent diffusion model for realistic food image generation. arXiv preprint arXiv:2312.03540 (2023)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413765"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413636"},{"key":"e_1_3_2_1_18_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In The Twelfth International Conference on Learning Representations.","author":"Podell Dustin","year":"2024","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2024. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_19_1","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021a. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, Vol. 139. 8748-8763."},{"key":"e_1_3_2_1_20_1","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021b. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, Vol. 139. 8748-8763."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00246"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_23_1","volume-title":"ImageRAG: Dynamic Image Retrieval for Reference-Guided Image Generation. arXiv preprint arXiv:2502.09411","author":"Shalev-Arkushin Rotem","year":"2025","unstructured":"Rotem Shalev-Arkushin, Rinon Gal, Amit H Bermano, and Ohad Fried. 2025. ImageRAG: Dynamic Image Retrieval for Reference-Guided Image Generation. arXiv preprint arXiv:2502.09411 (2025)."},{"key":"e_1_3_2_1_24_1","volume-title":"MakeAnything: Harnessing Diffusion Transformers for Multi-Domain Procedural Sequence Generation. arXiv preprint arXiv:2502.01572","author":"Song Yiren","year":"2025","unstructured":"Yiren Song, Cheng Liu, and Mike Zheng Shou. 2025. MakeAnything: Harnessing Diffusion Transformers for Multi-Domain Procedural Sequence Generation. arXiv preprint arXiv:2502.01572 (2025)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_26_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10887583"},{"key":"e_1_3_2_1_28_1","volume-title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models. In The Eleventh International Conference on Learning Representations.","author":"Wang Xuezhi","year":"2023","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc V Le, Ed H Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou. 2023. Self-Consistency Improves Chain of Thought Reasoning in Language Models. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_29_1","volume-title":"CookingDiffusion: Cooking Procedural Image Generation with Stable Diffusion. arXiv preprint arXiv:2501.09042","author":"Wang Yuan","year":"2025","unstructured":"Yuan Wang, Bin Xhu, Yanbin Hao, Chong-Wah Ngo, Yi Tan, and Xiang Wang. 2025b. CookingDiffusion: Cooking Procedural Image Generation with Stable Diffusion. arXiv preprint arXiv:2501.09042 (2025)."},{"key":"e_1_3_2_1_30_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_31_1","volume-title":"SD-Prompt: Learnable and Adaptive Prompts for Enhancing Subject-driven Text-to-Image Synthesis","author":"Xu Mengling","year":"2025","unstructured":"Mengling Xu, Ming Tao, Jie Wang, and Bing-Kun Bao. 2025. SD-Prompt: Learnable and Adaptive Prompts for Enhancing Subject-driven Text-to-Image Synthesis. IEEE MultiMedia (2025)."},{"key":"e_1_3_2_1_32_1","volume-title":"CookGALIP: Recipe Controllable Generative Adversarial CLIPs With Sequential Ingredient Prompts for Food Image Generation","author":"Xu Mengling","year":"2024","unstructured":"Mengling Xu, Jie Wang, Ming Tao, Bing-Kun Bao, and Changsheng Xu. 2024. CookGALIP: Recipe Controllable Generative Adversarial CLIPs With Sequential Ingredient Prompts for Food Image Generation. IEEE Transactions on Multimedia (2024), 1-11."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1166"},{"key":"e_1_3_2_1_34_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Yang Ling","year":"2024","unstructured":"Ling Yang, Zhaochen Yu, Chenlin Meng, Minkai Xu, Stefano Ermon, and Bin Cui. 2024. Mastering text-to-image diffusion: Recaptioning, planning, and generating with multimodal llms. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_35_1","first-page":"105345","article-title":". Diffusion of Thought: Chain-of-Thought Reasoning in Diffusion Language Models","volume":"37","author":"Ye Jiacheng","year":"2025","unstructured":"Jiacheng Ye, Shansan Gong, Liheng Chen, Lin Zheng, Jiahui Gao, Han Shi, Chuan Wu, Xin Jiang, Zhenguo Li, Wei Bi, et al., 2025. Diffusion of Thought: Chain-of-Thought Reasoning in Diffusion Language Models. Advances in Neural Information Processing Systems, Vol. 37 (2025), 105345-105374.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552485.3554938"},{"key":"e_1_3_2_1_37_1","unstructured":"Zhuosheng Zhang Aston Zhang Mu Li George Karypis Alex Smola et al. 2024. Multimodal Chain-of-Thought Reasoning in Language Models. Transactions on Machine Learning Research (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00556"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754727","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:06:25Z","timestamp":1765343185000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754727"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":38,"alternative-id":["10.1145\/3746027.3754727","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754727","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}