{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T18:48:32Z","timestamp":1755802112394,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","funder":[{"name":"Hubei Provincial Key Research and Development Program","award":["2024BAB039"],"award-info":[{"award-number":["2024BAB039"]}]},{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276196 and 62271361"],"award-info":[{"award-number":["62276196 and 62271361"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733438","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:29:43Z","timestamp":1750876183000},"page":"1377-1385","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Step-wise Soft Alignment Enhanced Procedural Text Generation from Long Instructional Videos"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-7622-0320","authenticated-orcid":false,"given":"Zhihao","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science and Artificial Intelligence, Wuhan University of Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7553-6916","authenticated-orcid":false,"given":"Lin","family":"Li","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Transportation Internet of Things, Wuhan University of Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5242-0467","authenticated-orcid":false,"given":"Xian","family":"Zhong","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Transportation Internet of Things, Wuhan University of Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0020-077X","authenticated-orcid":false,"given":"Xiaohui","family":"Tao","sequence":"additional","affiliation":[{"name":"University of Southern Queensland, Springfield, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4303-9020","authenticated-orcid":false,"given":"Jianquan","family":"Liu","sequence":"additional","affiliation":[{"name":"NEC Corporation, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proc. Annu. Meeting Assoc. Comput. Linguist. Workshop. 65--72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In Proc. Annu. Meeting Assoc. Comput. Linguist. Workshop. 65--72."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3679203"},{"volume-title":"Proc. Int. Conf. Mach. Learn. 1597--1607","author":"Chen Ting","key":"e_1_3_2_1_3_1","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey E. Hinton. 2020. A Simple Framework for Contrastive Learning of Visual Representations. In Proc. Int. Conf. Mach. Learn. 1597--1607."},{"key":"e_1_3_2_1_4_1","volume-title":"Sinkhorn Distances: Lightspeed Computation of Optimal Transport. In Adv. Neural Inf. Process. Syst. 2292--2300.","author":"Cuturi Marco","year":"2013","unstructured":"Marco Cuturi. 2013. Sinkhorn Distances: Lightspeed Computation of Optimal Transport. In Adv. Neural Inf. Process. Syst. 2292--2300."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"volume-title":"Proc. Assoc. Comput. Linguist. Findings. 1792--1804","author":"Faghihi Hossein Rajaby","key":"e_1_3_2_1_6_1","unstructured":"Hossein Rajaby Faghihi, Parisa Kordjamshidi, Choh Man Teng, and James F. Allen. 2023. The Role of Semantic Parsing in Understanding Procedural Text. In Proc. Assoc. Comput. Linguist. Findings. 1792--1804."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.27955"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00037"},{"key":"e_1_3_2_1_9_1","volume-title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning. In Adv. Neural Inf. Process. Syst.","author":"Ging Simon","year":"2020","unstructured":"Simon Ging, Mohammadreza Zolfaghari, Hamed Pirsiavash, and Thomas Brox. 2020. COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning. In Adv. Neural Inf. Process. Syst."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_12_1","volume-title":"Proc. Int. Conf. Mach. Learn. 448--456","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy. 2015. Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. In Proc. Int. Conf. Mach. Learn. 448--456."},{"volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. 20105--20115","author":"Ko Dohwan","key":"e_1_3_2_1_13_1","unstructured":"Dohwan Ko, Joonmyung Choi, Hyeong Kyu Choi, Kyoung-Woon On, Byungseok Roh, and Hyunwoo J. Kim. 2023. MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models. In Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. 20105--20115."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.233"},{"key":"e_1_3_2_1_16_1","volume-title":"William Yang Wang","author":"Li Linjie","year":"2021","unstructured":"Linjie Li, Jie Lei, Zhe Gan, Licheng Yu, Yen-Chun Chen, Rohit Pillai, Yu Cheng, Luowei Zhou, Xin Wang, William Yang Wang, Tamara L. Berg, Mohit Bansal, Jingjing Liu, Lijuan Wang, and Zicheng Liu. 2021. VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation. In Adv. Neural Inf. Process. Syst."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095662"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.3115\/1218955.1219032"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01742"},{"volume-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. 21663--21673","author":"Lin Wei","key":"e_1_3_2_1_20_1","unstructured":"Wei Lin and Antoni B. Chan. 2023. Optimal Transport Minimization: Crowd Localization on Density Maps for Semi-Supervised Counting. In Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. 21663--21673."},{"key":"e_1_3_2_1_21_1","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Lin Yijie","year":"2024","unstructured":"Yijie Lin, Jie Zhang, Zhenyu Huang, Jia Liu, Zujie Wen, and Xi Peng. 2024. Multi-granularity Correspondence Learning from Long-term Noisy Videos. In Proc. Int. Conf. Learn. Represent."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3652583.3658061"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475322"},{"key":"e_1_3_2_1_24_1","volume-title":"Proc. Annu. Meeting Assoc. Comput. Linguist. 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. BLEU: A Method for Automatic Evaluation of Machine Translation. In Proc. Annu. Meeting Assoc. Comput. Linguist. 311--318."},{"volume-title":"Proc. Conf. Empirical Methods Nat. Lang. Process. 1532--1543","author":"Pennington Jeffrey","key":"e_1_3_2_1_25_1","unstructured":"Jeffrey Pennington, Richard Socher, and Christopher D. Manning. 2014. Glove: Global Vectors for Word Representation. In Proc. Conf. Empirical Methods Nat. Lang. Process. 1532--1543."},{"key":"e_1_3_2_1_26_1","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proc. Int. Conf. Mach. Learn., Vol. 139. 8748--8763."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3652161"},{"key":"e_1_3_2_1_28_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In Adv. Neural Inf. Process. Syst. 5998--6008."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_30_1","first-page":"3363","article-title":"Learning Structural Representations for Recipe Generation and Food Retrieval","volume":"45","author":"Wang Hao","year":"2023","unstructured":"Hao Wang, Guosheng Lin, Steven C. H. Hoi, and Chunyan Miao. 2023a. Learning Structural Representations for Recipe Generation and Food Retrieval. IEEE Trans. Pattern Anal. Mach. Intell., Vol. 45, 3 (2023), 3363--3377.","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103799"},{"key":"e_1_3_2_1_32_1","unstructured":"Junke Wang Dongdong Chen Zuxuan Wu Chong Luo Luowei Zhou Yucheng Zhao Yujia Xie Ce Liu Yu-Gang Jiang and Lu Yuan. 2022a. OmniVL: One Foundation Model for Image-Language and Video-Language Tasks. In Adv. Neural Inf. Process. Syst."},{"key":"e_1_3_2_1_33_1","article-title":"GIT: A Generative Image-to-text Transformer for Vision and","volume":"2022","author":"Wang Jianfeng","year":"2022","unstructured":"Jianfeng Wang, Zhengyuan Yang, Xiaowei Hu, Linjie Li, Kevin Lin, Zhe Gan, Zicheng Liu, Ce Liu, and Lijuan Wang. 2022b. GIT: A Generative Image-to-text Transformer for Vision and Language. Trans. Mach. Learn. Res., Vol. 2022 (2022).","journal-title":"Language. Trans. Mach. Learn. Res."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00677"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2024.103954"},{"key":"e_1_3_2_1_36_1","volume-title":"Kankanhalli","author":"Wong Yongkang","year":"2022","unstructured":"Yongkang Wong, Shaojing Fan, Yangyang Guo, Ziwei Xu, Karen Stephen, Rishabh Sheoran, Anusha Bhamidipati, Vivek Barsopia, Jianquan Liu, and Mohan S. Kankanhalli. 2022. Compute to Tell the Tale: Goal-Driven Narrative Generation. In Proc. ACM Multimedia. 6875--6882."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2021.3098834"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3490519"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00445"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25412"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i15.29564"},{"key":"e_1_3_2_1_42_1","article-title":"CoCa: Contrastive Captioners are Image-Text Foundation","volume":"2022","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Zirui Wang, Vijay Vasudevan, Legg Yeung, Mojtaba Seyedhosseini, and Yonghui Wu. 2022b. CoCa: Contrastive Captioners are Image-Text Foundation Models. Trans. Mach. Learn. Res., Vol. 2022 (2022).","journal-title":"Models. Trans. Mach. Learn. Res."},{"key":"e_1_3_2_1_43_1","volume-title":"Proc. Int. Conf. Comput. Linguist. 2363--2373","author":"Yu Weijie","year":"2022","unstructured":"Weijie Yu, Liang Pang, Jun Xu, Bing Su, Zhenhua Dong, and Ji-Rong Wen. 2022a. Optimal Partial Transport Based Sentence Selection for Long-form Document Matching. In Proc. Int. Conf. Comput. Linguist. 2363--2373."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25484"},{"volume-title":"Proc. AAAI Conf. Artif. Intell. 7590--7598","author":"Zhou Luowei","key":"e_1_3_2_1_46_1","unstructured":"Luowei Zhou, Chenliang Xu, and Jason J. Corso. 2018. Towards Automatic Learning of Procedures From Web Instructional Videos. In Proc. AAAI Conf. Artif. Intell. 7590--7598."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611793"}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Chicago IL USA","acronym":"ICMR '25"},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733438","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:08:37Z","timestamp":1755749317000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733438"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":47,"alternative-id":["10.1145\/3731715.3733438","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733438","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}