{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,13]],"date-time":"2026-03-13T22:12:25Z","timestamp":1773439945824,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Sichuan Science and Technology Program","award":["2023NSFSC1392"],"award-info":[{"award-number":["2023NSFSC1392"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62102070,62220106008"],"award-info":[{"award-number":["62102070,62220106008"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Cisco-NUS Accelerated Digital Economy Corporate Laboratory","award":["I21001E0002"],"award-info":[{"award-number":["I21001E0002"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681677","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"4630-4639","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Leveraging Weak Cross-Modal Guidance for Coherence Modelling via Iterative Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9714-8738","authenticated-orcid":false,"given":"Yi","family":"Bin","sequence":"first","affiliation":[{"name":"Tongji University &amp; National University of Singapore, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9180-1867","authenticated-orcid":false,"given":"Junrong","family":"Liao","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2945-1107","authenticated-orcid":false,"given":"Yujuan","family":"Ding","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University, Hong Kong SAR, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1355-6256","authenticated-orcid":false,"given":"HaoXuan","family":"Li","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5070-4511","authenticated-orcid":false,"given":"Yang","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6565-7511","authenticated-orcid":false,"given":"See-Kiong","family":"Ng","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2999-2088","authenticated-orcid":false,"given":"Heng Tao","family":"Shen","sequence":"additional","affiliation":[{"name":"Tongji University &amp; University of Electronic Science and Technology of China, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Sort story: Sorting jumbled images and captions into stories. arXiv preprint arXiv:1606.07493","author":"Agrawal Harsh","year":"2016","unstructured":"Harsh Agrawal, Arjun Chandrasekaran, Dhruv Batra, Devi Parikh, and Mohit Bansal. 2016. Sort story: Sorting jumbled images and captions into stories. arXiv preprint arXiv:1606.07493 (2016)."},{"key":"e_1_3_2_1_2_1","first-page":"9758","article-title":"Self-supervised learning by cross-modal audio-video clustering","volume":"33","author":"Alwassel Humam","year":"2020","unstructured":"Humam Alwassel, Dhruv Mahajan, Bruno Korbar, Lorenzo Torresani, Bernard Ghanem, and Du Tran. 2020. Self-supervised learning by cross-modal audio-video clustering. Advances in Neural Information Processing Systems, Vol. 33 (2020), 9758--9770.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli.2008.34.1.1"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3063297"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612427"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475173"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.277"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548184"},{"key":"e_1_3_2_1_10_1","volume-title":"Philip K McGuire, Peter WR Woodruff, Susan D Iversen, and Anthony S David.","author":"Calvert Gemma A","year":"1997","unstructured":"Gemma A Calvert, Edward T Bullmore, Michael J Brammer, Ruth Campbell, Steven CR Williams, Philip K McGuire, Peter WR Woodruff, Susan D Iversen, and Anthony S David. 1997. Activation of auditory cortex during silent lipreading. science, Vol. 276, 5312 (1997), 593--596."},{"key":"e_1_3_2_1_11_1","first-page":"684","article-title":"Exploring the Potential of ChatGPT on Sentence Level Relations: A Focus on Temporal, Causal, and Discourse Relations","volume":"2024","author":"Chan Chunkit","year":"2024","unstructured":"Chunkit Chan, Cheng Jiayang, Weiqi Wang, Yuxin Jiang, Tianqing Fang, Xin Liu, and Yangqiu Song. 2024. Exploring the Potential of ChatGPT on Sentence Level Relations: A Focus on Temporal, Causal, and Discourse Relations. In Findings of the Association for Computational Linguistics: EACL 2024. 684--721.","journal-title":"Findings of the Association for Computational Linguistics: EACL"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16184"},{"key":"e_1_3_2_1_13_1","volume-title":"Is everything in order? a simple way to order sentences. arXiv preprint arXiv:2104.07064","author":"Roy Chowdhury Somnath Basu","year":"2021","unstructured":"Somnath Basu Roy Chowdhury, Faeze Brahman, and Snigdha Chaturvedi. 2021. Is everything in order? a simple way to order sentences. arXiv preprint arXiv:2104.07064 (2021)."},{"key":"e_1_3_2_1_14_1","unstructured":"Baiyun Cui Yingming Li Ming Chen and Zhongfei Zhang. 2018. Deep attentive sentence ordering network. In EMNLP. 4340--4349."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.511"},{"key":"e_1_3_2_1_16_1","volume-title":"Implications for educational practice of the science of learning and development. Applied developmental science","author":"Darling-Hammond Linda","year":"2020","unstructured":"Linda Darling-Hammond, Lisa Flook, Channa Cook-Harvey, Brigid Barron, and David Osher. 2020. Implications for educational practice of the science of learning and development. Applied developmental science, Vol. 24, 2 (2020), 97--140."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612583"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2023.103434"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies. 125--129","author":"Elsner Micha","year":"2011","unstructured":"Micha Elsner and Eugene Charniak. 2011. Extending the entity grid with entity-specific features. In Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies. 125--129."},{"key":"e_1_3_2_1_20_1","volume-title":"Towards bridged vision and language: Learning cross-modal knowledge representation for relation extraction","author":"Feng Junhao","year":"2023","unstructured":"Junhao Feng, Guohua Wang, Changmeng Zheng, Yi Cai, Ze Fu, Yaowei Wang, Xiao-Yong Wei, and Qing Li. 2023. Towards bridged vision and language: Learning cross-modal knowledge representation for relation extraction. IEEE Transactions on Circuits and Systems for Video Technology (2023)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3120867"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.27955"},{"key":"e_1_3_2_1_23_1","volume-title":"Pyramidclip: Hierarchical feature alignment for vision-language model pretraining. Advances in neural information processing systems","author":"Gao Yuting","year":"2022","unstructured":"Yuting Gao, Jinfeng Liu, Zihan Xu, Jun Zhang, Ke Li, Rongrong Ji, and Chunhua Shen. 2022. Pyramidclip: Hierarchical feature alignment for vision-language model pretraining. Advances in neural information processing systems, Vol. 35 (2022), 35959--35970."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1382"},{"key":"e_1_3_2_1_25_1","first-page":"6704","article-title":"Cyclip: Cyclic contrastive language-image pretraining","volume":"35","author":"Goel Shashank","year":"2022","unstructured":"Shashank Goel, Hritik Bansal, Sumit Bhatia, Ryan Rossi, Vishwa Vinay, and Aditya Grover. 2022. Cyclip: Cyclic contrastive language-image pretraining. Advances in Neural Information Processing Systems, Vol. 35 (2022), 6704--6719.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_26_1","volume-title":"End-to-end neural sentence ordering using pointer network. arXiv preprint arXiv:1611.04953","author":"Gong Jingjing","year":"2016","unstructured":"Jingjing Gong, Xinchi Chen, Xipeng Qiu, and Xuanjing Huang. 2016. End-to-end neural sentence ordering using pointer network. arXiv preprint arXiv:1611.04953 (2016)."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 51st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 93--103","author":"Guinaudeau Camille","year":"2013","unstructured":"Camille Guinaudeau and Michael Strube. 2013. Graph-based local coherence modeling. In Proceedings of the 51st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 93--103."},{"key":"e_1_3_2_1_28_1","volume-title":"Benchmarking Micro-action Recognition: Dataset, Method, and Application","author":"Guo Dan","year":"2024","unstructured":"Dan Guo, Kun Li, Bin Hu, Yan Zhang, and Meng Wang. 2024. Benchmarking Micro-action Recognition: Dataset, Method, and Application. IEEE Transactions on Circuits and Systems for Video Technology (2024)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3585088.3593867"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-1147"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00244"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.360"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6780"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.543"},{"key":"e_1_3_2_1_35_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2015","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In ICLR."},{"key":"e_1_3_2_1_36_1","volume-title":"Multimodal images in the brain. The neurophysiological foundations of mental and motor imagery","author":"Kosslyn Stephen M","year":"2010","unstructured":"Stephen M Kosslyn, Giorgio Ganis, and William L Thompson. 2010. Multimodal images in the brain. The neurophysiological foundations of mental and motor imagery (2010), 3--16."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-022-10255-9"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Pawan Kumar Dhanajit Brahma Harish Karnick and Piyush Rai. 2020. Deep Attentive Ranking Networks for Learning to Order Sentences.. In AAAI. 8115--8122.","DOI":"10.1609\/aaai.v34i05.6323"},{"key":"e_1_3_2_1_39_1","first-page":"1085","article-title":"Automatic evaluation of text coherence: Models and representations","volume":"5","author":"Lapata Mirella","year":"2005","unstructured":"Mirella Lapata, Regina Barzilay, et al. 2005. Automatic evaluation of text coherence: Models and representations. In Ijcai, Vol. 5. 1085--1090.","journal-title":"Ijcai"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612101"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01852"},{"key":"e_1_3_2_1_42_1","volume-title":"Visual Storytelling with Question-Answer Plans. arXiv preprint arXiv:2310.05295","author":"Liu Danyang","year":"2023","unstructured":"Danyang Liu, Mirella Lapata, and Frank Keller. 2023. Visual Storytelling with Question-Answer Plans. arXiv preprint arXiv:2310.05295 (2023)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11997"},{"key":"e_1_3_2_1_44_1","first-page":"4","article-title":"The art of storytelling: How do children learn it","volume":"38","author":"Magee Mary Ann","year":"1983","unstructured":"Mary Ann Magee and Brian Sutton-Smith. 1983. The art of storytelling: How do children learn it? Young Children, Vol. 38, 4 (1983), 4--12.","journal-title":"Young Children"},{"key":"e_1_3_2_1_45_1","volume-title":"Shaping visual representations with language for few-shot classification. arXiv preprint arXiv:1911.02683","author":"Mu Jesse","year":"2019","unstructured":"Jesse Mu, Percy Liang, and Noah Goodman. 2019. Shaping visual representations with language for few-shot classification. arXiv preprint arXiv:1911.02683 (2019)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cortex.2017.07.006"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1232"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475193"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Shrimai Prabhumoye Ruslan Salakhutdinov and Alan W Black. 2020. Topological Sort for Sentence Ordering. In ACL. 2783--2792.","DOI":"10.18653\/v1\/2020.acl-main.248"},{"key":"e_1_3_2_1_50_1","volume-title":"Xi Peng, and Peng Hu.","author":"Qin Yang","year":"2024","unstructured":"Yang Qin, Yuan Sun, Dezhong Peng, Joey Tianyi Zhou, Xi Peng, and Peng Hu. 2024. Cross-modal Active Complementary Learning with Self-refining Correspondence. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_51_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.278"},{"key":"e_1_3_2_1_53_1","volume-title":"Pointer networks. Advances in neural information processing systems","author":"Vinyals Oriol","year":"2015","unstructured":"Oriol Vinyals, Meire Fortunato, and Navdeep Jaitly. 2015. Pointer networks. Advances in neural information processing systems, Vol. 28 (2015)."},{"key":"e_1_3_2_1_54_1","volume-title":"Adaptive cross-modal few-shot learning. Advances in neural information processing systems","author":"Xing Chen","year":"2019","unstructured":"Chen Xing, Negar Rostamzadeh, Boris Oreshkin, and Pedro O O Pinheiro. 2019. Adaptive cross-modal few-shot learning. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16410"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3235495"},{"key":"e_1_3_2_1_57_1","volume-title":"The dawn of lmms: Preliminary explorations with gpt-4v (ision). arXiv preprint arXiv:2309.17421","author":"Yang Zhengyuan","year":"2023","unstructured":"Zhengyuan Yang, Linjie Li, Kevin Lin, Jianfeng Wang, Chung-Ching Lin, Zicheng Liu, and Lijuan Wang. 2023. The dawn of lmms: Preliminary explorations with gpt-4v (ision). arXiv preprint arXiv:2309.17421, Vol. 9, 1 (2023), 1."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6492"},{"key":"e_1_3_2_1_59_1","volume-title":"Graph-based neural sentence ordering. arXiv preprint arXiv:1912.07225","author":"Yin Yongjing","year":"2019","unstructured":"Yongjing Yin, Linfeng Song, Jinsong Su, Jiali Zeng, Chulun Zhou, and Jiebo Luo. 2019. Graph-based neural sentence ordering. arXiv preprint arXiv:1912.07225 (2019)."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3205212"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611714"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00632"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17722"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681677","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681677","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:50Z","timestamp":1750295870000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681677"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":63,"alternative-id":["10.1145\/3664647.3681677","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681677","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}