{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T13:07:42Z","timestamp":1765544862463,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3548010","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:42:35Z","timestamp":1665416555000},"page":"4887-4898","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Boosting Video-Text Retrieval with Explicit High-Level Semantics"],"prefix":"10.1145","author":[{"given":"Haoran","family":"Wang","sequence":"first","affiliation":[{"name":"Department of Computer Vision Technology (VIS), Baidu Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Di","family":"Xu","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongliang","family":"He","sequence":"additional","affiliation":[{"name":"Department of Computer Vision Technology (VIS), Baidu Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fu","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Computer Vision Technology (VIS), Baidu Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhong","family":"Ji","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jungong","family":"Han","sequence":"additional","affiliation":[{"name":"Computer Science Department, Aberystwyth University, SY23 3FL, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Errui","family":"Ding","sequence":"additional","affiliation":[{"name":"Department of Computer Vision Technology (VIS), Baidu Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6077--6086","author":"Anderson P.","key":"e_1_3_2_2_1_1","unstructured":"P. Anderson , X. He , C. Buehler , D. Teney , M. Johnson , S. Gould , and L. Zhang . 2018. Bottom-up and top-down attention for image captioning and VQA . In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6077--6086 . P. Anderson, X. He, C. Buehler, D. Teney, M. Johnson, S. Gould, and L. Zhang. 2018. Bottom-up and top-down attention for image captioning and VQA. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6077--6086."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"volume-title":"IEEE International Conference on Computer Vision. 2425--2433","author":"Antol S.","key":"e_1_3_2_2_3_1","unstructured":"S. Antol , A. Agrawal , J. Lu , M. Mitchell , D. Batra , Z. C. Lawrence , and D. Parikh . 2015. VQA: Visual question answering . In IEEE International Conference on Computer Vision. 2425--2433 . S. Antol, A. Agrawal, J. Lu, M. Mitchell, D. Batra, Z. C. Lawrence, and D. Parikh. 2015. VQA: Visual question answering. In IEEE International Conference on Computer Vision. 2425--2433."},{"key":"e_1_3_2_2_4_1","volume-title":"Deep Attention Neural Tensor Network for Visual Question Answering. In European Conference on Computer Vision. 20--35","author":"Bai Yalong","year":"2018","unstructured":"Yalong Bai , Jianlong Fu , Tiejun Zhao , and Tao Mei . 2018 . Deep Attention Neural Tensor Network for Visual Question Answering. In European Conference on Computer Vision. 20--35 . Yalong Bai, Jianlong Fu, Tiejun Zhao, and Tao Mei. 2018. Deep Attention Neural Tensor Network for Visual Question Answering. In European Conference on Computer Vision. 20--35."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"e_1_3_2_2_6_1","unstructured":"Shuqiang Cao Bairui Wang Wei Zhang and Lin Ma. 2022. Visual Consensus Modeling for Video-Text Retrieval. In AAAI.  Shuqiang Cao Bairui Wang Wei Zhang and Lin Ma. 2022. Visual Consensus Modeling for Video-Text Retrieval. In AAAI."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/2002472.2002497"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01065"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01138"},{"key":"e_1_3_2_2_10_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ArXiv abs\/2010.11929","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy , Lucas Beyer , Alexander Kolesnikov , Dirk Weissenborn , Xiaohua Zhai , Thomas Unterthiner , Mostafa Dehghani , Matthias Minderer , Georg Heigold , Sylvain Gelly , Jakob Uszkoreit , and Neil Houlsby . 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ArXiv abs\/2010.11929 ( 2021 ). Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ArXiv abs\/2010.11929 (2021)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00374"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298754"},{"key":"e_1_3_2_2_13_1","volume-title":"Clip2video: Mastering video-text retrieval via image clip. arXiv preprint arXiv:2106.11097","author":"Fang Han","year":"2021","unstructured":"Han Fang , Pengfei Xiong , Luhui Xu , and Yu Chen . 2021. Clip2video: Mastering video-text retrieval via image clip. arXiv preprint arXiv:2106.11097 ( 2021 ). Han Fang, Pengfei Xiong, Luhui Xu, and Yu Chen. 2021. Clip2video: Mastering video-text retrieval via image clip. arXiv preprint arXiv:2106.11097 (2021)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","author":"Huang Y.","key":"e_1_3_2_2_16_1","unstructured":"Y. Huang , Q. Wu , C. Song , and L. Wang . 2018. Learning semantic concepts and order for image and sentence matching . In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. Y. Huang, Q. Wu, C. Song, and L. Wang. 2018. Learning semantic concepts and order for image and sentence matching. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_2_17_1","volume-title":"Learning by abstraction: The neural state machine. Advances in Neural Information Processing Systems","author":"Hudson Drew","year":"2019","unstructured":"Drew Hudson and Christopher D Manning . 2019. Learning by abstraction: The neural state machine. Advances in Neural Information Processing Systems ( 2019 ). Drew Hudson and Christopher D Manning. 2019. Learning by abstraction: The neural state machine. Advances in Neural Information Processing Systems (2019)."},{"key":"e_1_3_2_2_18_1","volume-title":"Step-Wise Hierarchical Alignment Network for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 765--771","author":"Ji Zhong","year":"2021","unstructured":"Zhong Ji , Kexin Chen , and Haoran Wang . 2021 . Step-Wise Hierarchical Alignment Network for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 765--771 . Zhong Ji, Kexin Chen, and Haoran Wang. 2021. Step-Wise Hierarchical Alignment Network for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 765--771."},{"key":"e_1_3_2_2_19_1","volume-title":"International Conference on Learning Representations.","author":"Kipf Thomas N","year":"2016","unstructured":"Thomas N Kipf and Max Welling . 2016 . Semi-supervised classification with graph convolutional networks . In International Conference on Learning Representations. Thomas N Kipf and Max Welling. 2016. Semi-supervised classification with graph convolutional networks. In International Conference on Learning Representations."},{"volume-title":"Stacked Cross Attention for Image-Text Matching. In European Conference on Computer Vision. 201--216","author":"Lee K.-H.","key":"e_1_3_2_2_20_1","unstructured":"K.-H. Lee , X. Chen , G. Hua , H. Hu , and X. He . 2018 . Stacked Cross Attention for Image-Text Matching. In European Conference on Computer Vision. 201--216 . K.-H. Lee, X. Chen, G. Hua, H. Hu, and X. He. 2018. Stacked Cross Attention for Image-Text Matching. In European Conference on Computer Vision. 201--216."},{"key":"e_1_3_2_2_21_1","volume-title":"Proceedings of the 27th ACM International Conference on Multimedia. 1786--1794","author":"Li Xirong","year":"2019","unstructured":"Xirong Li , Chaoxi Xu , Gang Yang , Zhineng Chen , and Jianfeng Dong . 2019 . W2vv fully deep learning for ad-hoc video search . In Proceedings of the 27th ACM International Conference on Multimedia. 1786--1794 . Xirong Li, Chaoxi Xu, Gang Yang, Zhineng Chen, and Jianfeng Dong. 2019. W2vv fully deep learning for ad-hoc video search. In Proceedings of the 27th ACM International Conference on Multimedia. 1786--1794."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3042067"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3478331"},{"key":"e_1_3_2_2_24_1","volume-title":"VrR-VG: Refocusing Visually-Relevant Relationships. In IEEE\/CVF International Conference on Computer Vision. 10402--10411","author":"Liang Yuanzhi","year":"2019","unstructured":"Yuanzhi Liang , Yalong Bai , Wei Zhang , Xueming Qian , Li Zhu , and Tao Mei . 2019 . VrR-VG: Refocusing Visually-Relevant Relationships. In IEEE\/CVF International Conference on Computer Vision. 10402--10411 . Yuanzhi Liang, Yalong Bai, Wei Zhang, Xueming Qian, Li Zhu, and Tao Mei. 2019. VrR-VG: Refocusing Visually-Relevant Relationships. In IEEE\/CVF International Conference on Computer Vision. 10402--10411."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6823"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01170"},{"key":"e_1_3_2_2_27_1","unstructured":"Yang Liu Samuel Albanie Arsha Nagrani and Andrew Zisserman. 2019. Use what you have: Video retrieval using representations from collaborative experts. In arXiv preprint arXiv:1907.13487.  Yang Liu Samuel Albanie Arsha Nagrani and Andrew Zisserman. 2019. Use what you have: Video retrieval using representations from collaborative experts. In arXiv preprint arXiv:1907.13487."},{"key":"e_1_3_2_2_28_1","volume-title":"Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860","author":"Luo Huaishao","year":"2021","unstructured":"Huaishao Luo , Lei Ji , Ming Zhong , Yang Chen , Wen Lei , Nan Duan , and Tianrui Li. 2021. Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860 ( 2021 ). Huaishao Luo, Lei Ji, Ming Zhong, Yang Chen, Wen Lei, Nan Duan, and Tianrui Li. 2021. Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860 (2021)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"Lin Ma Zhengdong Lu and Hang Li. 2016. Learning to Answer Questions From Image Using Convolutional Neural Network. In AAAI.  Lin Ma Zhengdong Lu and Hang Li. 2016. Learning to Answer Questions From Image Using Convolutional Neural Network. In AAAI.","DOI":"10.1609\/aaai.v30i1.10442"},{"key":"e_1_3_2_2_30_1","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"van der Maaten Laurens","year":"2008","unstructured":"Laurens van der Maaten and Geoffrey Hinton . 2008 . Visualizing data using t-SNE . Journal of machine learning research 9 (2008), 2579 -- 2605 . Laurens van der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-SNE. Journal of machine learning research 9 (2008), 2579--2605.","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_2_31_1","volume-title":"Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516","author":"Miech Antoine","year":"2018","unstructured":"Antoine Miech , Ivan Laptev , and Josef Sivic . 2018. Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516 ( 2018 ). Antoine Miech, Ivan Laptev, and Josef Sivic. 2018. Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516 (2018)."},{"key":"e_1_3_2_2_32_1","volume-title":"Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516","author":"Miech Antoine","year":"2018","unstructured":"Antoine Miech , Ivan Laptev , and Josef Sivic . 2018. Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516 ( 2018 ). Antoine Miech, Ivan Laptev, and Josef Sivic. 2018. Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516 (2018)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3206025.3206064"},{"key":"e_1_3_2_2_34_1","volume-title":"Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824","author":"Patrick Mandela","year":"2020","unstructured":"Mandela Patrick , Po-Yao Huang , Yuki Asano , Florian Metze , Alexander Hauptmann , Joao Henriques , and Andrea Vedaldi . 2020. Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824 ( 2020 ). Mandela Patrick, Po-Yao Huang, Yuki Asano, Florian Metze, Alexander Hauptmann, Joao Henriques, and Andrea Vedaldi. 2020. Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824 (2020)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-77004-4_1"},{"key":"e_1_3_2_2_36_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International conference on machine learning. 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford , Jong Wook Kim , Chris Hallacy , Aditya Ramesh , Gabriel Goh , Sandhini Agarwal , Girish Sastry , Amanda Askell , Pamela Mishkin , Jack Clark , Gretchen Krueger , and Ilya Sutskever . 2021 . Learning Transferable Visual Models From Natural Language Supervision. In International conference on machine learning. 8748--8763 . Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International conference on machine learning. 8748--8763."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-93417-4_38"},{"key":"e_1_3_2_2_38_1","volume-title":"Lin","author":"Shi Peng","year":"2019","unstructured":"Peng Shi and Jimmy J . Lin . 2019 . Simple BERT Models for Relation Extraction and Semantic Role Labeling. ArXiv abs\/1904.05255 (2019). Peng Shi and Jimmy J. Lin. 2019. Simple BERT Models for Relation Extraction and Semantic Role Labeling. ArXiv abs\/1904.05255 (2019)."},{"key":"e_1_3_2_2_39_1","unstructured":"A. Vaswani N. Shazeer N. Parmar J. Uszkoreit L. Jones A. N. Gomez K. Lukasz and I. Polosukhin. 2017. Attention is all you need. In Advances in Neural Information Processing Systems.  A. Vaswani N. Shazeer N. Parmar J. Uszkoreit L. Jones A. N. Gomez K. Lukasz and I. Polosukhin. 2017. Attention is all you need. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00795"},{"key":"e_1_3_2_2_41_1","volume-title":"Adversarial Cross-Modal Retrieval. In ACM international conference on Multimedia.","author":"Wang Bokun","year":"2017","unstructured":"Bokun Wang , Yang Yang , Xing Xu , Alan Hanjalic , and Heng Tao Shen . 2017 . Adversarial Cross-Modal Retrieval. In ACM international conference on Multimedia. Bokun Wang, Yang Yang, Xing Xu, Alan Hanjalic, and Heng Tao Shen. 2017. Adversarial Cross-Modal Retrieval. In ACM international conference on Multimedia."},{"key":"e_1_3_2_2_42_1","volume-title":"Consensus-Aware Visual-Semantic Embedding for Image-Text Matching. In European Conference on Computer Vision. 18--34","author":"Wang Haoran","year":"2020","unstructured":"Haoran Wang , Ying Zhang , Zhong Ji , Yanwei Pang , and Lin Ma . 2020 . Consensus-Aware Visual-Semantic Embedding for Image-Text Matching. In European Conference on Computer Vision. 18--34 . Haoran Wang, Ying Zhang, Zhong Ji, Yanwei Pang, and Lin Ma. 2020. Consensus-Aware Visual-Semantic Embedding for Image-Text Matching. In European Conference on Computer Vision. 18--34."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00751"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00468"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475515"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.29"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.217"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.524"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58592-1_36"},{"key":"e_1_3_2_2_52_1","unstructured":"Michael Zhang James Lucas Jimmy Ba and Geoffrey E Hinton. 2019. Looka-head optimizer: k steps forward 1 step back. In Advances in neural information processing systems.  Michael Zhang James Lucas Jimmy Ba and Geoffrey E Hinton. 2019. Looka-head optimizer: k steps forward 1 step back. In Advances in neural information processing systems."},{"key":"e_1_3_2_2_53_1","volume-title":"Memory Enhanced Embedding Learning for Cross-Modal Video-Text Retrieval. arXiv preprint arXiv:2103.15686","author":"Zhao Rui","year":"2021","unstructured":"Rui Zhao , Kecheng Zheng , Zheng-Jun Zha , Hongtao Xie , and Jiebo Luo . 2021. Memory Enhanced Embedding Learning for Cross-Modal Video-Text Retrieval. arXiv preprint arXiv:2103.15686 ( 2021 ). Rui Zhao, Kecheng Zheng, Zheng-Jun Zha, Hongtao Xie, and Jiebo Luo. 2021. Memory Enhanced Embedding Learning for Cross-Modal Video-Text Retrieval. arXiv preprint arXiv:2103.15686 (2021)."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"crossref","unstructured":"Shuai Zhao Linchao Zhu Xiaohan Wang and Yi Yang. 2022. CenterCLIP: Token Clustering for Efficient Text-Video Retrieval. In SIGIR.  Shuai Zhao Linchao Zhu Xiaohan Wang and Yi Yang. 2022. CenterCLIP: Token Clustering for Efficient Text-Video Retrieval. In SIGIR.","DOI":"10.1145\/3477495.3531950"},{"key":"e_1_3_2_2_55_1","volume-title":"End-to-End Dense Video Captioning with Masked Transformer. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8739--8748","author":"Zhou Luowei","year":"2018","unstructured":"Luowei Zhou , Yingbo Zhou , Jason J. Corso , Richard Socher , and Caiming Xiong . 2018 . End-to-End Dense Video Captioning with Masked Transformer. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8739--8748 . Luowei Zhou, Yingbo Zhou, Jason J. Corso, Richard Socher, and Caiming Xiong. 2018. End-to-End Dense Video Captioning with Masked Transformer. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8739--8748."}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Lisboa Portugal","acronym":"MM '22"},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548010","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3548010","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:29Z","timestamp":1750186949000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548010"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":55,"alternative-id":["10.1145\/3503161.3548010","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3548010","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}