{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T16:28:47Z","timestamp":1784910527059,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Key Project of Science and Technology Innovation 2030","award":["2018AAA0101302"],"award-info":[{"award-number":["2018AAA0101302"]}]},{"name":"General Program of National Natural Science Foundation of China (NSFC)","award":["61773300"],"award-info":[{"award-number":["61773300"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475515","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T06:09:05Z","timestamp":1634537345000},"page":"3518-3527","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":62,"title":["HANet"],"prefix":"10.1145","author":[{"given":"Peng","family":"Wu","sequence":"first","affiliation":[{"name":"Xidian University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangteng","family":"He","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingqian","family":"Tang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiliang","family":"Lv","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jing","family":"Liu","sequence":"additional","affiliation":[{"name":"Xidian University, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_2_1","volume-title":"2020 a. Interclass-Relativity-Adaptive Metric Learning for Cross-Modal Matching and Beyond","author":"Chen Feiyu","year":"2020"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351055"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01065"},{"key":"e_1_3_2_1_5_1","volume-title":"Caglar Gulcehre, Dzmitry Bahdanau, Fethi Bougares, Holger Schwenk, and Yoshua Bengio.","author":"Cho Kyunghyun","year":"2014"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16209"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2832602"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00957"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3059295"},{"key":"e_1_3_2_1_10_1","volume-title":"Jamie Ryan Kiros, and Sanja Fidler","author":"Faghri Fartash","year":"2017"},{"key":"e_1_3_2_1_11_1","volume-title":"Exploiting Visual Semantic Reasoning for Video-Text Retrieval. arXiv preprint arXiv:2006.08889","author":"Feng Zerun","year":"2020"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"e_1_3_2_1_13_1","volume-title":"Coot: Cooperative hierarchical transformer for video-text representation learning. arXiv preprint arXiv:2011.00597","author":"Ging Simon","year":"2020"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350974"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_17_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014"},{"key":"e_1_3_2_1_18_1","volume-title":"Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539","author":"Kiros Ryan","year":"2014"},{"key":"e_1_3_2_1_19_1","volume-title":"Minh-Triet Tran, Yuki Watanabe, Martin Klinkigt, et al.","author":"Le Duy-Dinh","year":"2016"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"e_1_3_2_1_21_1","volume-title":"Less is more: Clipbert for video-and-language learning via sparse sampling. arXiv preprint arXiv:2102.06183","author":"Lei Jie","year":"2021"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350906"},{"key":"e_1_3_2_1_23_1","volume-title":"SEA: Sentence Encoder Assembly for Video Retrieval by Textual Queries","author":"Li Xirong","year":"2020"},{"key":"e_1_3_2_1_24_1","volume-title":"HiT: Hierarchical Transformer with Momentum Contrast for Video-Text Retrieval. arXiv preprint arXiv:2103.15049","author":"Liu Song","year":"2021"},{"key":"e_1_3_2_1_25_1","volume-title":"Use what you have: Video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487","author":"Liu Yang","year":"2019"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3078971.3079041"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"e_1_3_2_1_28_1","volume-title":"Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516","author":"Miech Antoine","year":"2018"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3206025.3206064"},{"key":"e_1_3_2_1_30_1","unstructured":"Phuong Anh Nguyen Qing Li Zhi-Qi Cheng Yi-Jie Lu Hao Zhang Xiao Wu and Chong-Wah Ngo. 2017. VIREO@ TRECVID 2017: Video-to-Text Ad-hoc Video Search and Video hyperlinking.. In TRECVID.  Phuong Anh Nguyen Qing Li Zhi-Qi Cheng Yi-Jie Lu Hao Zhang Xiao Wu and Chong-Wah Ngo. 2017. VIREO@ TRECVID 2017: Video-to-Text Ad-hoc Video Search and Video hyperlinking.. In TRECVID."},{"key":"e_1_3_2_1_31_1","volume-title":"Jo ao Henriques, and Andrea Vedaldi","author":"Patrick Mandela","year":"2020"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_35"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_1_34_1","volume-title":"Ivan Titov, and Max Welling.","author":"Schlichtkrull Michael","year":"2018"},{"key":"e_1_3_2_1_35_1","volume-title":"Simple bert models for relation extraction and semantic role labeling. arXiv preprint arXiv:1904.05255","author":"Shi Peng","year":"2019"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00208"},{"key":"e_1_3_2_1_37_1","unstructured":"Kazuya Ueki Koji Hirakawa Kotaro Kikuchi Tetsuji Ogawa and Tetsunori Kobayashi. 2017. Waseda_Meisei at TRECVID 2017: Ad-hoc Video Search.. In TRECVID.  Kazuya Ueki Koji Hirakawa Kotaro Kikuchi Tetsuji Ogawa and Tetsunori Kobayashi. 2017. Waseda_Meisei at TRECVID 2017: Ad-hoc Video Search.. In TRECVID."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00468"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01302"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00365"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00054"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413916"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3062192"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_20"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6941"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.5555\/2886521.2886647"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401151"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.347"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_23"},{"key":"e_1_3_2_1_51_1","volume-title":"Stacked Convolutional Deep Encoding Network For Video-Text Retrieval. In 2020 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1--6.","author":"Zhao Rui","year":"2020"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475515","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475515","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:49:10Z","timestamp":1750193350000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475515"}},"subtitle":["Hierarchical Alignment Networks for Video-Text Retrieval"],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":51,"alternative-id":["10.1145\/3474085.3475515","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475515","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}