{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T10:48:07Z","timestamp":1725706087263},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T00:00:00Z","timestamp":1665878400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T00:00:00Z","timestamp":1665878400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,10,16]]},"DOI":"10.1109\/icip46576.2022.9897206","type":"proceedings-article","created":{"date-parts":[[2022,11,3]],"date-time":"2022-11-03T21:27:24Z","timestamp":1667510844000},"page":"386-390","source":"Crossref","is-referenced-by-count":0,"title":["Context-Aware Hierarchical Transformer for Fine-Grained Video-Text Retrieval"],"prefix":"10.1109","author":[{"given":"Mingliang","family":"Chen","sequence":"first","affiliation":[{"name":"Peking University Shenzhen Graduate School,School of Electronic and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weimin","family":"Zhang","sequence":"additional","affiliation":[{"name":"AVS Industry Alliance,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yurui","family":"Ren","sequence":"additional","affiliation":[{"name":"Peking University Shenzhen Graduate School,School of Electronic and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ge","family":"Li","sequence":"additional","affiliation":[{"name":"Peking University Shenzhen Graduate School,School of Electronic and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Unifying visual-semantic embeddings with multimodal neural language models","year":"2014","author":"Kiros","key":"ref1"},{"article-title":"Vse++: Improving visual-semantic embeddings with hard negatives","volume-title":"Proceedings of the British Machine Vision Conference","author":"Faghri","key":"ref2"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00208"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00957"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01065"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475515"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"ref8","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","volume-title":"International Conference on Machine Learning","author":"Kim"},{"key":"ref9","first-page":"5998","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01035"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"article-title":"The kinetics human action video dataset","year":"2017","author":"Kay","key":"ref15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/140"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP42928.2021.9506697"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.502"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICME51207.2021.9428215"}],"event":{"name":"2022 IEEE International Conference on Image Processing (ICIP)","start":{"date-parts":[[2022,10,16]]},"location":"Bordeaux, France","end":{"date-parts":[[2022,10,19]]}},"container-title":["2022 IEEE International Conference on Image Processing (ICIP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9897158\/9897159\/09897206.pdf?arnumber=9897206","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T20:59:46Z","timestamp":1705957186000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9897206\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,16]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/icip46576.2022.9897206","relation":{},"subject":[],"published":{"date-parts":[[2022,10,16]]}}}