{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,31]],"date-time":"2025-05-31T09:24:16Z","timestamp":1748683456202,"version":"3.28.0"},"reference-count":22,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,6,28]],"date-time":"2021-06-28T00:00:00Z","timestamp":1624838400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,6,28]],"date-time":"2021-06-28T00:00:00Z","timestamp":1624838400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,6,28]],"date-time":"2021-06-28T00:00:00Z","timestamp":1624838400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,6,28]]},"DOI":"10.1109\/cbmi50038.2021.9461890","type":"proceedings-article","created":{"date-parts":[[2021,6,24]],"date-time":"2021-06-24T20:11:33Z","timestamp":1624565493000},"page":"1-6","source":"Crossref","is-referenced-by-count":6,"title":["Towards Efficient Cross-Modal Visual Textual Retrieval using Transformer-Encoder Deep Features"],"prefix":"10.1109","author":[{"given":"Nicola","family":"Messina","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giuseppe","family":"Amato","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fabrizio","family":"Falchi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Claudio","family":"Gennaro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stephane","family":"Marchand-Maillet","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Transformer reasoning network for image-text matching and retrieval","year":"2020","author":"messina","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-012-1271-1"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46759-7_7"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2019.102100"},{"key":"ref14","article-title":"Particular object retrieval with integral max-pooling of CNN activations","author":"tolias","year":"2016","journal-title":"ICLR 2016"},{"key":"ref15","first-page":"647","article-title":"Decaf: A deep convolutional activation feature for generic visual recognition","volume":"32","author":"donahue","year":"2014","journal-title":"ICML 2014"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2014.131"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.96"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-018-6210-3"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2003.1238663"},{"key":"ref4","first-page":"12","article-title":"VSE++: improving visual-semantic embeddings with hard negatives","author":"faghri","year":"2018","journal-title":"BMVC 2018"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00475"},{"key":"ref6","first-page":"13","article-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","author":"lu","year":"2019","journal-title":"NeurIPS 2019"},{"key":"ref5","article-title":"Learning visual relation priors for image-text matching and image captioning with neural scene graph generators","volume":"abs 1909 9953","author":"lee","year":"2019","journal-title":"CoRR"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00636"},{"key":"ref1","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","volume":"37","author":"xu","year":"2015","journal-title":"ICML 2015"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/s10791-017-9318-6"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0285-2"},{"key":"ref22","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"NeurIPS 2017"},{"key":"ref21","first-page":"4171","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"NAACL HLT 2019"}],"event":{"name":"2021 International Conference on Content-Based Multimedia Indexing (CBMI)","start":{"date-parts":[[2021,6,28]]},"location":"Lille, France","end":{"date-parts":[[2021,6,30]]}},"container-title":["2021 International Conference on Content-Based Multimedia Indexing (CBMI)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9461872\/9461873\/09461890.pdf?arnumber=9461890","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:42:52Z","timestamp":1652197372000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9461890\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,28]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/cbmi50038.2021.9461890","relation":{},"subject":[],"published":{"date-parts":[[2021,6,28]]}}}