{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T19:38:27Z","timestamp":1740166707207,"version":"3.37.3"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,5,11]],"date-time":"2022-05-11T00:00:00Z","timestamp":1652227200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,5,11]],"date-time":"2022-05-11T00:00:00Z","timestamp":1652227200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R &D Program of China","doi-asserted-by":"crossref","award":["2019YFC1521204"],"award-info":[{"award-number":["2019YFC1521204"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2022,9]]},"DOI":"10.1007\/s13735-022-00237-6","type":"journal-article","created":{"date-parts":[[2022,5,11]],"date-time":"2022-05-11T11:05:15Z","timestamp":1652267115000},"page":"369-382","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Semantic-enhanced discriminative embedding learning for cross-modal retrieval"],"prefix":"10.1007","volume":"11","author":[{"given":"Hao","family":"Pan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4939-3880","authenticated-orcid":false,"given":"Jun","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,5,11]]},"reference":[{"key":"237_CR1","unstructured":"Andrew G, Arora R, Bilmes J, Livescu K (2013) Deep canonical correlation analysis. In: International conference on machine learning. PMLR"},{"key":"237_CR2","doi-asserted-by":"crossref","unstructured":"Biten AF, Mafla A, G\u00f3mez L, Karatzas D (2022) Is an image worth five sentences? a new look into semantics for image-text matching. In: Proceedings of the IEEE winter conference on applications of computer vision","DOI":"10.1109\/WACV51458.2022.00254"},{"key":"237_CR3","doi-asserted-by":"crossref","unstructured":"Chen T, Deng J, Luo J (2020) Adaptive offline quintuplet loss for image-text matching. In: European conference on computer vision. Springer","DOI":"10.1007\/978-3-030-58601-0_33"},{"key":"237_CR4","unstructured":"Faghri F, Fleet DJ, Kiros JR, Fidler S (2018) VSE++: improving visual-semantic embeddings with hard negatives. In: British machine vision conference 2018, BMVC"},{"key":"237_CR5","doi-asserted-by":"crossref","unstructured":"Hadsell R, Chopra S, LeCun Y (2006) Dimensionality reduction by learning an invariant mapping. In: 2006 IEEE computer society conference on computer vision and pattern recognition (CVPR\u201906), vol\u00a02. IEEE","DOI":"10.1109\/CVPR.2006.100"},{"key":"237_CR6","doi-asserted-by":"crossref","unstructured":"He K, Fan H, Wu Y, Xie S, Girshick R (2020) Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"237_CR7","doi-asserted-by":"crossref","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8)","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"237_CR8","doi-asserted-by":"crossref","unstructured":"Hoffer E, Ailon N (2015) Deep metric learning using triplet network. In: International workshop on similarity-based pattern recognition. Springer","DOI":"10.1007\/978-3-319-24261-3_7"},{"key":"237_CR9","doi-asserted-by":"crossref","unstructured":"Karpathy A,\u00a0Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"237_CR10","unstructured":"Kiros R, Salakhutdinov R, Zemel RS (2014) Unifying visual-semantic embeddings with multimodal neural language models. arXiv:1411.2539"},{"key":"237_CR11","doi-asserted-by":"crossref","unstructured":"Lee K-H,\u00a0Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for image-text matching. In: Proceedings of the European conference on computer vision (ECCV)","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"237_CR12","doi-asserted-by":"crossref","unstructured":"Li K, Zhang Y, Li K, Li Y, Fu Y (2019) Visual semantic reasoning for image-text matching. In: Proceedings of the IEEE\/CVF international conference on computer vision","DOI":"10.1109\/ICCV.2019.00475"},{"key":"237_CR13","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: common objects in context. In: European conference on computer vision. Springer, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"237_CR14","doi-asserted-by":"crossref","unstructured":"Liu F, Ye R, Wang X, Li S (2020) Hal: improved text-image matching by mitigating visual semantic hubs. In: Proceedings of the AAAI conference on artificial intelligence, vol\u00a034","DOI":"10.1609\/aaai.v34i07.6823"},{"key":"237_CR15","doi-asserted-by":"crossref","unstructured":"Liu X, Wang Z, Shao J, Wang X, Li H (2019) Improving referring expression grounding with cross-modal attention-guided erasing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2019.00205"},{"key":"237_CR16","doi-asserted-by":"crossref","unstructured":"Mithun NC, Li J, Metze F, Roy-Chowdhury AK (2018) Learning joint embedding with multimodal cues for cross-modal video-text retrieval. In: Proceedings of the 2018 ACM on international conference on multimedia retrieval","DOI":"10.1145\/3206025.3206064"},{"key":"237_CR17","doi-asserted-by":"crossref","unstructured":"Nam H, Ha J-W, Kim J (2017) Dual attention networks for multimodal reasoning and matching. In: Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2017.232"},{"key":"237_CR18","unstructured":"Oord AVD, Li Y, Vinyals O (2018) Representation learning with contrastive predictive coding. arXiv:1807.03748"},{"key":"237_CR19","doi-asserted-by":"crossref","unstructured":"Song HO,\u00a0Xiang Y, Jegelka S, Savarese S (2016) Deep metric learning via lifted structured feature embedding. In: Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2016.434"},{"key":"237_CR20","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: towards real-time object detection with region proposal networks. In: Advances in neural information processing systems, pp 28"},{"key":"237_CR21","doi-asserted-by":"crossref","unstructured":"Wang H, Zhang Y, Ji Z, Pang Y, Ma L (2020) Consensus-aware visual-semantic embedding for image-text matching. In: European conference on computer vision. Springer","DOI":"10.1007\/978-3-030-58586-0_2"},{"key":"237_CR22","doi-asserted-by":"crossref","unstructured":"Wang L, Li Y, Huang J, Lazebnik S (2018) Learning two-branch neural networks for image-text matching tasks. IEEE Trans Pattern Anal Mach Intell 41(2)","DOI":"10.1109\/TPAMI.2018.2797921"},{"key":"237_CR23","doi-asserted-by":"crossref","unstructured":"Wang L, Li Y, Lazebnik S (2016) Learning deep structure-preserving image-text embeddings. In: Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2016.541"},{"key":"237_CR24","doi-asserted-by":"crossref","unstructured":"Wang X, Han X, Huang W, Dong D, Scott MR (2019) Multi-similarity loss with general pair weighting for deep metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2019.00516"},{"key":"237_CR25","doi-asserted-by":"crossref","unstructured":"Wei J, Xu X, Yang Y, Ji Y, Wang Z, Shen HT (2020) Universal weighting metric learning for cross-modal matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR42600.2020.01302"},{"key":"237_CR26","doi-asserted-by":"crossref","unstructured":"Wei X, Zhang T, Li Y, Zhang Y, Wu F (2020) Multi-modality cross attention network for image and sentence matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"237_CR27","doi-asserted-by":"crossref","unstructured":"Wen K, Gu X, Cheng Q (2020) Learning dual semantic relations with graph attention for image-text matching. In: IEEE transactions on circuits and systems for video technology","DOI":"10.1109\/TCSVT.2020.3030656"},{"key":"237_CR28","doi-asserted-by":"crossref","unstructured":"Wu Z, Xiong Y, Yu SX, Lin D (2018) Unsupervised feature learning via non-parametric instance discrimination. In: Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2018.00393"},{"key":"237_CR29","doi-asserted-by":"crossref","unstructured":"Young P, Lai A, Hodosh M, Hockenmaier J (2014) From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans Assoc Comput Linguist 2","DOI":"10.1162\/tacl_a_00166"},{"key":"237_CR30","doi-asserted-by":"crossref","unstructured":"Yu R, Dou Z, Bai S, Zhang Z, Xu Y, Bai X (2018) Hard-aware point-to-set deep metric for person re-identification. In: Proceedings of the European conference on computer vision (ECCV)","DOI":"10.1007\/978-3-030-01270-0_12"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00237-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-022-00237-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00237-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,24]],"date-time":"2024-09-24T11:09:11Z","timestamp":1727176151000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-022-00237-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,5,11]]},"references-count":30,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,9]]}},"alternative-id":["237"],"URL":"https:\/\/doi.org\/10.1007\/s13735-022-00237-6","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"type":"print","value":"2192-6611"},{"type":"electronic","value":"2192-662X"}],"subject":[],"published":{"date-parts":[[2022,5,11]]},"assertion":[{"value":"3 February 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 April 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 May 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}