{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T10:44:35Z","timestamp":1758278675407,"version":"3.37.3"},"reference-count":68,"publisher":"Springer Science and Business Media LLC","issue":"19","license":[{"start":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T00:00:00Z","timestamp":1688428800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T00:00:00Z","timestamp":1688428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["62262006"],"award-info":[{"award-number":["62262006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Zhejiang Lab","award":["2021KE0AB01"],"award-info":[{"award-number":["2021KE0AB01"]}]},{"name":"Open Fund of Key Laboratory of Monitoring, Evaluation and Early Warning of Territorial Spatial Planning Implementation, Ministry of Natural Resources","award":["LMEE-KF2021008"],"award-info":[{"award-number":["LMEE-KF2021008"]}]},{"DOI":"10.13039\/100016086","name":"Key Laboratory in Science and Technology Development Project of Suzhou","doi-asserted-by":"publisher","award":["cstc2021jscx-gksbX0058"],"award-info":[{"award-number":["cstc2021jscx-gksbX0058"]}],"id":[{"id":"10.13039\/100016086","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Guangxi Key Laboratory of Trusted Software","award":["kx202006"],"award-info":[{"award-number":["kx202006"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s10489-023-04669-3","type":"journal-article","created":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T06:02:22Z","timestamp":1688450542000},"page":"22898-22916","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["TIAR: Text-Image-Audio Retrieval with weighted multimodal re-ranking"],"prefix":"10.1007","volume":"53","author":[{"given":"Peide","family":"Chi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8820-8388","authenticated-orcid":false,"given":"Yong","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingliang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xian-cai","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong-heng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bao-hua","family":"Qiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,7,4]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6077\u20136086","key":"4669_CR1","DOI":"10.1109\/CVPR.2018.00636"},{"key":"4669_CR2","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M (2020) wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems 33:12449\u201312460","journal-title":"Advances in Neural Information Processing Systems"},{"unstructured":"Brock A, De S, Smith SL, Simonyan K (2021) High-performance large-scale image recognition without normalization. In: International Conference on Machine Learning, PMLR, pp 1059\u20131071","key":"4669_CR3"},{"doi-asserted-by":"crossref","unstructured":"Chen H, Ding G, Liu X, Lin Z, Liu J, Han J (2020a) Imram: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12655\u201312663","key":"4669_CR4","DOI":"10.1109\/CVPR42600.2020.01267"},{"issue":"13","key":"4669_CR5","doi-asserted-by":"publisher","first-page":"15193","DOI":"10.1007\/s10489-021-03075-x","volume":"52","author":"L Chen","year":"2022","unstructured":"Chen L, Ren J, Chen P, Mao X, Zhao Q (2022) Limited text speech synthesis with electroglottograph based on bi-lstm and modified tacotron-2. Applied Intelligence 52(13):15193\u201315209","journal-title":"Applied Intelligence"},{"doi-asserted-by":"crossref","unstructured":"Chen YC, Li L, Yu L, El\u00a0Kholy A, Ahmed F, Gan Z, Cheng Y, Liu J (2020b) Uniter: Universal image-text representation learning. In: European conference on computer vision, Springer, pp 104\u2013120","key":"4669_CR6","DOI":"10.1007\/978-3-030-58577-8_7"},{"doi-asserted-by":"crossref","unstructured":"Cheng M, Sun Y, Wang L, Zhu X, Yao K, Chen J, Song G, Han J, Liu J, Ding E, et\u00a0al. (2022) Vista: Vision and scene text aggregation for cross-modal retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 5184\u20135193","key":"4669_CR7","DOI":"10.1109\/CVPR52688.2022.00512"},{"doi-asserted-by":"crossref","unstructured":"Cho J, Lu J, Schwenk D, Hajishirzi H, Kembhavi A (2020) X-LXMERT: Paint, Caption and Answer Questions with Multi-Modal Transformers. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), Association for Computational Linguistics, Online, pp 8785\u20138805","key":"4669_CR8","DOI":"10.18653\/v1\/2020.emnlp-main.707"},{"doi-asserted-by":"crossref","unstructured":"Chung YA, Zhang Y, Han W, Chiu CC, Qin J, Pang R, Wu Y (2021) W2v-bert: Combining contrastive learning and masked language modeling for self-supervised speech pre-training. In: 2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), IEEE, pp 244\u2013250","key":"4669_CR9","DOI":"10.1109\/ASRU51503.2021.9688253"},{"doi-asserted-by":"crossref","unstructured":"Conneau A, Khandelwal K, Goyal N, Chaudhary V, Wenzek G, Guzm\u00e1n F, Grave E, Ott M, Zettlemoyer L, Stoyanov V (2020) Unsupervised cross-lingual representation learning at scale. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, Association for Computational Linguistics, Online, pp 8440\u20138451","key":"4669_CR10","DOI":"10.18653\/v1\/2020.acl-main.747"},{"unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2019) BERT: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), Association for Computational Linguistics, Minneapolis, Minnesota, pp 4171\u20134186","key":"4669_CR11"},{"key":"4669_CR12","doi-asserted-by":"publisher","first-page":"1218","DOI":"10.1609\/aaai.v35i2.16209","volume":"35","author":"H Diao","year":"2021","unstructured":"Diao H, Zhang Y, Ma L, Lu H (2021) Similarity reasoning and filtration for image-text matching. Proceedings of the AAAI conference on artificial intelligence 35:1218\u20131226","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"4669_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.107138","volume":"226","author":"X Dong","year":"2021","unstructured":"Dong X, Zhang H, Dong X, Lu X (2021) Iterative graph attention memory network for cross-modal retrieval. Knowledge-Based Systems 226:107138","journal-title":"Knowledge-Based Systems"},{"doi-asserted-by":"crossref","unstructured":"Dou ZY, Xu Y, Gan Z, Wang J, Wang S, Wang L, Zhu C, Zhang P, Yuan L, Peng N, et\u00a0al. (2022) An empirical study of training end-to-end vision-and-language transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18166\u201318176","key":"4669_CR14","DOI":"10.1109\/CVPR52688.2022.01763"},{"doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","key":"4669_CR15","DOI":"10.1109\/CVPR.2016.90"},{"doi-asserted-by":"crossref","unstructured":"He K, Chen X, Xie S, Li Y, Doll\u00e1r P, Girshick R (2022) Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 16000\u201316009","key":"4669_CR16","DOI":"10.1109\/CVPR52688.2022.01553"},{"issue":"4","key":"4669_CR17","doi-asserted-by":"publisher","first-page":"4257","DOI":"10.1007\/s10489-022-03653-7","volume":"53","author":"P He","year":"2023","unstructured":"He P, Wang M, Tu D, Wang Z (2023) Dual discriminant adversarial cross-modal retrieval. Applied Intelligence 53(4):4257\u20134267","journal-title":"Applied Intelligence"},{"key":"4669_CR18","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"WN Hsu","year":"2021","unstructured":"Hsu WN, Bolte B, Tsai YHH, Lakhotia K, Salakhutdinov R, Mohamed A (2021) Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 29:3451\u20133460","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"doi-asserted-by":"crossref","unstructured":"Huang Z, Zeng Z, Huang Y, Liu B, Fu D, Fu J (2021) Seeing out of the box: End-to-end pre-training for vision-language representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 12976\u201312985","key":"4669_CR19","DOI":"10.1109\/CVPR46437.2021.01278"},{"unstructured":"Jia C, Yang Y, Xia Y, Chen YT, Parekh Z, Pham H, Le Q, Sung YH, Li Z, Duerig T (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, PMLR, pp 4904\u20134916","key":"4669_CR20"},{"doi-asserted-by":"crossref","unstructured":"Jiang H, Misra I, Rohrbach M, Learned-Miller E, Chen X (2020) In defense of grid features for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10267\u201310276","key":"4669_CR21","DOI":"10.1109\/CVPR42600.2020.01028"},{"key":"4669_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.108354","volume":"242","author":"M Jin","year":"2022","unstructured":"Jin M, Zhang H, Zhu L, Sun J, Liu L (2022) Coarse-to-fine dual-level attention for video-text cross modal retrieval. Knowledge-Based Systems 242:108354","journal-title":"Knowledge-Based Systems"},{"issue":"1","key":"4669_CR23","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/s10489-021-02308-3","volume":"52","author":"P Kang","year":"2022","unstructured":"Kang P, Lin Z, Yang Z, Fang X, Bronstein AM, Li Q, Liu W (2022) Intra-class low-rank regularization for supervised and semi-supervised cross-modal retrieval. Applied Intelligence 52(1):33\u201354","journal-title":"Applied Intelligence"},{"unstructured":"Kenton JDMWC, Toutanova LK (2019) Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of NAACL-HLT, pp 4171\u20134186","key":"4669_CR24"},{"key":"4669_CR25","first-page":"7463","volume-title":"ICASSP 2021\u20132021 IEEE International Conference on Acoustics","author":"S Kim","year":"2021","unstructured":"Kim S, Kim G, Shin S, Lee S (2021) Two-stage textual knowledge distillation for end-to-end spoken language understanding. ICASSP 2021\u20132021 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 7463\u20137467"},{"unstructured":"Kim W, Son B, Kim I (2021b) Vilt: Vision-and-language transformer without convolution or region supervision. In: Meila M, Zhang T (eds) Proceedings of the 38th International Conference on Machine Learning, PMLR, Proceedings of Machine Learning Research, vol 139, pp 5583\u20135594,","key":"4669_CR26"},{"doi-asserted-by":"crossref","unstructured":"Kong D, Li X, Wang S, Li J, Yin B (2022) Learning visual-and-semantic knowledge embedding for zero-shot image classification. Applied Intelligence pp 1\u201315","key":"4669_CR27","DOI":"10.1007\/s10489-022-03443-1"},{"key":"4669_CR28","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li J, Selvaraju R, Gotmare A, Joty S, Xiong C, Hoi SCH (2021) Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems 34:9694\u20139705","journal-title":"Advances in neural information processing systems"},{"unstructured":"Li J, Li D, Xiong C, Hoi S (2022) Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, PMLR, pp 12888\u201312900","key":"4669_CR29"},{"doi-asserted-by":"crossref","unstructured":"Lin TY, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: Common objects in context. In: European conference on computer vision, Springer, pp 740\u2013755","key":"4669_CR30","DOI":"10.1007\/978-3-319-10602-1_48"},{"doi-asserted-by":"crossref","unstructured":"Liu C, Mao Z, Zhang T, Xie H, Wang B, Zhang Y (2020) Graph structured network for image-text matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10921\u201310930","key":"4669_CR31","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"4669_CR32","doi-asserted-by":"publisher","first-page":"1802","DOI":"10.1007\/s10489-020-01930-x","volume":"51","author":"H Liu","year":"2021","unstructured":"Liu H, Feng Y, Zhou M, Qiang B (2021) Semantic ranking structure preserving for cross-modal retrieval. Applied Intelligence 51:1802\u20131812","journal-title":"Applied Intelligence"},{"doi-asserted-by":"crossref","unstructured":"Liu Y, Ji S, Fu Q, Zhao J, Zhao Z, Gong M (2022) Latent semantic-enhanced discrete hashing for cross-modal retrieval. Applied Intelligence pp 1\u201317","key":"4669_CR33","DOI":"10.1007\/s10489-021-03143-2"},{"doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021b) Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 10012\u201310022","key":"4669_CR34","DOI":"10.1109\/ICCV48922.2021.00986"},{"issue":"22","key":"4669_CR35","doi-asserted-by":"publisher","first-page":"7665","DOI":"10.3390\/s21227665","volume":"21","author":"C Luna-Jim\u00e9nez","year":"2021","unstructured":"Luna-Jim\u00e9nez C, Griol D, Callejas Z, Kleinlein R, Montero JM, Fern\u00e1ndez-Mart\u00ednez F (2021) Multimodal emotion recognition on ravdess dataset using transfer learning. Sensors 21(22):7665","journal-title":"Sensors"},{"issue":"1","key":"4669_CR36","doi-asserted-by":"publisher","first-page":"327","DOI":"10.3390\/app12010327","volume":"12","author":"C Luna-Jim\u00e9nez","year":"2021","unstructured":"Luna-Jim\u00e9nez C, Kleinlein R, Griol D, Callejas Z, Montero JM, Fern\u00e1ndez-Mart\u00ednez F (2021) A proposal for multimodal emotion recognition using aural transformers and action units on ravdess dataset. Applied Sciences 12(1):327","journal-title":"Applied Sciences"},{"key":"4669_CR37","doi-asserted-by":"publisher","first-page":"2998","DOI":"10.1109\/TMM.2021.3091888","volume":"24","author":"X Ma","year":"2021","unstructured":"Ma X, Yang X, Gao J, Xu C (2021) The model may fit you: User-generalized cross-modal retrieval. IEEE Transactions on Multimedia 24:2998\u20133012","journal-title":"IEEE Transactions on Multimedia"},{"issue":"6","key":"4669_CR38","doi-asserted-by":"publisher","first-page":"9411","DOI":"10.1007\/s11042-020-10073-7","volume":"80","author":"M Malik","year":"2021","unstructured":"Malik M, Malik MK, Mehmood K, Makhdoom I (2021) Automatic speech recognition: a survey. Multimedia Tools and Applications 80(6):9411\u20139457","journal-title":"Multimedia Tools and Applications"},{"unstructured":"Mei X, Liu X, Huang Q, Plumbley MD, Wang W (2021) Audio captioning transformer. In: Proceedings of the 6th Detection and Classification of Acoustic Scenes and Events 2021, Barcelona, Spain, pp 211\u2013215","key":"4669_CR39"},{"doi-asserted-by":"crossref","unstructured":"Plummer BA, Wang L, Cervantes CM, Caicedo JC, Hockenmaier J, Lazebnik S (2015) Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE international conference on computer vision, pp 2641\u20132649","key":"4669_CR40","DOI":"10.1109\/ICCV.2015.303"},{"doi-asserted-by":"crossref","unstructured":"Pont-Tuset J, Uijlings J, Changpinyo S, Soricut R, Ferrari V (2020) Connecting vision and language with localized narratives. In: European conference on computer vision, Springer, pp 647\u2013664","key":"4669_CR41","DOI":"10.1007\/978-3-030-58558-7_38"},{"key":"4669_CR42","doi-asserted-by":"publisher","first-page":"2989","DOI":"10.1109\/TIP.2020.3048680","volume":"30","author":"M Qi","year":"2021","unstructured":"Qi M, Qin J, Yang Y, Wang Y, Luo J (2021) Semantics-aware spatial-temporal binaries for cross-modal video retrieval. IEEE Transactions on Image Processing 30:2989\u20133004","journal-title":"IEEE Transactions on Image Processing"},{"key":"4669_CR43","first-page":"7458","volume-title":"ICASSP 2021\u20132021 IEEE International Conference on Acoustics","author":"Y Qian","year":"2021","unstructured":"Qian Y, Bianv X, Shi Y, Kanda N, Shen L, Xiao Z, Zeng M (2021) Speech-language pre-training for end-to-end spoken language understanding. ICASSP 2021\u20132021 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 7458\u20137462"},{"unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, et\u00a0al. (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, PMLR, pp 8748\u20138763","key":"4669_CR44"},{"unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems 28","key":"4669_CR45"},{"key":"4669_CR46","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1162\/tacl_a_00353","volume":"9","author":"A Roy","year":"2021","unstructured":"Roy A, Saffar M, Vaswani A, Grangier D (2021) Efficient content-based sparse attention with routing transformers. Transactions of the Association for Computational Linguistics 9:53\u201368","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"4669_CR47","first-page":"3465","volume":"2019","author":"S Schneider","year":"2019","unstructured":"Schneider S, Baevski A, Collobert R, Auli M (2019) wav2vec: Unsupervised pre-training for speech recognition. Proc Interspeech 2019:3465\u20133469","journal-title":"Proc Interspeech"},{"key":"4669_CR48","first-page":"7152","volume-title":"ICASSP 2022\u20132022 IEEE International Conference on Acoustics","author":"S Seo","year":"2022","unstructured":"Seo S, Kwak D, Lee B (2022) Integration of pre-trained networks with continuous token interface for end-to-end spoken language understanding. ICASSP 2022\u20132022 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 7152\u20137156"},{"unstructured":"Stein BE, Meredith MA (1993) The merging of the senses. The MIT press","key":"4669_CR49"},{"issue":"1","key":"4669_CR50","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1162\/jocn.1989.1.1.12","volume":"1","author":"BE Stein","year":"1989","unstructured":"Stein BE, Meredith MA, Huneycutt WS, McDade L (1989) Behavioral indices of multisensory integration: orientation to visual cues is affected by auditory stimuli. Journal of Cognitive Neuroscience 1(1):12\u201324","journal-title":"Journal of Cognitive Neuroscience"},{"doi-asserted-by":"crossref","unstructured":"Tang C, Ma K, Cui B, Ji K, Abraham A (2022) Long text feature extraction network with data augmentation. Applied Intelligence pp 1\u201316","key":"4669_CR51","DOI":"10.1007\/s10489-022-03185-0"},{"unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Advances in neural information processing systems 30","key":"4669_CR52"},{"issue":"3","key":"4669_CR53","doi-asserted-by":"publisher","first-page":"2872","DOI":"10.1007\/s10489-021-02573-2","volume":"52","author":"L Wang","year":"2022","unstructured":"Wang L, He K, Feng X, Ma X (2022) Multilayer feature fusion with parallel convolutional block for fine-grained image classification. Applied Intelligence 52(3):2872\u20132883","journal-title":"Applied Intelligence"},{"doi-asserted-by":"crossref","unstructured":"Wang T, Xu X, Yang Y, Hanjalic A, Shen HT, Song J (2019) Matching images and text with multi-modal tensor fusion and re-ranking. In: Proceedings of the 27th ACM international conference on multimedia, pp 12\u201320","key":"4669_CR54","DOI":"10.1145\/3343031.3350875"},{"issue":"13","key":"4669_CR55","doi-asserted-by":"publisher","first-page":"14839","DOI":"10.1007\/s10489-022-03227-7","volume":"52","author":"X Wu","year":"2022","unstructured":"Wu X, Ji S, Wang J, Guo Y (2022) Speech synthesis with face embeddings. Applied Intelligence 52(13):14839\u201314852","journal-title":"Applied Intelligence"},{"key":"4669_CR56","first-page":"3030","volume-title":"ICASSP 2021\u20132021 IEEE International Conference on Acoustics","author":"Q Xu","year":"2021","unstructured":"Xu Q, Baevski A, Likhomanenko T, Tomasello P, Conneau A, Collobert R, Synnaeve G, Auli M (2021) Self-training and pre-training are complementary for speech recognition. ICASSP 2021\u20132021 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP), IEEE, pp 3030\u20133034"},{"doi-asserted-by":"crossref","unstructured":"Yang J, Duan J, Tran S, Xu Y, Chanda S, Chen L, Zeng B, Chilimbi T, Huang J (2022) Vision-language pre-training with triple contrastive learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 15671\u201315680","key":"4669_CR57","DOI":"10.1109\/CVPR52688.2022.01522"},{"key":"4669_CR58","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.109511","volume":"253","author":"L You","year":"2022","unstructured":"You L, Han F, Peng J, Jin H, Claramunt C (2022) Ask-roberta: A pretraining model for aspect-based sentiment classification via sentiment knowledge mining. Knowledge-Based Systems 253:109511","journal-title":"Knowledge-Based Systems"},{"doi-asserted-by":"crossref","unstructured":"Zeng D, Yu Y, Oyama K (2020) Deep triplet neural networks with cluster-cca for audio-visual cross-modal retrieval. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) 16(3):1\u201323","key":"4669_CR59","DOI":"10.1145\/3387164"},{"unstructured":"Zeng Y, Zhang X, Li H (2022) Multi-grained vision language pre-training: Aligning texts with visual concepts. In: Chaudhuri K, Jegelka S, Song L, Szepesvari C, Niu G, Sabato S (eds) Proceedings of the 39th International Conference on Machine Learning, PMLR, Proceedings of Machine Learning Research, vol 162, pp 25994\u201326009,","key":"4669_CR60"},{"doi-asserted-by":"crossref","unstructured":"Zhai X, Kolesnikov A, Houlsby N, Beyer L (2022) Scaling vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 12104\u201312113","key":"4669_CR61","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"4669_CR62","doi-asserted-by":"publisher","first-page":"1000","DOI":"10.1109\/TIP.2021.3138302","volume":"31","author":"F Zhang","year":"2021","unstructured":"Zhang F, Xu M, Xu C (2021) Geometry sensitive cross-modal reasoning for composed query based image retrieval. IEEE Transactions on Image Processing 31:1000\u20131011","journal-title":"IEEE Transactions on Image Processing"},{"key":"4669_CR63","doi-asserted-by":"publisher","first-page":"7154","DOI":"10.1109\/TIP.2022.3220051","volume":"31","author":"L Zhang","year":"2022","unstructured":"Zhang L, Wu X (2022) Latent space semantic supervision based on knowledge distillation for cross-modal retrieval. IEEE Transactions on Image Processing 31:7154\u20137164","journal-title":"IEEE Transactions on Image Processing"},{"doi-asserted-by":"crossref","unstructured":"Zhang P, Li X, Hu X, Yang J, Zhang L, Wang L, Choi Y, Gao J (2021b) Vinvl: Revisiting visual representations in vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 5579\u20135588","key":"4669_CR64","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"4669_CR65","doi-asserted-by":"publisher","first-page":"617","DOI":"10.1109\/TIP.2020.3038354","volume":"30","author":"Y Zhang","year":"2020","unstructured":"Zhang Y, Zhou W, Wang M, Tian Q, Li H (2020) Deep relation embedding for cross-modal retrieval. IEEE Transactions on Image Processing 30:617\u2013627","journal-title":"IEEE Transactions on Image Processing"},{"doi-asserted-by":"crossref","unstructured":"Zhao J, Zhou X, Shi G, Xiao N, Song K, Zhao J, Hao R, Li K (2022) Semantic consistency generative adversarial network for cross-modality domain adaptation in ultrasound thyroid nodule classification. Applied Intelligence pp 1\u201315","key":"4669_CR66","DOI":"10.1007\/s10489-021-03025-7"},{"issue":"3","key":"4669_CR67","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/s11263-018-1140-0","volume":"127","author":"B Zhou","year":"2019","unstructured":"Zhou B, Zhao H, Puig X, Xiao T, Fidler S, Barriuso A, Torralba A (2019) Semantic understanding of scenes through the ade20k dataset. International Journal of Computer Vision 127(3):302-321","journal-title":"International Journal of Computer Vision"},{"key":"4669_CR68","doi-asserted-by":"publisher","first-page":"5927","DOI":"10.1007\/s10489-020-02137-w","volume":"51","author":"L Zhu","year":"2021","unstructured":"Zhu L, Tian G, Wang B, Wang W, Zhang D, Li C (2021) Multi-attention based semantic deep hashing for cross-modal retrieval. Applied Intelligence 51:5927\u20135939","journal-title":"Applied Intelligence"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-023-04669-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-023-04669-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-023-04669-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,18]],"date-time":"2023-10-18T13:29:35Z","timestamp":1697635775000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-023-04669-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,4]]},"references-count":68,"journal-issue":{"issue":"19","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["4669"],"URL":"https:\/\/doi.org\/10.1007\/s10489-023-04669-3","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2023,7,4]]},"assertion":[{"value":"25 April 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 July 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing fnancial interests or personal relationships that could have appeared to infuence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}