{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T16:05:34Z","timestamp":1786982734854,"version":"3.56.0"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,2,22]],"date-time":"2023-02-22T00:00:00Z","timestamp":1677024000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,2,22]],"date-time":"2023-02-22T00:00:00Z","timestamp":1677024000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62003065"],"award-info":[{"award-number":["62003065"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science and Technology Project of Chongqing Education Commission of China","award":["KJQN201900520"],"award-info":[{"award-number":["KJQN201900520"]}]},{"name":"Chongqing Normal University Fund","award":["NO.22XLB003"],"award-info":[{"award-number":["NO.22XLB003"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s13735-023-00268-7","type":"journal-article","created":{"date-parts":[[2023,2,22]],"date-time":"2023-02-22T07:04:22Z","timestamp":1677049462000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":34,"title":["CLIP-based fusion-modal reconstructing hashing for large-scale unsupervised cross-modal retrieval"],"prefix":"10.1007","volume":"12","author":[{"given":"Li","family":"Mingyong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Yewen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ge","family":"Mingyuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ma","family":"Longfei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,2,22]]},"reference":[{"issue":"4","key":"268_CR1","doi-asserted-by":"publisher","first-page":"1445","DOI":"10.1109\/TPAMI.2020.2975798","volume":"43","author":"C Yan","year":"2020","unstructured":"Yan C, Gong B, Wei Y, Gao Y (2020) Deep multi-view enhancement hashing for image retrieval. IEEE Trans Pattern Anal Mach Intell 43(4):1445\u20131451","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"268_CR2","unstructured":"Gong Q, Wang L, Lai H, Pan Y, Yin J (2022) Vit2hash: unsupervised information-preserving hashing. arXiv preprint arXiv:2201.05541"},{"key":"268_CR3","doi-asserted-by":"crossref","unstructured":"Dubey SR, Singh SK, Chu W-T (2021) Vision transformer hashing for image retrieval. arXiv preprint arXiv:2109.12564","DOI":"10.1109\/ICME52920.2022.9859900"},{"key":"268_CR4","doi-asserted-by":"crossref","unstructured":"Wang B, Yang Y, Xu X, Hanjalic A, Shen HT (2017) Adversarial cross-modal retrieval. In: Proceedings of the 25th ACM international conference on multimedia, pp 154\u2013162","DOI":"10.1145\/3123266.3123326"},{"key":"268_CR5","doi-asserted-by":"crossref","unstructured":"Gu W, Gu X, Gu J, Li B, Xiong Z, Wang W (2019) Adversary guided asymmetric hashing for cross-modal retrieval. In: Proceedings of the 2019 on international conference on multimedia retrieval, pp 159\u2013167","DOI":"10.1145\/3323873.3325045"},{"key":"268_CR6","doi-asserted-by":"publisher","first-page":"4284","DOI":"10.1109\/JSTARS.2021.3070872","volume":"14","author":"Q Cheng","year":"2021","unstructured":"Cheng Q, Zhou Y, Fu P, Xu Y, Zhang L (2021) A deep semantic alignment network for the cross-modal image-text retrieval in remote sensing. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing 14:4284\u20134297","journal-title":"IEEE J Sel Top Appl Earth Obs Remote Sens"},{"key":"268_CR7","unstructured":"Kumar S, Udupa R (2011) Learning hash functions for cross-view similarity search. In: Twenty-second international joint conference on artificial intelligence"},{"key":"268_CR8","doi-asserted-by":"crossref","unstructured":"Song J, Yang Y, Yang Y, Huang Z, Shen HT (2013) Inter-media hashing for large-scale retrieval from heterogeneous data sources. In: Proceedings of the 2013 ACM SIGMOD international conference on management of data, pp 785\u2013796","DOI":"10.1145\/2463676.2465274"},{"key":"268_CR9","doi-asserted-by":"crossref","unstructured":"Bai C, Zeng C, Ma Q, Zhang J, Chen S (2020) Deep adversarial discrete hashing for cross-modal retrieval. In: Proceedings of the 2020 international conference on multimedia retrieval, pp 525\u2013531","DOI":"10.1145\/3372278.3390711"},{"key":"268_CR10","doi-asserted-by":"crossref","unstructured":"Lu X, Zhu L, Cheng Z, Li J, Nie X, Zhang H (2019) Flexible online multi-modal hashing for large-scale multimedia retrieval. In: Proceedings of the 27th ACM international conference on multimedia, pp 1129\u20131137","DOI":"10.1145\/3343031.3350999"},{"key":"268_CR11","doi-asserted-by":"crossref","unstructured":"Fan L, Ng KW, Ju C, Zhang T, Chan CS (2020) Deep polarized network for supervised learning of accurate binary hashing codes. In: IJCAI, pp 825\u2013831","DOI":"10.24963\/ijcai.2020\/115"},{"key":"268_CR12","doi-asserted-by":"crossref","unstructured":"Su S, Zhong Z, Zhang C (2019) Deep joint-semantics reconstructing hashing for large-scale unsupervised cross-modal retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3027\u20133035","DOI":"10.1109\/ICCV.2019.00312"},{"key":"268_CR13","doi-asserted-by":"crossref","unstructured":"Li C, Deng C, Wang L, Xie D, Liu X (2019) Coupled cyclegan: Unsupervised hashing network for cross-modal retrieval. In: Proceedings of the AAAI conference on artificial intelligence, vol. 33, pp 176\u2013183","DOI":"10.1609\/aaai.v33i01.3301176"},{"key":"268_CR14","doi-asserted-by":"publisher","first-page":"466","DOI":"10.1109\/TMM.2021.3053766","volume":"24","author":"P-F Zhang","year":"2021","unstructured":"Zhang P-F, Li Y, Huang Z, Xu X-S (2021) Aggregation-based graph convolutional hashing for unsupervised cross-modal retrieval. IEEE Trans Multimedia 24:466\u2013479","journal-title":"IEEE Trans Multimed"},{"key":"268_CR15","doi-asserted-by":"crossref","unstructured":"Shen X, Zhang H, Li L, Liu L (2021) Attention-guided semantic hashing for unsupervised cross-modal retrieval. In: 2021 IEEE international conference on multimedia and expo (ICME), pp 1\u20136. IEEE","DOI":"10.1109\/ICME51207.2021.9428330"},{"issue":"12","key":"268_CR16","doi-asserted-by":"publisher","first-page":"3034","DOI":"10.1109\/TPAMI.2018.2789887","volume":"40","author":"F Shen","year":"2018","unstructured":"Shen, F., Xu, Y., Liu, L., Yang, Y., Huang, Z., Shen, H.T.: Unsupervised deep hashing with similarity-adaptive and discrete optimization. IEEE transactions on pattern analysis and machine intelligence 40(12), 3034\u20133044 (2018)","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"268_CR17","doi-asserted-by":"crossref","unstructured":"Zhang X, Lai H, Feng J (2018) Attention-aware deep adversarial hashing for cross-modal retrieval. In: Proceedings of the European conference on computer vision (ECCV), pp 591\u2013606","DOI":"10.1007\/978-3-030-01267-0_36"},{"issue":"2","key":"268_CR18","first-page":"1365","volume":"35","author":"D Zhang","year":"2021","unstructured":"Zhang D, Wu X-J, Xu T, Yin H (2021) Dah: discrete asymmetric hashing for efficient cross-media retrieval. IEEE Trans Knowl Data Eng 35(2):1365\u20131378","journal-title":"IEEE Trans Knowl Data Eng"},{"issue":"3","key":"268_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3446774","volume":"17","author":"D Zhang","year":"2021","unstructured":"Zhang D, Wu X-J, Yu J (2021) Label consistent flexible matrix factorization hashing for efficient cross-modal retrieval. ACM Transactions on Multimedia Computing, Communications, and Applications TOMM 17(3):1\u201318","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"268_CR20","doi-asserted-by":"crossref","unstructured":"Zhang D, Wu X-J, Yu J (2021) Discrete bidirectional matrix factorization hashing for zero-shot cross-media retrieval. In: Chinese conference on pattern recognition and computer vision (PRCV), pp 524\u2013536. Springer","DOI":"10.1007\/978-3-030-88007-1_43"},{"key":"268_CR21","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"issue":"10s","key":"268_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3505244","volume":"54","author":"S Khan","year":"2021","unstructured":"Khan S, Naseer M, Hayat M, Zamir SW, Khan FS, Shah M (2021) Transformers in vision: a survey. ACM Comput Surv (CSUR) 54(10s):1\u201341","journal-title":"ACM Comput Surv (CSUR)"},{"key":"268_CR23","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"268_CR24","doi-asserted-by":"crossref","unstructured":"Zhang J, Peng Y, Yuan M (2018) Unsupervised generative adversarial cross-modal hashing. In: Proceedings of the AAAI conference on artificial intelligence, p 32","DOI":"10.1609\/aaai.v32i1.11263"},{"key":"268_CR25","doi-asserted-by":"crossref","unstructured":"Yu J, Zhou H, Zhan Y, Tao D (2021) Deep graph-neighbor coherence preserving network for unsupervised cross-modal hashing. In: Proceedings of the AAAI conference on artificial intelligence, vol. 35. pp 4626\u20134634","DOI":"10.1609\/aaai.v35i5.16592"},{"key":"268_CR26","doi-asserted-by":"crossref","unstructured":"Wu G, Lin Z, Han J, Liu L, Ding G, Zhang B, Shen J (2018) Unsupervised deep hashing via binary latent factor models for large-scale cross-modal retrieval. In: IJCAI, pp 1:5","DOI":"10.24963\/ijcai.2018\/396"},{"key":"268_CR27","doi-asserted-by":"crossref","unstructured":"Yang D, Wu D, Zhang W, Zhang H, Li B, Wang W (2020) Deep semantic-alignment hashing for unsupervised cross-modal retrieval. In: Proceedings of the 2020 international conference on multimedia retrieval, pp 44\u201352","DOI":"10.1145\/3372278.3390673"},{"key":"268_CR28","doi-asserted-by":"crossref","unstructured":"Liu S, Qian S, Guan Y, Zhan J, Ying L (2020) Joint-modal distribution-based similarity hashing for large-scale unsupervised deep cross-modal retrieval. In: Proceedings of the 43rd international ACM SIGIR conference on research and development in information retrieval, pp 1379\u20131388","DOI":"10.1145\/3397271.3401086"},{"key":"268_CR29","doi-asserted-by":"publisher","first-page":"466","DOI":"10.1109\/TMM.2021.3053766","volume":"24","author":"P-F Zhang","year":"2021","unstructured":"Zhang P-F, Li Y, Huang Z, Xu X-S (2021) Aggregation-based graph convolutional hashing for unsupervised cross-modal retrieval. IEEE Trans Multimedia 24:466\u2013479","journal-title":"IEEE Trans Multimed"},{"key":"268_CR30","doi-asserted-by":"crossref","unstructured":"Jiang Q-Y, Li W-J (2017) Deep cross-modal hashing. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3232\u20133240","DOI":"10.1109\/CVPR.2017.348"},{"key":"268_CR31","doi-asserted-by":"crossref","unstructured":"Li C, Deng C, Li N, Liu W, Gao X, Tao D (2018) Self-supervised adversarial hashing networks for cross-modal retrieval. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4242\u20134251","DOI":"10.1109\/CVPR.2018.00446"},{"key":"268_CR32","doi-asserted-by":"crossref","unstructured":"Zhang D, Wu X-J (2022) Robust and discrete matrix factorization hashing for cross-modal retrieval. Pattern Recogn 122:108343","DOI":"10.1016\/j.patcog.2021.108343"},{"key":"268_CR33","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3159131","author":"D Zhang","year":"2022","unstructured":"Zhang D, Wu X-J, Xu T, Kittler J (2022) Watch: two-stage discrete cross-media hashing. IEEE Trans Knowl Data Eng. https:\/\/doi.org\/10.1109\/TKDE.2022.3159131","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"268_CR34","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1016\/j.neucom.2020.03.019","volume":"400","author":"X Wang","year":"2020","unstructured":"Wang, X., Zou, X., Bakker, E.M., Wu, S.: Self-constraining and attention-based hashing network for bit-scalable cross-modal retrieval. Neurocomputing 400, 255\u2013271 (2020)","journal-title":"Neurocomputing"},{"issue":"2","key":"268_CR35","doi-asserted-by":"publisher","first-page":"563","DOI":"10.1007\/s11280-020-00859-y","volume":"24","author":"P-F Zhang","year":"2021","unstructured":"Zhang P-F, Luo Y, Huang Z, Xu X-S, Song J (2021) High-order nonlocal hashing for unsupervised cross-modal retrieval. World Wide Web 24(2):563\u2013583","journal-title":"World Wide Web"},{"key":"268_CR36","doi-asserted-by":"crossref","unstructured":"Mikriukov G, Ravanbakhsh M, Demir B (2022) Deep unsupervised contrastive hashing for large-scale cross-modal text-image retrieval in remote sensing. arXiv preprint arXiv:2201.08125","DOI":"10.1109\/ICASSP43922.2022.9746251"},{"key":"268_CR37","doi-asserted-by":"crossref","unstructured":"Cao J, Gan Z, Cheng Y, Yu L, Chen Y-C, Liu J (2020) Behind the scene: Revealing the secrets of pre-trained vision-and-language models. In: European conference on computer vision, pp 565\u2013580. Springer","DOI":"10.1007\/978-3-030-58539-6_34"},{"key":"268_CR38","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763. PMLR"},{"key":"268_CR39","unstructured":"Su W, Zhu X, Cao Y, Li B, Lu L, Wei F, Dai J (2019) Vl-bert: pre-training of generic visual-linguistic representations. arXiv preprint arXiv:1908.08530"},{"key":"268_CR40","doi-asserted-by":"crossref","unstructured":"Chen Y-C, Li L, Yu L, El\u00a0Kholy A, Ahmed F, Gan Z, Cheng Y, Liu J (2020) Uniter: universal image-text representation learning. In: European conference on computer vision, pp 104\u2013120. Springer","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"268_CR41","unstructured":"Lu J, Batra D, Parikh D, Lee S (2019) Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv Neural Inf Process Syst 32"},{"key":"268_CR42","unstructured":"Zeng Z, Mao W (2022) A comprehensive empirical study of vision-language pre-trained model for supervised cross-modal retrieval. arXiv preprint arXiv:2201.02772"},{"key":"268_CR43","doi-asserted-by":"crossref","unstructured":"Zhuo Y, Li Y, Hsiao J, Ho C, Li B (2022) Clip4hashing: unsupervised deep hashing for cross-modal video-text retrieval. In: Proceedings of the 2022 international conference on multimedia retrieval, pp 158\u2013166","DOI":"10.1145\/3512527.3531381"},{"issue":"7","key":"268_CR44","doi-asserted-by":"publisher","first-page":"3210","DOI":"10.1109\/TIP.2018.2814344","volume":"27","author":"J Song","year":"2018","unstructured":"Song, J., Zhang, H., Li, X., Gao, L., Wang, M., Hong, R.: Self-supervised video hashing with hierarchical binary auto-encoder. IEEE Transactions on Image Processing 27(7), 3210\u20133221 (2018)","journal-title":"IEEE Trans Image Process"},{"issue":"7","key":"268_CR45","doi-asserted-by":"publisher","first-page":"5947","DOI":"10.1109\/TCYB.2020.3032017","volume":"52","author":"D Zhang","year":"2020","unstructured":"Zhang D, Wu X-J (2020) Scalable discrete matrix factorization and semantic autoencoder for cross-media retrieval. IEEE Transactions on Cybernetics 52(7):5947\u201360","journal-title":"IEEE Trans Cybernet"},{"key":"268_CR46","doi-asserted-by":"crossref","unstructured":"Liu H, Lin M, Zhang S, Wu Y, Huang F, Ji R (2018) Dense auto-encoder hashing for robust cross-modality retrieval. In: Proceedings of the 26th ACM international conference on multimedia, pp 1589\u20131597","DOI":"10.1145\/3240508.3240684"},{"key":"268_CR47","doi-asserted-by":"crossref","unstructured":"Bai S, Bai X, Tian Q, Latecki LJ (2017) Regularized diffusion process for visual retrieval. In: Proceedings of the AAAI conference on artificial intelligence, p 31","DOI":"10.1609\/aaai.v31i1.11216"},{"issue":"5","key":"268_CR48","doi-asserted-by":"publisher","first-page":"1213","DOI":"10.1109\/TPAMI.2018.2828815","volume":"41","author":"S Bai","year":"2018","unstructured":"Bai, S., Bai, X., Tian, Q., Latecki, L.J.: Regularized diffusion process on bidirectional context for object retrieval. IEEE Transactions on Pattern Analysis and Machine Intelligence 41(5), 1213\u20131226 (2018)","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"268_CR49","doi-asserted-by":"crossref","unstructured":"Rasiwasia N, Costa\u00a0Pereira J, Coviello E, Doyle G, Lanckriet GR, Levy R, Vasconcelos N (2010) A new approach to cross-modal multimedia retrieval. In: Proceedings of the 18th ACM international conference on multimedia, pp 251\u2013260","DOI":"10.1145\/1873951.1873987"},{"key":"268_CR50","doi-asserted-by":"crossref","unstructured":"Huiskes MJ, Lew MS (2008) The mir flickr retrieval evaluation. In: Proceedings of the 1st ACM international conference on multimedia information retrieval, pp 39\u201343","DOI":"10.1145\/1460096.1460104"},{"key":"268_CR51","doi-asserted-by":"crossref","unstructured":"Chua T-S, Tang J, Hong R, Li H, Luo Z, Zheng Y (2009) Nus-wide: a real-world web image database from national university of singapore. In: Proceedings of the ACM international conference on image and video retrieval, pp 1\u20139","DOI":"10.1145\/1646396.1646452"},{"key":"268_CR52","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: common objects in context. In: European conference on computer vision, pp 740\u2013755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"268_CR53","doi-asserted-by":"crossref","unstructured":"Shi Y, Zhao Y, Liu X, Zheng F, Ou W, You X, Peng Q (2022) Deep adaptively-enhanced hashing with discriminative similarity guidance for unsupervised cross-modal retrieval. IEEE Transactions on Circuits and Systems for Video Technology 32(10):7255\u201368","DOI":"10.1109\/TCSVT.2022.3172716"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00268-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-023-00268-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00268-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,6,14]],"date-time":"2023-06-14T11:24:00Z","timestamp":1686741840000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-023-00268-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,22]]},"references-count":53,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["268"],"URL":"https:\/\/doi.org\/10.1007\/s13735-023-00268-7","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,2,22]]},"assertion":[{"value":"20 September 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 October 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 November 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 February 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"2"}}