{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T00:41:52Z","timestamp":1768351312700,"version":"3.49.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2023,7,31]],"date-time":"2023-07-31T00:00:00Z","timestamp":1690761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,7,31]],"date-time":"2023-07-31T00:00:00Z","timestamp":1690761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16164-5","type":"journal-article","created":{"date-parts":[[2023,7,31]],"date-time":"2023-07-31T09:01:56Z","timestamp":1690794116000},"page":"20243-20263","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Learning visual similarity for image retrieval with global descriptors and capsule networks"],"prefix":"10.1007","volume":"83","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5356-5943","authenticated-orcid":false,"given":"Duygu","family":"Durmu\u015f","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2462-6959","authenticated-orcid":false,"given":"U\u011fur","family":"G\u00fcd\u00fckbay","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6887-3778","authenticated-orcid":false,"given":"\u00d6zg\u00fcr","family":"Ulusoy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,7,31]]},"reference":[{"key":"16164_CR1","doi-asserted-by":"crossref","unstructured":"Azizpour H, Razavian AS, Sullivan J, Maki A, Carlsson S (2015) From generic to specific deep representations for visual recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. CVPRW \u201915, pp. 36\u201345","DOI":"10.1109\/CVPRW.2015.7301270"},{"key":"16164_CR2","doi-asserted-by":"publisher","first-page":"584","DOI":"10.1007\/978-3-319-10590-1_38","volume-title":"Computer Vision - ECCV 2014","author":"A Babenko","year":"2014","unstructured":"Babenko A, Slesarev A, Chigorin A, Lempitsky V (2014) Neural codes for image retrieval. In: Fleet D, Pajdla T, Schiele B, Tuytelaars T (eds) Computer Vision - ECCV 2014. Springer, Cham, pp 584\u2013599"},{"key":"16164_CR3","unstructured":"Babenko A, Lempitsky V (2015) Aggregating local deep features for image retrieval. In: Proceedings of the IEEE International Conference on Computer Vision. ICCV \u201915, pp. 1269\u20131277"},{"key":"16164_CR4","doi-asserted-by":"crossref","unstructured":"Bell S, Bala K (2015) Learning visual similarity for product design with convolutional neural networks. ACM Trans Graph 34(4):10. Article no. 98,","DOI":"10.1145\/2766959"},{"key":"16164_CR5","unstructured":"Berman M, J\u00e9gou H, Vedaldi A, Kokkinos I, Douze M (2019) Multi- Grain: a unified image embedding for classes and instances. CoRR abs\/1902.05509"},{"key":"16164_CR6","doi-asserted-by":"crossref","unstructured":"Bucher M, Herbin S, Jurie F (2016) Hard negative mining for metric learning based zero-shot classification. CoRR abs\/1608.07441. 1608.07441","DOI":"10.1007\/978-3-319-49409-8_45"},{"key":"16164_CR7","doi-asserted-by":"crossref","unstructured":"Chatfield K, Simonyan K, Vedaldi A, Zisserman A (2014) Return of the devil in the details: Delving deep into convolutional nets. In: Proceedings of the British Machine Vision Conference. BMVC \u201914. BMVA Press, Durham, UK","DOI":"10.5244\/C.28.6"},{"key":"16164_CR8","doi-asserted-by":"crossref","unstructured":"Dai Z, Chen M, Gu X, Zhu S, Tan P (2019) Batch dropblock network for person re-identification and beyond. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV \u201919. Seoul, South Korea, pp. 3690\u20133700","DOI":"10.1109\/ICCV.2019.00379"},{"key":"16164_CR9","doi-asserted-by":"crossref","unstructured":"Galdran A, Dolz J, Chakor H, Lombaert H, Ayed IB (2020) Cost-sensitive regularization for diabetic retinopathy grading from eye fundus images. In: A. L. Martel et al. (ed.) Proceedings of the 23rd International Conference on Medical Image Computing and Computer Assisted Intervention, MICCAI \u201920, Part IV. Lecture Notes in Computer Science, vol. 12265. Springer, Lima, Peru, pp. 665\u2013674","DOI":"10.1007\/978-3-030-59722-1_64"},{"key":"16164_CR10","doi-asserted-by":"crossref","unstructured":"Ge W, Huang W, Dong D, Scott MR (2018) Deep metric learning with hierarchical triplet loss. In: Ferrari V, Hebert M, Sminchisescu C, Weiss Y (eds) Computer Vision \u2013 ECCV 2018. Springer, Cham, pp 272\u2013288","DOI":"10.1007\/978-3-030-01231-1_17"},{"key":"16164_CR11","unstructured":"Gu Y, Li C, Xie J (2018) Attention-aware generalized mean pooling for image retrieval. CoRR abs\/1811.00202"},{"key":"16164_CR12","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2015) Delving deep into rectifiers: Surpassing human-level performance on ImageNet classification. In: Proceedings of the IEEE International Conference on Computer Vision. ICCV \u201915, pp. 1026\u20131034","DOI":"10.1109\/ICCV.2015.123"},{"key":"16164_CR13","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. CVPR \u201916, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"16164_CR14","unstructured":"Hinton GE, Sabour S, Frosst N (2018) Matrix capsules with EM routing. In: Proceedings of the International Conference on Learning Representations. ICLR \u201918"},{"key":"16164_CR15","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR \u201918, pp. 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"key":"16164_CR16","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: Accelerating deep network training by reducing internal covariate shift. In: Bach, F., Blei, D. (eds.) Proceedings of the 32nd International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 37. PMLR, Lille, France, pp. 448\u2013456"},{"key":"16164_CR17","doi-asserted-by":"crossref","unstructured":"J\u00e9gou H, Douze M, Schmid C, P\u00e9rez P (2010) Aggregating local descriptors into a compact image representation. In: Proceedings of the 23rd IEEE Conference on Computer Vision and Pattern Recognition. CVPR \u201910. IEEE Computer Society, San Francisco, USA, pp. 3304\u20133311","DOI":"10.1109\/CVPR.2010.5540039"},{"key":"16164_CR18","unstructured":"Jun H, Ko B, Kim Y, Kim I, Kim J (2020) Combination of multiple global descriptors for image retrieval. arXiv:1903.10663"},{"key":"16164_CR19","doi-asserted-by":"crossref","unstructured":"Kalantidis Y, Mellina C, Osindero S (2016) Cross-dimensional weighting for aggregated deep convolutional features. In: Hua G, J\u00e9gou H (eds) Computer Vision \u2013 ECCV 2016 Workshops. Springer, Cham, pp 685\u2013701","DOI":"10.1007\/978-3-319-46604-0_48"},{"key":"16164_CR20","doi-asserted-by":"crossref","unstructured":"Kapoor R, Sharma D, Gulati T (2021) State of the art content based image retrieval techniques using deep learning: a survey 80(29561\u201329583)","DOI":"10.1007\/s11042-021-11045-1"},{"issue":"21\u201323","key":"16164_CR21","doi-asserted-by":"publisher","first-page":"32763","DOI":"10.1007\/s11042-021-11217-z","volume":"80","author":"N Kayhan","year":"2021","unstructured":"Kayhan N, Fekri-Ershad S (2021) Content based image retrieval based on weighted fusion of texture and color features derived from modified local binary patterns and local neighborhood difference patterns. Multimed Tools Appl 80(21\u201323):32763\u201332790","journal-title":"Multimed Tools Appl"},{"key":"16164_CR22","doi-asserted-by":"crossref","unstructured":"Kim W, Goyal B, Chawla K, Lee J, Kwon K (2018) Attention-based ensemble for deep metric learning. In: Ferrari V, Hebert M, Sminchisescu C, Weiss Y (eds) Computer Vision \u2013 ECCV 2018. Springer, Cham, pp 760\u2013777","DOI":"10.1007\/978-3-030-01246-5_45"},{"key":"16164_CR23","unstructured":"Kingma DP, Ba J (2015) Adam: A method for stochastic optimization. In: Bengio, Y., LeCun, Y. (eds.) Proceedings of the 3rd International Conference on Learning Representations. ICLR \u201915, San Diego, CA, USA"},{"key":"16164_CR24","doi-asserted-by":"crossref","unstructured":"Kinli F, Ozcan B, Kirac F (2019) Fashion image retrieval with capsule networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops. ICCVW \u201919","DOI":"10.1109\/ICCVW.2019.00376"},{"key":"16164_CR25","unstructured":"Kosiorek A, Sabour S, Teh YW, Hinton GE (2019) Stacked capsule autoencoders. In: Wallach H, Larochelle H, Beygelzimer A, d\u2019 Alch\u00e9- Buc F, Fox E, Garnett R (eds.) Advances in Neural Information Processing Systems, vol. 33. Curran Associates, Inc., Red Hook, NY, USA, pp. 15512\u201315522"},{"issue":"6","key":"16164_CR26","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2017) ImageNet classification with deep convolutional neural networks. Commun ACM 60(6):84\u201390","journal-title":"Commun ACM"},{"key":"16164_CR27","unstructured":"LeCun Y, Cortes C, Burges C (2010) MNIST Handwritten Digit Database. ATT Labs [Online] 2 . Available at http:\/\/yann.lecun.com\/exdb\/ mnist"},{"key":"16164_CR28","unstructured":"Li X, Hu X, Yang J (2019) Spatial group-wise enhance: Improving semantic feature learning in convolutional networks. CoRR abs\/1905.09646. arXiv:1905.09646"},{"issue":"2","key":"16164_CR29","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1023\/B:VISI.0000029664.99615.94","volume":"60","author":"DG Lowe","year":"2004","unstructured":"Lowe DG (2004) Distinctive image features from scale-invariant keypoints. Int J Comput Vision 60(2):91\u2013110","journal-title":"Int J Comput Vision"},{"key":"16164_CR30","doi-asserted-by":"crossref","unstructured":"Ma N, Zhang X, Zheng H-T, Sun J (2018) Shufflenet v2: Practical guidelines for efficient cnn architecture design. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 116\u2013131","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"16164_CR31","doi-asserted-by":"crossref","unstructured":"Melekhov I, Kannala J, Rahtu E (2016) Siamese network features for image matching. In: Proceedings of the 23rd International Conference on Pattern Recognition. ICPR \u201916, pp. 378\u2013383","DOI":"10.1109\/ICPR.2016.7899663"},{"key":"16164_CR32","doi-asserted-by":"crossref","unstructured":"Opitz M, Waltner G, Possegger H, Bischof H (2017) BIER - Boosting Independent Embeddings Robustly. In: Proceedings of the IEEE International Conference on Computer Vision. ICCV \u201917, pp. 5199\u20135208","DOI":"10.1109\/ICCV.2017.555"},{"key":"16164_CR33","unstructured":"\u00d6zcan B, Kinli F, Kira\u00e7 F (2020) Quaternion capsule networks. CoRR abs\/2007.04389. 2007.04389"},{"issue":"7","key":"16164_CR34","doi-asserted-by":"publisher","first-page":"1655","DOI":"10.1109\/TPAMI.2018.2846566","volume":"41","author":"F Radenovic","year":"2019","unstructured":"Radenovic F, Tolias G, Chum O (2019) Fine-tuning CNN image retrieval with no human annotation. IEEE Trans Pattern Anal Mach Intell 41(7):1655\u20131668","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"16164_CR35","doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R, Farhadi A (2016) You only look once: Unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. CVPR \u201916, pp. 779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"key":"16164_CR36","doi-asserted-by":"publisher","first-page":"3749","DOI":"10.1609\/aaai.v34i04.5785","volume":"34","author":"F Ribeiro","year":"2020","unstructured":"Ribeiro F, Leontidis G, Kollias S (2020) Capsule routing via variational bayes. Proceedings of the AAAI Conference on Artificial Intelligence 34:3749\u20133756","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"issue":"3","key":"16164_CR37","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein M, Berg AC, Fei-Fei L (2015) ImageNet Large Scale Visual Recognition Challenge. Int J Comput Vision 115(3):211\u2013252","journal-title":"Int J Comput Vision"},{"key":"16164_CR38","unstructured":"Sabour S, Frosst N, Hinton GE (2017) Dynamic routing between capsules. In: Guyon I, Luxburg UV, Bengio S, Wallach H, Fergus R, Vishwanathan S, Garnett R (eds.) Advances in Neural Information Processing Systems. NIPS \u201917, vol. 30. Curran Associates, Inc., Red Hook, NY, USA, pp. 3856\u20133866"},{"key":"16164_CR39","doi-asserted-by":"crossref","unstructured":"Schroff F, Kalenichenko D, Philbin J (2015) FaceNet: a unified embedding for face recognition and clustering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. CVPR \u201915, pp. 815\u2013823","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"16164_CR40","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for largescale image recognition. In: Bengio Y, LeCun Y (eds.) Proceedings of the 3rd International Conference on Learning Representations. ICLR \u201915, San Diego, CA, USA"},{"key":"16164_CR41","doi-asserted-by":"crossref","unstructured":"Sivic J, Zisserman A (2003) Video Google: A text retrieval approach to object matching in videos. In: Proceedings of the Ninth IEEE International Conference on Computer Vision. ICCV \u201903, vol. 2. IEEE Computer Society, Nice, France, p. 1470","DOI":"10.1109\/ICCV.2003.1238663"},{"key":"16164_CR42","series-title":"Curran Associates Inc","first-page":"1857","volume-title":"Advances in Neural Information Processing Systems","author":"K Sohn","year":"2016","unstructured":"Sohn K (2016) Improved deep metric learning with multi-class n-pair loss objective. In: Lee D, Sugiyama M, Luxburg U, Guyon I, Garnett R (eds) Advances in Neural Information Processing Systems, vol 29. Curran Associates Inc. Red Hook, NY, USA, pp 1857\u20131865"},{"key":"16164_CR43","doi-asserted-by":"crossref","unstructured":"Song HO, Xiang Y, Jegelka S, Savarese S (2016) Deep metric learning via lifted structured feature embedding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. CVPR \u201916, pp. 4004\u20134012","DOI":"10.1109\/CVPR.2016.434"},{"key":"16164_CR44","doi-asserted-by":"crossref","unstructured":"Sun Y, Cheng C, Zhang Y, Zhang C, Zheng L, Wang Z, Wei Y (2020) Circle loss: A unified perspective of pair similarity optimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR \u201920, pp. 6397\u20136406","DOI":"10.1109\/CVPR42600.2020.00643"},{"key":"16164_CR45","unstructured":"Sun W, Tagliasacchi A, Deng B, Sabour S, Yazdani S, Hinton GE, Yi KM (2021) Canonical capsules: Unsupervised capsules in canonical pose. In: Advances in Neural Information Processing Systems. NeurIPS \u201921, vol.34"},{"key":"16164_CR46","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. CVPR \u201915, pp. 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"16164_CR47","unstructured":"Tolias G, Sicre R, J\u00e9gou H (2016) Particular object retrieval with integral max-pooling of CNN activations. In: Proceedings of the International Conference on Learning Representations. ICL \u201916, San Juan, Puerto Rico, pp. 1\u201312"},{"key":"16164_CR48","doi-asserted-by":"crossref","unstructured":"Wu C, Manmatha R, Smola AJ, Kr\u00e4henb\u00fchl P (2017) Sampling matters in deep embedding learning. In: Proceedings of the IEEE International Conference on Computer Vision. ICCV \u201917, pp. 2859\u20132867","DOI":"10.1109\/ICCV.2017.309"},{"key":"16164_CR49","doi-asserted-by":"crossref","unstructured":"Yuan Y, Yang K, Zhang C (2017) Hard-aware deeply cascaded embedding. In: Proceedings of the IEEE International Conference on Computer Vision. ICCV \u201917, pp. 814\u2013823","DOI":"10.1109\/ICCV.2017.94"},{"key":"16164_CR50","unstructured":"Zhang H, Goodfellow IJ, Metaxas DN, Odena A (2019) Self-Attention Generative Adversarial Networks. In: Chaudhuri K, Salakhutdinov R (eds.) Proceedings of the 36th International Conference on Machine Learning. ICML \u201919, vol. 97. PMLR, Long Beach, CA, USA, pp. 7354\u20137363"},{"issue":"5","key":"16164_CR51","doi-asserted-by":"publisher","first-page":"1224","DOI":"10.1109\/TPAMI.2017.2709749","volume":"40","author":"L Zheng","year":"2016","unstructured":"Zheng L, Yang Y, Tian Q (2016) SIFT Meets CNN: A decade survey of instance retrieval. IEEE Trans Pattern Anal Mach Intell 40(5):1224\u20131244","journal-title":"IEEE Trans Pattern Anal Mach Intell"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16164-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16164-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16164-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,15]],"date-time":"2024-02-15T10:28:05Z","timestamp":1707992885000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16164-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,31]]},"references-count":51,"journal-issue":{"issue":"7","published-online":{"date-parts":[[2024,2]]}},"alternative-id":["16164"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16164-5","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,7,31]]},"assertion":[{"value":"26 July 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 May 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 July 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 July 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There is no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}