{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T13:46:14Z","timestamp":1782481574229,"version":"3.54.5"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100016901","name":"Posts and Telecommunications Institute of Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100016901","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computers and Electrical Engineering"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.compeleceng.2026.111336","type":"journal-article","created":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T13:41:46Z","timestamp":1782222106000},"page":"111336","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["Clustered Graph Transformer for fast and efficient image retrieval"],"prefix":"10.1016","volume":"139","author":[{"given":"Van Khanh","family":"Nguyen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Thi Thuy Quynh","family":"Dao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quynh Nguyen","family":"Huu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.compeleceng.2026.111336_b1","doi-asserted-by":"crossref","unstructured":"Liu S, Deng W. Very deep convolutional neural network based image classification using small training sample size. In: 2015 3rd IAPR Asian conference on pattern recognition. ACPR, 2015, p. 730\u20134. http:\/\/dx.doi.org\/10.1109\/ACPR.2015.7486599.","DOI":"10.1109\/ACPR.2015.7486599"},{"key":"10.1016\/j.compeleceng.2026.111336_b2","series-title":"2016 IEEE conference on computer vision and pattern recognition","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.compeleceng.2026.111336_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.compeleceng.2023.108647","article-title":"Shuffled-Xception-DarkNet-53: A content-based image retrieval model based on deep learning algorithm","volume":"107","author":"Pathak","year":"2023","journal-title":"Comput Electr Eng"},{"key":"10.1016\/j.compeleceng.2026.111336_b4","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.compeleceng.2026.111336_b5","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B. Swin Transformer: Hierarchical Vision Transformer using Shifted Windows. 2021, p. 9992\u201310002. http:\/\/dx.doi.org\/10.1109\/ICCV48922.2021.00986.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.compeleceng.2026.111336_b6","doi-asserted-by":"crossref","unstructured":"Song CH, Yoon J, Choi S, Avrithis Y. Boosting vision transformers for image retrieval. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision. 2023, p. 107\u201317.","DOI":"10.1109\/WACV56688.2023.00019"},{"key":"10.1016\/j.compeleceng.2026.111336_b7","series-title":"2025 international conference on multimedia analysis and pattern recognition","first-page":"1","article-title":"A multi-loss hybrid CNN-ViT model for efficient image retrieval","author":"Thi","year":"2025"},{"key":"10.1016\/j.compeleceng.2026.111336_b8","series-title":"2025 IEEE\/CVF winter conference on applications of computer vision","first-page":"7029","article-title":"Uncertainty-guided metric learning without labels","author":"Devalraju","year":"2025"},{"key":"10.1016\/j.compeleceng.2026.111336_b9","doi-asserted-by":"crossref","unstructured":"Shi G, Liang X, Li W, Lin X. Learning Separable Fine-Grained Representation via Dendrogram Construction from Coarse Labels for Fine-grained Visual Recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision. 2025, p. 870\u20139.","DOI":"10.1109\/ICCV51701.2025.00089"},{"key":"10.1016\/j.compeleceng.2026.111336_b10","doi-asserted-by":"crossref","unstructured":"Xiao Z, Suma P, Sachdeva A, Wang H-J, Kordopatis-Zilos G, Tolias G, Ordonez V. LOCORE: Image Re-ranking with Long-Context Sequence Modeling. In: Proceedings of the computer vision and pattern recognition conference. 2025, p. 9580\u201390.","DOI":"10.1109\/CVPR52734.2025.00895"},{"key":"10.1016\/j.compeleceng.2026.111336_b11","doi-asserted-by":"crossref","unstructured":"Liu H, Wang R, Shan S, Chen X. Deep supervised hashing for fast image retrieval. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2016, p. 2064\u201372.","DOI":"10.1109\/CVPR.2016.227"},{"key":"10.1016\/j.compeleceng.2026.111336_b12","doi-asserted-by":"crossref","unstructured":"Dizaji KG, Zheng F, Sadoughi N, Yang Y, Deng C, Huang H. Unsupervised deep generative adversarial hashing network. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2018, p. 3664\u201373.","DOI":"10.1109\/CVPR.2018.00386"},{"key":"10.1016\/j.compeleceng.2026.111336_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.compeleceng.2024.109799","article-title":"A gradual approach to knowledge distillation in deep supervised hashing for large-scale image retrieval","volume":"120","author":"Hussain","year":"2024","journal-title":"Comput Electr Eng"},{"key":"10.1016\/j.compeleceng.2026.111336_b14","series-title":"2025 IEEE\/CVF winter conference on applications of computer vision","first-page":"1537","article-title":"Metric compatible training for online backfilling in large-scale retrieval","author":"Seo","year":"2025"},{"issue":"1","key":"10.1016\/j.compeleceng.2026.111336_b15","doi-asserted-by":"crossref","first-page":"28847","DOI":"10.1038\/s41598-025-14576-x","article-title":"Enhancing image retrieval through optimal barcode representation","volume":"15","author":"Khosrowshahli","year":"2025","journal-title":"Sci Rep"},{"issue":"6","key":"10.1016\/j.compeleceng.2026.111336_b16","first-page":"1","article-title":"Deep hashing with semantic hash centers for image retrieval","volume":"43","author":"Chen","year":"2025","journal-title":"ACM Trans Inf Syst"},{"key":"10.1016\/j.compeleceng.2026.111336_b17","series-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"10.1016\/j.compeleceng.2026.111336_b18","series-title":"Caltech-UCSD birds-200\u20132011","author":"Wah","year":"2025"},{"key":"10.1016\/j.compeleceng.2026.111336_b19","doi-asserted-by":"crossref","unstructured":"Krause J, Stark M, Deng J, Fei-Fei L. 3D Object Representations for Fine-Grained Categorization. In: Proceedings of the IEEE international conference on computer vision (ICCV) workshops. 2013.","DOI":"10.1109\/ICCVW.2013.77"},{"issue":"7","key":"10.1016\/j.compeleceng.2026.111336_b20","doi-asserted-by":"crossref","first-page":"1655","DOI":"10.1109\/TPAMI.2018.2846566","article-title":"Fine-tuning CNN image retrieval with no human annotation","volume":"41","author":"Radenovi\u0107","year":"2018","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10.1016\/j.compeleceng.2026.111336_b21","doi-asserted-by":"crossref","unstructured":"Feng Z, Zhang R, Nie Z. Improving composed image retrieval via contrastive learning with scaling positives and negatives. In: Proceedings of the 32nd ACM international conference on multimedia. 2024, p. 1632\u201341.","DOI":"10.1145\/3664647.3680808"},{"key":"10.1016\/j.compeleceng.2026.111336_b22","doi-asserted-by":"crossref","unstructured":"D\u2019Innocente A, Garg N, Zhang Y, Bazzani L, Donoser M. Localized triplet loss for fine-grained fashion image retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2021, p. 3910\u20135.","DOI":"10.1109\/CVPRW53098.2021.00435"},{"issue":"4","key":"10.1016\/j.compeleceng.2026.111336_b23","doi-asserted-by":"crossref","first-page":"1964","DOI":"10.1109\/TPAMI.2023.3312311","article-title":"Introspective deep metric learning","volume":"46","author":"Wang","year":"2023","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10.1016\/j.compeleceng.2026.111336_b24","doi-asserted-by":"crossref","unstructured":"Jang YK, Cho NI. Self-supervised product quantization for deep unsupervised image retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision. 2021, p. 12085\u201394.","DOI":"10.1109\/ICCV48922.2021.01187"},{"key":"10.1016\/j.compeleceng.2026.111336_b25","series-title":"Self-supervised consistent quantization for fully unsupervised image retrieval","author":"Wu","year":"2022"},{"issue":"3","key":"10.1016\/j.compeleceng.2026.111336_b26","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3617597","article-title":"Composed image retrieval using contrastive learning and task-oriented clip-based features","volume":"20","author":"Baldrati","year":"2023","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"issue":"1","key":"10.1016\/j.compeleceng.2026.111336_b27","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","article-title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","volume":"123","author":"Krishna","year":"2017","journal-title":"Int J Comput Vis"},{"key":"10.1016\/j.compeleceng.2026.111336_b28","doi-asserted-by":"crossref","unstructured":"Johnson J, Krishna R, Stark M, Li L-J, Shamma D, Bernstein M, Fei-Fei L. Image retrieval using scene graphs. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2015, p. 3668\u201378.","DOI":"10.1109\/CVPR.2015.7298990"},{"key":"10.1016\/j.compeleceng.2026.111336_b29","doi-asserted-by":"crossref","unstructured":"Schroeder B, Tripathi S. Structured query-based image retrieval using scene graphs. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops. 2020, p. 178\u20139.","DOI":"10.1109\/CVPRW50498.2020.00097"},{"key":"10.1016\/j.compeleceng.2026.111336_b30","doi-asserted-by":"crossref","unstructured":"Yoon S, Kang WY, Jeon S, Lee S, Han C, Park J, Kim E-S. Image-to-image retrieval by learning similarity between scene graphs. In: Proceedings of the AAAI conference on artificial intelligence. Vol. 35, 2021, p. 10718\u201326.","DOI":"10.1609\/aaai.v35i12.17281"},{"key":"10.1016\/j.compeleceng.2026.111336_b31","doi-asserted-by":"crossref","unstructured":"Jiang B, Zhang Z, Lin D, Tang J, Luo B. Semi-supervised learning with graph learning-convolutional networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2019, p. 11313\u201320.","DOI":"10.1109\/CVPR.2019.01157"},{"key":"10.1016\/j.compeleceng.2026.111336_b32","series-title":"Graph attention networks","author":"Veli\u010dkovi\u0107","year":"2017"},{"key":"10.1016\/j.compeleceng.2026.111336_b33","doi-asserted-by":"crossref","unstructured":"Caron M, Bojanowski P, Joulin A, Douze M. Deep clustering for unsupervised learning of visual features. In: Proceedings of the European conference on computer vision. ECCV, 2018, p. 132\u201349.","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"10.1016\/j.compeleceng.2026.111336_b34","doi-asserted-by":"crossref","unstructured":"Chiang W-L, Liu X, Si S, Li Y, Bengio S, Hsieh C-J. Cluster-gcn: An efficient algorithm for training deep and large graph convolutional networks. In: Proceedings of the 25th ACM SIGKDD international conference on knowledge discovery & data mining. 2019, p. 257\u201366.","DOI":"10.1145\/3292500.3330925"},{"key":"10.1016\/j.compeleceng.2026.111336_b35","series-title":"How attentive are graph attention networks?","author":"Brody","year":"2021"},{"key":"10.1016\/j.compeleceng.2026.111336_b36","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv Neural Inf Process Syst"},{"key":"10.1016\/j.compeleceng.2026.111336_b37","doi-asserted-by":"crossref","unstructured":"Wang X, Han X, Huang W, Dong D, Scott MR. Multi-similarity loss with general pair weighting for deep metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2019, p. 5022\u201330.","DOI":"10.1109\/CVPR.2019.00516"},{"key":"10.1016\/j.compeleceng.2026.111336_b38","doi-asserted-by":"crossref","unstructured":"Zheng W, Wang C, Lu J, Zhou J. Deep compositional metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2021, p. 9320\u20139.","DOI":"10.1109\/CVPR46437.2021.00920"},{"key":"10.1016\/j.compeleceng.2026.111336_b39","doi-asserted-by":"crossref","unstructured":"Sarkar R, Kak A. Dual Pose-invariant Embeddings: Learning Category and Object-specific Discriminative Representations for Recognition and Retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2024, p. 17077\u201385.","DOI":"10.1109\/CVPR52733.2024.01616"},{"key":"10.1016\/j.compeleceng.2026.111336_b40","doi-asserted-by":"crossref","unstructured":"Kim S, Kim D, Cho M, Kwak S. Proxy anchor loss for deep metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2020, p. 3238\u201347.","DOI":"10.1109\/CVPR42600.2020.00330"},{"key":"10.1016\/j.compeleceng.2026.111336_b41","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"10.1016\/j.compeleceng.2026.111336_b42","doi-asserted-by":"crossref","unstructured":"Han D, Kim J, Kim J. Deep pyramidal residual networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2017, p. 5927\u201335.","DOI":"10.1109\/CVPR.2017.668"},{"key":"10.1016\/j.compeleceng.2026.111336_b43","doi-asserted-by":"crossref","unstructured":"Jeong Y, Kim Y, Song HO. End-to-End Efficient Representation Learning via Cascading Combinatorial Optimization. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2019, p. 11379\u201387.","DOI":"10.1109\/CVPR.2019.01164"},{"key":"10.1016\/j.compeleceng.2026.111336_b44","series-title":"Class anchor margin loss for content-based image retrieval","author":"Ghita","year":"2023"},{"key":"10.1016\/j.compeleceng.2026.111336_b45","doi-asserted-by":"crossref","unstructured":"Feng C, Patras I. Maskcon: Masked contrastive learning for coarse-labelled dataset. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2023, p. 19913\u201322.","DOI":"10.1109\/CVPR52729.2023.01907"},{"key":"10.1016\/j.compeleceng.2026.111336_b46","series-title":"European conference on computer vision","first-page":"491","article-title":"Big transfer (bit): General visual representation learning","author":"Kolesnikov","year":"2020"},{"key":"10.1016\/j.compeleceng.2026.111336_b47","series-title":"European conference on computer vision","first-page":"630","article-title":"Identity mappings in deep residual networks","author":"He","year":"2016"},{"key":"10.1016\/j.compeleceng.2026.111336_b48","doi-asserted-by":"crossref","unstructured":"Wang C, Zheng W, Li J, Zhou J, Lu J. Deep factorized metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2023, p. 7672\u201382.","DOI":"10.1109\/CVPR52729.2023.00741"},{"key":"10.1016\/j.compeleceng.2026.111336_b49","doi-asserted-by":"crossref","unstructured":"Li A, Sato I, Ishikawa K, Kawakami R, Yokota R. Informative sample-aware proxy for deep metric learning. In: Proceedings of the 4th ACM international conference on multimedia in Asia. 2022, p. 1\u201311.","DOI":"10.1145\/3551626.3564942"},{"key":"10.1016\/j.compeleceng.2026.111336_b50","doi-asserted-by":"crossref","unstructured":"Shu Y, Van den Hengel A, Liu L. Learning common rationale to improve self-supervised representation for fine-grained visual recognition problems. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2023, p. 11392\u2013401.","DOI":"10.1109\/CVPR52729.2023.01096"},{"key":"10.1016\/j.compeleceng.2026.111336_b51","doi-asserted-by":"crossref","unstructured":"Moskvyak O, Maire F, Dayoub F, Baktashmotlagh M. Keypoint-aligned embeddings for image retrieval and re-identification. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision. 2021, p. 676\u201385.","DOI":"10.1109\/WACV48630.2021.00072"},{"key":"10.1016\/j.compeleceng.2026.111336_b52","doi-asserted-by":"crossref","unstructured":"Yang B, Sun H, Li FW, Chen Z, Cai J, Song C. HSE: Hybrid species embedding for deep metric learning. In: Proceedings of the IEEE\/CVF international conference on computer vision. 2023, p. 11047\u201357.","DOI":"10.1109\/ICCV51070.2023.01014"},{"key":"10.1016\/j.compeleceng.2026.111336_b53","doi-asserted-by":"crossref","unstructured":"Duan J, Lin Y-L, Tran S, Davis LS, Kuo C-CJ. Slade: A self-training framework for distance metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2021, p. 9644\u201353.","DOI":"10.1109\/CVPR46437.2021.00952"},{"key":"10.1016\/j.compeleceng.2026.111336_b54","first-page":"28877","article-title":"Do transformers really perform badly for graph representation?","volume":"34","author":"Ying","year":"2021","journal-title":"Adv Neural Inf Process Syst"}],"container-title":["Computers and Electrical Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0045790626004064?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0045790626004064?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T12:51:15Z","timestamp":1782478275000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0045790626004064"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":54,"alternative-id":["S0045790626004064"],"URL":"https:\/\/doi.org\/10.1016\/j.compeleceng.2026.111336","relation":{},"ISSN":["0045-7906"],"issn-type":[{"value":"0045-7906","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Clustered Graph Transformer for fast and efficient image retrieval","name":"articletitle","label":"Article Title"},{"value":"Computers and Electrical Engineering","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.compeleceng.2026.111336","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"111336"}}