{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T02:05:25Z","timestamp":1780365925193,"version":"3.54.1"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031936906","type":"print"},{"value":"9783031936913","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T00:00:00Z","timestamp":1752969600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T00:00:00Z","timestamp":1752969600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-93691-3_4","type":"book-chapter","created":{"date-parts":[[2025,7,19]],"date-time":"2025-07-19T07:58:39Z","timestamp":1752911919000},"page":"40-54","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["AutoTagGen: A Semantic Approach for\u00a0Image Tagging Utilising Large Language Models and\u00a0Community Verified Integrative Knowledge"],"prefix":"10.1007","author":[{"given":"Swati Sampada","family":"Parida","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Raghav","family":"Khullar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manav","family":"Aggarwal","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gerard","family":"Deepak","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"A.","family":"Santhanavijayan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,20]]},"reference":[{"key":"4_CR1","doi-asserted-by":"publisher","first-page":"2107","DOI":"10.1007\/s11042-013-1491-z","volume":"72","author":"H Bannour","year":"2014","unstructured":"Bannour, H., Hudelot, C.: Building and using fuzzy multimedia ontologies for semantic image annotation. Multimedia Tools Appl. 72, 2107\u20132141 (2014)","journal-title":"Multimedia Tools Appl."},{"key":"4_CR2","doi-asserted-by":"crossref","unstructured":"Chatfield, K., Simonyan, K., Vedaldi, A., Zisserman, A.: Return of the devil in the details: delving deep into convolutional nets. arXiv preprint arXiv:1405.3531 (2014)","DOI":"10.5244\/C.28.6"},{"issue":"4","key":"4_CR3","doi-asserted-by":"publisher","first-page":"897","DOI":"10.1109\/TMM.2019.2937181","volume":"22","author":"C Chaudhary","year":"2019","unstructured":"Chaudhary, C., Goyal, P., Prasad, D.N., Chen, Y.-P.P.: Enhancing the quality of image tagging using a visio-textual knowledge base. IEEE Trans. Multimedia 22(4), 897\u2013911 (2019)","journal-title":"IEEE Trans. Multimedia"},{"key":"4_CR4","doi-asserted-by":"crossref","unstructured":"Durand, T., Mehrasa, N., Mori, G.: Learning a deep convnet for multi-label classification with partial labels (2019)","DOI":"10.1109\/CVPR.2019.00074"},{"key":"4_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"424","DOI":"10.1007\/978-3-319-10584-0_28","volume-title":"Computer Vision \u2013 ECCV 2014","author":"Z Feng","year":"2014","unstructured":"Feng, Z., Feng, S., Jin, R., Jain, A.K.: Image tag completion by noisy matrix recovery. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8695, pp. 424\u2013438. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10584-0_28"},{"key":"4_CR6","doi-asserted-by":"crossref","unstructured":"Gaur, S.: Generation of a short narrative caption for an image using the suggested hashtag. In: 2019 IEEE 35th International Conference on Data Engineering Workshops (ICDEW), pp. 331\u2013337. IEEE (2019)","DOI":"10.1109\/ICDEW.2019.00060"},{"issue":"6","key":"4_CR7","doi-asserted-by":"publisher","first-page":"1452","DOI":"10.1109\/TNNLS.2016.2514360","volume":"28","author":"C Gong","year":"2016","unstructured":"Gong, C., Tao, D., Liu, W., Liu, L., Yang, J.: Label propagation via teaching-to-learn and learning-to-teach. IEEE Trans. Neural Netw. Learn. Syst. 28(6), 1452\u20131465 (2016)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"4_CR8","doi-asserted-by":"publisher","first-page":"174","DOI":"10.1016\/j.neucom.2019.08.089","volume":"370","author":"T Gong","year":"2019","unstructured":"Gong, T., Liu, B., Chu, Q., Nenghai, Yu.: Using multi-label classification to improve object detection. Neurocomputing 370, 174\u2013185 (2019)","journal-title":"Neurocomputing"},{"key":"4_CR9","unstructured":"Gong, Y., Jia, Y., Leung, T., Toshev, A., Ioffe, S.: Deep convolutional ranking for multilabel image annotation. arXiv preprint arXiv:1312.4894 (2013)"},{"key":"4_CR10","doi-asserted-by":"crossref","unstructured":"Guillaumin, M., Mensink, T., Verbeek, J., Schmid, C.: Tagprop: discriminative metric learning in nearest neighbor models for image auto-annotation. In: 2009 IEEE 12th International Conference on Computer Vision, pp. 309\u2013316. IEEE (2009)","DOI":"10.1109\/ICCV.2009.5459266"},{"key":"4_CR11","doi-asserted-by":"crossref","unstructured":"Guillaumin, M., Verbeek, J., Schmid, C.: Multimodal semi-supervised learning for image classification. In: 2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, pp. 902\u2013909. IEEE (2010)","DOI":"10.1109\/CVPR.2010.5540120"},{"key":"4_CR12","doi-asserted-by":"crossref","unstructured":"He, S., et al.: Open-vocabulary multi-label classification via multi-modal knowledge transfer. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 808\u2013816 (2023)","DOI":"10.1609\/aaai.v37i1.25159"},{"key":"4_CR13","doi-asserted-by":"crossref","unstructured":"Hu, H., Zhou, G.-T., Deng, Z., Liao, Z., Mori, G.: Learning structured inference neural networks with label relations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2960\u20132968 (2016)","DOI":"10.1109\/CVPR.2016.323"},{"key":"4_CR14","unstructured":"Huang, X., et al.: Open-set image tagging with multi-grained text supervision (2023)"},{"key":"4_CR15","unstructured":"Huang, X., et al.: Tag2text: guiding vision-language model via image tagging. arXiv preprint arXiv:2303.05657 (2023)"},{"key":"4_CR16","doi-asserted-by":"crossref","unstructured":"Jin, J., Nakayama, H.: Annotation order matters: Recurrent image annotator for arbitrary length image tagging. In: 2016 23rd International Conference on Pattern Recognition (ICPR), pp. 2452\u20132457. IEEE (2016)","DOI":"10.1109\/ICPR.2016.7900004"},{"key":"4_CR17","doi-asserted-by":"crossref","unstructured":"Johnson, J., Ballan, L., Fei-Fei, L.: Love thy neighbors: image annotation by exploiting image metadata. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4624\u20134632 (2015)","DOI":"10.1109\/ICCV.2015.525"},{"key":"4_CR18","doi-asserted-by":"crossref","unstructured":"Juyal, P., Kundaliya, A.: Multilabel image classification using the CNN and DC-CNN model on pascal VOC 2012 dataset. In: 2023 International Conference on Sustainable Computing and Smart Systems (ICSCSS), pp. 452\u2013459. IEEE (2023)","DOI":"10.1109\/ICSCSS57650.2023.10169541"},{"key":"4_CR19","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"issue":"7","key":"4_CR20","doi-asserted-by":"publisher","first-page":"1310","DOI":"10.1109\/TMM.2009.2030598","volume":"11","author":"X Li","year":"2009","unstructured":"Li, X., Snoek, C.G.M., Worring, M.: Learning social tag relevance by neighbor voting. IEEE Trans. Multimedia 11(7), 1310\u20131322 (2009)","journal-title":"IEEE Trans. Multimedia"},{"issue":"1","key":"4_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2906152","volume":"49","author":"X Li","year":"2016","unstructured":"Li, X., Uricchio, T., Ballan, L., Bertini, M., Snoek, C.G.M., Del Bimbo, A.: Socializing the semantic gap: a comparative survey on image tag assignment, refinement, and retrieval. ACM Comput. Surv. (CSUR) 49(1), 1\u201339 (2016)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"4_CR22","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1007\/s10462-016-9502-x","volume":"48","author":"V Maihami","year":"2017","unstructured":"Maihami, V., Yaghmaee, F.: A review on the application of structured sparse representation at image annotation. Artif. Intell. Rev. 48, 331\u2013348 (2017)","journal-title":"Artif. Intell. Rev."},{"issue":"5","key":"4_CR23","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1145\/3191513","volume":"61","author":"T Mitchell","year":"2018","unstructured":"Mitchell, T., et al.: Never-ending learning. Commun. ACM 61(5), 103\u2013115 (2018)","journal-title":"Commun. ACM"},{"key":"4_CR24","doi-asserted-by":"crossref","unstructured":"Murthy, V.N., Maji, S., Manmatha, R.: Automatic image annotation using deep learning representations. In: Proceedings of the 5th ACM on International Conference on Multimedia Retrieval, pp. 603\u2013606 (2015)","DOI":"10.1145\/2671188.2749391"},{"key":"4_CR25","doi-asserted-by":"crossref","unstructured":"Murthy, V.N., Sharma, A., Chari, V., Manmatha, R.: Image annotation using multi-scale hypergraph heat diffusion framework. In: Proceedings of the 2016 ACM on International Conference on Multimedia Retrieval, pp. 299\u2013303 (2016)","DOI":"10.1145\/2911996.2912055"},{"key":"4_CR26","doi-asserted-by":"crossref","unstructured":"Niu, Y., Lu, Z., Wen, J.-R., Xiang, T., Chang, S.-F.: Multi-modal multi-scale deep learning for large-scale image annotation (2018)","DOI":"10.1109\/TIP.2018.2881928"},{"key":"4_CR27","unstructured":"R\u00a0OpenAI. Gpt-4 technical report. arxiv:2303.08774. View in Article 2(5) (2023)"},{"key":"4_CR28","unstructured":"Oquab, M., Bottou, L., Laptev, I., Sivic, J., et al.: Weakly supervised object recognition with convolutional neural networks. In: Proceedings of NIPS, vol. 2014, pp. 1545\u20135963. Citeseer (2014)"},{"key":"4_CR29","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"4_CR30","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"4_CR31","doi-asserted-by":"crossref","unstructured":"Ridnik, T., Sharir, G., Ben-Cohen, A., Ben-Baruch, E., Noy, A.: ML-decoder: scalable and versatile classification head. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 32\u201341 (2023)","DOI":"10.1109\/WACV56688.2023.00012"},{"key":"4_CR32","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"4_CR33","unstructured":"Srivastava, N., Salakhutdinov, R.R.: Multimodal learning with deep Boltzmann machines. In: Advances in Neural Information Processing Systems, vol. 25 (2012)"},{"issue":"3","key":"4_CR34","doi-asserted-by":"publisher","first-page":"1028","DOI":"10.1109\/TIP.2014.2298978","volume":"23","author":"F Sun","year":"2014","unstructured":"Sun, F., Tang, J., Li, H., Qi, G.-J., Huang, T.S.: Multi-label image categorization with sparse factor representation. IEEE Trans. Image Process. 23(3), 1028\u20131037 (2014)","journal-title":"IEEE Trans. Image Process."},{"key":"4_CR35","doi-asserted-by":"crossref","unstructured":"Tang, J., Shu, X., Li, Z., Qi, G.-J., Wang, J.: Generalized deep transfer networks for knowledge propagation in heterogeneous domains (2016)","DOI":"10.1145\/2998574"},{"key":"4_CR36","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1016\/j.imavis.2017.11.002","volume":"69","author":"A Tariq","year":"2018","unstructured":"Tariq, A., Foroosh, H.: Designing a symmetric classifier for image annotation using multi-layer sparse coding. Image Vis. Comput. 69, 33\u201343 (2018)","journal-title":"Image Vis. Comput."},{"key":"4_CR37","unstructured":"Anil, R., et al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"4_CR38","doi-asserted-by":"publisher","first-page":"8337","DOI":"10.1007\/s00500-021-05756-8","volume":"25","author":"S Tiwari","year":"2021","unstructured":"Tiwari, S., Al-Aswadi, F.N., Gaurav, D.: Recent trends in knowledge graphs: theory and practice. Soft Comput. 25, 8337\u20138355 (2021)","journal-title":"Soft Comput."},{"key":"4_CR39","doi-asserted-by":"publisher","first-page":"144","DOI":"10.1016\/j.patcog.2017.05.019","volume":"71","author":"T Uricchio","year":"2017","unstructured":"Uricchio, T., Ballan, L., Seidenari, L., Del Bimbo, A.: Automatic image annotation via label transfer in the semantic space. Pattern Recogn. 71, 144\u2013157 (2017)","journal-title":"Pattern Recogn."},{"key":"4_CR40","doi-asserted-by":"publisher","first-page":"126","DOI":"10.1007\/s11263-016-0927-0","volume":"121","author":"Y Verma","year":"2017","unstructured":"Verma, Y., Jawahar, C.V.: Image annotation by propagating labels from semantic neighbourhoods. Int. J. Comput. Vision 121, 126\u2013148 (2017)","journal-title":"Int. J. Comput. Vision"},{"issue":"4","key":"4_CR41","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/TPAMI.2016.2587640","volume":"39","author":"O Vinyals","year":"2016","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: lessons learned from the 2015 MSCOCO image captioning challenge. IEEE Trans. Pattern Anal. Mach. Intell. 39(4), 652\u2013663 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4_CR42","doi-asserted-by":"crossref","unstructured":"Wang, J., Yang, Y., Mao, J., Huang, Z., Huang, C., Xu, W.: CNN-RNN: a unified framework for multi-label image classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2285\u20132294 (2016)","DOI":"10.1109\/CVPR.2016.251"},{"key":"4_CR43","doi-asserted-by":"publisher","first-page":"616","DOI":"10.1016\/j.procs.2021.02.105","volume":"183","author":"W Wei","year":"2021","unstructured":"Wei, W., et al.: Automatic image annotation based on an improved nearest neighbor technique with tag semantic extension model. Procedia Comput. Sci. 183, 616\u2013623 (2021)","journal-title":"Procedia Comput. Sci."},{"key":"4_CR44","unstructured":"Wei, Y., et al.: CNN: single-label to multi-label. arXiv preprint arXiv:1406.5726 (2014)"},{"key":"4_CR45","doi-asserted-by":"crossref","unstructured":"Wu, B., Chen, W., Sun, P., Liu, W., Ghanem, B., Lyu, S.: Tagging like humans: diverse and distinct image annotation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7967\u20137975 (2018)","DOI":"10.1109\/CVPR.2018.00831"},{"key":"4_CR46","doi-asserted-by":"crossref","unstructured":"Yang, H., Tianyi Zhou, J., Zhang, Y., Gao, B.-B., Wu, J., Cai, J.: Exploit bounding box annotations for multi-label object recognition (2016)","DOI":"10.1109\/CVPR.2016.37"},{"key":"4_CR47","unstructured":"Zhang, Y., et al.: Recognize anything: a strong image tagging model. arXiv preprint arXiv:2306.03514 (2023)"}],"container-title":["Communications in Computer and Information Science","Computer Vision and Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-93691-3_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T01:38:53Z","timestamp":1780364333000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-93691-3_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,20]]},"ISBN":["9783031936906","9783031936913"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-93691-3_4","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,20]]},"assertion":[{"value":"20 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CVIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computer Vision and Image Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chennai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cvip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/cvip2024.iiitdm.ac.in\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}