{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T18:15:24Z","timestamp":1785953724950,"version":"3.56.0"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031590566","type":"print"},{"value":"9783031590573","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-59057-3_20","type":"book-chapter","created":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T22:02:19Z","timestamp":1715119339000},"page":"316-333","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["MDC-Net: Multimodal Detection and Captioning Network for Steel Surface Defects"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6673-4363","authenticated-orcid":false,"given":"Anthony Ashwin Peter","family":"Chazhoor","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8698-7605","authenticated-orcid":false,"given":"Shanfeng","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3377-6895","authenticated-orcid":false,"given":"Bin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0709-8384","authenticated-orcid":false,"given":"Wai Lok","family":"Woo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,5,8]]},"reference":[{"key":"20_CR1","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 779\u2013788 (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"issue":"2","key":"20_CR3","doi-asserted-by":"publisher","first-page":"565","DOI":"10.1007\/s12559-022-10061-z","volume":"15","author":"Y Xie","year":"2023","unstructured":"Xie, Y., Hu, W., Xie, S., He, L.: Surface defect detection algorithm based on feature-enhanced yolo. Cogn. Comput. 15(2), 565\u2013579 (2023)","journal-title":"Cogn. Comput."},{"key":"20_CR4","unstructured":"Chen, T., Saxena, S., Li, L., Fleet, D.J., Hinton, G.: Pix2seq: a language modeling framework for object detection. arXiv preprint arXiv:2109.10852 (2021)"},{"key":"20_CR5","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1016\/j.rcim.2015.09.008","volume":"38","author":"Q Luo","year":"2016","unstructured":"Luo, Q., He, Y.: A cost-effective and automatic surface defect inspection system for hot-rolled flat steel. Robot. Comput.-Integr. Manuf. 38, 16\u201330 (2016)","journal-title":"Robot. Comput.-Integr. Manuf."},{"issue":"5","key":"20_CR6","doi-asserted-by":"publisher","first-page":"384","DOI":"10.1016\/j.ndteint.2005.11.004","volume":"39","author":"B Helifa","year":"2006","unstructured":"Helifa, B., Oulhadj, A., Benbelghit, A., Lefkaier, I., Boubenider, F., Boutassouna, D.: Detection and measurement of surface cracks in ferromagnetic materials using eddy current testing. Ndt & E Int. 39(5), 384\u2013390 (2006)","journal-title":"Ndt & E Int."},{"issue":"2","key":"20_CR7","doi-asserted-by":"publisher","first-page":"412","DOI":"10.1109\/JSEN.2016.2625815","volume":"17","author":"X Li","year":"2016","unstructured":"Li, X., Gao, B., Woo, W.L., Tian, G.Y., Qiu, X., Gu, L.: Quantitative surface crack evaluation based on eddy current pulsed thermography. IEEE Sens. J. 17(2), 412\u2013421 (2016)","journal-title":"IEEE Sens. J."},{"key":"20_CR8","doi-asserted-by":"publisher","first-page":"676","DOI":"10.1016\/j.infrared.2016.04.033","volume":"76","author":"R Shrestha","year":"2016","unstructured":"Shrestha, R., Park, J., Kim, W.: Application of thermal wave imaging and phase shifting method for defect detection in stainless steel. Infrared Phys. Technol. 76, 676\u2013683 (2016)","journal-title":"Infrared Phys. Technol."},{"key":"20_CR9","doi-asserted-by":"crossref","unstructured":"Gao, B., Li, X., Woo, W.L., Yun Tian, G.: Physics-based image segmentation using first order statistical properties and genetic algorithm for inductive thermography imaging. IEEE Trans. Image Process. 27(5) (2017) 2160\u20132175","DOI":"10.1109\/TIP.2017.2783627"},{"key":"20_CR10","doi-asserted-by":"crossref","unstructured":"Li, X.G., Miao, C.Y., Wang, J., Zhang, Y.: Automatic defect detection method for the steel cord conveyor belt based on its x-ray images. In: 2011 International Conference on Control, Automation and Systems Engineering (CASE), pp. 1\u20134. IEEE (2011)","DOI":"10.1109\/ICCASE.2011.5997624"},{"key":"20_CR11","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105628","volume":"117","author":"Y Zhang","year":"2023","unstructured":"Zhang, Y., et al.: Development of a cross-scale weighted feature fusion network for hot-rolled steel surface defect detection. Eng. Appl. Artif. Intell. 117, 105628 (2023)","journal-title":"Eng. Appl. Artif. Intell."},{"issue":"11","key":"20_CR12","doi-asserted-by":"publisher","first-page":"8389","DOI":"10.1007\/s00521-022-08112-5","volume":"35","author":"K Demir","year":"2023","unstructured":"Demir, K., Ay, M., Cavas, M., Demir, F.: Automated steel surface defect detection and classification using a new deep learning-based approach. Neural Comput. Appl. 35(11), 8389\u20138406 (2023)","journal-title":"Neural Comput. Appl."},{"key":"20_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119388","volume":"215","author":"L Yang","year":"2023","unstructured":"Yang, L., Xu, S., Fan, J., Li, E., Liu, Y.: A pixel-level deep segmentation network for automatic defect detection. Expert Syst. Appl. 215, 119388 (2023)","journal-title":"Expert Syst. Appl."},{"key":"20_CR14","doi-asserted-by":"crossref","unstructured":"Ji, A., Thee, Q.Y., Woo, W.L., Wong, E.: Experimental investigations of a convolutional neural network model for detecting railway track anomalies. In: IECON 2023-49th Annual Conference of the IEEE Industrial Electronics Society, pp. 1\u20137. IEEE (2023)","DOI":"10.1109\/IECON51785.2023.10312404"},{"key":"20_CR15","doi-asserted-by":"crossref","unstructured":"Chazhoor, A.A.P., Zhu, M., Ho, E.S.L., Gao, B., Woo, W.L.: Classification of different types of plastics using deep transfer learning. In: ROBOVIS, SciTePress, Science and Technology Publications, pp. 190\u2013195 (2021)","DOI":"10.5220\/0010716500003061"},{"key":"20_CR16","first-page":"1","volume":"2","author":"AAP Chazhoor","year":"2022","unstructured":"Chazhoor, A.A.P., Ho, E.S., Gao, B., Woo, W.L.: Deep transfer learning benchmark for plastic waste classification. Intell. Robot. 2, 1\u201319 (2022)","journal-title":"Intell. Robot."},{"issue":"4","key":"20_CR17","doi-asserted-by":"publisher","first-page":"1493","DOI":"10.1109\/TIM.2019.2915404","volume":"69","author":"Y He","year":"2019","unstructured":"He, Y., Song, K., Meng, Q., Yan, Y.: An end-to-end steel surface defect detection approach via fusing multiple hierarchical features. IEEE Trans. Instrum. Meas. 69(4), 1493\u20131504 (2019)","journal-title":"IEEE Trans. Instrum. Meas."},{"issue":"1","key":"20_CR18","doi-asserted-by":"publisher","first-page":"114","DOI":"10.1007\/s42979-023-02436-2","volume":"5","author":"AAP Chazhoor","year":"2023","unstructured":"Chazhoor, A.A.P., Ho, E.S., Gao, B., Woo, W.L.: A review and benchmark on state-of-the-art steel defects detection. SN Comput. Sci. 5(1), 114 (2023)","journal-title":"SN Comput. Sci."},{"key":"20_CR19","first-page":"1","volume":"2021","author":"W Zhao","year":"2021","unstructured":"Zhao, W., Chen, F., Huang, H., Li, D., Cheng, W.: A new steel defect detection algorithm based on deep learning. Comput. Intell. Neurosci. 2021, 1\u201313 (2021)","journal-title":"Comput. Intell. Neurosci."},{"issue":"1","key":"20_CR20","doi-asserted-by":"publisher","first-page":"4","DOI":"10.3390\/info15010004","volume":"15","author":"S Mirzaei","year":"2023","unstructured":"Mirzaei, S., Mao, H., Al-Nima, R.R.O., Woo, W.L.: Explainable AI evaluation: a top-down approach for selecting optimal explanations for black box models. Information 15(1), 4 (2023)","journal-title":"Information"},{"key":"20_CR21","doi-asserted-by":"crossref","unstructured":"Wang, J., Madhyastha, P., Specia, L.: Object counts! bringing explicit detections back into image captioning. arXiv preprint arXiv:1805.00314 (2018)","DOI":"10.18653\/v1\/N18-1198"},{"issue":"11","key":"20_CR22","doi-asserted-by":"publisher","first-page":"1387","DOI":"10.1111\/mice.12793","volume":"37","author":"PJ Chun","year":"2022","unstructured":"Chun, P.J., Yamane, T., Maemura, Y.: A deep learning-based image captioning method to automatically generate comprehensive explanations of bridge damage. Comput.-Aided Civ. Infrastruct. Eng. 37(11), 1387\u20131401 (2022)","journal-title":"Comput.-Aided Civ. Infrastruct. Eng."},{"issue":"4","key":"20_CR23","doi-asserted-by":"publisher","first-page":"1270","DOI":"10.3390\/s21041270","volume":"21","author":"K Iwamura","year":"2021","unstructured":"Iwamura, K., Louhi Kasahara, J.Y., Moro, A., Yamashita, A., Asama, H.: Image captioning using motion-CNN with object detection. Sensors 21(4), 1270 (2021)","journal-title":"Sensors"},{"issue":"8","key":"20_CR24","doi-asserted-by":"publisher","first-page":"2075","DOI":"10.1049\/ipr2.12470","volume":"16","author":"X Shao","year":"2022","unstructured":"Shao, X., Xiang, Z., Li, Y., Zhang, M.: Variational joint self-attention for image captioning. IET Image Process. 16(8), 2075\u20132086 (2022)","journal-title":"IET Image Process."},{"issue":"17","key":"20_CR25","doi-asserted-by":"publisher","first-page":"6419","DOI":"10.3390\/s22176419","volume":"22","author":"D Wei","year":"2022","unstructured":"Wei, D., Wei, X., Jia, L.: Automatic defect description of railway track line image based on dense captioning. Sensors 22(17), 6419 (2022)","journal-title":"Sensors"},{"key":"20_CR26","doi-asserted-by":"crossref","unstructured":"Yong, C., Yingchi, M., Yi, W., Ping, P., Longbao, W.: Keywords-based dam defect image caption generation. In: 2021 IEEE Seventh International Conference on Big Data Computing Service and Applications (BigDataService), pp. 214\u2013221. IEEE (2021)","DOI":"10.1109\/BigDataService52369.2021.00034"},{"key":"20_CR27","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"20_CR28","doi-asserted-by":"crossref","unstructured":"Tsaniya, H., Fatichah, C., Suciati, N.: Transformer approaches in image captioning: a literature review. In: 2022 14th International Conference on Information Technology and Electrical Engineering (ICITEE), pp. 1\u20136. IEEE (2022)","DOI":"10.1109\/ICITEE56407.2022.9954086"},{"key":"20_CR29","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xu, J., Sun, Y.: End-to-end transformer based model for image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 2585\u20132594 (2022)","DOI":"10.1609\/aaai.v36i3.20160"},{"key":"20_CR30","doi-asserted-by":"crossref","unstructured":"Dittakan, K., Prompitak, K., Thungklang, P., Wongwattanakit, C.: Image caption generation using transformer learning methods: a case study on instagram image. Multimed. Tools Appl. 1\u201321 (2023)","DOI":"10.1007\/s11042-023-17275-9"},{"key":"20_CR31","doi-asserted-by":"publisher","first-page":"858","DOI":"10.1016\/j.apsusc.2013.09.002","volume":"285","author":"K Song","year":"2013","unstructured":"Song, K., Yan, Y.: A noise robust method based on completed local binary patterns for hot-rolled steel strip surface defects. Appl. Surf. Sci. 285, 858\u2013864 (2013)","journal-title":"Appl. Surf. Sci."},{"key":"20_CR32","doi-asserted-by":"publisher","unstructured":"Touvron, H., Cord, M., J\u00e9gou, H.: DeiT III: revenge of the ViT. In: Avidan, S., Brostow, G., Cisse, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022. ECCV 2022. LNCS, vol. 13684, pp. 516\u2013533. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20053-3_30","DOI":"10.1007\/978-3-031-20053-3_30"},{"key":"20_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, Z.: Improved adam optimizer for deep neural networks. In: IEEE\/ACM 26th International Symposium on Quality of Service (IWQoS). IEEE 2018, pp. 1\u20132 (2018)","DOI":"10.1109\/IWQoS.2018.8624183"},{"key":"20_CR34","doi-asserted-by":"crossref","unstructured":"Yao, Z., Gholami, A., Shen, S., Mustafa, M., Keutzer, K., Mahoney, M.: Adahessian: an adaptive second order optimizer for machine learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 10665\u201310673 (2021)","DOI":"10.1609\/aaai.v35i12.17275"},{"key":"20_CR35","unstructured":"Zhang, Z., Sabuncu, M.: Generalized cross entropy loss for training deep neural networks with noisy labels. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"key":"20_CR36","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"}],"container-title":["Communications in Computer and Information Science","Robotics, Computer Vision and Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-59057-3_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T22:06:16Z","timestamp":1715119576000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-59057-3_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031590566","9783031590573"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-59057-3_20","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"8 May 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors declare that there are no competing interests to disclose that are relevant to the content of this article. No funding has been allocated for this project.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ROBOVIS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Robotics, Computer Vision and Intelligent Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Rome","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 February 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 February 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"robovis2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/robovis.scitevents.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}