{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T17:17:56Z","timestamp":1770830276181,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,5,19]],"date-time":"2023-05-19T00:00:00Z","timestamp":1684454400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100020593","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["202165007"],"award-info":[{"award-number":["202165007"]}],"id":[{"id":"10.13039\/100020593","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,5,19]]},"DOI":"10.1145\/3604078.3604096","type":"proceedings-article","created":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T16:46:07Z","timestamp":1698338767000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["UATR-MSG-Transformer: A Deep Learning Network for Underwater Acoustic Target Recognition Based on Spectrogram Feature Fusion and Transformer with Messenger Tokens"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1548-1623","authenticated-orcid":false,"given":"Hao","family":"Zhou","sequence":"first","affiliation":[{"name":"Department of Information Science and Engineering, Ocean University of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2698-2348","authenticated-orcid":false,"given":"Xuening","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Information Science and Engineering, Ocean University of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7746-8061","authenticated-orcid":false,"given":"Peishun","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Information Science and Engineering, Ocean University of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2909-7320","authenticated-orcid":false,"given":"Liang","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Information Science and Engineering, Ocean University of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4273-8623","authenticated-orcid":false,"given":"Ruichun","family":"Tang","sequence":"additional","affiliation":[{"name":"Department of Information Science and Engineering, Ocean University of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,26]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"An overview of text-independent speaker recognition: From features to supervectors[J]. Speech communication","author":"Kinnunen T","year":"2010","unstructured":"Kinnunen T , Li H. An overview of text-independent speaker recognition: From features to supervectors[J]. Speech communication , 2010 , 52(1): 12-40. Kinnunen T, Li H. An overview of text-independent speaker recognition: From features to supervectors[J]. Speech communication, 2010, 52(1): 12-40."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2018.2885636"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.115270"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2021.107989"},{"key":"e_1_3_2_1_5_1","volume-title":"Underwater acoustic target recognition with resnet18 on shipsear dataset[C]\/\/2021 IEEE 4th International Conference on Electronics Technology (ICET)","author":"Hong F","year":"2021","unstructured":"Hong F , Liu C , Guo L , Underwater acoustic target recognition with resnet18 on shipsear dataset[C]\/\/2021 IEEE 4th International Conference on Electronics Technology (ICET) . IEEE , 2021 : 1240-1244. Hong F, Liu C, Guo L, Underwater acoustic target recognition with resnet18 on shipsear dataset[C]\/\/2021 IEEE 4th International Conference on Electronics Technology (ICET). IEEE, 2021: 1240-1244."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2016.06.008"},{"key":"e_1_3_2_1_7_1","volume-title":"Multi-scale context aggregation by dilated convolutions[J]. arXiv preprint arXiv:1511.07122","author":"Yu F","year":"2015","unstructured":"Yu F , Koltun V. Multi-scale context aggregation by dilated convolutions[J]. arXiv preprint arXiv:1511.07122 , 2015 . Yu F, Koltun V. Multi-scale context aggregation by dilated convolutions[J]. arXiv preprint arXiv:1511.07122, 2015."},{"key":"e_1_3_2_1_8_1","first-page":"1451","article-title":"Understanding convolution for semantic segmentation[C]\/\/2018 IEEE winter conference on applications of computer vision (WACV)","volume":"2018","author":"Wang P","unstructured":"Wang P , Chen P , Yuan Y , Understanding convolution for semantic segmentation[C]\/\/2018 IEEE winter conference on applications of computer vision (WACV) . Ieee , 2018 : 1451 - 1460 . Wang P, Chen P, Yuan Y, Understanding convolution for semantic segmentation[C]\/\/2018 IEEE winter conference on applications of computer vision (WACV). Ieee, 2018: 1451-1460.","journal-title":"Ieee"},{"key":"e_1_3_2_1_9_1","volume-title":"Glass J. Ast: Audio spectrogram transformer[J]. arXiv preprint arXiv:2104.01778","author":"Gong Y","year":"2021","unstructured":"Gong Y , Chung Y A , Glass J. Ast: Audio spectrogram transformer[J]. arXiv preprint arXiv:2104.01778 , 2021 . Gong Y, Chung Y A, Glass J. Ast: Audio spectrogram transformer[J]. arXiv preprint arXiv:2104.01778, 2021."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Yuan L Chen Y Wang T Tokens-to-token vit: Training vision transformers from scratch on imagenet[C]\/\/Proceedings of the IEEE\/CVF international conference on computer vision. 2021: 558-567.  Yuan L Chen Y Wang T Tokens-to-token vit: Training vision transformers from scratch on imagenet[C]\/\/Proceedings of the IEEE\/CVF international conference on computer vision. 2021: 558-567.","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2020.3020896"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Yin X Sun X Liu P Underwater acoustic target classification based on LOFAR spectrum and convolutional neural network[C]\/\/Proceedings of the 2nd International Conference on Artificial Intelligence and Advanced Manufacture. 2020: 59-63.  Yin X Sun X Liu P Underwater acoustic target classification based on LOFAR spectrum and convolutional neural network[C]\/\/Proceedings of the 2nd International Conference on Artificial Intelligence and Advanced Manufacture. 2020: 59-63.","DOI":"10.1145\/3421766.3421890"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.3390\/fi13100265"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.3390\/s19071733"},{"key":"e_1_3_2_1_15_1","volume-title":"Deep convolution stack for waveform in underwater acoustic target recognition[J]. Scientific reports","author":"Tian S","year":"2021","unstructured":"Tian S , Chen D , Wang H , Deep convolution stack for waveform in underwater acoustic target recognition[J]. Scientific reports , 2021 , 11(1): 9614. Tian S, Chen D, Wang H, Deep convolution stack for waveform in underwater acoustic target recognition[J]. Scientific reports, 2021, 11(1): 9614."},{"key":"e_1_3_2_1_16_1","first-page":"91","volume":"202","author":"Xu C","unstructured":"Xu C , Li Y , Zhang M , Underwater Acoustic Target Recognition Based on Feature Fusion and Self-attention Mechanism[J] . Mobile Communications , 202 ,46(06): 91 - 98 . Xu C, Li Y, Zhang M, Underwater Acoustic Target Recognition Based on Feature Fusion and Self-attention Mechanism[J]. Mobile Communications,202,46(06): 91-98.","journal-title":"Mobile Communications"},{"key":"e_1_3_2_1_17_1","volume-title":"Feature Fusion Methods Based on Channel Domain Attention Mechanism[J]. Journal of northeast normal university (natural science edition)","author":"Luo D","year":"2021","unstructured":"Luo D , Fang J , Liu Y. Feature Fusion Methods Based on Channel Domain Attention Mechanism[J]. Journal of northeast normal university (natural science edition) , 2021 does (03): 44-48. Luo D, Fang J, Liu Y. Feature Fusion Methods Based on Channel Domain Attention Mechanism[J]. Journal of northeast normal university (natural science edition), 2021 does (03): 44-48."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Hu J Shen L Sun G. Squeeze-and-excitation networks[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2018: 7132-7141.  Hu J Shen L Sun G. Squeeze-and-excitation networks[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2018: 7132-7141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Dai Y Gieseke F Oehmcke S Attentional feature fusion[C]\/\/Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. 2021: 3560-3569.  Dai Y Gieseke F Oehmcke S Attentional feature fusion[C]\/\/Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. 2021: 3560-3569.","DOI":"10.1109\/WACV48630.2021.00360"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Wang Q Wu B Zhu P ECA-Net: Efficient channel attention for deep convolutional neural networks[C]\/\/Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2020: 11534-11542.  Wang Q Wu B Zhu P ECA-Net: Efficient channel attention for deep convolutional neural networks[C]\/\/Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2020: 11534-11542.","DOI":"10.1109\/CVPR42600.2020.01155"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18178\/joig.10.1.10-16"},{"key":"e_1_3_2_1_22_1","volume-title":"Attention is all you need[J]. Advances in neural information processing systems","author":"Vaswani A","year":"2017","unstructured":"Vaswani A , Shazeer N , Parmar N , Attention is all you need[J]. Advances in neural information processing systems , 2017 , 30. Vaswani A, Shazeer N, Parmar N, Attention is all you need[J]. Advances in neural information processing systems, 2017, 30."},{"key":"e_1_3_2_1_23_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale[J]. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy A","year":"2020","unstructured":"Dosovitskiy A , Beyer L , Kolesnikov A , An image is worth 16x16 words: Transformers for image recognition at scale[J]. arXiv preprint arXiv:2010.11929 , 2020 . Dosovitskiy A, Beyer L, Kolesnikov A, An image is worth 16x16 words: Transformers for image recognition at scale[J]. arXiv preprint arXiv:2010.11929, 2020."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Liu Z Lin Y Cao Y Swin transformer: Hierarchical vision transformer using shifted windows[C]\/\/Proceedings of the IEEE\/CVF international conference on computer vision. 2021: 10012-10022.  Liu Z Lin Y Cao Y Swin transformer: Hierarchical vision transformer using shifted windows[C]\/\/Proceedings of the IEEE\/CVF international conference on computer vision. 2021: 10012-10022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_25_1","volume-title":"Wang X","author":"Fang J","year":"2022","unstructured":"Fang J , Xie L , Wang X , Msg-transformer : Exchanging local spatial information by manipulating messenger tokens[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition . 2022 : 12063-12072. Fang J, Xie L, Wang X, Msg-transformer: Exchanging local spatial information by manipulating messenger tokens[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022: 12063-12072."},{"key":"e_1_3_2_1_26_1","volume-title":"Specaugment: A simple data augmentation method for automatic speech recognition[J]. arXiv preprint arXiv:1904.08779","author":"Park D S","year":"2019","unstructured":"Park D S , Chan W , Zhang Y , Specaugment: A simple data augmentation method for automatic speech recognition[J]. arXiv preprint arXiv:1904.08779 , 2019 . Park D S, Chan W, Zhang Y, Specaugment: A simple data augmentation method for automatic speech recognition[J]. arXiv preprint arXiv:1904.08779, 2019."},{"key":"e_1_3_2_1_27_1","volume-title":"PMLR","author":"Tan M","unstructured":"Tan M , Le Q. Efficientnet : Rethinking model scaling for convolutional neural networks[C]\/\/International conference on machine learning . PMLR , 2019: 6105-6114. Tan M, Le Q. Efficientnet: Rethinking model scaling for convolutional neural networks[C]\/\/International conference on machine learning. PMLR, 2019: 6105-6114."},{"key":"e_1_3_2_1_28_1","volume-title":"Efficient computation of depthwise separable convolution in MoblieNet deep neural network models[C]\/\/2021 IEEE International Conference on Consumer Electronics-Taiwan (ICCE-TW)","author":"Hsiao S F","year":"2021","unstructured":"Hsiao S F , Tsai B C . Efficient computation of depthwise separable convolution in MoblieNet deep neural network models[C]\/\/2021 IEEE International Conference on Consumer Electronics-Taiwan (ICCE-TW) . IEEE , 2021 : 1-2. Hsiao S F, Tsai B C. Efficient computation of depthwise separable convolution in MoblieNet deep neural network models[C]\/\/2021 IEEE International Conference on Consumer Electronics-Taiwan (ICCE-TW). IEEE, 2021: 1-2."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"He K Zhang X Ren S Deep residual learning for image recognition[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2016: 770-778.  He K Zhang X Ren S Deep residual learning for image recognition[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2016: 770-778.","DOI":"10.1109\/CVPR.2016.90"}],"event":{"name":"ICDIP 2023: The 15th International Conference on Digital Image Processing","location":"Nanjing China","acronym":"ICDIP 2023"},"container-title":["Proceedings of the 15th International Conference on Digital Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3604078.3604096","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3604078.3604096","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:16Z","timestamp":1750178176000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3604078.3604096"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,19]]},"references-count":29,"alternative-id":["10.1145\/3604078.3604096","10.1145\/3604078"],"URL":"https:\/\/doi.org\/10.1145\/3604078.3604096","relation":{},"subject":[],"published":{"date-parts":[[2023,5,19]]},"assertion":[{"value":"2023-10-26","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}