{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,17]],"date-time":"2026-05-17T06:05:30Z","timestamp":1778997930777,"version":"3.51.4"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2024,2,23]],"date-time":"2024-02-23T00:00:00Z","timestamp":1708646400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,23]],"date-time":"2024-02-23T00:00:00Z","timestamp":1708646400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"The National Key Technologies Research and Development Program of China","award":["2021YFC2801001"],"award-info":[{"award-number":["2021YFC2801001"]}]},{"name":"The National Social Science Foundation of China","award":["20&ZD130"],"award-info":[{"award-number":["20&ZD130"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s11227-024-05932-1","type":"journal-article","created":{"date-parts":[[2024,2,23]],"date-time":"2024-02-23T18:01:57Z","timestamp":1708711317000},"page":"12863-12890","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["MEDMCN: a novel multi-modal EfficientDet with multi-scale CapsNet for object detection"],"prefix":"10.1007","volume":"80","author":[{"given":"Xingye","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengyu","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongdai","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,23]]},"reference":[{"issue":"7553","key":"5932_CR1","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521(7553):436\u2013444","journal-title":"Nature"},{"key":"5932_CR2","doi-asserted-by":"publisher","first-page":"1197","DOI":"10.1007\/s00371-013-0886-1","volume":"30","author":"R Jafri","year":"2014","unstructured":"Jafri R, Ali SA, Arabnia HR, Fatima S (2014) Computer vision-based object recognition for the visually impaired in an indoors environment: a survey. Vis Comput 30:1197\u20131222","journal-title":"Vis Comput"},{"key":"5932_CR3","doi-asserted-by":"publisher","first-page":"111126","DOI":"10.1016\/j.knosys.2023.111126","volume":"283","author":"X Li","year":"2024","unstructured":"Li X, Liu J, Xie Y, Gong P, Zhang X, He H (2024) Magdra: a multi-modal attention graph network with dynamic routing-by-agreement for multi-label emotion recognition. Knowl-Based Syst 283:111126","journal-title":"Knowl-Based Syst"},{"key":"5932_CR4","doi-asserted-by":"crossref","unstructured":"Wang H, Liu J, Duan M, Gong P, Wu Z, Wang J, Han B (2023) Cross-modal knowledge guided model for abstractive summarization. Complex Intell Syst. pp 1\u201318","DOI":"10.1007\/s40747-023-01170-9"},{"key":"5932_CR5","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Goyal P, Girshick R, He K, Doll\u00e1r P (2017) Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 2980\u20132988","DOI":"10.1109\/ICCV.2017.324"},{"key":"5932_CR6","doi-asserted-by":"crossref","unstructured":"Tan M, Pang R, Le QV (2020) Efficientdet: scalable and efficient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10781\u201310790","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"5932_CR7","doi-asserted-by":"crossref","unstructured":"Law H, Deng J (2018) Cornernet: detecting objects as paired keypoints. In: Proceedings of the European Conference on Computer Vision (ECCV). pp. 734\u2013750","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"5932_CR8","doi-asserted-by":"crossref","unstructured":"Gong P, Liu J, Xie Y, Liu M, Zhang X (2023) Enhancing context representations with part-of-speech information and neighboring signals for question classification. Complex Intell Syst. pp 1\u201319","DOI":"10.1007\/s40747-023-01067-7"},{"issue":"1","key":"5932_CR9","doi-asserted-by":"publisher","first-page":"101","DOI":"10.3390\/app10010101","volume":"10","author":"Y Yang","year":"2019","unstructured":"Yang Y, Xu C, Dong F, Wang X (2019) A new multi-scale convolutional model based on multiple attention for image classification. Appl Sci 10(1):101","journal-title":"Appl Sci"},{"key":"5932_CR10","doi-asserted-by":"crossref","unstructured":"Liu J, Yang Y, Lv S, Wang J, Chen H (2019) Attention-based BiGRU-CNN for Chinese question classification. J Ambient Intell Hum Comput. pp 1\u201312","DOI":"10.1007\/s12652-019-01344-9"},{"issue":"9","key":"5932_CR11","doi-asserted-by":"publisher","first-page":"1939","DOI":"10.3390\/app9091939","volume":"9","author":"Y Yang","year":"2019","unstructured":"Yang Y, Wang X, Zhao Q, Sui T (2019) Two-level attentions and grouping attention convolutional network for fine-grained image classification. Appl Sci 9(9):1939","journal-title":"Appl Sci"},{"key":"5932_CR12","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1016\/j.neucom.2020.04.056","volume":"403","author":"J Liu","year":"2020","unstructured":"Liu J, Yang Y, He H (2020) Multi-level semantic representation enhancement network for relationship extraction. Neurocomputing 403:282\u2013293","journal-title":"Neurocomputing"},{"key":"5932_CR13","doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R, Farhadi A (2016) You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"key":"5932_CR14","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D, Szegedy C, Reed S, Fu C-Y, Berg AC (2016) Ssd: single shot multibox detector. In: European Conference on Computer Vision. pp. 21\u201337. Springer","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"5932_CR15","doi-asserted-by":"crossref","unstructured":"Girshick R, Donahue J, Darrell T, Malik J (2014) Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 580\u2013587","DOI":"10.1109\/CVPR.2014.81"},{"key":"5932_CR16","doi-asserted-by":"crossref","unstructured":"Girshick R (2015) Fast r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 1440\u20131448","DOI":"10.1109\/ICCV.2015.169"},{"key":"5932_CR17","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: towards real-time object detection with region proposal networks. Adv Neural Inf Process Syst. Vol. 28"},{"key":"5932_CR18","unstructured":"Wagner J, Fischer V, Herman M, Behnke S et\u00a0al (2016) Multispectral pedestrian detection using deep fusion convolutional neural networks. In: ESANN, vol. 587. pp. 509\u2013514"},{"key":"5932_CR19","doi-asserted-by":"crossref","unstructured":"Liu J, Zhang S, Wang S, Metaxas DN (2016) Multispectral deep neural networks for pedestrian detection. arXiv preprint arXiv:1611.02644","DOI":"10.5244\/C.30.73"},{"key":"5932_CR20","doi-asserted-by":"publisher","first-page":"376","DOI":"10.1016\/j.patcog.2018.08.007","volume":"86","author":"H Chen","year":"2019","unstructured":"Chen H, Li Y, Su D (2019) Multi-modal fusion network with multi-scale multi-path and cross-modal interactions for RGB-D salient object detection. Pattern Recogn 86:376\u2013385","journal-title":"Pattern Recogn"},{"issue":"6","key":"5932_CR21","doi-asserted-by":"publisher","first-page":"2825","DOI":"10.1109\/TIP.2019.2891104","volume":"28","author":"H Chen","year":"2019","unstructured":"Chen H, Li Y (2019) Three-stream attention-aware network for RGB-D salient object detection. IEEE Trans Image Process 28(6):2825\u20132835","journal-title":"IEEE Trans Image Process"},{"key":"5932_CR22","doi-asserted-by":"crossref","unstructured":"Mees O, Eitel A, Burgard W (2016) Choosing smartly: adaptive multimodal fusion for object detection in changing environments. In: 2016 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). pp. 151\u2013156. IEEE","DOI":"10.1109\/IROS.2016.7759048"},{"key":"5932_CR23","doi-asserted-by":"publisher","first-page":"55277","DOI":"10.1109\/ACCESS.2019.2913107","volume":"7","author":"N Wang","year":"2019","unstructured":"Wang N, Gong X (2019) Adaptive fusion for RGB-D salient object detection. IEEE Access 7:55277\u201355284","journal-title":"IEEE Access"},{"issue":"12","key":"5932_CR24","doi-asserted-by":"publisher","first-page":"1850","DOI":"10.1109\/LSP.2018.2873892","volume":"25","author":"C Xiang","year":"2018","unstructured":"Xiang C, Zhang L, Tang Y, Zou W, Xu C (2018) MS-CApsNet: a novel multi-scale capsule network. IEEE Signal Process Lett 25(12):1850\u20131854","journal-title":"IEEE Signal Process Lett"},{"key":"5932_CR25","unstructured":"Sabour S, Frosst N, Hinton GE (2017) Dynamic routing between capsules. Adv Neural Inf Process Syst. Vol. 30"},{"key":"5932_CR26","doi-asserted-by":"crossref","unstructured":"Valverde FR, Hurtado JV, Valada A (2021) There is more than meets the eye: self-supervised multi-object detection and tracking with sound by distilling multimodal knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11612\u201311621","DOI":"10.1109\/CVPR46437.2021.01144"},{"key":"5932_CR27","doi-asserted-by":"crossref","unstructured":"Patterson G, Hays J (2016) Coco attributes: attributes for people, animals, and objects. In: European Conference on Computer Vision, pp. 85\u2013100. Springer","DOI":"10.1007\/978-3-319-46466-4_6"},{"issue":"11","key":"5932_CR28","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. Proc IEEE 86(11):2278\u20132324","journal-title":"Proc IEEE"},{"key":"5932_CR29","doi-asserted-by":"crossref","unstructured":"Viola P, Jones M (2001) Rapid object detection using a boosted cascade of simple features. In: Proceedings of the 2001 IEEE Computer Society Conference on Computer Vision and Pattern Recognition. CVPR 2001, vol. 1, p. IEEE","DOI":"10.1109\/CVPR.2001.990517"},{"key":"5932_CR30","doi-asserted-by":"crossref","unstructured":"Dalal N, Triggs B (2005) Histograms of oriented gradients for human detection. In: 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201905), vol. 1, pp. 886\u2013893. IEEE","DOI":"10.1109\/CVPR.2005.177"},{"key":"5932_CR31","doi-asserted-by":"crossref","unstructured":"Felzenszwalb P, McAllester D, Ramanan D (2008) A discriminatively trained, multiscale, deformable part model. In: 2008 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20138. IEEE","DOI":"10.1109\/CVPR.2008.4587597"},{"issue":"9","key":"5932_CR32","doi-asserted-by":"publisher","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","volume":"37","author":"K He","year":"2015","unstructured":"He K, Zhang X, Ren S, Sun J (2015) Spatial pyramid pooling in deep convolutional networks for visual recognition. IEEE Trans Pattern Anal Mach Intell 37(9):1904\u20131916","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5932_CR33","doi-asserted-by":"crossref","unstructured":"Duan K, Bai S, Xie L, Qi H, Huang Q, Tian Q (2019) Centernet: keypoint triplets for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6569\u20136578","DOI":"10.1109\/ICCV.2019.00667"},{"key":"5932_CR34","doi-asserted-by":"crossref","unstructured":"Zhu C, He Y, Savvides M (2019) Feature selective anchor-free module for single-shot object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 840\u2013849","DOI":"10.1109\/CVPR.2019.00093"},{"key":"5932_CR35","doi-asserted-by":"crossref","unstructured":"Tian Z, Shen C, Chen H, He T (2019) Fcos: Fully convolutional one-stage object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9627\u20139636","DOI":"10.1109\/ICCV.2019.00972"},{"key":"5932_CR36","doi-asserted-by":"crossref","unstructured":"Zong Z, Song G, Liu Y (2023) Detrs with collaborative hybrid assignments training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6748\u20136758","DOI":"10.1109\/ICCV51070.2023.00621"},{"key":"5932_CR37","unstructured":"Zhu X, Su W, Lu L, Li B, Wang X, Dai J (2020) Deformable detr: deformable transformers for end-to-end object detection. In: International Conference on Learning Representations"},{"key":"5932_CR38","doi-asserted-by":"crossref","unstructured":"Fang Y, Wang W, Xie B, Sun Q, Wu L, Wang X, Huang T, Wang X, Cao Y (2023) Eva: exploring the limits of masked visual representation learning at scale. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19358\u201319369","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"5932_CR39","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold, G Gelly S, et\u00a0al (2020) An image is worth 16x16 words: transformers for image recognition at scale. In: International Conference on Learning Representations"},{"key":"5932_CR40","doi-asserted-by":"crossref","unstructured":"Kim JU, Ro YM (2023) Enabling visual object detection with object sounds via visual modality recalling memory. IEEE Trans Neural Netw Learn Syst","DOI":"10.1109\/TNNLS.2023.3323560"},{"key":"5932_CR41","unstructured":"Tan M, Le Q (2019) Efficientnet: rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning. pp. 6105\u20136114. PMLR"},{"key":"5932_CR42","doi-asserted-by":"publisher","first-page":"160708","DOI":"10.1109\/ACCESS.2021.3132050","volume":"9","author":"NS Syazwany","year":"2021","unstructured":"Syazwany NS, Nam J-H, Lee S-C (2021) MM-BiFPN: multi-modality fusion network with Bi-FPN for MRI brain tumor segmentation. IEEE Access 9:160708\u2013160720","journal-title":"IEEE Access"},{"key":"5932_CR43","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"key":"5932_CR44","unstructured":"Nair V, Hinton GE (2010) Rectified linear units improve restricted Boltzmann machines. In: ICML"},{"key":"5932_CR45","doi-asserted-by":"crossref","unstructured":"Chollet F (2017) Xception: deep learning with depthwise separable convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1251\u20131258","DOI":"10.1109\/CVPR.2017.195"},{"key":"5932_CR46","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: accelerating deep network training by reducing internal covariate shift. In: International Conference on Machine Learning, pp. 448\u2013456. PMLR"},{"key":"5932_CR47","unstructured":"Ramachandran P, Zoph B, Le QV (2017) Searching for activation functions. arXiv preprint arXiv:1710.05941 (2017)"},{"key":"5932_CR48","doi-asserted-by":"crossref","unstructured":"Chen J, Mai H, Luo L, Chen X, Wu K (2021) Effective feature fusion network in BIFPN for small object detection. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 699\u2013703. IEEE","DOI":"10.1109\/ICIP42928.2021.9506347"},{"key":"5932_CR49","doi-asserted-by":"publisher","first-page":"79876","DOI":"10.1109\/ACCESS.2020.2990700","volume":"8","author":"S Chang","year":"2020","unstructured":"Chang S, Liu J (2020) Multi-lane capsule network for classifying images with complex background. IEEE Access 8:79876\u201379886","journal-title":"IEEE Access"},{"key":"5932_CR50","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham M, Van Gool L, Williams CK, Winn J, Zisserman A (2010) The pascal visual object classes (VOC) challenge. Int J Comput Vis 88:303\u2013338","journal-title":"Int J Comput Vis"},{"key":"5932_CR51","doi-asserted-by":"crossref","unstructured":"Valverde FR, Hurtado JV, Valada A (2021) There is more than meets the eye: self-supervised multi-object detection and tracking with sound by distilling multimodal knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11612\u201311621","DOI":"10.1109\/CVPR46437.2021.01144"},{"key":"5932_CR52","doi-asserted-by":"crossref","unstructured":"Xie S, Girshick R, Doll\u00e1r P, Tu Z, He K (2017) Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1492\u20131500","DOI":"10.1109\/CVPR.2017.634"},{"key":"5932_CR53","unstructured":"Redmon J, Farhadi A (2018) Yolov3: an incremental improvement. arXiv preprint arXiv:1804.02767"},{"key":"5932_CR54","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2017) Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969","DOI":"10.1109\/ICCV.2017.322"},{"key":"5932_CR55","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"5932_CR56","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"5932_CR57","doi-asserted-by":"crossref","unstructured":"Woo S, Park J, Lee J-Y, Kweon IS (2018) Cbam: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV). pp. 3\u201319","DOI":"10.1007\/978-3-030-01234-2_1"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-05932-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-05932-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-05932-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,12]],"date-time":"2024-11-12T17:45:28Z","timestamp":1731433528000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-05932-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,23]]},"references-count":57,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["5932"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-05932-1","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,23]]},"assertion":[{"value":"26 January 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 February 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflicts of interest regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}