{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,25]],"date-time":"2025-11-25T06:57:33Z","timestamp":1764053853484,"version":"3.37.3"},"reference-count":72,"publisher":"Springer Science and Business Media LLC","issue":"28","license":[{"start":{"date-parts":[[2023,8,5]],"date-time":"2023-08-05T00:00:00Z","timestamp":1691193600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,5]],"date-time":"2023-08-05T00:00:00Z","timestamp":1691193600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 61973066"],"award-info":[{"award-number":["No. 61973066"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Major Science and Technology Projects of Liaoning Province","award":["No. 2021JH1\/10400049"],"award-info":[{"award-number":["No. 2021JH1\/10400049"]}]},{"name":"Fundation of Key Laboratory of Equipment Reliability","award":["No. WD2C20205500306"],"award-info":[{"award-number":["No. WD2C20205500306"]}]},{"name":"Fundation of Key Laboratory of Aerospace System Simulation","award":["No. 6142002200301"],"award-info":[{"award-number":["No. 6142002200301"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s00521-023-08886-2","type":"journal-article","created":{"date-parts":[[2023,8,5]],"date-time":"2023-08-05T04:01:15Z","timestamp":1691208075000},"page":"21309-21330","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["TRF-Net: a transformer-based RGB-D fusion network for desktop object instance segmentation"],"prefix":"10.1007","volume":"35","author":[{"given":"He","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0610-3732","authenticated-orcid":false,"given":"Yunzhou","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dexing","family":"Shan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaozheng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaqi","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,5]]},"reference":[{"issue":"20","key":"8886_CR1","doi-asserted-by":"publisher","first-page":"17963","DOI":"10.1007\/s00521-022-07446-4","volume":"34","author":"C Yin","year":"2022","unstructured":"Yin C, Zhang Q (2022) Object affordance detection with boundary-preserving network for robotic manipulation tasks. Neural Comput Appl 34(20):17963\u201317980","journal-title":"Neural Comput Appl"},{"issue":"6","key":"8886_CR2","doi-asserted-by":"publisher","first-page":"5984","DOI":"10.1109\/TIE.2021.3090707","volume":"69","author":"S Liu","year":"2021","unstructured":"Liu S, Tian G, Zhang Y, Zhang M, Liu S (2021) Active object detection based on a novel deep q-learning network and long-term learning strategy for service robot. IEEE Trans Ind Electron 69(6):5984\u20135993","journal-title":"IEEE Trans Ind Electron"},{"key":"8886_CR3","doi-asserted-by":"crossref","unstructured":"Sundermeyer M, Mousavian A, Triebel R, Fox D (2021) \u201cContact-graspnet: efficient 6-DOF grasp generation in cluttered scenes. In: 2021 IEEE international conference on robotics and automation (ICRA). IEEE, pp 13438\u201313444","DOI":"10.1109\/ICRA48506.2021.9561877"},{"key":"8886_CR4","doi-asserted-by":"crossref","unstructured":"Li Y, Kong T, Chu R, Li Y, Wang P, Li L (2021) Simultaneous semantic and collision learning for 6-DOF grasp pose estimation. In: 2021 IEEE\/RSJ international conference on intelligent robots and systems (IROS). IEEE, pp 3571\u20133578","DOI":"10.1109\/IROS51168.2021.9636012"},{"key":"8886_CR5","doi-asserted-by":"publisher","DOI":"10.1016\/j.rcim.2020.102086","volume":"68","author":"C Zhuang","year":"2021","unstructured":"Zhuang C, Wang Z, Zhao H, Ding H (2021) Semantic part segmentation method based 3D object pose estimation with RGB-D images for bin-picking. Robot Comput-Integr Manuf 68:102086","journal-title":"Robot Comput-Integr Manuf"},{"key":"8886_CR6","doi-asserted-by":"crossref","unstructured":"Hu Y, Hugonot J, Fua P, Salzmann M (2019) Segmentation-driven 6d object pose estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3385\u20133394","DOI":"10.1109\/CVPR.2019.00350"},{"key":"8886_CR7","doi-asserted-by":"crossref","unstructured":"Silberman N, Hoiem D, Kohli P, Fergus R (2012) Indoor segmentation and support inference from RGBD images. In: European conference on computer vision. Springer, pp 746\u2013760","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"8886_CR8","doi-asserted-by":"crossref","unstructured":"Song S, Lichtenberg SP, Xiao J (2015) Sun RGB-D: A RGB-D scene understanding benchmark suite. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 567\u2013576","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"8886_CR9","doi-asserted-by":"crossref","unstructured":"Cordts M, Omran M, Ramos S, Rehfeld T, Enzweiler M, Benenson R, Franke U, Roth S, Schiele B (2016) The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3213\u20133223","DOI":"10.1109\/CVPR.2016.350"},{"key":"8886_CR10","doi-asserted-by":"crossref","unstructured":"Zhou B, Zhao H, Puig X, Fidler S, Barriuso A, Torralba A (2017) Scene parsing through ade20k dataset. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 633\u2013641","DOI":"10.1109\/CVPR.2017.544"},{"issue":"2","key":"8886_CR11","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham M, Van Gool L, Williams CK, Winn J, Zisserman A (2010) The pascal visual object classes (VOC) challenge. Int J Comput Vis 88(2):303\u2013338","journal-title":"Int J Comput Vis"},{"key":"8886_CR12","doi-asserted-by":"crossref","unstructured":"Richtsfeld A, M\u00f6rwald T, Prankl J, Zillich M, Vincze M (2012) Segmentation of unknown objects in indoor environments. In: 2012 IEEE\/RSJ international conference on intelligent robots and systems. IEEE, pp 4791\u20134796","DOI":"10.1109\/IROS.2012.6385661"},{"key":"8886_CR13","unstructured":"Platt J (1998) Sequential minimal optimization: a fast algorithm for training support vector machines. Advances in Kernel Methods. Support Vector Learning, MIT Press, Boston"},{"key":"8886_CR14","unstructured":"Xie C, Xiang Y, Mousavian A, Fox D (2020) The best of both modes: separately leveraging RGB and depth for unseen object instance segmentation. In: Conference on robot learning. PMLR, pp 1369\u20131378"},{"issue":"5","key":"8886_CR15","doi-asserted-by":"publisher","first-page":"1343","DOI":"10.1109\/TRO.2021.3060341","volume":"37","author":"C Xie","year":"2021","unstructured":"Xie C, Xiang Y, Mousavian A, Fox D (2021) Unseen object instance segmentation for robotic environments. IEEE Trans Robot 37(5):1343\u20131359","journal-title":"IEEE Trans Robot"},{"key":"8886_CR16","unstructured":"Xiang Y, Xie C, Mousavian A, Fox D (2020) Learning RGB-D feature embeddings for unseen object instance segmentation. In: Conference on robot learning. PMLR, pp 461\u2013470"},{"key":"8886_CR17","doi-asserted-by":"crossref","unstructured":"Back S, Lee J, Kim T, Noh S, Kang R, Bak S, Lee K (2022) Unseen object amodal instance segmentation via hierarchical occlusion modeling. In: 2022 international conference on robotics and automation (ICRA). IEEE, pp 5085\u20135092","DOI":"10.1109\/ICRA46639.2022.9811646"},{"issue":"19","key":"8886_CR18","doi-asserted-by":"publisher","first-page":"12283","DOI":"10.1007\/s00521-020-05644-6","volume":"33","author":"S Zabihifar","year":"2021","unstructured":"Zabihifar S, Semochkin A, Seliverstova E, Efimov A (2021) Unreal mask: one-shot multi-object class-based pose estimation for robotic manipulation using keypoints with a synthetic dataset. Neural Comput Appl 33(19):12283\u201312300","journal-title":"Neural Comput Appl"},{"key":"8886_CR19","unstructured":"Coumans E, Bai Y (2016) Pybullet, a python module for physics simulation for games, robotics and machine learning. http:\/\/pybullet.org\/"},{"key":"8886_CR20","unstructured":"Denninger M, Sundermeyer M, Winkelbauer D, Zidan Y, Olefir D, Elbadrawy M, Lodhi A, Katam H (2019) Blenderproc. arXiv preprint arXiv:1911.01911"},{"key":"8886_CR21","doi-asserted-by":"crossref","unstructured":"Danielczuk M, Matl M, Gupta S, Li A, Lee A, Mahler J, Goldberg K (2019) Segmenting unknown 3d objects from real depth images using mask r-CNN trained on synthetic data. In: 2019 international conference on robotics and automation (ICRA). IEEE, pp 7283\u20137290","DOI":"10.1109\/ICRA.2019.8793744"},{"key":"8886_CR22","doi-asserted-by":"crossref","unstructured":"Song S, Yu F, Zeng A, Chang AX, Savva M, Funkhouser T (2017) Semantic scene completion from a single depth image. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1746\u20131754","DOI":"10.1109\/CVPR.2017.28"},{"key":"8886_CR23","unstructured":"Chang A X, Funkhouser T, Guibas L, Hanrahan P, Huang Q, Li Z, Savarese S, Savva M, Song S, Su H et al (2015) Shapenet: an information-rich 3D model repository. arXiv preprint arXiv:1512.03012"},{"key":"8886_CR24","doi-asserted-by":"crossref","unstructured":"Calli B, Singh A, Walsman A, Srinivasa S, Abbeel P, Dollar AM (2015) The YCB object and model set: towards common benchmarks for manipulation research. In: International conference on advanced robotics (ICAR). IEEE, pp 510\u2013517","DOI":"10.1109\/ICAR.2015.7251504"},{"issue":"3","key":"8886_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3486678","volume":"18","author":"D Yuan","year":"2022","unstructured":"Yuan D, Chang X, Li Z, He Z (2022) Learning adaptive spatial-temporal context-aware correlation filters for UAV tracking. ACM Trans Multimedia Comput, Commun, Appl (TOMM) 18(3):1\u201318","journal-title":"ACM Trans Multimedia Comput, Commun, Appl (TOMM)"},{"key":"8886_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109257","volume":"136","author":"X Shu","year":"2023","unstructured":"Shu X, Yang Y, Liu J, Chang X, Wu B (2023) Alvls: adaptive local variances-based levelset framework for medical images segmentation. Pattern Recogn 136:109257","journal-title":"Pattern Recogn"},{"key":"8886_CR27","doi-asserted-by":"crossref","unstructured":"Cen J, Yun P, Cai J, Wang M Y, Liu M (2021) Deep metric learning for open world semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 15333\u201315342","DOI":"10.1109\/ICCV48922.2021.01505"},{"key":"8886_CR28","doi-asserted-by":"crossref","unstructured":"Chen Y, Pont-Tuset J, Montes A, Van Gool L (2018) Blazingly fast video object segmentation with pixel-wise metric learning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1189\u20131198","DOI":"10.1109\/CVPR.2018.00130"},{"key":"8886_CR29","doi-asserted-by":"crossref","unstructured":"Milioto A, Mandtler L, Stachniss C (2019) Fast instance and semantic segmentation exploiting local connectivity, metric learning, and one-shot detection for robotics. In: 2019 international conference on robotics and automation (ICRA). IEEE, pp 5481\u20135487","DOI":"10.1109\/ICRA.2019.8793593"},{"key":"8886_CR30","doi-asserted-by":"crossref","unstructured":"Wang K, Liew J H, Zou Y, Zhou D, Feng J (2019) Panet: few-shot image semantic segmentation with prototype alignment. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 9197\u20139206","DOI":"10.1109\/ICCV.2019.00929"},{"key":"8886_CR31","doi-asserted-by":"crossref","unstructured":"Zhang M, Shi M, Li L (2021) Mfnet: multi-class few-shot segmentation network with pixel-wise metric learning. arXiv preprint arXiv:2111.00232","DOI":"10.1109\/TCSVT.2022.3193612"},{"key":"8886_CR32","unstructured":"Chen J, Lu Y, Yu Q, Luo X, Adeli E, Wang Y, Lu L, Yuille AL, Zhou Y (2021) Transunet: transformers make strong encoders for medical image segmentation. arXiv preprint arXiv:2102.04306"},{"key":"8886_CR33","doi-asserted-by":"crossref","unstructured":"Hatamizadeh A, Tang Y, Nath V, Yang D, Myronenko A, Landman B, Roth H R, Xu D (2022) Unetr: transformers for 3D medical image segmentation. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 574\u2013584","DOI":"10.1109\/WACV51458.2022.00181"},{"key":"8886_CR34","doi-asserted-by":"crossref","unstructured":"Hazirbas C, Ma L, Domokos C, Cremers D (2016) Fusenet: incorporating depth into semantic segmentation via fusion-based CNN architecture. In: Asian conference on computer vision. Springer, pp 213\u2013228","DOI":"10.1007\/978-3-319-54181-5_14"},{"key":"8886_CR35","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2017) Mask r-CNN. In: Proceedings of the IEEE international conference on computer vision, pp 2961\u20132969","DOI":"10.1109\/ICCV.2017.322"},{"key":"8886_CR36","doi-asserted-by":"crossref","unstructured":"Chen X, Lin K-Y, Wang J, Wu W, Qian C, Li H, Zeng G (2020) Bi-directional cross-modality feature propagation with separation-and-aggregation gate for RGB-D semantic segmentation. In: European conference on computer vision. Springer, pp 561\u2013577","DOI":"10.1007\/978-3-030-58621-8_33"},{"issue":"5","key":"8886_CR37","doi-asserted-by":"publisher","first-page":"1239","DOI":"10.1007\/s11263-019-01188-y","volume":"128","author":"A Valada","year":"2020","unstructured":"Valada A, Mohan R, Burgard W (2020) Self-supervised model adaptation for multimodal semantic segmentation. Int J Comput Vis 128(5):1239\u20131285","journal-title":"Int J Comput Vis"},{"key":"8886_CR38","unstructured":"Zhang Y, Yang Y, Xiong C, Sun G, Guo Y (2022) Attention-based dual supervised decoder for RGBD semantic segmentation. arXiv preprint arXiv:2201.01427"},{"issue":"1","key":"8886_CR39","doi-asserted-by":"publisher","first-page":"595","DOI":"10.1007\/s00521-022-07772-7","volume":"35","author":"SK Singh","year":"2022","unstructured":"Singh SK, Srivastava R (2022) Sl-net: self-learning and mutual attention-based distinguished window for RGBD complex salient object detection. Neural Comput Appl 35(1):595\u2013609","journal-title":"Neural Comput Appl"},{"key":"8886_CR40","unstructured":"Jiang J, Zheng L, Luo F, Zhang Z (2018) Rednet: residual encoder-decoder network for indoor RGB-D semantic segmentation. arXiv preprint arXiv:1806.01054"},{"key":"8886_CR41","doi-asserted-by":"crossref","unstructured":"Seichter D, K\u00f6hler M, Lewandowski B, Wengefeld T, Gross H M (2021Efficient RGB-D semantic segmentation for indoor scene analysis. In: 2021 IEEE international conference on robotics and automation (ICRA). IEEE, pp 13525\u201313531","DOI":"10.1109\/ICRA48506.2021.9561675"},{"issue":"10","key":"8886_CR42","doi-asserted-by":"publisher","first-page":"7547","DOI":"10.1007\/s00521-021-06845-3","volume":"34","author":"T Chen","year":"2022","unstructured":"Chen T, Hu X, Xiao J, Zhang G, Wang S (2022) Cfidnet: cascaded feature interaction decoder for RGB-D salient object detection. Neural Comput Appl 34(10):7547\u20137563","journal-title":"Neural Comput Appl"},{"key":"8886_CR43","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108468","volume":"124","author":"H Zhou","year":"2022","unstructured":"Zhou H, Qi L, Huang H, Yang X, Wan Z, Wen X (2022) Canet: co-attention network for RGB-D semantic segmentation. Pattern Recogn 124:108468","journal-title":"Pattern Recogn"},{"issue":"8","key":"8886_CR44","doi-asserted-by":"publisher","first-page":"11836","DOI":"10.1109\/TITS.2021.3107672","volume":"23","author":"Y Qian","year":"2021","unstructured":"Qian Y, Deng L, Li T, Wang C, Yang M (2021) Gated-residual block for semantic segmentation using RGB-D data. IEEE Trans Intell Transp Syst 23(8):11836\u201311844","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"8886_CR45","doi-asserted-by":"publisher","first-page":"1115","DOI":"10.1109\/LSP.2021.3084855","volume":"28","author":"Y Yue","year":"2021","unstructured":"Yue Y, Zhou W, Lei J, Yu L (2021) Two-stage cascaded decoder for semantic segmentation of RGB-D images. IEEE Signal Process Lett 28:1115\u20131119","journal-title":"IEEE Signal Process Lett"},{"key":"8886_CR46","unstructured":"Hermans A, Beyer L, Leibe B (2017) \u201cIn defense of the triplet loss for person re-identification,\u201d arXiv preprint arXiv:1703.07737"},{"key":"8886_CR47","doi-asserted-by":"crossref","unstructured":"Lee J, Abu-El-Haija S, Varadarajan B, Natsev A (2018) Collaborative deep metric learning for video understanding. In: Proceedings of the 24th ACM SIGKDD international conference on knowledge discovery and data mining, pp 481\u2013490","DOI":"10.1145\/3219819.3219856"},{"key":"8886_CR48","doi-asserted-by":"crossref","unstructured":"Xie C, Xiang Y, Harchaoui Z, Fox D (2019) Object discovery in videos as foreground motion clustering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9994\u201310003","DOI":"10.1109\/CVPR.2019.01023"},{"issue":"7","key":"8886_CR49","doi-asserted-by":"publisher","first-page":"926","DOI":"10.1109\/LSP.2018.2822810","volume":"25","author":"F Wang","year":"2018","unstructured":"Wang F, Cheng J, Liu W, Liu H (2018) Additive margin softmax for face verification. IEEE Signal Process Lett 25(7):926\u2013930","journal-title":"IEEE Signal Process Lett"},{"key":"8886_CR50","doi-asserted-by":"crossref","unstructured":"Roth K, Brattoli B, Ommer B (2019) Mic: mining interclass characteristics for improved metric learning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 8000\u20138009","DOI":"10.1109\/ICCV.2019.00809"},{"key":"8886_CR51","doi-asserted-by":"crossref","unstructured":"Jeeveswaran K, Kathiresan S, Varma A, Magdy O, Zonooz B, Arani E (2022) A comprehensive study of vision transformers on dense prediction tasks. arXiv preprint arXiv:2201.08683","DOI":"10.5220\/0010917800003124"},{"key":"8886_CR52","doi-asserted-by":"crossref","unstructured":"Zheng S, Lu J, Zhao H, Zhu X, Luo Z, Wang Y, Fu Y, Feng J, Xiang T, Torr PH et al. (2021) Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6881\u20136890","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"8886_CR53","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Proceedings of the 31st international conference on neural information processing system, pp 6000\u20136010"},{"key":"8886_CR54","doi-asserted-by":"crossref","unstructured":"Strudel R, Garcia R, Laptev I, Schmid C (2021) Segmenter: transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7262\u20137272","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"8886_CR55","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie E, Wang W, Yu Z, Anandkumar A, Alvarez JM, Luo P (2021) Segformer: simple and efficient design for semantic segmentation with transformers. Adv Neural Inf Process Syst 34:12077\u201312090","journal-title":"Adv Neural Inf Process Syst"},{"key":"8886_CR56","unstructured":"Wu S, Wu T, Lin F, Tian S, Guo G (2021) Fully transformer networks for semantic image segmentation. arXiv preprint arXiv:2106.04108"},{"key":"8886_CR57","doi-asserted-by":"crossref","unstructured":"Li Y, Chen Y, Wang N, Zhang Z (2019) Scale-aware trident networks for object detection. In: Proceedings of the IEEE\/CVF international conference on computer vision pp 6054\u20136063","DOI":"10.1109\/ICCV.2019.00615"},{"key":"8886_CR58","doi-asserted-by":"crossref","unstructured":"Zhang F, Li M, Zhai G, Liu Y (2021) Multi-branch and multi-scale attention learning for fine-grained visual categorization. In: MultiMedia modeling: 27th international conference, MMM (2021) Prague, Czech Republic, June 22\u201324, 2021, Proceedings, Part I, vol 27. Springer, pp 136\u2013147","DOI":"10.1007\/978-3-030-67832-6_12"},{"key":"8886_CR59","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108792","volume":"130","author":"H Tang","year":"2022","unstructured":"Tang H, Yuan C, Li Z, Tang J (2022) Learning attention-guided pyramidal features for few-shot fine-grained recognition. Pattern Recogn 130:108792","journal-title":"Pattern Recogn"},{"issue":"6","key":"8886_CR60","doi-asserted-by":"publisher","first-page":"1856","DOI":"10.1109\/TMI.2019.2959609","volume":"39","author":"Z Zhou","year":"2019","unstructured":"Zhou Z, Siddiquee MMR, Tajbakhsh N, Liang J (2019) Unet++: redesigning skip connections to exploit multiscale features in image segmentation. IEEE Trans Med Imaging 39(6):1856\u20131867","journal-title":"IEEE Trans Med Imaging"},{"key":"8886_CR61","first-page":"1","volume":"71","author":"N Zeng","year":"2022","unstructured":"Zeng N, Wu P, Wang Z, Li H, Liu W, Liu X (2022) A small-sized object detection oriented multi-scale feature fusion approach with application to defect detection. IEEE Trans Instrum Meas 71:1\u201314","journal-title":"IEEE Trans Instrum Meas"},{"key":"8886_CR62","first-page":"1","volume":"60","author":"Z Wang","year":"2022","unstructured":"Wang Z, Guo J, Zhang C, Wang B (2022) Multiscale feature enhancement network for salient object detection in optical remote sensing images. IEEE Trans Geosci Remote Sens 60:1\u201319","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"8886_CR63","doi-asserted-by":"crossref","unstructured":"Danielczuk M, Mousavian A, Eppner C, Fox D (2021) Object rearrangement using learned implicit collision functions. In: 2021 IEEE international conference on robotics and automation (ICRA). IEEE, pp 6010\u20136017","DOI":"10.1109\/ICRA48506.2021.9561516"},{"key":"8886_CR64","doi-asserted-by":"crossref","unstructured":"Goyal A, Mousavian A, Paxton C, Chao Y-W, Okorn B, Deng J, Fox D (2022) Ifor: iterative flow minimization for robotic object rearrangement. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14787\u201314797","DOI":"10.1109\/CVPR52688.2022.01437"},{"key":"8886_CR65","doi-asserted-by":"crossref","unstructured":"Serhan B, Pandya H, Kucukyilmaz A, Neumann G (2022) Push-to-see: learning non-prehensile manipulation to enhance instance segmentation via deep q-learning. In: 2022 international conference on robotics and automation (ICRA). IEEE, pp 1513\u20131519","DOI":"10.1109\/ICRA46639.2022.9811645"},{"key":"8886_CR66","doi-asserted-by":"crossref","unstructured":"Long J, Shelhamer E, Darrell T (2015) Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3431\u20133440","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"8886_CR67","doi-asserted-by":"crossref","unstructured":"Ronneberger O, Fischer P, Brox T, U-net: convolutional networks for biomedical image segmentation. In: Medical image computing and computer-assisted intervention-MICCAI, (2015) 18th international conference, Munich, Germany, October 5\u20139, 2015, Proceedings, Part III, vol 18. Springer, pp 234\u2013241","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"8886_CR68","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"issue":"1","key":"8886_CR69","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1109\/TITS.2017.2750080","volume":"19","author":"E Romera","year":"2017","unstructured":"Romera E, Alvarez JM, Bergasa LM, Arroyo R (2017) Erfnet: efficient residual factorized convnet for real-time semantic segmentation. IEEE Trans Intell Transp Syst 19(1):263\u2013272","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"4","key":"8886_CR70","doi-asserted-by":"publisher","first-page":"1207","DOI":"10.1007\/s00521-020-05009-z","volume":"33","author":"M Gutoski","year":"2021","unstructured":"Gutoski M, Lazzaretti AE, Lopes HS (2021) Deep metric learning for open-set human action recognition in videos. Neural Comput Appl 33(4):1207\u20131220","journal-title":"Neural Comput Appl"},{"issue":"5","key":"8886_CR71","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1109\/34.1000236","volume":"24","author":"D Comaniciu","year":"2002","unstructured":"Comaniciu D, Meer P (2002) Mean shift: a robust approach toward feature space analysis. IEEE Trans Pattern Anal Mach Intell 24(5):603\u2013619","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"8886_CR72","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: IEEE conference on computer vision and pattern recognition. IEEE, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08886-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-023-08886-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08886-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,30]],"date-time":"2023-08-30T00:33:38Z","timestamp":1693355618000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-023-08886-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,5]]},"references-count":72,"journal-issue":{"issue":"28","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["8886"],"URL":"https:\/\/doi.org\/10.1007\/s00521-023-08886-2","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2023,8,5]]},"assertion":[{"value":"28 November 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 July 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 August 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}