{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T04:06:27Z","timestamp":1783656387583,"version":"3.55.0"},"reference-count":82,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2021,5,5]],"date-time":"2021-05-05T00:00:00Z","timestamp":1620172800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,5,5]],"date-time":"2021-05-05T00:00:00Z","timestamp":1620172800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"Research Grants Council of Hong Kong","award":["CityU 11205015"],"award-info":[{"award-number":["CityU 11205015"]}]},{"name":"Research Grants Council of Hong Kong","award":["CityU 11255716"],"award-info":[{"award-number":["CityU 11255716"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,7]]},"DOI":"10.1007\/s11263-021-01452-0","type":"journal-article","created":{"date-parts":[[2021,5,5]],"date-time":"2021-05-05T09:03:54Z","timestamp":1620205434000},"page":"2076-2096","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["CNN-Based RGB-D Salient Object Detection: Learn, Select, and Fuse"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3138-505X","authenticated-orcid":false,"given":"Hao","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Youfu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongjian","family":"Deng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guosheng","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,5,5]]},"reference":[{"key":"1452_CR1","doi-asserted-by":"crossref","unstructured":"Alpert, S., Galun, M., Basri, R., & Brandt, A. (2007). Image segmentation by probabilistic bottom-up aggregation and cue integration. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 1\u20138).","DOI":"10.1109\/CVPR.2007.383017"},{"issue":"12","key":"1452_CR2","doi-asserted-by":"publisher","first-page":"5706","DOI":"10.1109\/TIP.2015.2487833","volume":"24","author":"A Borji","year":"2015","unstructured":"Borji, A., Cheng, M. M., Jiang, H., & Li, J. (2015). Salient object detection: A benchmark. IEEE Transactions on Image Processing, 24(12), 5706\u20135722.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR3","doi-asserted-by":"crossref","unstructured":"Camplani, M., Hannuna, S.L., Mirmehdi, M., Damen, D., Paiement, A., Tao, L., & Burghardt, T. (2015). Real-time rgb-d tracking with depth scaling kernelised correlation filters and occlusion handling. In Proceedings of the British machine vision conference (pp. 145\u20131).","DOI":"10.5244\/C.29.145"},{"key":"1452_CR4","doi-asserted-by":"crossref","unstructured":"Chen, H., & Li, Y. (2018). Progressively complementarity-aware fusion network for rgb-d salient object detection. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 3051\u20133060).","DOI":"10.1109\/CVPR.2018.00322"},{"key":"1452_CR5","doi-asserted-by":"crossref","unstructured":"Chen, H., Li, Y., & Su, D. (2018). Multi-modal fusion network with multi-scale multi-path and cross-modal interactions for RGB-D salient object detection. Pattern Recognition.","DOI":"10.1016\/j.patcog.2018.08.007"},{"issue":"6","key":"1452_CR6","doi-asserted-by":"publisher","first-page":"2825","DOI":"10.1109\/TIP.2019.2891104","volume":"28","author":"H Chen","year":"2019","unstructured":"Chen, H., & Li, Y. (2019). Three-stream attention-aware network for RGB-D salient object detection. IEEE Transactions on Image Processing, 28(6), 2825\u20132835.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR7","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Cai, R., Li, Z., Zhao, X., & Huang, K. (2017). Locality sensitive deconvolution networks with gated fusion for RGB-D indoor semantic segmentation. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (Vol.\u00a03).","DOI":"10.1109\/CVPR.2017.161"},{"key":"1452_CR8","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Fu, H., Wei, X., Xiao, J., & Cao, X. (2014) Depth enhanced saliency detection method. In Proceedings of international conference on internet multimedia computing and service (ICIMCS) (pp. 23\u201327).","DOI":"10.1145\/2632856.2632866"},{"issue":"3","key":"1452_CR9","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1109\/TPAMI.2014.2345401","volume":"37","author":"MM Cheng","year":"2015","unstructured":"Cheng, M. M., Mitra, N. J., Huang, X., Torr, P. H., & Hu, S. M. (2015). Global contrast based salient region detection. IEEE Transactions on Pattern Analysis and Machine Intelligence, 37(3), 569\u2013582.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1452_CR10","doi-asserted-by":"crossref","unstructured":"Christoudias, C.M., Urtasun, R., Salzmann, M., & Darrell, T. (2010). Learning to recognize objects from unseen modalities. In Proceedings of European conference on computer vision (pp. 677\u2013691).","DOI":"10.1007\/978-3-642-15549-9_49"},{"key":"1452_CR11","doi-asserted-by":"crossref","unstructured":"Ciptadi, A., Hermans, T., & Rehg, J. M. (2013). An in depth view of saliency. In Proceedings of the British machine vision conference.","DOI":"10.5244\/C.27.112"},{"key":"1452_CR12","unstructured":"Cong, R., Lei, J., Fu, H., Lin, W., Huang, Q., Cao, X., & Hou, C. (2017). An iterative co-saliency framework for RGBD images. IEEE Transactions on Cybernetics"},{"issue":"2","key":"1452_CR13","doi-asserted-by":"publisher","first-page":"568","DOI":"10.1109\/TIP.2017.2763819","volume":"27","author":"R Cong","year":"2018","unstructured":"Cong, R., Lei, J., Fu, H., Huang, Q., Cao, X., & Hou, C. (2018). Co-saliency detection for RGBD images based on multi-constraint feature matching and cross label propagation. IEEE Transactions on Image Processing, 27(2), 568\u2013579.","journal-title":"IEEE Transactions on Image Processing"},{"issue":"6","key":"1452_CR14","doi-asserted-by":"publisher","first-page":"819","DOI":"10.1109\/LSP.2016.2557347","volume":"23","author":"R Cong","year":"2016","unstructured":"Cong, R., Lei, J., Zhang, C., Huang, Q., Cao, X., & Hou, C. (2016). Saliency detection for stereoscopic images based on depth confidence analysis and multiple cues fusion. Signal Processing Letters, 23(6), 819\u2013823.","journal-title":"Signal Processing Letters"},{"key":"1452_CR15","doi-asserted-by":"crossref","unstructured":"Desingh, K., Krishna, K. M., Rajan, D., & Jawahar, C. (2013). Depth really matters: Improving visual salient region detection with depth. In Proceedings of the British machine vision conference.","DOI":"10.5244\/C.27.98"},{"key":"1452_CR16","doi-asserted-by":"crossref","unstructured":"Du, D., Wang, L., Wang, H., Zhao, K, & Wu, G. (2019). Translate-to-recognize networks for RGB-D scene recognition. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 11836\u201311845.","DOI":"10.1109\/CVPR.2019.01211"},{"key":"1452_CR17","doi-asserted-by":"crossref","unstructured":"Fan, D. P., Cheng, M. M., Liu, Y., Li, T., & Borji, A. (2017). Structure-measure: A new way to evaluate foreground maps. In Proceedings of the IEEE computer society conference on computer vision (pp. 4548\u20134557).","DOI":"10.1109\/ICCV.2017.487"},{"key":"1452_CR18","doi-asserted-by":"crossref","unstructured":"Fan, D. P., Gong, C., Cao, Y., Ren, B., Cheng, M. M., & Borji, A. (2018). Enhanced-alignment measure for binary foreground map evaluation. In Proceedings of IJCAI.","DOI":"10.24963\/ijcai.2018\/97"},{"key":"1452_CR19","doi-asserted-by":"crossref","unstructured":"Fan, D. P., Lin, Z., Zhang, Z., Zhu, M., & Cheng, M. M. (2020). Rethinking RGB-D salient object detection: Models, data sets, and large-scale benchmarks. IEEE Transactions on Neural Networking Learning Systems.","DOI":"10.1109\/TNNLS.2020.2996406"},{"key":"1452_CR20","doi-asserted-by":"crossref","unstructured":"Fan, X., Liu, Z., & Sun, G. (2014). Salient region detection for stereoscopic images. In Proceedings of international conference on digital signal process (pp. 454\u2013458).","DOI":"10.1109\/ICDSP.2014.6900706"},{"key":"1452_CR21","doi-asserted-by":"crossref","unstructured":"Feng, D., Barnes, N., You, S., & McCarthy, C. (2016). Local background enclosure for rgb-d salient object detection. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 2343\u20132350).","DOI":"10.1109\/CVPR.2016.257"},{"key":"1452_CR22","doi-asserted-by":"crossref","unstructured":"Fu, H., Xu, D., Lin, S., & Liu, J. (2015). Object-based rgbd image co-segmentation with mutex constraint. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 4428\u20134436).","DOI":"10.1109\/CVPR.2015.7299072"},{"issue":"3","key":"1452_CR23","doi-asserted-by":"publisher","first-page":"1418","DOI":"10.1109\/TIP.2017.2651369","volume":"26","author":"H Fu","year":"2017","unstructured":"Fu, H., Xu, D., & Lin, S. (2017). Object-based multiple foreground segmentation in RGBD video. IEEE Transactions on Image Processing, 26(3), 1418\u20131427.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR24","doi-asserted-by":"crossref","unstructured":"Garcia, N.C., Morerio, P., & Murino, V. (2018). Modality distillation with multiple stream networks for action recognition. In Proceedings of European conference on computer vision (pp. 103\u2013118).","DOI":"10.1007\/978-3-030-01237-3_7"},{"key":"1452_CR25","doi-asserted-by":"crossref","unstructured":"Guo, J., Ren, T., & Bei, J. (2016). Salient object detection for RGB-D image via saliency evolution. In Proceedings of IEEE international conference on multimedia and expo (pp. 1\u20136).","DOI":"10.1109\/ICME.2016.7552907"},{"key":"1452_CR26","doi-asserted-by":"crossref","unstructured":"Gupta, S., Girshick, R., Arbel\u00e1ez, P., & Malik, J. (2014). Learning rich features from RGB-D images for object detection and segmentation. In Proceedings of European conference on computer vision (pp. 345\u2013360).","DOI":"10.1007\/978-3-319-10584-0_23"},{"key":"1452_CR27","doi-asserted-by":"crossref","unstructured":"Gupta, S., Hoffman, J., & Malik, J. (2016). Cross modal distillation for supervision transfer. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 2827\u20132836).","DOI":"10.1109\/CVPR.2016.309"},{"issue":"2","key":"1452_CR28","doi-asserted-by":"publisher","first-page":"133","DOI":"10.1007\/s11263-014-0777-6","volume":"112","author":"S Gupta","year":"2015","unstructured":"Gupta, S., Arbel\u00e1ez, P., Girshick, R., & Malik, J. (2015). Indoor scene understanding with rgb-d images: Bottom-up segmentation, object detection and semantic segmentation. International Journal of Computer Vision, 112(2), 133\u2013149.","journal-title":"International Journal of Computer Vision"},{"key":"1452_CR29","doi-asserted-by":"crossref","unstructured":"Han, J., Chen, H., Liu, N., Yan, C., & Li, X. (2017). Cnns-based rgb-d saliency detection via cross-view transfer and multiview fusion. IEEE Transactions on Cybernetics.","DOI":"10.1109\/TCYB.2017.2761775"},{"issue":"5","key":"1452_CR30","doi-asserted-by":"publisher","first-page":"1318","DOI":"10.1109\/TCYB.2013.2265378","volume":"43","author":"J Han","year":"2013","unstructured":"Han, J., Shao, L., Xu, D., & Shotton, J. (2013). Enhanced computer vision with microsoft kinect sensor: A review. IEEE Transactions on Cybernetics, 43(5), 1318\u20131334.","journal-title":"IEEE Transactions on Cybernetics"},{"key":"1452_CR31","doi-asserted-by":"crossref","unstructured":"Harel, J., Koch, C., & Perona, P. (2007). Graph-based visual saliency. In Proceedings of advances in neural information processing systems (pp. 545\u2013552).","DOI":"10.7551\/mitpress\/7503.003.0073"},{"key":"1452_CR32","doi-asserted-by":"crossref","unstructured":"Hazirbas, C., Ma, L., Domokos, C., & Cremers, D. (2016). Fusenet: Incorporating depth into semantic segmentation via fusion-based cnn architecture. In Proceedings of Asian conference computer vision (pp. 213\u2013228). Springer.","DOI":"10.1007\/978-3-319-54181-5_14"},{"key":"1452_CR33","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531."},{"key":"1452_CR34","doi-asserted-by":"crossref","unstructured":"Hoffman, J., Gupta, S., & Darrell, T. (2016). Learning with side information through modality hallucination. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 826\u2013834).","DOI":"10.1109\/CVPR.2016.96"},{"key":"1452_CR35","doi-asserted-by":"crossref","unstructured":"Hou, Q., Cheng, M.M., Hu, X., Borji, A., Tu, Z., & Torr, P. (2017). Deeply supervised salient object detection with short connections. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 5300\u20135309).","DOI":"10.1109\/CVPR.2017.563"},{"key":"1452_CR36","doi-asserted-by":"crossref","unstructured":"Hou, J., Dai, A., & Nie\u00dfner, M. (2019). 3D-SIS: 3D semantic instance segmentation of RGB-D scans. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 4421\u20134430).","DOI":"10.1109\/CVPR.2019.00455"},{"key":"1452_CR37","unstructured":"Huang, Z., & Wang, N. (2017). Like what you like: Knowledge distill via neuron selectivity transfer. arXiv preprint arXiv:1707.01219."},{"key":"1452_CR38","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., & Darrell, T. (2014). Caffe: Convolutional architecture for fast feature embedding. In Proceedings of ACM international conference on multimedia (pp. 675\u2013678).","DOI":"10.1145\/2647868.2654889"},{"key":"1452_CR39","doi-asserted-by":"crossref","unstructured":"Ju, R., Ge, L., Geng, W., Ren, T., & Wu, G. (2014). Depth saliency based on anisotropic center-surround difference. In Proceedings of European conference on image process (pp. 1115\u20131119).","DOI":"10.1109\/ICIP.2014.7025222"},{"key":"1452_CR40","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In Proceedings of advances in neural information processing systems (pp. 1097\u20131105)."},{"key":"1452_CR41","doi-asserted-by":"crossref","unstructured":"Lang, C., Nguyen, T. V., Katti, H., Yadati, K., Kankanhalli, M., & Yan, S. (2012). Depth matters: Influence of depth cues on visual saliency. In Proceedings of European conference on computer vision (pp. 101\u2013115).","DOI":"10.1007\/978-3-642-33709-3_8"},{"key":"1452_CR42","doi-asserted-by":"crossref","unstructured":"Lenc, K., & Vedaldi, A. (2015). Understanding image representations by measuring their equivariance and equivalence. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 991\u2013999).","DOI":"10.1109\/CVPR.2015.7298701"},{"key":"1452_CR43","doi-asserted-by":"crossref","unstructured":"Li, Q., Jin, S., & Yan, J. (2017). Mimicking very efficient network for object detection. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 7341\u20137349).","DOI":"10.1109\/CVPR.2017.776"},{"key":"1452_CR44","doi-asserted-by":"crossref","unstructured":"Li, J., Liu, Y., Gong, D., Shi, Q., Yuan, X., Zhao, C., & Reid, I. (2019). RGBD based dimensional decomposition residual network for 3D semantic scene completion. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 7693\u20137702).","DOI":"10.1109\/CVPR.2019.00788"},{"key":"1452_CR45","doi-asserted-by":"crossref","unstructured":"Li, G., Liu, Z., Ye, L., Wang, Y., & Ling, H. (2020b). Cross-modal weighting network for RGB-D salient object detection. In Proceedings of European conference on computer vision (pp. 665\u2013681).","DOI":"10.1007\/978-3-030-58520-4_39"},{"key":"1452_CR46","doi-asserted-by":"crossref","unstructured":"Li, N., Ye, J., Ji, Y., Ling, H., & Yu, J. (2014). Saliency detection on light field. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 2806\u20132813).","DOI":"10.1109\/CVPR.2014.359"},{"issue":"4","key":"1452_CR47","doi-asserted-by":"publisher","first-page":"1591","DOI":"10.1109\/TIP.2018.2878956","volume":"28","author":"G Li","year":"2018","unstructured":"Li, G., Gan, Y., Wu, H., Xiao, N., & Lin, L. (2018). Cross-modal attentional context learning for RGB-D object detection. IEEE Transactions on Image Processing, 28(4), 1591\u20131601.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR48","doi-asserted-by":"publisher","first-page":"4873","DOI":"10.1109\/TIP.2020.2976689","volume":"29","author":"G Li","year":"2020","unstructured":"Li, G., Liu, Z., & Ling, H. (2020a). ICNet: Information conversion network for RGB-D based salient object detection. IEEE Transactions on Image Processing, 29, 4873\u20134884.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR49","doi-asserted-by":"crossref","unstructured":"Lin, D., Chen, G., Cohen-Or. D., Heng. P. A., & Huang, H. (2017a). Cascaded feature network for semantic segmentation of RGB-D images. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 1320\u20131328).","DOI":"10.1109\/ICCV.2017.147"},{"key":"1452_CR50","doi-asserted-by":"crossref","unstructured":"Lin, G., Milan, A., Shen, C., & Reid, I. (2017b). Refinenet: Multi-path refinement networks for high-resolution semantic segmentation. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (Vol.\u00a01, p.\u00a03).","DOI":"10.1109\/CVPR.2017.549"},{"key":"1452_CR51","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., & Darrell, T. (2015). Fully convolutional networks for semantic segmentation. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 3431\u20133440).","DOI":"10.1109\/CVPR.2015.7298965"},{"issue":"3","key":"1452_CR52","doi-asserted-by":"publisher","first-page":"541","DOI":"10.1109\/TPAMI.2012.98","volume":"35","author":"V Mahadevan","year":"2013","unstructured":"Mahadevan, V., Vasconcelos, N., et al. (2013). Biologically inspired object tracking using center-surround saliency mechanisms. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(3), 541\u2013554.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1452_CR53","doi-asserted-by":"crossref","unstructured":"Margolin, R., Zelnik-Manor, L., & Tal, A. (2014). How to evaluate foreground maps? In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 248\u2013255).","DOI":"10.1109\/CVPR.2014.39"},{"key":"1452_CR54","doi-asserted-by":"crossref","unstructured":"Misra, I., Shrivastava, A., Gupta, A., & Hebert, M. (2016). Cross-stitch networks for multi-task learning. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 3994\u20134003).","DOI":"10.1109\/CVPR.2016.433"},{"key":"1452_CR55","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., & Ng, A. Y. (2011). Multimodal deep learning. In Proceedings of international conference on machine learning (pp. 689\u2013696)."},{"key":"1452_CR56","unstructured":"Niu, Y., Geng, Y., Li, X., & Liu, F. (2012). Leveraging stereopsis for saliency analysis. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 454\u2013461)."},{"key":"1452_CR57","unstructured":"Park, S.J., Hong, K.S., & Lee, S. (2017). RDFNet: RGB-D multi-level residual feature fusion for indoor semantic segmentation. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition."},{"key":"1452_CR58","doi-asserted-by":"crossref","unstructured":"Peng, H., Li, B., Xiong, W., Hu, W., & Ji, R. (2014). RGBD salient object detection: a benchmark and algorithms. In Proceedings of European conference on computer vision (pp. 92\u2013109).","DOI":"10.1007\/978-3-319-10578-9_7"},{"key":"1452_CR59","doi-asserted-by":"crossref","unstructured":"Piao, Y., Ji, W., Li, J., Zhang, M., & Lu, H. (2019). Depth-induced multi-scale recurrent attention network for saliency detection. In Proceedings of the IEEE computer society conference on computer vision (pp. 7254\u20137263).","DOI":"10.1109\/ICCV.2019.00735"},{"key":"1452_CR60","doi-asserted-by":"crossref","unstructured":"Piao, Y., Rong, Z., Zhang, M., & Lu, H. (2020). Exploit and replace: An asymmetrical two-stream architecture for versatile light field saliency detection. In AAAI (pp. 11865\u201311873).","DOI":"10.1609\/aaai.v34i07.6860"},{"key":"1452_CR61","doi-asserted-by":"crossref","unstructured":"Qi, X., Liao, R., Jia, J., Fidler, S., & Urtasun, R. (2017). 3D graph neural networks for RGBD semantic segmentation. In Proceedings of the IEEE computer society conference on computer vision (pp. 5199\u20135208).","DOI":"10.1109\/ICCV.2017.556"},{"issue":"5","key":"1452_CR62","doi-asserted-by":"publisher","first-page":"2274","DOI":"10.1109\/TIP.2017.2682981","volume":"26","author":"L Qu","year":"2017","unstructured":"Qu, L., He, S., Zhang, J., Tian, J., Tang, Y., & Yang, Q. (2017). Rgbd salient object detection via deep fusion. IEEE Transactions on Image Processing, 26(5), 2274\u20132285.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR63","doi-asserted-by":"crossref","unstructured":"Ren, J., Gong, X., Yu, L., Zhou, W., & Ying\u00a0Yang, M. (2015). Exploiting global priors for RGB-D saliency detection. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 25\u201332).","DOI":"10.1109\/CVPRW.2015.7301391"},{"key":"1452_CR64","unstructured":"Romero, A., Ballas, N., Kahou, S. E., Chassang, A., Gatta, C., & Bengio, Y. (2014). Fitnets: Hints for thin deep nets. arXiv preprint arXiv:1412.6550."},{"issue":"10","key":"1452_CR65","doi-asserted-by":"publisher","first-page":"1932","DOI":"10.1016\/j.patcog.2006.04.010","volume":"39","author":"L Shao","year":"2006","unstructured":"Shao, L., & Brady, M. (2006). Specific object retrieval based on salient regions. Pattern Recognition, 39(10), 1932\u20131948.","journal-title":"Pattern Recognition"},{"key":"1452_CR66","unstructured":"Socher, R., Ganjoo, M., Manning, C. D., & Ng, A. (2013). Zero-shot learning through cross-modal transfer. In Proceedings of advances in neural information processing systems (pp. 935\u2013943)."},{"key":"1452_CR67","doi-asserted-by":"crossref","unstructured":"Song, S., Lichtenberg, S.P., & Xiao, J. (2015). SUN RGB-D: A RGB-D scene understanding benchmark suite. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 567\u2013576).","DOI":"10.1109\/CVPR.2015.7298655"},{"issue":"9","key":"1452_CR68","doi-asserted-by":"publisher","first-page":"4204","DOI":"10.1109\/TIP.2017.2711277","volume":"26","author":"H Song","year":"2017","unstructured":"Song, H., Liu, Z., Du, H., Sun, G., Le Meur, O., & Ren, T. (2017). Depth-aware salient object detection and segmentation via multiscale discriminative saliency fusion and bootstrap learning. IEEE Transactions on Image Processing, 26(9), 4204\u20134216.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR69","doi-asserted-by":"crossref","unstructured":"Wang, W., & Neumann, U. (2018). Depth-aware CNN for RGB-D segmentation. In Proceedings of the IEEE computer society conference on computer vision (pp. 135\u2013150).","DOI":"10.1007\/978-3-030-01252-6_9"},{"key":"1452_CR70","doi-asserted-by":"crossref","unstructured":"Wang, A., Cai, J., Lu, J., & Cham, T. J. (2015). Mmss: Multi-modal sharable and specific feature learning for RGB-D object recognition. In Proceedings of the IEEE computer society conference on computer vision (pp. 1125\u20131133).","DOI":"10.1109\/ICCV.2015.134"},{"key":"1452_CR71","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1016\/j.patcog.2017.07.026","volume":"72","author":"X Xu","year":"2017","unstructured":"Xu, X., Li, Y., Wu, G., & Luo, J. (2017). Multi-modal deep feature learning for RGB-D object detection. Pattern Recognition, 72, 300\u2013313.","journal-title":"Pattern Recognition"},{"key":"1452_CR72","doi-asserted-by":"crossref","unstructured":"Yan, Q., Xu, L., Shi, J., & Jia, J. (2013). Hierarchical saliency detection. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 1155\u20131162).","DOI":"10.1109\/CVPR.2013.153"},{"issue":"3","key":"1452_CR73","doi-asserted-by":"publisher","first-page":"576","DOI":"10.1109\/TPAMI.2016.2547384","volume":"39","author":"J Yang","year":"2017","unstructured":"Yang, J., & Yang, M. H. (2017). Top-down visual saliency via joint crf and dictionary learning. IEEE Transactions on Pattern Analysis and Machine Intelligence, 39(3), 576\u2013588.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1452_CR74","doi-asserted-by":"crossref","unstructured":"Zeng, J., Tong, Y., Huang, Y., Yan, Q., Sun, W., Chen, J., & Wang, Y. (2019). Deep surface normal estimation with hierarchical RGB-D fusion. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 6153\u20136162).","DOI":"10.1109\/CVPR.2019.00631"},{"key":"1452_CR75","unstructured":"Zhang, M., Li, J., Wei, J., Piao, Y., & Lu, H. (2019). Memory-oriented decoder for light field salient object detection. In Proceedings of advances in neural information processing systems (pp. 898\u2013908)."},{"key":"1452_CR76","doi-asserted-by":"publisher","first-page":"6276","DOI":"10.1109\/TIP.2020.2990341","volume":"29","author":"M Zhang","year":"2020","unstructured":"Zhang, M., Ji, W., Piao, Y., Li, J., Zhang, Y., Xu, S., et al. (2020). LFNet: Light field fusion network for salient object detection. IEEE Transactions on Image Processing, 29, 6276\u20136287.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1452_CR77","doi-asserted-by":"crossref","unstructured":"Zhao, J. X., Cao, Y., Fan, D. P., Cheng, M. M., Li, X. Y., & Zhang, L. (2019). Contrast prior and fluid pyramid integration for RGBD salient object detection. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 3927\u20133936).","DOI":"10.1109\/CVPR.2019.00405"},{"key":"1452_CR78","doi-asserted-by":"crossref","unstructured":"Zhao, X., Pang, Y., Zhang, L., Lu, H., & Zhang, L. (2020a). Suppress and balance: A simple gated network for salient object detection. In Proceedings of European conference on computer vision.","DOI":"10.1007\/978-3-030-58536-5_3"},{"key":"1452_CR79","doi-asserted-by":"crossref","unstructured":"Zhao, X., Zhang, L., Pang, Y., Lu, H., & Zhang, L. (2020b). A single stream network for robust and real-time RGB-D salient object detection. In Proceedings of European conference on computer vision.","DOI":"10.1007\/978-3-030-58542-6_39"},{"key":"1452_CR80","doi-asserted-by":"crossref","unstructured":"Zhou, T., Fan, D. P., Cheng, M. M., Shen, J., & Shao, L. (2020). RGB-D salient object detection: A survey. Computational Visual Media, pp. 1\u201333","DOI":"10.1007\/s41095-020-0199-z"},{"key":"1452_CR81","unstructured":"Zhu, C., & Li, G. (2017). A three-pathway psychobiological framework of salient object detection using stereoscopic technology. In Proceedings of the IEEE computer society conference on computer vision (pp. 3008\u20133014)."},{"key":"1452_CR82","doi-asserted-by":"crossref","unstructured":"Zhu, H., Weibel, J. B., & Lu, S. (2016). Discriminative multi-modal feature fusion for RGBD indoor scene recognition. In Proceedings of the IEEE computer society conference on computer vision and pattern recognition (pp. 2969\u20132976).","DOI":"10.1109\/CVPR.2016.324"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01452-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-021-01452-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01452-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,3]],"date-time":"2023-11-03T08:14:05Z","timestamp":1698999245000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-021-01452-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,5]]},"references-count":82,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2021,7]]}},"alternative-id":["1452"],"URL":"https:\/\/doi.org\/10.1007\/s11263-021-01452-0","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,5,5]]},"assertion":[{"value":"17 January 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 February 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 May 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}