{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T07:16:42Z","timestamp":1785395802027,"version":"3.56.0"},"reference-count":109,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2021,6,21]],"date-time":"2021-06-21T00:00:00Z","timestamp":1624233600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2021,6,21]],"date-time":"2021-06-21T00:00:00Z","timestamp":1624233600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2021,12]]},"DOI":"10.1007\/s00371-021-02200-8","type":"journal-article","created":{"date-parts":[[2021,6,21]],"date-time":"2021-06-21T06:02:35Z","timestamp":1624255355000},"page":"2931-2949","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":58,"title":["Image and video processing on mobile devices: a survey"],"prefix":"10.1007","volume":"37","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7502-534X","authenticated-orcid":false,"given":"Chamin","family":"Morikawa","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michihiro","family":"Kobayashi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masaki","family":"Satoh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yasuhiro","family":"Kuroda","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Teppei","family":"Inomata","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hitoshi","family":"Matsuo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Takeshi","family":"Miura","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masaki","family":"Hilaga","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,6,21]]},"reference":[{"key":"2200_CR1","doi-asserted-by":"crossref","unstructured":"Adams, M.D.: Coaxial range measurement current trends for mobile robotic applications (2002)","DOI":"10.1109\/7361.987055"},{"key":"2200_CR2","doi-asserted-by":"publisher","DOI":"10.1186\/s13640-018-0278-6","author":"R Angulu","year":"2018","unstructured":"Angulu, R., Tapamo, J.R., Adewumi, A.: Age estimation via face images: a survey. EURASIP J. Image Video Process. (2018). https:\/\/doi.org\/10.1186\/s13640-018-0278-6","journal-title":"EURASIP J. Image Video Process."},{"key":"2200_CR3","unstructured":"Apple Inc.: Core ML - Apple Developer. https:\/\/developer.apple.com\/documentation\/coreml (2016). Accessed 19 Apr 2021"},{"key":"2200_CR4","unstructured":"Apple Inc.: Tracking and visualizing faces. https:\/\/developer.apple.com\/documentation\/arkit\/content_anchors\/tracking_and_visualizing_faces (2019). Accessed 20 Apr 2021"},{"key":"2200_CR5","unstructured":"Apple Inc.: Apple unveils all-new iPad Air with A14 Bionic, Apple\u2019s most advanced chip. https:\/\/www.apple.com\/newsroom\/2020\/09\/apple-unveils-all-new-ipad-air-with-a14-bionic-apples-most-advanced-chip\/ (2020). Accessed 16 Apr 2021"},{"key":"2200_CR6","unstructured":"Apple Insider: Hands on with Apple\u2019s FaceTime Attention Correction feature in iOS 13. https:\/\/appleinsider.com\/articles\/19\/07\/03\/hands-on-with-apples-facetime-attention-correction-feature-in-ios-13 (2019). Accessed 21 Apr 2021"},{"key":"2200_CR7","unstructured":"ARM Inc.: Arm Compute Library. https:\/\/developer.arm.com\/ip-products\/processors\/machine-learning\/compute-library (2017). Accessed 19 Apr 2021"},{"key":"2200_CR8","unstructured":"Ba, L.J., Kiros, J.R., Hinton, G.E.: Layer normalization. CoRR arXiv:1607.06450 (2016)"},{"key":"2200_CR9","doi-asserted-by":"publisher","first-page":"228","DOI":"10.1007\/978-3-642-10520-3_21","volume-title":"Advances in Visual Computing","author":"B Bartczak","year":"2009","unstructured":"Bartczak, B., Koch, R.: Dense depth maps from low resolution time-of-flight depth and high resolution color views. In: Bebis, G., Boyle, R., Parvin, B., Koracin, D., Kuno, Y., Wang, J., Pajarola, R., Lindstrom, P., Hinkenjann, A., Encarna\u00e7\u00e3o, M.L., Silva, C.T., Coming, D. (eds.) Advances in Visual Computing, pp. 228\u2013239. Springer, Berlin (2009)"},{"key":"2200_CR10","doi-asserted-by":"publisher","DOI":"10.14569\/IJACSA.2017.081144","author":"E Bebeselea-Sterp","year":"2017","unstructured":"Bebeselea-Sterp, E., Brad, R., Brad, R.: A comparative study of stereovision algorithms. Int. J. Adv. Comput. Sci. Appl. (2017). https:\/\/doi.org\/10.14569\/IJACSA.2017.081144","journal-title":"Int. J. Adv. Comput. Sci. Appl."},{"key":"2200_CR11","doi-asserted-by":"publisher","unstructured":"Bleiweiss, A., Werman, M.: Robust head pose estimation by fusing time-of-flight depth and color. In: 2010 IEEE International Workshop on Multimedia Signal Processing, MMSP 2010, Saint Malo, France, October 4\u20136, 2010, pp. 116\u2013121. IEEE (2010). https:\/\/doi.org\/10.1109\/MMSP.2010.5662004","DOI":"10.1109\/MMSP.2010.5662004"},{"key":"2200_CR12","unstructured":"Bochkovskiy, A., Wang, C., Liao, H.M.: Yolov4: Optimal speed and accuracy of object detection. CoRR. arXiv:2004.10934 (2020)"},{"key":"2200_CR13","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer, Berlin (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2200_CR14","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., Schiele, B.: The cityscapes dataset for semantic urban scene understanding. In: Proc. of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"2200_CR15","doi-asserted-by":"publisher","unstructured":"Dalal, N., Triggs, B.: Histograms of oriented gradients for human detection. In: 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201905). vol.\u00a01, pp. 886\u2013893 (2005). https:\/\/doi.org\/10.1109\/CVPR.2005.177","DOI":"10.1109\/CVPR.2005.177"},{"key":"2200_CR16","doi-asserted-by":"publisher","unstructured":"Debevec, P.E., Malik, J.: Recovering high dynamic range radiance maps from photographs. In: Owen, G.S., Whitted, T., Mones-Hattal, B. (eds.) Proceedings of the 24th Annual Conference on Computer Graphics and Interactive Techniques, SIGGRAPH 1997, Los Angeles, CA, USA, August 3\u20138, 1997, pp. 369\u2013378. ACM (1997). https:\/\/doi.org\/10.1145\/258734.258884","DOI":"10.1145\/258734.258884"},{"key":"2200_CR17","doi-asserted-by":"publisher","unstructured":"Deng, J., Dong, W., Socher, R., Li, L., Li, K., Li, F.-F.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255 (2009). https:\/\/doi.org\/10.1109\/CVPR.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2200_CR18","unstructured":"Devries, T., Taylor, G.W.: Improved regularization of convolutional neural networks with cutout. CoRR. arXiv:1708.04552 (2017)"},{"key":"2200_CR19","doi-asserted-by":"publisher","unstructured":"Dew, P., Fuchs, H., Kunii, T., Wozny, M.: Parallel processing for computer vision and display (panel session). In: Proc. ACM Siggraph Computer Graphics, vol.\u00a022, p.\u00a0346 (1988). https:\/\/doi.org\/10.1145\/54852.378545","DOI":"10.1145\/54852.378545"},{"key":"2200_CR20","unstructured":"DigitalTrends: A complete history of the camera phone. https:\/\/www.digitaltrends.com\/mobile\/camera-phone-history\/ (2013). Accessed 16 Apr 2021"},{"key":"2200_CR21","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16x16 words: transformers for image recognition at scale (2020)"},{"key":"2200_CR22","doi-asserted-by":"publisher","unstructured":"Duan, K., Bai, S., Xie, L., Qi, H., Huang, Q., Tian, Q.: Centernet: Keypoint triplets for object detection. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 6568\u20136577 (2019). https:\/\/doi.org\/10.1109\/ICCV.2019.00667","DOI":"10.1109\/ICCV.2019.00667"},{"key":"2200_CR23","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network (2014)"},{"issue":"1","key":"2200_CR24","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1111\/j.1475-6781.2007.00103.x","volume":"16","author":"K Endo","year":"2007","unstructured":"Endo, K.: Personal, portable, pedestrian: mobile phones in Japanese life edited by Mizuko Ito, Daisuke Okabe and Misa Matsuda. Int. J. Jpn. Sociol. 16(1), 115\u2013116 (2007). https:\/\/doi.org\/10.1111\/j.1475-6781.2007.00103.x","journal-title":"Int. J. Jpn. Sociol."},{"issue":"2","key":"2200_CR25","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C.K.I., Winn, J., Zisserman, A.: The pascal visual object classes (VOC) challenge. Int. J. Comput. Vis. 88(2), 303\u2013338 (2010)","journal-title":"Int. J. Comput. Vis."},{"issue":"1","key":"2200_CR26","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/s11263-014-0733-5","volume":"111","author":"M Everingham","year":"2015","unstructured":"Everingham, M., Eslami, S.M.A., Van Gool, L., Williams, C.K.I., Winn, J., Zisserman, A.: The pascal visual object classes challenge: a retrospective. Int. J. Comput. Vis. 111(1), 98\u2013136 (2015)","journal-title":"Int. J. Comput. Vis."},{"key":"2200_CR27","doi-asserted-by":"crossref","unstructured":"Fairchild, M.: The HDR photographic survey. In: Color Imaging Conference (2007)","DOI":"10.2352\/CIC.2007.15.1.art00044"},{"key":"2200_CR28","doi-asserted-by":"publisher","unstructured":"Felzenszwalb, P., McAllester, D., Ramanan, D.: A discriminatively trained, multiscale, deformable part model. In: 2008 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20138 (2008). https:\/\/doi.org\/10.1109\/CVPR.2008.4587597","DOI":"10.1109\/CVPR.2008.4587597"},{"key":"2200_CR29","doi-asserted-by":"publisher","unstructured":"Geiger, A., Lenz, P., Stiller, C., Urtasun, R.: Vision meets robotics: the KITTI dataset. Int. J. Robot. Res. 32(11), 1231\u20131237 (2013). https:\/\/doi.org\/10.1177\/0278364913491297","DOI":"10.1177\/0278364913491297"},{"key":"2200_CR30","doi-asserted-by":"publisher","unstructured":"Georgescu, M.D.: Evolution of mobile processors. In: 2003 IEEE Pacific Rim Conference on Communications Computers and Signal Processing (PACRIM 2003) (Cat. No.03CH37490), vol. 2, pp. 638\u2013641 (2003). https:\/\/doi.org\/10.1109\/PACRIM.2003.1235862","DOI":"10.1109\/PACRIM.2003.1235862"},{"key":"2200_CR31","doi-asserted-by":"publisher","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp. 580\u2013587 (2014). https:\/\/doi.org\/10.1109\/CVPR.2014.81","DOI":"10.1109\/CVPR.2014.81"},{"key":"2200_CR32","doi-asserted-by":"publisher","unstructured":"Godard, C., Aodha, O.M., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, Honolulu, HI, USA, July 21\u201326, 2017, pp. 6602\u20136611. IEEE Computer Society (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.699","DOI":"10.1109\/CVPR.2017.699"},{"key":"2200_CR33","unstructured":"Gonzalez, R., Woods, R.: Digital Image Processing. Pearson (2018). https:\/\/books.google.co.jp\/books?id=0F05vgAACAAJ"},{"key":"2200_CR34","unstructured":"Google Inc.: Portrait Light: Enhancing Portrait Lighting with Machine Learning . https:\/\/ai.googleblog.com\/2020\/12\/portrait-light-enhancing-portrait.html (2020). Accessed 20 Apr 2021"},{"key":"2200_CR35","doi-asserted-by":"publisher","first-page":"220","DOI":"10.1016\/j.inffus.2019.09.003","volume":"55","author":"B Goyal","year":"2020","unstructured":"Goyal, B., Dogra, A., Agrawal, S., Sohi, B., Sharma, A.: Image denoising review: from classical to state-of-the-art approaches. Inf. Fusion 55, 220\u2013244 (2020)","journal-title":"Inf. Fusion"},{"key":"2200_CR36","unstructured":"Han, S., Mao, H., Dally, W.J.: Deep compression: compressing deep neural networks with pruning, trained quantization and Huffman coding. In: Proc. ICLR 2016 (2016). http:\/\/arxiv.org\/abs\/1510.00149"},{"key":"2200_CR37","unstructured":"Han, S., Pool, J., Tran, J., Dally, W.J.: Learning both weights and connections for efficient neural networks. In: Proceedings of the 28th International Conference on Neural Information Processing Systems, NIPS\u201915, vol. 1, pp. 1135\u20131143. MIT Press, Cambridge (2015)"},{"key":"2200_CR38","doi-asserted-by":"publisher","DOI":"10.1145\/2980179.2980254","author":"SW Hasinoff","year":"2016","unstructured":"Hasinoff, S.W., Sharlet, D., Geiss, R., Adams, A., Barron, J.T., Kainz, F., Chen, J., Levoy, M.: Burst photography for high dynamic range and low-light imaging on mobile cameras. ACM Trans. Graph. (2016). https:\/\/doi.org\/10.1145\/2980179.2980254","journal-title":"ACM Trans. Graph."},{"key":"2200_CR39","volume-title":"Portable Electronics Product Design and Development","author":"B Haskell","year":"2004","unstructured":"Haskell, B.: Portable Electronics Product Design and Development. Electronic engineering, McGraw-Hill Education, McGraw-Hill professional engineering (2004)"},{"key":"2200_CR40","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"2200_CR41","doi-asserted-by":"publisher","unstructured":"He, K., Gkioxari, G., Dollar, P., Girshick, R.: Mask r-cnn. In: 2017 IEEE International Conference on Computer Vision (ICCV), Los Alamitos, CA, USA, pp. 2980\u20132988. IEEE Computer Society (2017). https:\/\/doi.org\/10.1109\/ICCV.2017.322","DOI":"10.1109\/ICCV.2017.322"},{"key":"2200_CR42","doi-asserted-by":"publisher","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 2980\u20132988 (2017). https:\/\/doi.org\/10.1109\/ICCV.2017.322","DOI":"10.1109\/ICCV.2017.322"},{"key":"2200_CR43","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1109\/AVSS.2003.1217942","volume":"2003","author":"E Hemayed","year":"2003","unstructured":"Hemayed, E.: A survey of camera self-calibration. Proceedings of the IEEE Conference on Advanced Video and Signal Based Surveillance 2003, 351\u2013357 (2003). https:\/\/doi.org\/10.1109\/AVSS.2003.1217942","journal-title":"Proceedings of the IEEE Conference on Advanced Video and Signal Based Surveillance"},{"key":"2200_CR44","unstructured":"Higher Intellect Vintage Wiki: Connectix QuickCam. https:\/\/wiki.preterhuman.net\/Connectix_QuickCam (2020). Accessed 16 Apr 2021"},{"key":"2200_CR45","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. In: NIPS Deep Learning and Representation Learning Workshop (2015). arXiv:1503.02531"},{"issue":"3","key":"2200_CR46","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1006\/cviu.2001.0921","volume":"83","author":"E Hjelm\u00e5s","year":"2001","unstructured":"Hjelm\u00e5s, E., Low, B.K.: Face detection: a survey. Comput. Vis. Image Underst. 83(3), 236\u2013274 (2001)","journal-title":"Comput. Vis. Image Underst."},{"key":"2200_CR47","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., Adam, H.: Mobilenets: efficient convolutional neural networks for mobile vision applications. CoRR. arXiv:1704.04861 (2017)"},{"key":"2200_CR48","doi-asserted-by":"publisher","unstructured":"Howard, A., Pang, R., Adam, H., Le, Q.V., Sandler, M., Chen, B., Wang, W., Chen, L., Tan, M., Chu, G., Vasudevan, V., Zhu, Y.: Searching for mobilenetv3. In: 2019 IEEE\/CVF International Conference on Computer Vision, ICCV 2019, Seoul, Korea (South), October 27\u2013November 2, 2019, pp. 1314\u20131324. IEEE (2019). https:\/\/doi.org\/10.1109\/ICCV.2019.00140","DOI":"10.1109\/ICCV.2019.00140"},{"key":"2200_CR49","doi-asserted-by":"publisher","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00745","DOI":"10.1109\/CVPR.2018.00745"},{"key":"2200_CR50","unstructured":"Hubara, I., Courbariaux, M., Soudry, D., El-Yaniv, R., Bengio, Y.: Quantized neural networks: training neural networks with low precision weights and activations. J. Mach. Learn. Res. 18(187), 1\u201330 (2018). http:\/\/jmlr.org\/papers\/v18\/16-456.html"},{"key":"2200_CR51","unstructured":"Inc., G.: TensorFlow Lite\u2014ML for Mobile and Edge Devices. https:\/\/www.tensorflow.org\/lite\/ (2018). Accessed 19 Apr 2021"},{"key":"2200_CR52","unstructured":"Insider Magazine: How popular smartphones make your skin look \u2019whiter\u2019 in selfies. https:\/\/www.insider.com\/samsung-huawei-smartphone-beauty-filters-whiten-your-skin-2017-8 (2017). Accessed 21 Apr 2021"},{"key":"2200_CR53","unstructured":"Intel Inc.: Beginner\u2019s guide to depth. https:\/\/www.intelrealsense.com\/beginners-guide-to-depth (2019). Accessed 20 Apr 2021"},{"key":"2200_CR54","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift. In: Bach, F., Blei, D. (eds.) Proceedings of the 32nd International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a037, pp. 448\u2013456. PMLR, Lille (2015)"},{"key":"2200_CR55","unstructured":"Kartynnik, Y., Ablavatski, A., Grishchenko, I., Grundmann, M.: Real-time facial surface geometry from monocular video on mobile GPUs (2019)"},{"key":"2200_CR56","doi-asserted-by":"crossref","unstructured":"Khabarlak, K., Koriashkina, L.: Fast facial landmark detection and applications: a survey (2021)","DOI":"10.24215\/16666038.22.e02"},{"key":"2200_CR57","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1016\/j.jpdc.2018.11.012","volume":"127","author":"M Khairy","year":"2019","unstructured":"Khairy, M., Wassal, A.G., Zahran, M.: A survey of architectural approaches for improving GPGPU performance, programmability and heterogeneity. J. Parallel Distrib. Comput. 127, 65\u201388 (2019). https:\/\/doi.org\/10.1016\/j.jpdc.2018.11.012","journal-title":"J. Parallel Distrib. Comput."},{"key":"2200_CR58","unstructured":"Kim, Y., Park, E., Yoo, S., Choi, T., Yang, L., Shin, D.: Compression of deep convolutional neural networks for fast and low power mobile applications. In: Bengio, Y., LeCun, Y. (eds.) 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2\u20134, 2016, Conference Track Proceedings (2016). arxiv:1511.06530"},{"key":"2200_CR59","doi-asserted-by":"publisher","unstructured":"Kirillov, A., He, K., Girshick, R.B., Rother, C., Doll\u00e1r, P.: Panoptic segmentation. In: IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2019, Long Beach, CA, USA, June 16\u201320, 2019. pp. 9404\u20139413. Computer Vision Foundation\/IEEE (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00963","DOI":"10.1109\/CVPR.2019.00963"},{"key":"2200_CR60","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Pereira, F., Burges, C.J.C., Bottou, L., Weinberger, K.Q. (eds.) Advances in Neural Information Processing Systems. vol.\u00a025. Curran Associates, Inc. (2012). https:\/\/proceedings.neurips.cc\/paper\/2012\/file\/c399862d3b9d6b76c8436e924a68c45b-Paper.pdf"},{"key":"2200_CR61","doi-asserted-by":"publisher","unstructured":"Kumari, J., Rajesh, R., Pooja, K.: Facial expression recognition: a survey. Procedia Comput. Sci. 58, 486\u2013491 (2015). https:\/\/doi.org\/10.1016\/j.procs.2015.08.011. Second International Symposium on Computer Vision and the Internet (VisionNet\u201915)","DOI":"10.1016\/j.procs.2015.08.011"},{"key":"2200_CR62","doi-asserted-by":"crossref","unstructured":"Laina, I., Rupprecht, C., Belagiannis, V., Tombari, F., Navab, N.: Deeper depth prediction with fully convolutional residual networks (2016)","DOI":"10.1109\/3DV.2016.32"},{"issue":"3","key":"2200_CR63","doi-asserted-by":"publisher","first-page":"642","DOI":"10.1007\/s11263-019-01204-1","volume":"128","author":"H Law","year":"2020","unstructured":"Law, H., Deng, J.: Cornernet: detecting objects as paired keypoints. Int. J. Comput. Vis. 128(3), 642\u2013656 (2020). https:\/\/doi.org\/10.1007\/s11263-019-01204-1","journal-title":"Int. J. Comput. Vis."},{"key":"2200_CR64","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision\u2013ECCV 2014","author":"TY Lin","year":"2014","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) Computer Vision\u2013ECCV 2014, pp. 740\u2013755. Springer, Cham (2014)"},{"issue":"3941","key":"2200_CR65","doi-asserted-by":"publisher","first-page":"166","DOI":"10.1126\/science.169.3941.166","volume":"169","author":"L Lipkin","year":"1970","unstructured":"Lipkin, L.: Picture processing by computer. Azriel rosenfeld. Science 169(3941), 166\u2013167 (1970). https:\/\/doi.org\/10.1126\/science.169.3941.166","journal-title":"Azriel rosenfeld. Science"},{"key":"2200_CR66","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/978-3-319-46448-0_2","volume-title":"Computer Vision\u2013ECCV 2016","author":"W Liu","year":"2016","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.Y., Berg, A.C.: SSD: single shot multibox detector. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) Computer Vision\u2013ECCV 2016, pp. 21\u201337. Springer, Cham (2016)"},{"key":"2200_CR67","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2200_CR68","doi-asserted-by":"publisher","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE Computer Society, Los Alamitos, CA, USA, pp. 3431\u20133440 (2015). https:\/\/doi.org\/10.1109\/CVPR.2015.7298965","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"2200_CR69","doi-asserted-by":"crossref","unstructured":"Lowe, D.: Distinctive image features from scale-invariant keypoints. Int. J. Comput. Vis. 91\u2013110, (2004)","DOI":"10.1023\/B:VISI.0000029664.99615.94"},{"key":"2200_CR70","doi-asserted-by":"crossref","unstructured":"Lumsdaine, A., Georgiev, T.: The focused plenoptic camera. In: 2009 IEEE International Conference on Computational Photography (ICCP), pp. 1\u20138 (2009)","DOI":"10.1109\/ICCPHOT.2009.5559008"},{"key":"2200_CR71","doi-asserted-by":"publisher","unstructured":"Mahjourian, R., Wicke, M., Angelova, A.: Unsupervised learning of depth and ego-motion from monocular video using 3d geometric constraints. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5667\u20135675 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00594","DOI":"10.1109\/CVPR.2018.00594"},{"key":"2200_CR72","unstructured":"Microsoft Windows Blog: Make a more personal connection with Eye Contact, now generally available. https:\/\/blogs.windows.com\/devices\/2020\/08\/20\/make-a-more-personal-connection-with-eye-contact-now-generally-available\/ (2021). Accessed 21 Apr 2021"},{"key":"2200_CR73","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1016\/j.image.2015.11.008","volume":"41","author":"S Milani","year":"2016","unstructured":"Milani, S., Calvagno, G.: Correction and interpolation of depth maps from structured light infrared sensors. Signal Process. Image Commun. 41, 28\u201339 (2016)","journal-title":"Signal Process. Image Commun."},{"key":"2200_CR74","unstructured":"Morpho Inc.: Featuring Technology: SoftNeuro. https:\/\/www.morphoinc.com\/en\/featuringtechnology\/softneuro (2018). Accessed 19 Apr 2021"},{"key":"2200_CR75","unstructured":"Morpho Inc.: Featuring Technology: Semantic Filtering. https:\/\/www.morphoinc.com\/en\/featuringtechnology\/semanticfiltering (2020). Accessed 21 Apr 2021"},{"key":"2200_CR76","unstructured":"Morpho Inc.: Morpho: Technology. https:\/\/www.morphoinc.com\/en\/technology#tab-02 (2020). Accessed 18 May 2021"},{"key":"2200_CR77","doi-asserted-by":"publisher","unstructured":"Nair, R., Ruhl, K., Lenzen, F., Meister, S., Sch\u00e4fer, H., Garbe, C.S., Eisemann, M., Magnor, M., Kondermann, D.: A Survey on Time-of-Flight Stereo Fusion, pp. 105\u2013127. Springer, Berlin (2013). https:\/\/doi.org\/10.1007\/978-3-642-44964-2_6","DOI":"10.1007\/978-3-642-44964-2_6"},{"key":"2200_CR78","doi-asserted-by":"publisher","unstructured":"Narioka, K., Nishimura, H., Itamochi, T., Inomata, T.: Understanding 3d semantic structure around the vehicle with monocular cameras. In: 2018 IEEE Intelligent Vehicles Symposium (IV), pp. 132\u2013137 (2018). https:\/\/doi.org\/10.1109\/IVS.2018.8500397","DOI":"10.1109\/IVS.2018.8500397"},{"key":"2200_CR79","unstructured":"Neumann, J.V.: Introduction to \u201cThe First Draft Report on the EDVAC\u201d. http:\/\/qss.stanford.edu\/~godfrey\/vonNeumann\/vnedvac.pdf (1945). Accessed 16 Apr 2021"},{"key":"2200_CR80","unstructured":"Ng, R., Levoy, M., Br\u00e9dif, M., Duval, G., Horowitz, M., Hanrahan, P.: Light field photography with a hand-held plenoptic camera. In: Stanford Tech Report CTSR 2005-02 (2005)"},{"key":"2200_CR81","doi-asserted-by":"publisher","unstructured":"Park, J., Lee, C., Kim, C.S.: Deep learning approach to video frame rate up-conversion using bilateral motion estimation. In: 2019 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), pp. 1970\u20131975 (2019). https:\/\/doi.org\/10.1109\/APSIPAASC47483.2019.9023270","DOI":"10.1109\/APSIPAASC47483.2019.9023270"},{"key":"2200_CR82","first-page":"286","volume":"2","author":"M Poulose","year":"2013","unstructured":"Poulose, M.: Literature survey on image deblurring techniques. Int. J. Comput. Appl. Technol. Res. 2, 286\u2013288 (2013)","journal-title":"Int. J. Comput. Appl. Technol. Res."},{"key":"2200_CR83","unstructured":"Qualcomm Inc.: Qualcomm Neural Processing SDK. https:\/\/developer.qualcomm.com\/software\/qualcomm-neural-processing-sdk (2016). Accessed 19 Apr 2021"},{"key":"2200_CR84","unstructured":"Qualcomm Inc.: Mobile AI\u2014On-Device AI\u2014Qualcomm. https:\/\/www.qualcomm.com\/products\/smartphones\/mobile-ai (2020). Accessed 16-Apr 2021"},{"key":"2200_CR85","doi-asserted-by":"crossref","unstructured":"Rastegari, M., Ordonez, V., Redmon, J., Farhadi, A.: Xnor-net: imagenet classification using binary convolutional neural networks. In: European Conference on Computer Vision (2016)","DOI":"10.1007\/978-3-319-46493-0_32"},{"key":"2200_CR86","doi-asserted-by":"publisher","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 779\u2013788 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.91","DOI":"10.1109\/CVPR.2016.91"},{"key":"2200_CR87","unstructured":"Redmon, J., Farhadi, A.: Yolov3: an incremental improvement. Preprint at arXiv:1804.02767 (2018)"},{"issue":"8","key":"2200_CR88","first-page":"31","volume":"58","author":"SR Reeja","year":"2012","unstructured":"Reeja, S.R., Kavya, N.P.: Noise reduction in video sequences: the state of art and the technique for motion detection. Int. J. Comput. Appl. 58(8), 31\u201336 (2012)","journal-title":"Int. J. Comput. Appl."},{"issue":"6","key":"2200_CR89","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2017). https:\/\/doi.org\/10.1109\/TPAMI.2016.2577031","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2200_CR90","doi-asserted-by":"publisher","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.: Mobilenetv2: inverted residuals and linear bottlenecks. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00474","DOI":"10.1109\/CVPR.2018.00474"},{"issue":"9","key":"2200_CR91","doi-asserted-by":"publisher","first-page":"994","DOI":"10.1109\/34.713364","volume":"20","author":"Y Shinagawa","year":"1998","unstructured":"Shinagawa, Y., Kunii, T.: Unconstrained automatic image matching using multiresolutional critical-point filters. IEEE Trans. Pattern Anal. Mach. Intell. 20(9), 994\u20131010 (1998). https:\/\/doi.org\/10.1109\/34.713364","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2200_CR92","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1007\/978-3-642-33715-4_54","volume-title":"Computer Vision\u2013ECCV 2012","author":"N Silberman","year":"2012","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) Computer Vision\u2013ECCV 2012, pp. 746\u2013760. Springer, Berlin (2012)"},{"key":"2200_CR93","doi-asserted-by":"publisher","DOI":"10.5120\/15564-4339","author":"M Singh","year":"2014","unstructured":"Singh, M., Kumar, M.: Evolution of processor architecture in mobile phones. Int. J. Comput. Appl. (2014). https:\/\/doi.org\/10.5120\/15564-4339","journal-title":"Int. J. Comput. Appl."},{"key":"2200_CR94","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.: Dropout: a simple way to prevent neural networks from overfitting. J. Mach. Learn. Res. 15(56), 1929\u20131958 (2014). http:\/\/jmlr.org\/papers\/v15\/srivastava14a.html"},{"key":"2200_CR95","doi-asserted-by":"publisher","unstructured":"Szegedy, C., Wei Liu, Yangqing Jia, Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A.: Going deeper with convolutions. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1\u20139 (2015). https:\/\/doi.org\/10.1109\/CVPR.2015.7298594","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"2200_CR96","unstructured":"Tan, M., Le, Q.: EfficientNet: rethinking model scaling for convolutional neural networks. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 97, pp. 6105\u20136114. PMLR (2019)"},{"key":"2200_CR97","unstructured":"Tao, A., Sapra, K., Catanzaro, B.: Hierarchical multi-scale attention for semantic segmentation. arXiv:2005.10821 (2020)"},{"key":"2200_CR98","doi-asserted-by":"publisher","unstructured":"Tico, M., Vehvilainen, M.: Robust image fusion for image stabilization. In: 1988 International Conference on Acoustics, Speech, and Signal Processing, 1988. ICASSP-88, vol. 1, pp. I\u2013565 (2007). https:\/\/doi.org\/10.1109\/ICASSP.2007.365970","DOI":"10.1109\/ICASSP.2007.365970"},{"issue":"10","key":"2200_CR99","doi-asserted-by":"publisher","first-page":"1039","DOI":"10.1016\/j.imavis.2006.02.026","volume":"24","author":"J van Ouwerkerk","year":"2006","unstructured":"van Ouwerkerk, J.: Image super-resolution survey. Image Vis. Comput. 24(10), 1039\u20131052 (2006)","journal-title":"Image Vis. Comput."},{"key":"2200_CR100","doi-asserted-by":"publisher","unstructured":"Viola, P., Jones, M.: Rapid object detection using a boosted cascade of simple features. In: Proceedings of the 2001 IEEE Computer Society Conference on Computer Vision and Pattern Recognition. CVPR 2001, vol. 1, pp. I\u2013I (2001). https:\/\/doi.org\/10.1109\/CVPR.2001.990517","DOI":"10.1109\/CVPR.2001.990517"},{"key":"2200_CR101","unstructured":"Wan, L., Zeiler, M., Zhang, S., Cun, Y.L., Fergus, R.: Regularization of neural networks using dropconnect. In: Dasgupta, S., McAllester, D. (eds.) Proceedings of the 30th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 28, pp. 1058\u20131066. PMLR, Atlanta (2013)"},{"key":"2200_CR102","unstructured":"Wang, X., Zhang, R., Kong, T., Li, L., Shen, C.: Solov2: dynamic, faster and stronger. CoRR. arXiv:2003.10152 (2020)"},{"key":"2200_CR103","unstructured":"Wikipedia: Dulmont Magnum. https:\/\/en.wikipedia.org\/wiki\/Dulmont_Magnum (2015). Accessed 16 Apr 2021"},{"key":"2200_CR104","doi-asserted-by":"crossref","unstructured":"Xu, Y., Zhu, X., Shi, J., Zhang, G., Bao, H., Li, H.: Depth completion from sparse lidar data with depth-normal constraints. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00290"},{"key":"2200_CR105","doi-asserted-by":"publisher","unstructured":"Zhang,Y., Zhou, L., Liu, Z., Shang, Y.: A flexible online camera calibration using line segments. J. Sens. 2016, Article ID 2802343, 16 pages (2016). https:\/\/doi.org\/10.1155\/2016\/2802343","DOI":"10.1155\/2016\/2802343"},{"key":"2200_CR106","unstructured":"Zhang, H., Cisse, M., Dauphin, Y., Lopez-Paz, D.: mixup: Beyond empirical risk minimization. In: Proc. ICLR 2018 (2018)"},{"issue":"2","key":"2200_CR107","doi-asserted-by":"publisher","first-page":"225","DOI":"10.1109\/TCSVT.2015.2501941","volume":"27","author":"L Zhang","year":"2017","unstructured":"Zhang, L., Xu, Q.K., Huang, H.: A global approach to fast video stabilization. IEEE Trans. Circuits Syst. Video Technol. 27(2), 225\u2013235 (2017). https:\/\/doi.org\/10.1109\/TCSVT.2015.2501941","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"07","key":"2200_CR108","first-page":"13001","volume":"34","author":"Z Zhong","year":"2020","unstructured":"Zhong, Z., Zheng, L., Kang, G., Li, S., Yang, Y.: Random erasing data augmentation. Proc. AAAI Conf. Artif. Intell. 34(07), 13001\u201313008 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"2200_CR109","unstructured":"Zoph, B., Le, Q.V.: Neural architecture search with reinforcement learning. Preprint at arXiv:1611.01578 (2017)"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-021-02200-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-021-02200-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-021-02200-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,8]],"date-time":"2025-04-08T22:47:44Z","timestamp":1744152464000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-021-02200-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,21]]},"references-count":109,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2021,12]]}},"alternative-id":["2200"],"URL":"https:\/\/doi.org\/10.1007\/s00371-021-02200-8","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,6,21]]},"assertion":[{"value":"5 June 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 June 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors did not receive support from any organization for the submitted work. The authors have no conflicts of interest to declare that are relevant to the content of this article. Product names and organization names included in the publication are included solely for the purpose of surveying.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}