{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T11:20:47Z","timestamp":1762341647612},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2021,5,24]],"date-time":"2021-05-24T00:00:00Z","timestamp":1621814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,5,24]],"date-time":"2021-05-24T00:00:00Z","timestamp":1621814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2022,7]]},"DOI":"10.1007\/s00371-021-02120-7","type":"journal-article","created":{"date-parts":[[2021,5,24]],"date-time":"2021-05-24T20:02:41Z","timestamp":1621886561000},"page":"2417-2430","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":16,"title":["Two-stage multi-view deep network for 3D human pose reconstruction using images and its 2D joint heatmaps through enhanced stack-hourglass approach"],"prefix":"10.1007","volume":"38","author":[{"given":"Pratishtha","family":"Verma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rajeev","family":"Srivastava","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,5,24]]},"reference":[{"issue":"6","key":"2120_CR1","doi-asserted-by":"publisher","first-page":"671","DOI":"10.1007\/s00530-020-00677-2","volume":"26","author":"P Verma","year":"2020","unstructured":"Verma, P., Sah, A., Srivastava, R.: Deep learning-based multi-modal approach using RGB and skeleton sequences for human activity recognition. Multimed. Syst. 26(6), 671\u2013685 (2020)","journal-title":"Multimed. Syst."},{"issue":"5","key":"2120_CR2","doi-asserted-by":"publisher","first-page":"585","DOI":"10.1007\/s00530-020-00667-4","volume":"26","author":"SK Tripathy","year":"2020","unstructured":"Tripathy, S.K., Srivastava, R.: A real-time two-input stream multi-column multi-stage convolution neural network (TIS-MCMS-CNN) for efficient crowd congestion-level analysis. Multimed. Syst. 26(5), 585\u2013605 (2020)","journal-title":"Multimed. Syst."},{"key":"2120_CR3","doi-asserted-by":"crossref","unstructured":"Bo, L., Sminchisescu, C., Kanaujia, A., Metaxas, D.: Fast algorithms for large scale conditional 3D prediction. In: 2008 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20138. IEEE (2008)","DOI":"10.1109\/CVPR.2008.4587578"},{"issue":"7","key":"2120_CR4","doi-asserted-by":"publisher","first-page":"1052","DOI":"10.1109\/TPAMI.2006.149","volume":"28","author":"G Mori","year":"2006","unstructured":"Mori, G., Malik, J.: Recovering 3d human body configurations using shape contexts. IEEE Trans. Pattern Anal. Mach. Intell. 28(7), 1052\u20131062 (2006)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2120_CR5","doi-asserted-by":"crossref","unstructured":"Shakhnarovich, G., Viola, P., Darrell, T.: Fast pose estimation with parameter-sensitive hashing. In: Null, p. 750. IEEE (2003)","DOI":"10.1109\/ICCV.2003.1238424"},{"key":"2120_CR6","unstructured":"Agarwal, A., Triggs, B.: 3D human pose from silhouettes by relevance vector regression. In: Proceedings of the 2004 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, 2004. CVPR, vol. 2, pp. II\u2013II. IEEE (2004)"},{"key":"2120_CR7","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Zhou, X., Derpanis, K.G., Daniilidis, K.: Coarse-to-fine volumetric prediction for single-image 3D human pose. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7025\u20137034 (2017)","DOI":"10.1109\/CVPR.2017.139"},{"key":"2120_CR8","doi-asserted-by":"crossref","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.J.: A simple yet effective baseline for 3d human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2640\u20132649 (2017)","DOI":"10.1109\/ICCV.2017.288"},{"issue":"12","key":"2120_CR9","doi-asserted-by":"publisher","first-page":"1326","DOI":"10.1007\/s11263-018-1066-6","volume":"126","author":"I Katircioglu","year":"2018","unstructured":"Katircioglu, I., Tekin, B., Salzmann, M., Lepetit, V., Fua, P.: Learning latent representations of 3d human pose with deep neural networks. Int. J. Comput. Vis. 126(12), 1326\u20131341 (2018)","journal-title":"Int. J. Comput. Vis."},{"key":"2120_CR10","doi-asserted-by":"crossref","unstructured":"Popa, A.-I., Zanfir, M., Sminchisescu, C.: Deep multitask architecture for integrated 2d and 3d human sensing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6289\u20136298 (2017)","DOI":"10.1109\/CVPR.2017.501"},{"key":"2120_CR11","doi-asserted-by":"crossref","unstructured":"Tekin, B., Katircioglu, I., Salzmann, M., Lepetit, V., Fua, P.: Structured prediction of 3d human pose with deep neural networks. arXiv preprint arXiv:1605.05180 (2016)","DOI":"10.5244\/C.30.130"},{"key":"2120_CR12","doi-asserted-by":"crossref","unstructured":"Nibali, A., He, Z., Morgan, S., Prendergast, L.: 3d human pose estimation with 2d marginal heatmaps. In: 2019 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1477\u20131485. IEEE (2019)","DOI":"10.1109\/WACV.2019.00162"},{"key":"2120_CR13","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1016\/j.neucom.2018.10.009","volume":"323","author":"JC N\u00fa\u00f1ez","year":"2019","unstructured":"N\u00fa\u00f1ez, J.C., Cabido, R., V\u00e9lez, J.F., Montemayor, A.S., Pantrigo, J.J.: Multiview 3D human pose estimation using improved least-squares and LSTM networks. Neurocomputing 323, 335\u2013343 (2019)","journal-title":"Neurocomputing"},{"key":"2120_CR14","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. In: European Conference on Computer Vision, pp. 483\u2013499. Springer, Cham (2016)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"2120_CR15","doi-asserted-by":"crossref","unstructured":"Wei, S.-E., Ramakrishna, V., Kanade, T., Sheikh, Y.: Convolutional pose machines. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4724\u20134732 (2016)","DOI":"10.1109\/CVPR.2016.511"},{"key":"2120_CR16","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A.: Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"2120_CR17","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"key":"2120_CR18","doi-asserted-by":"crossref","unstructured":"Yang, W., Ouyang, W., Wang, X., Ren, J., Li, H., Wang, X.: 3d human pose estimation in the wild by adversarial learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5255\u20135264 (2018)","DOI":"10.1109\/CVPR.2018.00551"},{"key":"2120_CR19","doi-asserted-by":"crossref","unstructured":"Tekin, B., M\u00e1rquez-Neila, P., Salzmann, M., Fua, P.: Learning to fuse 2d and 3d image cues for monocular body pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3941\u20133950 (2017)","DOI":"10.1109\/ICCV.2017.425"},{"key":"2120_CR20","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1016\/j.cag.2019.09.002","volume":"85","author":"DC Luvizon","year":"2019","unstructured":"Luvizon, D.C., Tabia, H., Picard, D.: Human pose regression by combining indirect part detection and contextual information. Comput. Graph. 85, 15\u201322 (2019)","journal-title":"Comput. Graph."},{"issue":"12","key":"2120_CR21","doi-asserted-by":"publisher","first-page":"5659","DOI":"10.1109\/TIP.2015.2487860","volume":"24","author":"C Hong","year":"2015","unstructured":"Hong, C., Jun, Y., Wan, J., Tao, D., Wang, M.: Multimodal deep autoencoder for human pose recovery. IEEE Trans. Image Process. 24(12), 5659\u20135670 (2015)","journal-title":"IEEE Trans. Image Process."},{"issue":"6","key":"2120_CR22","first-page":"3742","volume":"62","author":"C Hong","year":"2014","unstructured":"Hong, C., Jun, Y., Tao, D., Wang, M.: Image-based three-dimensional human pose recovery by multiview locality-sensitive sparse retrieval. IEEE Trans. Ind. Electron. 62(6), 3742\u20133751 (2014)","journal-title":"IEEE Trans. Ind. Electron."},{"key":"2120_CR23","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1016\/j.sigpro.2015.10.004","volume":"124","author":"C Hong","year":"2016","unstructured":"Hong, C., Chen, X., Wang, X., Tang, C.: Hypergraph regularized autoencoder for image-based 3D human pose recovery. Sig. Process. 124, 132\u2013140 (2016)","journal-title":"Sig. Process."},{"key":"2120_CR24","first-page":"3","volume":"2","author":"M Trumble","year":"2017","unstructured":"Trumble, M., Gilbert, A., Malleson, C., Hilton, A., Collomosse, J.: Total capture: 3D human pose estimation fusing video and inertial sensors. BMVC 2, 3 (2017)","journal-title":"BMVC"},{"key":"2120_CR25","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Zhou, X., Derpanis, K.G., Daniilidis, K.: Harvesting multiple views for marker-less 3d human pose annotations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6988\u20136997 (2017)","DOI":"10.1109\/CVPR.2017.138"},{"issue":"12","key":"2120_CR26","doi-asserted-by":"publisher","first-page":"15573","DOI":"10.1007\/s11042-017-5133-8","volume":"77","author":"S Ershadi-Nasab","year":"2018","unstructured":"Ershadi-Nasab, S., Noury, E., Kasaei, S., Sanaei, E.: Multiple human 3d pose estimation from multiview images. Multimed. Tools Appl. 77(12), 15573\u201315601 (2018)","journal-title":"Multimed. Tools Appl."},{"key":"2120_CR27","doi-asserted-by":"crossref","unstructured":"Insafutdinov, E., Pishchulin, L., Andres, B., Andriluka, M., Schiele, B.: Deepercut: a deeper, stronger, and faster multi-person pose estimation model. In: European Conference on Computer Vision, pp. 34\u201350. Springer, Cham (2016)","DOI":"10.1007\/978-3-319-46466-4_3"},{"key":"2120_CR28","doi-asserted-by":"publisher","first-page":"102866","DOI":"10.1016\/j.jvcir.2020.102866","volume":"71","author":"P Verma","year":"2020","unstructured":"Verma, P., Srivastava, R.: Three stage deep network for 3D human pose reconstruction by exploiting spatial and temporal data via its 2D pose. J. Vis. Commun. Image Represent. 71, 102866 (2020)","journal-title":"J. Vis. Commun. Image Represent."},{"key":"2120_CR29","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Ioffe, S., Vanhoucke, V., Alemi, A.A.: Inception-v4, inception-resnet and the impact of residual connections on learning. In: Thirty-First AAAI Conference on Artificial Intelligence (2017)","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"2120_CR30","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift. arXiv preprint arXiv:1502.03167 (2015)"},{"issue":"1","key":"2120_CR31","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.: Dropout: a simple way to prevent neural networks from overfitting. J. Mach. Learn. Res. 15(1), 1929\u20131958 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"2120_CR32","unstructured":"Nair, V., Hinton, G.E.: Rectified linear units improve restricted Boltzmann machines. In: Proceedings of the 27th International Conference on Machine Learning (ICML-10), pp. 807\u2013814 (2010)"},{"key":"2120_CR33","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2120_CR34","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., Schiele, B.: 2d human pose estimation: new benchmark and state of the art analysis. In: Proceedings of the IEEE Conference on computer Vision and Pattern Recognition, pp. 3686\u20133693 (2014)","DOI":"10.1109\/CVPR.2014.471"},{"key":"2120_CR35","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Toshev, A., Jaitly, N.: Chained predictions using convolutional neural networks. In: European Conference on Computer Vision, pp. 728\u2013743. Springer, Cham (2016)","DOI":"10.1007\/978-3-319-46493-0_44"},{"key":"2120_CR36","first-page":"2","volume":"1","author":"U Rafi","year":"2016","unstructured":"Rafi, U., Leibe, B., Gall, J., Kostrikov, I.: An efficient convolutional network for human pose estimation. BMVC 1, 2 (2016)","journal-title":"BMVC"},{"key":"2120_CR37","doi-asserted-by":"crossref","unstructured":"Belagiannis, V., Zisserman, A.: Recurrent human pose estimation. In: 2017 12th IEEE International Conference on Automatic Face & Gesture Recognition (FG 2017), pp. 468\u2013475. IEEE (2017)","DOI":"10.1109\/FG.2017.64"},{"key":"2120_CR38","doi-asserted-by":"crossref","unstructured":"Chen, Y., Shen, C., Chen, H., Wei, X.-S., Liu, L., Yang, J.: Adversarial learning of structure-aware fully convolutional networks for landmark localization. In: IEEE Transactions on Pattern Analysis and Machine Intelligence (2019)","DOI":"10.1109\/TPAMI.2019.2901875"},{"key":"2120_CR39","doi-asserted-by":"crossref","unstructured":"Chou, C.-J., Chien, J.-T., Chen, H.-T.: Self adversarial training for human pose estimation. In: 2018 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), pp. 17\u201330. IEEE (2018)","DOI":"10.23919\/APSIPA.2018.8659538"},{"key":"2120_CR40","doi-asserted-by":"crossref","unstructured":"Bulat, A., Tzimiropoulos, G.: Human pose estimation via convolutional part heatmap regression. In: European Conference on Computer Vision, pp. 717\u2013732. Springer, Cham (2016)","DOI":"10.1007\/978-3-319-46478-7_44"},{"key":"2120_CR41","doi-asserted-by":"crossref","unstructured":"Chu, X., Yang, W., Ouyang, W., Ma, C., Yuille, A.L., Wang, X.: Multi-context attention for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1831\u20131840 (2017)","DOI":"10.1109\/CVPR.2017.601"},{"key":"2120_CR42","doi-asserted-by":"crossref","unstructured":"Ionescu, C., Papava, D., Olaru, V., Sminchisescu, C.: Human3.6m: Large scale datasets and predictive methods for 3d human sensing in natural environments. IEEE Trans. Pattern Anal. Mach. Intell. 36(7), 1325\u20131339 (2013)","DOI":"10.1109\/TPAMI.2013.248"},{"key":"2120_CR43","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"2120_CR44","doi-asserted-by":"publisher","first-page":"404","DOI":"10.1016\/j.patrec.2019.05.020","volume":"125","author":"X Zhang","year":"2019","unstructured":"Zhang, X., Tang, Z., Hou, J., Hao, Y.: 3D human pose estimation via human structure-aware fully connected network. Pattern Recogn. Lett. 125, 404\u2013410 (2019)","journal-title":"Pattern Recogn. Lett."},{"key":"2120_CR45","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2892452","volume-title":"3D human pose machines with self-supervised learning","author":"K Wang","year":"2019","unstructured":"Wang, K., Lin, L., Jiang, C., Qian, C., Wei, P.: 3D human pose machines with self-supervised learning. IEEE Trans. Pattern Anal. Mach, Intell (2019)"},{"key":"2120_CR46","doi-asserted-by":"crossref","unstructured":"Chen, X., Lin, K.-Y., Liu, W., Qian, C., Lin, L.: Weakly-supervised discovery of geometry-aware representation for 3d human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 10895\u201310904 (2019)","DOI":"10.1109\/CVPR.2019.01115"},{"key":"2120_CR47","doi-asserted-by":"crossref","unstructured":"Habibie, I., Xu, W., Mehta, D., Pons-Moll, G., Theobalt, C.: In the wild human pose estimation using explicit 2D features and intermediate 3D representations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 10905\u201310914 (2019)","DOI":"10.1109\/CVPR.2019.01116"},{"key":"2120_CR48","doi-asserted-by":"crossref","unstructured":"Pavllo, D., Feichtenhofer, C., Grangier, D., Auli, M.: 3D human pose estimation in video with temporal convolutions and semi-supervised training. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7753\u20137762 (2019)","DOI":"10.1109\/CVPR.2019.00794"},{"key":"2120_CR49","doi-asserted-by":"crossref","unstructured":"Lee, K., Lee, I., Lee, S.: Propagating lstm: 3d pose estimation based on joint interdependency. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 119\u2013135 (2018)","DOI":"10.1007\/978-3-030-01234-2_8"},{"key":"2120_CR50","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Zhou, X., Daniilidis, K.: Ordinal depth supervision for 3d human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7307\u20137316 (2018)","DOI":"10.1109\/CVPR.2018.00763"},{"key":"2120_CR51","doi-asserted-by":"crossref","unstructured":"Hossain, M.R.I., Little, J.J.: Exploiting temporal information for 3d human pose estimation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 68\u201384 (2018)","DOI":"10.1007\/978-3-030-01249-6_5"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-021-02120-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-021-02120-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-021-02120-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,28]],"date-time":"2022-12-28T12:04:14Z","timestamp":1672229054000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-021-02120-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,24]]},"references-count":51,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2022,7]]}},"alternative-id":["2120"],"URL":"https:\/\/doi.org\/10.1007\/s00371-021-02120-7","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,5,24]]},"assertion":[{"value":"22 March 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 May 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}