{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T16:17:48Z","timestamp":1780762668699,"version":"3.54.1"},"publisher-location":"Cham","reference-count":43,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030585198","type":"print"},{"value":"9783030585204","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-58520-4_36","type":"book-chapter","created":{"date-parts":[[2020,11,18]],"date-time":"2020-11-18T10:08:18Z","timestamp":1605694098000},"page":"609-625","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":29,"title":["Key Frame Proposal Network for Efficient Pose Estimation in Videos"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5012-5459","authenticated-orcid":false,"given":"Yuexi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6810-0962","authenticated-orcid":false,"given":"Yin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1945-9172","authenticated-orcid":false,"given":"Octavia","family":"Camps","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4439-3988","authenticated-orcid":false,"given":"Mario","family":"Sznaier","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,11,19]]},"reference":[{"key":"36_CR1","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., Schiele, B.: 2D human pose estimation: new benchmark and state of the art analysis. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2014","DOI":"10.1109\/CVPR.2014.471"},{"key":"36_CR2","doi-asserted-by":"crossref","unstructured":"Belagiannis, V., Zisserman, A.: Recurrent human pose estimation. In: 2017 12th IEEE International Conference on Automatic Face and Gesture Recognition (FG 2017), pp. 468\u2013475. IEEE (2017)","DOI":"10.1109\/FG.2017.64"},{"key":"36_CR3","unstructured":"Bertasius, G., Feichtenhofer, C., Tran, D., Shi, J., Torresani, L.: Learning temporal pose estimation from sparsely-labeled videos. In: Wallach, H., Larochelle, H., Beygelzimer, A., d\u00c1lch\u00e9-Buc, F., Fox, E., Garnett, R. (eds.) Advances in Neural Information Processing Systems 32, pp. 3027\u20133038. Curran Associates, Inc. (2019). http:\/\/papers.nips.cc\/paper\/8567-learning-temporal-pose-estimation-from-sparsely-labeled-videos.pdf"},{"key":"36_CR4","doi-asserted-by":"crossref","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.E., Sheikh, Y.: OpenPose: realtime multi-person 2D pose estimation using part affinity fields. arXiv preprint arXiv:1812.08008 (2018)","DOI":"10.1109\/CVPR.2017.143"},{"key":"36_CR5","doi-asserted-by":"crossref","unstructured":"Charles, J., Pfister, T., Magee, D., Hogg, D., Zisserman, A.: Personalizing human video pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3063\u20133072 (2016)","DOI":"10.1109\/CVPR.2016.334"},{"key":"36_CR6","unstructured":"Chen, X., Yuille, A.L.: Articulated pose estimation by a graphical model with image dependent pairwise relations. In: Advances in Neural Information Processing Systems, pp. 1736\u20131744 (2014)"},{"key":"36_CR7","doi-asserted-by":"crossref","unstructured":"Chu, X., Ouyang, W., Li, H., Wang, X.: Structured feature learning for pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4715\u20134723 (2016)","DOI":"10.1109\/CVPR.2016.510"},{"key":"36_CR8","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1016\/j.neucom.2011.12.038","volume":"100","author":"M Cristani","year":"2013","unstructured":"Cristani, M., Raghavendra, R., Del Bue, A., Murino, V.: Human behavior analysis in video surveillance: a social signal processing perspective. Neurocomputing 100, 86\u201397 (2013)","journal-title":"Neurocomputing"},{"key":"36_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"728","DOI":"10.1007\/978-3-319-46493-0_44","volume-title":"Computer Vision","author":"G Gkioxari","year":"2016","unstructured":"Gkioxari, G., Toshev, A., Jaitly, N.: Chained predictions using convolutional neural networks. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 728\u2013743. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_44"},{"key":"36_CR10","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. CoRR abs\/1512.03385 http:\/\/arxiv.org\/abs\/1512.03385 (2015)"},{"key":"36_CR11","doi-asserted-by":"crossref","unstructured":"Ilg, E., Mayer, N., Saikia, T., Keuper, M., Dosovitskiy, A., Brox, T.: FlowNet 2.0: evolution of optical flow estimation with deep networks. CoRR abs\/1612.01925 http:\/\/arxiv.org\/abs\/1612.01925 (2016)","DOI":"10.1109\/CVPR.2017.179"},{"key":"36_CR12","doi-asserted-by":"crossref","unstructured":"Iqbal, U., Garbade, M., Gall, J.: Pose for action-action for pose. In: 2017 12th IEEE International Conference on Automatic Face and Gesture Recognition (FG 2017), pp. 438\u2013445. IEEE (2017)","DOI":"10.1109\/FG.2017.61"},{"key":"36_CR13","unstructured":"Iqbal, U., Milan, A., Gall, J.: Pose-track: joint multi-person pose estimation and tracking. CoRR abs\/1611.07727 http:\/\/arxiv.org\/abs\/1611.07727 (2016)"},{"key":"36_CR14","doi-asserted-by":"crossref","unstructured":"Jhuang, H., Gall, J., Zuffi, S., Schmid, C., Black, M.J.: Towards understanding action recognition. In: International Conference on Computer Vision (ICCV), pp. 3192\u20133199, December 2013","DOI":"10.1109\/ICCV.2013.396"},{"key":"36_CR15","doi-asserted-by":"crossref","unstructured":"Kreiss, S., Bertoni, L., Alahi, A.: PifPaf: composite fields for human pose estimation. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2019","DOI":"10.1109\/CVPR.2019.01225"},{"key":"36_CR16","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1007\/978-3-642-17688-3_31","volume-title":"Advanced Concepts for Intelligent Vision Systems","author":"H-Y Lin","year":"2010","unstructured":"Lin, H.-Y., Chen, T.-W.: Augmented reality with human body interaction based on monocular 3D pose estimation. In: Blanc-Talon, J., Bone, D., Philips, W., Popescu, D., Scheunders, P. (eds.) ACIVS 2010. LNCS, vol. 6474, pp. 321\u2013331. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-17688-3_31"},{"key":"36_CR17","doi-asserted-by":"crossref","unstructured":"Lin, M., Lin, L., Liang, X., Wang, K., Cheng, H.: Recurrent 3D pose sequence machines. CoRR abs\/1707.09695 http:\/\/arxiv.org\/abs\/1707.09695 (2017)","DOI":"10.1109\/CVPR.2017.588"},{"key":"36_CR18","unstructured":"Lin, T., Doll\u00e1r, P., Girshick, R.B., He, K., Hariharan, B., Belongie, S.J.: Feature pyramid networks for object detection. CoRR abs\/1612.03144 http:\/\/arxiv.org\/abs\/1612.03144 (2016)"},{"key":"36_CR19","unstructured":"Liu, W., Sharma, A., Camps, O.I., Sznaier, M.: DYAN: a dynamical atoms network for video prediction. CoRR abs\/1803.07201 http:\/\/arxiv.org\/abs\/1803.07201 (2018)"},{"key":"36_CR20","doi-asserted-by":"crossref","unstructured":"Luo, Y., et al.: LSTM pose machines. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5207\u20135215 (2018)","DOI":"10.1109\/CVPR.2018.00546"},{"key":"36_CR21","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. CoRR abs\/1603.06937 http:\/\/arxiv.org\/abs\/1603.06937 (2016)"},{"key":"36_CR22","doi-asserted-by":"crossref","unstructured":"Nie, X., Feng, J., Yan, S.: Mutual learning to adapt for joint human parsing and pose estimation. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01228-1_31"},{"key":"36_CR23","doi-asserted-by":"crossref","unstructured":"Nie, X., Li, Y., Luo, L., Zhang, N., Feng, J.: Dynamic kernel distillation for efficient pose estimation in videos. In: The IEEE International Conference on Computer Vision (ICCV), October 2019","DOI":"10.1109\/ICCV.2019.00704"},{"key":"36_CR24","unstructured":"Papandreou, G., et al.: Towards accurate multi-person pose estimation in the wild. CoRR abs\/1701.01779 http:\/\/arxiv.org\/abs\/1701.01779 (2017)"},{"key":"36_CR25","doi-asserted-by":"crossref","unstructured":"Park, D., Ramanan, D.: N-best maximal decoders for part models. In: 2011 International Conference on Computer Vision, pp. 2627\u20132634. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126552"},{"issue":"1","key":"36_CR26","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1016\/j.cviu.2007.10.005","volume":"111","author":"S Park","year":"2008","unstructured":"Park, S., Trivedi, M.M.: Understanding human interactions with track and body synergies (TBS) captured from multiple views. Comput. Vis. Image Underst. 111(1), 2\u201320 (2008)","journal-title":"Comput. Vis. Image Underst."},{"key":"36_CR27","doi-asserted-by":"crossref","unstructured":"Pfister, T., Charles, J., Zisserman, A.: Flowing convnets for human pose estimation in videos. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1913\u20131921 (2015)","DOI":"10.1109\/ICCV.2015.222"},{"key":"36_CR28","doi-asserted-by":"crossref","unstructured":"Pishchulin, L., Andriluka, M., Gehler, P., Schiele, B.: Strong appearance and expressive spatial models for human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3487\u20133494 (2013)","DOI":"10.1109\/ICCV.2013.433"},{"key":"36_CR29","doi-asserted-by":"crossref","unstructured":"Shotton, J., et al.: Real-time human pose recognition in parts from single depth images. In: CVPR 2011, pp. 1297\u20131304. IEEE (2011)","DOI":"10.1109\/CVPR.2011.5995316"},{"key":"36_CR30","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. In: Advances in Neural Information Processing Systems, pp. 568\u2013576 (2014)"},{"key":"36_CR31","doi-asserted-by":"crossref","unstructured":"Song, J., Wang, L., Van Gool, L., Hilliges, O.: Thin-slicing network: a deep structured model for pose estimation in videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4220\u20134229 (2017)","DOI":"10.1109\/CVPR.2017.590"},{"key":"36_CR32","doi-asserted-by":"crossref","unstructured":"Tang, W., Yu, P., Wu, Y.: Deeply learned compositional models for human pose estimation. In: The European Conference on Computer Vision (ECCV), September 2018","DOI":"10.1007\/978-3-030-01219-9_12"},{"key":"36_CR33","doi-asserted-by":"crossref","unstructured":"Tempo, R., Bai, E.W., Dabbene, F.: Probabilistic robustness analysis: explicit bounds for the minimum number of samples. In: Proceedings of 35th IEEE Conference on Decision and Control, vol. 3, pp. 3424\u20133428, December 1996","DOI":"10.1109\/CDC.1996.573690"},{"key":"36_CR34","doi-asserted-by":"publisher","unstructured":"Toshev, A., Szegedy, C.: DeepPose: human pose estimation via deep neural networks. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1653\u20131660, June 2014. https:\/\/doi.org\/10.1109\/CVPR.2014.214","DOI":"10.1109\/CVPR.2014.214"},{"key":"36_CR35","doi-asserted-by":"crossref","unstructured":"Wei, S.E., Ramakrishna, V., Kanade, T., Sheikh, Y.: Convolutional pose machines. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4724\u20134732 (2016)","DOI":"10.1109\/CVPR.2016.511"},{"key":"36_CR36","unstructured":"Xiao, B., Wu, H., Wei, Y.: Simple baselines for human pose estimation and tracking. CoRR abs\/1804.06208 http:\/\/arxiv.org\/abs\/1804.06208 (2018)"},{"key":"36_CR37","doi-asserted-by":"crossref","unstructured":"Xiaohan Nie, B., Xiong, C., Zhu, S.C.: Joint action recognition and pose estimation from video. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1293\u20131301 (2015)","DOI":"10.1109\/CVPR.2015.7298734"},{"key":"36_CR38","doi-asserted-by":"crossref","unstructured":"Yang, J., et al.: Quantization networks. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2019","DOI":"10.1109\/CVPR.2019.00748"},{"key":"36_CR39","doi-asserted-by":"crossref","unstructured":"Yang, W., Li, S., Ouyang, W., Li, H., Wang, X.: Learning feature pyramids for human pose estimation. arXiv preprint arXiv:1708.01101 (2017)","DOI":"10.1109\/ICCV.2017.144"},{"key":"36_CR40","doi-asserted-by":"crossref","unstructured":"Yang, W., Ouyang, W., Li, H., Wang, X.: End-to-end learning of deformable mixture of parts and deep convolutional neural networks for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3073\u20133082 (2016)","DOI":"10.1109\/CVPR.2016.335"},{"key":"36_CR41","doi-asserted-by":"publisher","unstructured":"Yang, Y., Ramanan, D.: Articulated pose estimation with flexible mixtures-of-parts. In: CVPR 2011, pp. 1385\u20131392 (2011). https:\/\/doi.org\/10.1109\/CVPR.2011.5995741","DOI":"10.1109\/CVPR.2011.5995741"},{"key":"36_CR42","doi-asserted-by":"crossref","unstructured":"Zhang, F., Zhu, X., Ye, M.: Fast human pose estimation. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2019","DOI":"10.1109\/CVPR.2019.00363"},{"key":"36_CR43","doi-asserted-by":"publisher","unstructured":"Zhang, W., Zhu, M., Derpanis, K.G.: From actemes to action: a strongly-supervised representation for detailed action understanding. In: 2013 IEEE International Conference on Computer Vision, pp. 2248\u20132255, December 2013. https:\/\/doi.org\/10.1109\/ICCV.2013.280","DOI":"10.1109\/ICCV.2013.280"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2020"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-58520-4_36","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,18]],"date-time":"2024-11-18T00:25:55Z","timestamp":1731889555000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-58520-4_36"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030585198","9783030585204"],"references-count":43,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-58520-4_36","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"19 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Glasgow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2020.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OpenReview","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5025","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1360","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"27% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic. From the ECCV Workshops 249 full papers, 18 short papers, and 21 further contributions were published out of a total of 467 submissions.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}