{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,3]],"date-time":"2025-07-03T04:03:15Z","timestamp":1751515395095,"version":"3.41.0"},"reference-count":53,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"3","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2018]]},"DOI":"10.1587\/transinf.2017edp7204","type":"journal-article","created":{"date-parts":[[2018,3,1]],"date-time":"2018-03-01T22:26:12Z","timestamp":1519943172000},"page":"758-766","source":"Crossref","is-referenced-by-count":4,"title":["Pose Estimation with Action Classification Using Global-and-Pose Features and Fine-Grained Action-Specific Pose Models"],"prefix":"10.1587","volume":"E101.D","author":[{"given":"Norimichi","family":"UKITA","sequence":"first","affiliation":[{"name":"Toyota Technological Institute"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"doi-asserted-by":"crossref","unstructured":"[1] C. Thurau and V. Hlav\u00e1c, \u201cPose primitive based human action recognition in videos or still images,\u201d In CVPR, pp.1-8, 2008. 10.1109\/cvpr.2008.4587721","key":"1","DOI":"10.1109\/CVPR.2008.4587721"},{"doi-asserted-by":"crossref","unstructured":"[2] N. Ikizler-Cinbis, R.G. Cinbis, and S. Sclaroff, \u201cLearning actions from the web,\u201d In ICCV, pp.995-1002, 2009. 10.1109\/iccv.2009.5459368","key":"2","DOI":"10.1109\/ICCV.2009.5459368"},{"doi-asserted-by":"crossref","unstructured":"[3] W. Yang, Y. Wang, and G. Mori, \u201cRecognizing human actions from still images with latent poses,\u201d In CVPR, pp.2030-2037, 2010. 10.1109\/cvpr.2010.5539879","key":"3","DOI":"10.1109\/CVPR.2010.5539879"},{"doi-asserted-by":"crossref","unstructured":"[4] S. Maji, L.D. Bourdev, and J. Malik, \u201cAction recognition from a distributed representation of pose and appearance,\u201d In CVPR, pp.3177-3184, 2011. 10.1109\/cvpr.2011.5995631","key":"4","DOI":"10.1109\/CVPR.2011.5995631"},{"doi-asserted-by":"publisher","unstructured":"[5] F.S. Khan, R.M. Anwer, J. van de Weijer, A.D. Bagdanov, A.M. L\u00f3pez, and M. Felsberg, \u201cColoring action recognition in still images,\u201d International Journal of Computer Vision, vol.105, no.3, pp.205-221, 2013. 10.1007\/s11263-013-0633-0","key":"5","DOI":"10.1007\/s11263-013-0633-0"},{"doi-asserted-by":"crossref","unstructured":"[6] G. Gkioxari, R.B. Girshick, and J. Malik, \u201cContextual action recognition with r<sup>*<\/sup>cnn,\u201d In ICCV, pp.1080-1088, 2015. 10.1109\/iccv.2015.129","key":"6","DOI":"10.1109\/ICCV.2015.129"},{"doi-asserted-by":"publisher","unstructured":"[7] Y. Zhang, L. Cheng, J. Wu, J. Cai, M.N. Do, and J. Lu, \u201cAction recognition in still images with minimum annotation efforts,\u201d IEEE Trans. Image Processing, vol.25, no.11, pp.5479-5490, 2016. 10.1109\/tip.2016.2605305","key":"7","DOI":"10.1109\/TIP.2016.2605305"},{"doi-asserted-by":"crossref","unstructured":"[8] N. Ikizler-Cinbis and S. Sclaroff, \u201cObject, scene and actions: Combining multiple features for human action recognition,\u201d In ECCV (1), vol.6311, pp.494-507, 2010. 10.1007\/978-3-642-15549-9_36","key":"8","DOI":"10.1007\/978-3-642-15549-9_36"},{"doi-asserted-by":"crossref","unstructured":"[9] T. Lan, Y. Wang, and G. Mori, \u201cDiscriminative figure-centric models for joint action localization and recognition,\u201d In ICCV, pp.2003-2010, 2011. 10.1109\/iccv.2011.6126472","key":"9","DOI":"10.1109\/ICCV.2011.6126472"},{"doi-asserted-by":"crossref","unstructured":"[10] R. Xu, B. Zhang, Q. Ye, and J. Jiao, \u201cCascaded l1-norm minimization learning (clml) classifier for human detection,\u201d In CVPR, pp.89-96, 2010. 10.1109\/cvpr.2010.5540224","key":"10","DOI":"10.1109\/CVPR.2010.5540224"},{"doi-asserted-by":"crossref","unstructured":"[11] W.R. Schwartz, A. Kembhavi, D. Harwood, and L.S. Davis, \u201cHuman detection using partial least squares analysis,\u201d In ICCV, pp.24-31, 2009. 10.1109\/iccv.2009.5459205","key":"11","DOI":"10.1109\/ICCV.2009.5459205"},{"doi-asserted-by":"crossref","unstructured":"[12] P.F. Felzenszwalb, R.B. Girshick, D.A. McAllester, and D. Ramanan, \u201cObject detection with discriminatively trained part-based models,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.32, no.9, pp.1627-1645, 2010. 10.1109\/tpami.2009.167","key":"12","DOI":"10.1109\/TPAMI.2009.167"},{"doi-asserted-by":"crossref","unstructured":"[13] Y. Yang and D. Ramanan, \u201cArticulated pose estimation with flexible mixtures-of-parts,\u201d In CVPR, pp.1385-1392, 2011. 10.1109\/cvpr.2011.5995741","key":"13","DOI":"10.1109\/CVPR.2011.5995741"},{"doi-asserted-by":"crossref","unstructured":"[14] N. Ukita, \u201cArticulated pose estimation with parts connectivity using discriminative local oriented contours,\u201d In CVPR, pp.3154-3161, 2012. 10.1109\/cvpr.2012.6248049","key":"14","DOI":"10.1109\/CVPR.2012.6248049"},{"doi-asserted-by":"publisher","unstructured":"[15] N. Ukita, \u201cPart-segment features with optimized shape priors for articulated pose estimation,\u201d IEICE Transactions, vol.99-D, no.1, pp.248-256, 2016. 10.1587\/transinf.2015edp7228","key":"15","DOI":"10.1587\/transinf.2015EDP7228"},{"doi-asserted-by":"crossref","unstructured":"[16] N. Ukita, \u201cIterative action and pose recognition using global-and-pose features and action-specific models,\u201d In Workshop on Understanding Human Activities: Context and Interactions, pp.476-483, 2013. 10.1109\/iccvw.2013.68","key":"16","DOI":"10.1109\/ICCVW.2013.68"},{"doi-asserted-by":"publisher","unstructured":"[17] A. Gupta, A. Kembhavi, and L.S. Davis, \u201cObserving human-object interactions: Using spatial and functional compatibility for recognition,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.31, no.10, pp.1775-1789, 2009. 10.1109\/tpami.2009.83","key":"17","DOI":"10.1109\/TPAMI.2009.83"},{"doi-asserted-by":"crossref","unstructured":"[18] B. Yao, X. Jiang, A. Khosla, A.L. Lin, L.J. Guibas, and F.-F. Li, \u201cHuman action recognition by learning bases of action attributes and parts,\u201d In ICCV, pp.1331-1338, 2011. 10.1109\/iccv.2011.6126386","key":"18","DOI":"10.1109\/ICCV.2011.6126386"},{"unstructured":"[19] V. Delaitre, J. Sivic, and I. Laptev, \u201cLearning person-object interactions for action recognition in still images,\u201d In NIPS, pp.1503-1511, 2011.","key":"19"},{"doi-asserted-by":"crossref","unstructured":"[20] B. Yao and F.-F. Li, \u201cModeling mutual context of object and human pose in human-object interaction activities,\u201d In CVPR, pp.17-24, 2010. 10.1109\/cvpr.2010.5540235","key":"20","DOI":"10.1109\/CVPR.2010.5540235"},{"doi-asserted-by":"publisher","unstructured":"[21] Y. Wang and G. Mori, \u201cHidden part models for human action recognition: Probabilistic versus max margin,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.33, no.7, pp.1310-1323, 2011. 10.1109\/tpami.2010.214","key":"21","DOI":"10.1109\/TPAMI.2010.214"},{"doi-asserted-by":"publisher","unstructured":"[22] K.N. Tran, I.A. Kakadiaris, and S.K. Shah, \u201cPart-based motion descriptor image for human action recognition,\u201d Pattern Recognition, vol.45, no.7, pp.2562-2572, 2012. 10.1016\/j.patcog.2011.12.028","key":"22","DOI":"10.1016\/j.patcog.2011.12.028"},{"doi-asserted-by":"crossref","unstructured":"[23] P. Natarajan, V.K. Singh, and R. Nevatia, \u201cLearning 3d action models from a few 2d videos for view invariant action recognition,\u201d In CVPR, pp.2006-2013, 2010. 10.1109\/cvpr.2010.5539876","key":"23","DOI":"10.1109\/CVPR.2010.5539876"},{"doi-asserted-by":"crossref","unstructured":"[24] V.K. Singh and R. Nevatia, \u201cAction recognition in cluttered dynamic scenes using pose-specific part models,\u201d In ICCV, pp.113-120, 2011. 10.1109\/iccv.2011.6126232","key":"24","DOI":"10.1109\/ICCV.2011.6126232"},{"doi-asserted-by":"crossref","unstructured":"[25] A. Yao, J. Gall, G. Fanelli, and L.V. Gool, \u201cDoes human action recognition benefit from pose estimation?,\u201d In BMVC, pp.67.1-67.11, 2011. 10.5244\/c.25.67","key":"25","DOI":"10.5244\/C.25.67"},{"doi-asserted-by":"crossref","unstructured":"[26] J. Chen, M. Kim, Y. Wang, and Q. Ji, \u201cSwitching gaussian process dynamic models for simultaneous composite motion tracking and recognition,\u201d In CVPR, pp.2655-2662, 2009. 10.1109\/cvprw.2009.5206580","key":"26","DOI":"10.1109\/CVPR.2009.5206580"},{"doi-asserted-by":"crossref","unstructured":"[27] J. Gall, A. Yao, and L.J. Van Gool, \u201c2d action recognition serves 3d human pose estimation,\u201d In ECCV, vol.6313, pp.425-438, 2010. 10.1007\/978-3-642-15558-1_31","key":"27","DOI":"10.1007\/978-3-642-15558-1_31"},{"doi-asserted-by":"publisher","unstructured":"[28] N. Ukita, \u201cSimultaneous particle tracking in multi-action motion models with synthesized paths,\u201d Image Vision Comput., vol.31, no.6-7, pp.448-459, 2013. 10.1016\/j.imavis.2012.09.010","key":"28","DOI":"10.1016\/j.imavis.2012.09.010"},{"doi-asserted-by":"publisher","unstructured":"[29] N. Ukita and T. Kanade, \u201cGaussian process motion graph models for smooth transitions among multiple actions,\u201d Computer Vision and Image Understanding, vol.116, no.4, pp.500-509, 2012. 10.1016\/j.cviu.2011.11.005","key":"29","DOI":"10.1016\/j.cviu.2011.11.005"},{"doi-asserted-by":"crossref","unstructured":"[30] C. Desai and D. Ramanan, \u201cDetecting actions, poses, and objects with relational phraselets,\u201d In ECCV, vol.7575, pp.158-172, 2012. 10.1007\/978-3-642-33765-9_12","key":"30","DOI":"10.1007\/978-3-642-33765-9_12"},{"doi-asserted-by":"crossref","unstructured":"[31] N. Shapovalova, A. Vahdat, K. Cannons, T. Lan, and G. Mori, \u201cSimilarity constrained latent support vector machine: An application to weakly supervised action classification,\u201d In ECCV, vol.7578, pp.55-68, 2012. 10.1007\/978-3-642-33786-4_5","key":"31","DOI":"10.1007\/978-3-642-33786-4_5"},{"unstructured":"[32] M. Marszalek, I. Laptev, and C. Schmid, \u201cActions in context,\u201d In CVPR, pp.2929-2936, 2009.","key":"32"},{"doi-asserted-by":"publisher","unstructured":"[33] A. Oliva and A. Torralba, \u201cModeling the shape of the scene: A holistic representation of the spatial envelope,\u201d International Journal of Computer Vision, vol.42, no.3, pp.145-175, 2001. 10.1023\/a:1011139631724","key":"33","DOI":"10.1023\/A:1011139631724"},{"unstructured":"[34] C. Li, A. Kowdle, A. Saxena, and T. Chen, \u201cTowards holistic scene understanding: Feedback enabled cascaded classification models,\u201d In NIPS, pp.1351-1359, 2010.","key":"34"},{"doi-asserted-by":"crossref","unstructured":"[35] C. Wang, D.M. Blei, and F.-F. Li, \u201cSimultaneous image classification and annotation,\u201d In CVPR, pp.1903-1910, 2009. 10.1109\/cvprw.2009.5206800","key":"35","DOI":"10.1109\/CVPR.2009.5206800"},{"doi-asserted-by":"crossref","unstructured":"[36] L.-J. Li, R. Socher, and F.-F. Li, \u201cTowards total scene understanding: Classification, annotation and segmentation in an automatic framework,\u201d In CVPR, pp.2036-2043, 2009. 10.1109\/cvpr.2009.5206718","key":"36","DOI":"10.1109\/CVPR.2009.5206718"},{"doi-asserted-by":"crossref","unstructured":"[37] M. Pandey and S. Lazebnik, \u201cScene recognition and weakly supervised object localization with deformable part-based models,\u201d In ICCV, pp.1307-1314, 2011. 10.1109\/iccv.2011.6126383","key":"37","DOI":"10.1109\/ICCV.2011.6126383"},{"unstructured":"[38] L.-J. Li, H. Su, E.P. Xing, and F.-F. Li, \u201cObject bank: A high-level image representation for scene classification &amp; semantic feature sparsification,\u201d In NIPS, pp.1378-1386, 2010.","key":"38"},{"unstructured":"[39] X. Chen and A.L. Yuille, \u201cArticulated pose estimation by a graphical model with image dependent pairwise relations,\u201d In NIPS, 2014.","key":"39"},{"unstructured":"[40] A. Krizhevsky, I. Sutskever, and G.E. Hinton, \u201cImagenet classification with deep convolutional neural networks,\u201d In NIPS, pp.1106-1114, 2012.","key":"40"},{"doi-asserted-by":"crossref","unstructured":"[41] I. Tsochantaridis, T. Hofmann, T. Joachims, and Y. Altun, \u201cSupport vector machine learning for interdependent and structured output spaces,\u201d In ICML, ICML &apos;04, New York, NY, USA, p.104, ACM, 2004. 10.1145\/1015330.1015341","key":"41","DOI":"10.1145\/1015330.1015341"},{"unstructured":"[42] R.-E. Fan, K.-W. Chang, C.-J. Hsieh, X.-R. Wang, and C.J. Lin, \u201cLIBLINEAR: A library for large linear classification,\u201d Journal of Machine Learning Research, vol.9, pp.1871-1874, 2008.","key":"42"},{"doi-asserted-by":"publisher","unstructured":"[43] C.-C. Chang and C.-J. Lin, \u201cLibsvm: A library for support vector machines,\u201d ACM TIST, vol.2, no.3, p.27, 2011. 10.1145\/1961189.1961199","key":"43","DOI":"10.1145\/1961189.1961199"},{"unstructured":"[44] T.-F. Wu, C.-J. Lin, and R.C. Weng, \u201cProbability estimates for multi-class classification by pairwise coupling,\u201d Journal of Machine Learning Research, vol.5, pp.975-1005, 2004.","key":"44"},{"doi-asserted-by":"crossref","unstructured":"[45] S. Johnson and M. Everingham, \u201cClustered pose and nonlinear appearance models for human pose estimation,\u201d In BMVC, pp.1-11, 2010. 10.5244\/c.24.12","key":"45","DOI":"10.5244\/C.24.12"},{"doi-asserted-by":"crossref","unstructured":"[46] S. Johnson and M. Everingham, \u201cLearning effective human pose estimation from inaccurate annotation,\u201d In CVPR, pp.1465-1472, 2011. 10.1109\/cvpr.2011.5995318","key":"46","DOI":"10.1109\/CVPR.2011.5995318"},{"doi-asserted-by":"crossref","unstructured":"[47] V. Ferrari, M.J. Mar\u00edn-Jim\u00e9nez, and A. Zisserman, \u201cProgressive search space reduction for human pose estimation,\u201d In CVPR, pp.1-8, 2008. 10.1109\/cvpr.2008.4587468","key":"47","DOI":"10.1109\/CVPR.2008.4587468"},{"unstructured":"[48] L. Pishchulin, A. Jain, M. Andriluka, T. Thorm\u00e4hlen, and B. Schiele, \u201cArticulated people detection and pose estimation: Reshaping the future,\u201d In CVPR, pp.3178-3185, 2012.","key":"48"},{"doi-asserted-by":"crossref","unstructured":"[49] S.-E. Wei, V. Ramakrishna, T. Kanade, and Y. Sheikh, \u201cConvolutional pose machines,\u201d In CVPR, pp.4724-4732, 2016. 10.1109\/cvpr.2016.511","key":"49","DOI":"10.1109\/CVPR.2016.511"},{"doi-asserted-by":"crossref","unstructured":"[50] X. Yu, F. Zhou, and M. Chandraker, \u201cDeep deformation network for object landmark localization,\u201d In ECCV, vol.9909, pp.52-70, 2016. 10.1007\/978-3-319-46454-1_4","key":"50","DOI":"10.1007\/978-3-319-46454-1_4"},{"doi-asserted-by":"crossref","unstructured":"[51] U. Rafi, B. Leibe, J. Gall, and I. Kostrikov, \u201cAn efficient convolutional network for human pose estimation,\u201d In BMVC, pp.109.1-109.11, 2016. 10.5244\/c.30.109","key":"51","DOI":"10.5244\/C.30.109"},{"doi-asserted-by":"crossref","unstructured":"[52] J.R.R. Uijlings, K.E.A. van de Sande, T. Gevers, and A.W.M. Smeulders, \u201cSelective search for object recognition,\u201d International Journal of Computer Vision, vol.104, no.2, pp.154-171, 2013.","key":"52","DOI":"10.1007\/s11263-013-0620-5"},{"doi-asserted-by":"publisher","unstructured":"[53] J.C. Niebles, H. Wang, and F.-F. Li, \u201cUnsupervised learning of human action categories using spatial-temporal words,\u201d International Journal of Computer Vision, vol.79, no.3, pp.299-318, 2008. 10.1007\/s11263-007-0122-4","key":"53","DOI":"10.1007\/s11263-007-0122-4"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E101.D\/3\/E101.D_2017EDP7204\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,2]],"date-time":"2025-07-02T06:32:39Z","timestamp":1751437959000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E101.D\/3\/E101.D_2017EDP7204\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"references-count":53,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2018]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2017edp7204","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2018]]}}}