{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T22:22:48Z","timestamp":1784413368781,"version":"3.55.0"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2016,10,1]],"date-time":"2016-10-01T00:00:00Z","timestamp":1475280000000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"City University of Hong Kong (HK)","award":["7004417"],"award-info":[{"award-number":["7004417"]}]},{"name":"Research Grants Council of the Hong Kong Special Administrative Region, China","award":["CityU 123212"],"award-info":[{"award-number":["CityU 123212"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2017,3]]},"DOI":"10.1007\/s11263-016-0962-x","type":"journal-article","created":{"date-parts":[[2016,10,1]],"date-time":"2016-10-01T14:47:09Z","timestamp":1475333229000},"page":"149-168","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":19,"title":["Maximum-Margin Structured Learning with Deep Networks for 3D Human Pose Estimation"],"prefix":"10.1007","volume":"122","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0628-0336","authenticated-orcid":false,"given":"Sijin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weichen","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2886-2513","authenticated-orcid":false,"given":"Antoni B.","family":"Chan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2016,10,1]]},"reference":[{"key":"962_CR1","first-page":"1247","volume":"28","author":"G Andrew","year":"2013","unstructured":"Andrew, G., Arora, R., Bilmes, J., & Livescu, K. (2013). Deep canonical correlation analysis. ICML, 28, 1247\u20131255.","journal-title":"ICML"},{"key":"962_CR2","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., & Schiele, B. (2014). 2d human pose estimation: New benchmark and state of the art analysis. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 3686\u20133693).","DOI":"10.1109\/CVPR.2014.471"},{"key":"962_CR3","unstructured":"Bastien, F., Lamblin, P., Pascanu, R., Bergstra, J., Goodfellow, I. J., Bergeron, A., Bouchard, N., & Bengio, Y. (2012). Theano: new features and speed improvements. In NIPS: Deep learning and unsupervised feature learning workshop"},{"key":"962_CR4","unstructured":"Bengio, Y., Mesnil, G., Dauphin, Y., & Rifai, S. (2013). Better mixing via deep representations. In ICML (pp. 552\u2013560)."},{"issue":"3","key":"962_CR5","doi-asserted-by":"crossref","first-page":"179","DOI":"10.1023\/B:VISI.0000011203.00237.9b","volume":"56","author":"C Bregler","year":"2004","unstructured":"Bregler, C., Malik, J., & Pullen, K. (2004). Twist based acquisition and tracking of animal and human kinematics. International Journal of Computer Vision, 56(3), 179\u2013194.","journal-title":"International Journal of Computer Vision"},{"key":"962_CR6","doi-asserted-by":"crossref","unstructured":"Burenius, M., Sullivan, J., & Carlsson, S. (2013). 3d pictorial structures for multiple view articulated pose estimation. In CVPR (pp. 3618\u20133625).","DOI":"10.1109\/CVPR.2013.464"},{"issue":"1","key":"962_CR7","doi-asserted-by":"crossref","first-page":"93","DOI":"10.1007\/BF02592073","volume":"39","author":"PH Calamai","year":"1987","unstructured":"Calamai, P. H., & Mor\u00e9, J. J. (1987). Projected gradient methods for linearly constrained problems. Mathematical programming, 39(1), 93\u2013116.","journal-title":"Mathematical programming"},{"key":"962_CR8","doi-asserted-by":"crossref","unstructured":"Carreira, J., Agrawal, P., Fragkiadaki, K., & Malik, J. (2016). Human pose estimation with iterative error feedback. In The IEEE conference on computer vision and pattern recognition (CVPR)","DOI":"10.1109\/CVPR.2016.512"},{"key":"962_CR9","unstructured":"Chen, X. & Yuille, A. (2014). Articulated pose estimation by a graphical model with image dependent pairwise relations. In NIPS"},{"key":"962_CR10","doi-asserted-by":"crossref","unstructured":"Chu, X., Ouyang, W., Yang, W., & Wang, X. (2015). Multi-task recurrent neural network for immediacy prediction. In The IEEE international conference on computer vision (ICCV) (pp. 3352\u20133360).","DOI":"10.1109\/ICCV.2015.383"},{"issue":"2","key":"962_CR11","doi-asserted-by":"crossref","first-page":"185","DOI":"10.1023\/B:VISI.0000043757.18370.9c","volume":"61","author":"J Deutscher","year":"2005","unstructured":"Deutscher, J., & Reid, I. (2005). Articulated body motion capture by stochastic search. IJCV, 61(2), 185\u2013205.","journal-title":"IJCV"},{"key":"962_CR12","unstructured":"Dhungel, N., Carneiro, G., & Bradley, A. P. (2014). Deep structured learning for mass segmentation from mammograms. CoRR\u00a0 arXiv:1410.7454"},{"key":"962_CR13","doi-asserted-by":"crossref","unstructured":"Eichner, M. & Ferrari, V. (2009). Better appearance models for pictorial structures. In BMVC (pp 1\u201311)","DOI":"10.5244\/C.23.3"},{"issue":"1","key":"962_CR14","doi-asserted-by":"crossref","first-page":"55","DOI":"10.1023\/B:VISI.0000042934.15159.49","volume":"61","author":"PF Felzenszwalb","year":"2005","unstructured":"Felzenszwalb, P. F., & Huttenlocher, D. P. (2005). Pictorial structures for object recognition. IJCV, 61(1), 55\u201379.","journal-title":"IJCV"},{"key":"962_CR15","unstructured":"Goodfellow, I. J., Shlens, J., & Szegedy, C. (2015). Explaining and harnessing adversarial examples. In International conference on learning representations"},{"key":"962_CR16","doi-asserted-by":"crossref","unstructured":"Ionescu, C., Bo, L., & Sminchisescu, C. (2009). Structural SVM for visual localization and continuous state estimation. In ICCV (pp. 1157\u20131164).","DOI":"10.1109\/ICCV.2009.5459346"},{"key":"962_CR17","doi-asserted-by":"crossref","unstructured":"Ionescu, C., Li, F., & Sminchisescu, C. (2011). Latent structured models for human pose estimation. In ICCV (pp. 2220\u20132227).","DOI":"10.1109\/ICCV.2011.6126500"},{"issue":"7","key":"962_CR18","doi-asserted-by":"crossref","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","volume":"36","author":"C Ionescu","year":"2014","unstructured":"Ionescu, C., Papava, D., Olaru, V., & Sminchisescu, C. (2014). Human3.6m: Large scale datasets and predictive methods for 3d human sensing in natural environments. IEEE TPAMI, 36(7), 1325\u20131339.","journal-title":"IEEE TPAMI"},{"key":"962_CR19","unstructured":"Jaderberg, M., Simonyan, K., Vedaldi, A., & Zisserman, A. (2015). Deep structured output learning for unconstrained text recognition. ICLR"},{"key":"962_CR20","unstructured":"Jain, A., Tompson, J., Andriluka, M., Taylor, G. W., & Bregler, C. (2014). Learning human pose estimation features with convolutional networks. In ICLR"},{"issue":"1","key":"962_CR21","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1007\/s10994-009-5108-8","volume":"77","author":"T Joachims","year":"2009","unstructured":"Joachims, T., Finley, T., & Yu, C. N. J. (2009). Cutting-plane training of structural svms. Machine Learning, 77(1), 27\u201359.","journal-title":"Machine Learning"},{"key":"962_CR22","volume-title":"Probabilistic graphical models: Principles and techniques","author":"D Koller","year":"2009","unstructured":"Koller, D., & Friedman, N. (2009). Probabilistic graphical models: Principles and techniques. Cambridge: MIT Press."},{"key":"962_CR23","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In NIPS"},{"key":"962_CR24","doi-asserted-by":"crossref","unstructured":"Li, S. & Chan, A. B. (2014). 3d human pose estimation from monocular images with deep convolutional neural network. In ACCV","DOI":"10.1007\/978-3-319-16808-1_23"},{"key":"962_CR25","doi-asserted-by":"crossref","unstructured":"Li, S., Liu, Z. Q., & Chan, A. B. (2014). Heterogeneous multi-task learning for human pose estimation with deep convolutional neural network. In IJCV (pp 1\u201318).","DOI":"10.1109\/CVPRW.2014.78"},{"key":"962_CR26","doi-asserted-by":"crossref","unstructured":"Li, S., Zhang, W., & Chan, A. B. (2015). Maximum-margin structured learning with deep networks for 3d human pose estimation. In The IEEE international conference on computer vision (ICCV)","DOI":"10.1109\/ICCV.2015.326"},{"key":"962_CR27","volume-title":"A mathematical introduction to robotic manipulation","author":"RM Murray","year":"1994","unstructured":"Murray, R. M., Li, Z., & Sastry, S. S. (1994). A mathematical introduction to robotic manipulation (Vol. 29). Boca Raton: CRC press."},{"key":"962_CR28","unstructured":"Nair, V. & Hinton, G. E. (2010). Rectified linear units improve restricted boltzmann machines. In ICML"},{"key":"962_CR29","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., & Ng, A. Y. (2011). Multimodal deep learning. In ICML (pp. 689\u2013696)"},{"key":"962_CR30","first-page":"1197","volume":"8","author":"M Osadchy","year":"2007","unstructured":"Osadchy, M., LeCun, Y., & Miller, M. L. (2007). Synergistic face detection and pose estimation with energy-based models. Journal of Machine Learning Research, 8, 1197\u20131215.","journal-title":"Journal of Machine Learning Research"},{"key":"962_CR31","doi-asserted-by":"crossref","unstructured":"Razavian, A. S., Azizpour, H., Sullivan, J., & Carlsson, S. (2014). CNN features off-the-shelf: An astounding baseline for recognition. In CVPR (pp. 512\u2013519)","DOI":"10.1109\/CVPRW.2014.131"},{"key":"962_CR32","doi-asserted-by":"crossref","unstructured":"Rodr\u00edguez, J. A. & Perronnin, F. (2013). Label embedding for text recognition. In BMVC","DOI":"10.5244\/C.27.5"},{"key":"962_CR33","unstructured":"Rumelhart, D. E., Hinton, G. E., & Williams, R. J. (1988). Neurocomputing: Foundations of research, Chap Learning representations by back-propagating errors (pp. 696\u2013699). Cambridge, MA: MIT Press."},{"key":"962_CR34","doi-asserted-by":"crossref","unstructured":"Sapp, B. & Taskar, B. (2013). Modec: Multimodal decomposablemodels for human pose estimation. In Proceedings of the IEEE conference on CVPR","DOI":"10.1109\/CVPR.2013.471"},{"key":"962_CR35","unstructured":"Sermanet, P., Eigen, D., Zhang, X., Mathieu, M., Fergus, R., & LeCun, Y. (2013). Overfeat: Integrated recognition, localization and detection using convolutional networks. CoRR\u00a0 arXiv:1312.6229"},{"key":"962_CR36","unstructured":"Srivastava, N. & Salakhutdinov, R. R. (2012). Multimodal learning with deep boltzmann machines. In NIPS (pp. 2222\u20132230). Curran Associates Inc., Red Hook."},{"issue":"1","key":"962_CR37","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: A simple way to prevent neural networks from overfitting. Journal of Machine Learning Research, 15(1), 1929\u20131958.","journal-title":"Journal of Machine Learning Research"},{"key":"962_CR38","doi-asserted-by":"crossref","unstructured":"Sun, Y., Wang, X., & Tang, X. (2014). Deep learning face representation from predicting 10,000 classes. In CVPR, IEEE Computer Society","DOI":"10.1109\/CVPR.2014.244"},{"key":"962_CR39","unstructured":"Tompson, J., Jain, A., LeCun, Y., & Bregler, C. (2014). Joint training of a convolutional network and a graphical model for human pose estimation. In NIPS"},{"key":"962_CR40","doi-asserted-by":"crossref","unstructured":"Toshev, A. & Szegedy, C. (2014). Deeppose: Human pose estimation via deep neural networks. In CVPR","DOI":"10.1109\/CVPR.2014.214"},{"key":"962_CR41","doi-asserted-by":"crossref","unstructured":"Tsochantaridis, I., Hofmann, T., Joachims, T., & Altun, Y. (2004). Support vector machine learning for interdependent and structured output spaces. In ICML","DOI":"10.1145\/1015330.1015341"},{"key":"962_CR42","first-page":"1453","volume":"6","author":"I Tsochantaridis","year":"2005","unstructured":"Tsochantaridis, I., Joachims, T., Hofmann, T., & Altun, Y. (2005). Large margin methods for structured and interdependent output variables. Journal of Machine Learning Research, 6, 1453\u20131484.","journal-title":"Journal of Machine Learning Research"},{"key":"962_CR43","doi-asserted-by":"crossref","unstructured":"Yang, Y. & Ramanan, D. (2011). Articulated pose estimation with flexible mixtures-of-parts. In CVPR (pp. 1385 \u2013 1392)","DOI":"10.1109\/CVPR.2011.5995741"},{"key":"962_CR44","doi-asserted-by":"crossref","unstructured":"Zheng, S., Jayasumana, S., Romera-Paredes, B., Vineet, V., Su, Z., Du, D., Huang, C., & Torr, P. (2015). Conditional random fields as recurrent neural networks. In International Conference on Computer Vision (ICCV)","DOI":"10.1109\/ICCV.2015.179"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-016-0962-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-016-0962-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-016-0962-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T22:18:03Z","timestamp":1749593883000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-016-0962-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,10,1]]},"references-count":44,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2017,3]]}},"alternative-id":["962"],"URL":"https:\/\/doi.org\/10.1007\/s11263-016-0962-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,10,1]]}}}