{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T11:49:18Z","timestamp":1725882558542},"publisher-location":"Cham","reference-count":53,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319545257"},{"type":"electronic","value":"9783319545264"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-54526-4_37","type":"book-chapter","created":{"date-parts":[[2017,3,15]],"date-time":"2017-03-15T07:13:16Z","timestamp":1489561996000},"page":"500-514","source":"Crossref","is-referenced-by-count":0,"title":["Attributes and Action Recognition Based on Convolutional Neural Networks and Spatial Pyramid VLAD Encoding"],"prefix":"10.1007","author":[{"given":"Shiyang","family":"Yan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jeremy S.","family":"Smith","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bailing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,3,16]]},"reference":[{"key":"37_CR1","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., Perona, P.: A Bayesian hierarchical model for learning natural scene categories. In: 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2005), vol. 2, pp. 524\u2013531 (2005)","DOI":"10.1109\/CVPR.2005.16"},{"key":"37_CR2","series-title":"Communications in Computer and Information Science","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1007\/978-3-642-25382-9_2","volume-title":"Computer Vision, Imaging and Computer Graphics. Theory and Applications","author":"G Csurka","year":"2011","unstructured":"Csurka, G., Perronnin, F.: Fisher vectors: beyond bag-of-visual-words image representations. In: Richard, P., Braz, J. (eds.) VISIGRAPP 2010. CCIS, vol. 229, pp. 28\u201342. Springer, Heidelberg (2011). doi: 10.1007\/978-3-642-25382-9_2"},{"key":"37_CR3","doi-asserted-by":"crossref","unstructured":"J\u00e9gou, H., Douze, M., Schmid, C., P\u00e9rez, P.: Aggregating local descriptors into a compact image representation. In: 2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3304\u20133311. IEEE (2010)","DOI":"10.1109\/CVPR.2010.5540039"},{"key":"37_CR4","doi-asserted-by":"crossref","unstructured":"Sharma, G., Jurie, F., Schmid, C.: Discriminative spatial saliency for image classification. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3506\u20133513 (2012)","DOI":"10.1109\/CVPR.2012.6248093"},{"key":"37_CR5","doi-asserted-by":"publisher","unstructured":"Delaitre, V., Laptev, I., Sivic, J.: Recognizing human actions in still images: a study of bag-of-features and part-based representations. In: Proceedings of the British Machine Vision Conference, pp. 97.1\u201397.11. BMVA Press (2010). doi: 10.5244\/C.24.97.","DOI":"10.5244\/C.24.97."},{"key":"37_CR6","doi-asserted-by":"crossref","first-page":"60","DOI":"10.1007\/s11263-012-0594-8","volume":"103","author":"H Wang","year":"2013","unstructured":"Wang, H., Kl\u00e4ser, A., Schmid, C., Liu, C.L.: Dense trajectories and motion boundary descriptors for action recognition. Int. J. Comput. Vis. 103, 60\u201379 (2013)","journal-title":"Int. J. Comput. Vis."},{"key":"37_CR7","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/978-3-319-10602-1_38","volume-title":"Computer Vision \u2013 ECCV 2014","author":"X Peng","year":"2014","unstructured":"Peng, X., Zou, C., Qiao, Y., Peng, Q.: Action recognition with stacked fisher vectors. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 581\u2013595. Springer, Heidelberg (2014). doi: 10.1007\/978-3-319-10602-1_38"},{"key":"37_CR8","doi-asserted-by":"crossref","unstructured":"Lowe, D.G.: Object recognition from local scale-invariant features. In: The Proceedings of the Seventh IEEE International Conference on Computer Vision, vol. 2, pp. 1150\u20131157. IEEE (1999)","DOI":"10.1109\/ICCV.1999.790410"},{"key":"37_CR9","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Neural Information Processing Systems (2012)"},{"key":"37_CR10","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp. 580\u2013587 (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"37_CR11","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Girshick, R., Malik, J.: Contextual action recognition with R* CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1080\u20131088 (2015)","DOI":"10.1109\/ICCV.2015.129"},{"key":"37_CR12","doi-asserted-by":"crossref","unstructured":"Uricchio, T., Bertini, M., Seidenari, L., Bimbo, A.D.: Fisher encoded convolutional bag-of-windows for efficient image retrieval and social image tagging. In: 2015 IEEE International Conference on Computer Vision Workshop (ICCVW), pp. 1020\u20131026 (2015)","DOI":"10.1109\/ICCVW.2015.134"},{"key":"37_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1007\/978-3-319-10602-1_26","volume-title":"Computer Vision \u2013 ECCV 2014","author":"CL Zitnick","year":"2014","unstructured":"Zitnick, C.L., Doll\u00e1r, P.: Edge boxes: locating object proposals from edges. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 391\u2013405. Springer, Heidelberg (2014). doi: 10.1007\/978-3-319-10602-1_26"},{"key":"37_CR14","unstructured":"Shin, A., Yamaguchi, M., Ohnishi, K., Harada, T.: Dense image representation with spatial pyramid VLAD coding of CNN for locally robust captioning. arXiv preprint arXiv:1603.09046 (2016)"},{"key":"37_CR15","doi-asserted-by":"crossref","unstructured":"Bourdev, L., Maji, S., Malik, J.: Describing people: a poselet-based approach to attribute classification. In: 2011 International Conference on Computer Vision, pp. 1543\u20131550 (2011)","DOI":"10.1109\/ICCV.2011.6126413"},{"key":"37_CR16","doi-asserted-by":"crossref","unstructured":"Yao, B., Jiang, X., Khosla, A., Lin, A.L., Guibas, L., Fei-Fei, L.: Human action recognition by learning bases of action attributes and parts. In: 2011 IEEE International Conference on Computer Vision (ICCV), pp. 1331\u20131338. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126386"},{"key":"37_CR17","doi-asserted-by":"crossref","unstructured":"Kumar, N., Berg, A.C., Belhumeur, P.N., Nayar, S.K.: Attribute and simile classifiers for face verification. In: IEEE International Conference on Computer Vision (ICCV) (2009)","DOI":"10.1109\/ICCV.2009.5459250"},{"key":"37_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"609","DOI":"10.1007\/978-3-642-33712-3_44","volume-title":"Computer Vision \u2013 ECCV 2012","author":"H Chen","year":"2012","unstructured":"Chen, H., Gallagher, A., Girod, B.: Describing clothing by semantic attributes. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012. LNCS, vol. 7574, pp. 609\u2013623. Springer, Heidelberg (2012). doi: 10.1007\/978-3-642-33712-3_44"},{"key":"37_CR19","doi-asserted-by":"crossref","unstructured":"Cai, J., Zha, Z.J., Zhou, W., Tian, Q.: Attribute-assisted reranking for web image retrieval. In: Proceedings of the 20th ACM International Conference on Multimedia, pp. 873\u2013876. ACM (2012)","DOI":"10.1145\/2393347.2396335"},{"key":"37_CR20","unstructured":"Peng, X., Wang, L., Wang, X., Qiao, Y.: Bag of visual words and fusion methods for action recognition: comprehensive study and good practice. arXiv preprint arXiv:1405.4506 (2014)"},{"key":"37_CR21","doi-asserted-by":"crossref","unstructured":"Oneata, D., Verbeek, J., Schmid, C.: Action and event recognition with fisher vectors on a compact feature set. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1817\u20131824 (2013)","DOI":"10.1109\/ICCV.2013.228"},{"key":"37_CR22","doi-asserted-by":"crossref","unstructured":"Ullah, M.M., Parizi, S.N., Laptev, I.: Improving bag-of-features action recognition with non-local cues. In: BMVC, vol. 10, pp. 95\u20131. Citeseer (2010)","DOI":"10.5244\/C.24.95"},{"key":"37_CR23","unstructured":"Delaitre, V., Laptev, I., Sivic, J.: Recognizing human actions in still images: a study of bag-of-features and part-based representations (2010). http:\/\/www.di.ens.fr\/willow\/research\/stillactions\/"},{"key":"37_CR24","doi-asserted-by":"crossref","unstructured":"Sun, C., Nevatia, R.: Large-scale web video event classification by use of fisher vectors. In: 2013 IEEE Workshop on Applications of Computer Vision (WACV), pp. 15\u201322. IEEE (2013)","DOI":"10.1109\/WACV.2013.6474994"},{"key":"37_CR25","doi-asserted-by":"crossref","unstructured":"Jain, M., J\u00e9gou, H., Bouthemy, P.: Better exploiting motion for better action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2555\u20132562 (2013)","DOI":"10.1109\/CVPR.2013.330"},{"key":"37_CR26","doi-asserted-by":"crossref","first-page":"1627","DOI":"10.1109\/TPAMI.2009.167","volume":"32","author":"PF Felzenszwalb","year":"2010","unstructured":"Felzenszwalb, P.F., Girshick, R.B., McAllester, D., Ramanan, D.: Object detection with discriminatively trained part-based models. IEEE Trans. Pattern Anal. Mach. Intell. 32, 1627\u20131645 (2010)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"37_CR27","doi-asserted-by":"crossref","unstructured":"Bourdev, L., Malik, J.: Poselets: body part detectors trained using 3D human pose annotations. In: 2009 IEEE 12th International Conference on Computer Vision, pp. 1365\u20131372. IEEE (2009)","DOI":"10.1109\/ICCV.2009.5459303"},{"key":"37_CR28","doi-asserted-by":"crossref","unstructured":"Zhang, N., Paluri, M., Ranzato, M., Darrell, T., Bourdev, L.: PANDA: pose aligned networks for deep attribute modeling. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1637\u20131644 (2014)","DOI":"10.1109\/CVPR.2014.212"},{"key":"37_CR29","doi-asserted-by":"crossref","first-page":"1691","DOI":"10.1109\/TPAMI.2012.67","volume":"34","author":"B Yao","year":"2012","unstructured":"Yao, B., Fei-Fei, L.: Recognizing human-object interactions in still images by modeling the mutual context of objects and human poses. IEEE Trans. Pattern Anal. Mach. Intell. 34, 1691\u20131703 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"37_CR30","doi-asserted-by":"crossref","first-page":"601","DOI":"10.1109\/TPAMI.2011.158","volume":"34","author":"A Prest","year":"2012","unstructured":"Prest, A., Schmid, C., Ferrari, V.: Weakly supervised learning of interactions between humans and objects. IEEE Trans. Pattern Anal. Mach. Intell. 34, 601\u2013614 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"37_CR31","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN. In: 2015 IEEE International Conference on Computer Vision (ICCV), pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"37_CR32","doi-asserted-by":"crossref","unstructured":"Zheng, S., Jayasumana, S., Romera-Paredes, B., Vineet, V., Su, Z., Du, D., Huang, C., Torr, P.H.S.: Conditional random fields as recurrent neural networks. In: 2015 IEEE International Conference on Computer Vision (ICCV), pp. 1529\u20131537 (2015)","DOI":"10.1109\/ICCV.2015.179"},{"key":"37_CR33","doi-asserted-by":"crossref","unstructured":"Oquab, M., Bottou, L., Laptev, I., Sivic, J.: Learning and transferring mid-level image representations using convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1717\u20131724 (2014)","DOI":"10.1109\/CVPR.2014.222"},{"key":"37_CR34","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Girshick, R., Malik, J.: Actions and attributes from wholes and parts. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2470\u20132478 (2015)","DOI":"10.1109\/ICCV.2015.284"},{"key":"37_CR35","doi-asserted-by":"publisher","unstructured":"Diba, A., Pazandeh, A.M., Pirsiavash, H., Van Gool, L.: DeepCAMP: deep convolutional action & attribute mid-level patterns. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Las Vegas, NV, pp. 3557\u20133565 (2016). doi: 10.1109\/CVPR.2016.387","DOI":"10.1109\/CVPR.2016.387"},{"key":"37_CR36","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: All about VLAD. In: 2013 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1578\u20131585 (2013)","DOI":"10.1109\/CVPR.2013.207"},{"key":"37_CR37","doi-asserted-by":"crossref","first-page":"222","DOI":"10.1007\/s11263-013-0636-x","volume":"105","author":"J S\u00e1nchez","year":"2013","unstructured":"S\u00e1nchez, J., Perronnin, F., Mensink, T., Verbeek, J.: Image classification with the fisher vector: theory and practice. Int. J. Comput. Vis. 105, 222\u2013245 (2013)","journal-title":"Int. J. Comput. Vis."},{"key":"37_CR38","doi-asserted-by":"crossref","unstructured":"Dixit, M., Chen, S., Gao, D., Rasiwasia, N., Vasconcelos, N.: Scene classification with semantic fisher vectors. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2974\u20132983 (2015)","DOI":"10.1109\/CVPR.2015.7298916"},{"key":"37_CR39","doi-asserted-by":"crossref","unstructured":"Lazebnik, S., Schmid, C., Ponce, J.: Beyond bags of features: spatial pyramid matching for recognizing natural scene categories. In: 2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, vol. 2, pp. 2169\u20132178. IEEE (2006)","DOI":"10.1109\/CVPR.2006.68"},{"key":"37_CR40","doi-asserted-by":"crossref","unstructured":"Zhou, R., Yuan, Q., Gu, X., Zhang, D.: Spatial pyramid VLAD. In: 2014 IEEE Visual Communications and Image Processing Conference, pp. 342\u2013345. IEEE (2014)","DOI":"10.1109\/VCIP.2014.7051576"},{"key":"37_CR41","doi-asserted-by":"crossref","unstructured":"Hosang, J., Benenson, R., Schiele, B.: How good are detection proposals, really? In: 25th British Machine Vision Conference, pp. 1\u201312. BMVA Press (2014)","DOI":"10.5244\/C.28.24"},{"key":"37_CR42","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. CoRR abs\/1409.1556 (2014)"},{"key":"37_CR43","doi-asserted-by":"crossref","unstructured":"J\u00e9gou, H., Douze, M., Schmid, C., P\u00e9rez, P.: Aggregating local descriptors into a compact image representation. In: IEEE Conference on Computer Vision & Pattern Recognition (2010)","DOI":"10.1109\/CVPR.2010.5540039"},{"key":"37_CR44","doi-asserted-by":"crossref","first-page":"125","DOI":"10.1007\/s11263-007-0075-7","volume":"77","author":"DA Ross","year":"2008","unstructured":"Ross, D.A., Lim, J., Lin, R., Yang, M.: Incremental learning for robust visual tracking. Int. J. Comput. Vis. 77, 125\u2013141 (2008)","journal-title":"Int. J. Comput. Vis."},{"key":"37_CR45","unstructured":"Arthur, D., Vassilvitskii, S.: k-means++: the advantages of careful seeding. In: Proceedings of the Eighteenth Annual ACM-SIAM Symposium on Discrete Algorithms, pp. 1027\u20131035. Society for Industrial and Applied Mathematics (2007)"},{"key":"37_CR46","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., Darrell, T.: Caffe: convolutional architecture for fast feature embedding. In: Proceedings of the ACM International Conference on Multimedia, pp. 675\u2013678. ACM (2014)","DOI":"10.1145\/2647868.2654889"},{"key":"37_CR47","first-page":"2825","volume":"12","author":"F Pedregosa","year":"2011","unstructured":"Pedregosa, F., Varoquaux, G., Gramfort, A., Michel, V., Thirion, B., Grisel, O., Blondel, M., Prettenhofer, P., Weiss, R., Dubourg, V., Vanderplas, J., Passos, A., Cournapeau, D., Brucher, M., Perrot, M., Duchesnay, E.: Scikit-learn: machine learning in python. J. Mach. Learn. Res. 12, 2825\u20132830 (2011)","journal-title":"J. Mach. Learn. Res."},{"key":"37_CR48","unstructured":"Vedaldi, A., Fulkerson, B.: VLFeat: an open and portable library of computer vision algorithms (2008)"},{"key":"37_CR49","doi-asserted-by":"crossref","first-page":"27:1","DOI":"10.1145\/1961189.1961199","volume":"2","author":"CC Chang","year":"2011","unstructured":"Chang, C.C., Lin, C.J.: LIBSVM: a library for support vector machines. ACM Trans. Intell. Syst. Technol. 2, 27:1\u201327:27 (2011). http:\/\/www.csie.ntu.edu.tw\/ cjlin\/libsvm","journal-title":"ACM Trans. Intell. Syst. Technol."},{"key":"37_CR50","unstructured":"Li, L.J., Su, H., Fei-Fei, L., Xing, E.P.: Object bank: a high-level image representation for scene classification & semantic feature sparsification. In: Advances in Neural Information Processing Systems, pp. 1378\u20131386 (2010)"},{"key":"37_CR51","doi-asserted-by":"crossref","unstructured":"Wang, J., Yang, J., Yu, K., Lv, F., Huang, T., Gong, Y.: Locality-constrained linear coding for image classification. In: 2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3360\u20133367. IEEE (2010)","DOI":"10.1109\/CVPR.2010.5540018"},{"key":"37_CR52","doi-asserted-by":"crossref","unstructured":"Sharma, G., Jurie, F., Schmid, C.: Expanded parts model for human attribute and action recognition in still images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013659 (2013)","DOI":"10.1109\/CVPR.2013.90"},{"key":"37_CR53","doi-asserted-by":"crossref","first-page":"4422","DOI":"10.1109\/TIP.2015.2465147","volume":"24","author":"FS Khan","year":"2015","unstructured":"Khan, F.S., Xu, J., van de Weijer, J., Bagdanov, A.D., Anwer, R.M., Lopez, A.M.: Recognizing actions through action-specific person detection. IEEE Trans. Image Process. 24, 4422\u20134432 (2015)","journal-title":"IEEE Trans. Image Process."}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ACCV 2016 Workshops"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-54526-4_37","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,19]],"date-time":"2019-09-19T18:56:32Z","timestamp":1568919392000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-54526-4_37"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319545257","9783319545264"],"references-count":53,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-54526-4_37","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2017]]}}}