{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,9]],"date-time":"2025-04-09T14:40:03Z","timestamp":1744209603614,"version":"3.40.3"},"publisher-location":"Berlin, Heidelberg","reference-count":31,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642337086"},{"type":"electronic","value":"9783642337093"}],"license":[{"start":{"date-parts":[[2012,1,1]],"date-time":"2012-01-01T00:00:00Z","timestamp":1325376000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2012]]},"DOI":"10.1007\/978-3-642-33709-3_49","type":"book-chapter","created":{"date-parts":[[2012,9,26]],"date-time":"2012-09-26T08:05:20Z","timestamp":1348646720000},"page":"688-701","source":"Crossref","is-referenced-by-count":21,"title":["Scene Aligned Pooling for Complex Video Recognition"],"prefix":"10.1007","author":[{"given":"Liangliang","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yadong","family":"Mu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Apostol","family":"Natsev","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shih-Fu","family":"Chang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gang","family":"Hua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John R.","family":"Smith","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"49_CR1","doi-asserted-by":"crossref","unstructured":"Schuldt, C., Laptev, I., Caputo, B.: Recognizing human actions: A local SVM approach. In: ICPR (2004)","DOI":"10.1109\/ICPR.2004.1334462"},{"key":"49_CR2","doi-asserted-by":"crossref","unstructured":"Blank, M., Gorelick, L., Shechtman, E., Irani, M., Basri, R.: Actions as space-time shapes. In: ICCV (2005)","DOI":"10.1109\/ICCV.2005.28"},{"key":"49_CR3","doi-asserted-by":"crossref","unstructured":"Liu, J., Luo, J., Shah, M.: Recognizing realistic actions from videos \u201din the wild\u201d. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206744"},{"key":"49_CR4","doi-asserted-by":"crossref","unstructured":"Efros, A., Berg, A., Mori, G., Malik, J.: Recognizing action at a distance. In: ICCV, pp. 726\u2013733 (2003)","DOI":"10.1109\/ICCV.2003.1238420"},{"key":"49_CR5","doi-asserted-by":"publisher","first-page":"598","DOI":"10.1038\/33402","volume":"392","author":"R. Epstein","year":"1998","unstructured":"Epstein, R., Kanwisher, N.: A cortical representation of the local visual environment. Nature\u00a0392, 598\u2013601 (1998)","journal-title":"Nature"},{"key":"49_CR6","doi-asserted-by":"publisher","first-page":"316","DOI":"10.1037\/0096-3445.108.3.316","volume":"108","author":"A. Friedman","year":"1979","unstructured":"Friedman, A.: Framing pictures: The role of knowledge in automatized encoding and memory for gist. Journal of Experimental Psychology\u00a0108, 316\u2013355 (1979)","journal-title":"Journal of Experimental Psychology"},{"key":"49_CR7","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y. LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proceedings of the IEEE\u00a086, 2278\u20132324 (1998)","journal-title":"Proceedings of the IEEE"},{"key":"49_CR8","doi-asserted-by":"publisher","first-page":"299","DOI":"10.1007\/s11263-007-0122-4","volume":"79","author":"J.C. Niebles","year":"2008","unstructured":"Niebles, J.C., Wang, H., Fei-Fei, L.: Unsupervised learning of human action categories using spatial-temporal words. IJCV\u00a079, 299\u2013318 (2008)","journal-title":"IJCV"},{"key":"49_CR9","unstructured":"Doll\u00e1r, P., Rabaud, V., Cottrell, G., Belongie, S.: Behavior recognition via sparse spatio-temporal features. In: IEEE International Workshop on VS-PETS (2005)"},{"key":"49_CR10","doi-asserted-by":"crossref","unstructured":"Laptev, I., Marszalek, M., Schmid, C., Rozenfeld, B.: Learning realistic human actions from movies. In: CVPR (2008)","DOI":"10.1109\/CVPR.2008.4587756"},{"key":"49_CR11","doi-asserted-by":"crossref","unstructured":"Boureau, Y.L., Bach, F., LeCun, Y., Ponce, J.: Learning mid-level features for recognition. In: CVPR (2010)","DOI":"10.1109\/CVPR.2010.5539963"},{"key":"49_CR12","unstructured":"Yang, J., Yu, K., Gong, Y., Huang, T.: Linear pyramid matching using sparse coding for image classification. In: CVPR (2009)"},{"key":"49_CR13","doi-asserted-by":"crossref","unstructured":"J\u00e9gou, H., Douze, M., Schmid, C., P\u00e9rez, P.: Aggregating local descriptors into a compact image representation. In: 2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3304\u20133311. IEEE (2010)","DOI":"10.1109\/CVPR.2010.5540039"},{"key":"49_CR14","doi-asserted-by":"crossref","unstructured":"Lazebnik, S., Schmid, C., Ponce, J.: Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories. In: CVPR, vol.\u00a02, pp. 2169\u20132178 (2006)","DOI":"10.1109\/CVPR.2006.68"},{"key":"49_CR15","doi-asserted-by":"crossref","unstructured":"Boureau, Y., Le Roux, N., Bach, F., Ponce, J., LeCun, Y.: Ask the locals: multi-way local pooling for image recognition. In: ICCV (2011)","DOI":"10.1109\/ICCV.2011.6126555"},{"key":"49_CR16","unstructured":"Oliva, A., Torralba, A.: Modeling the shape of the scene: a holistic representation of the spatial envelope. IJCV\u00a042 (2004)"},{"key":"49_CR17","unstructured":"Fei-Fei, L., Perona, P.: A bayesian hierarchy model for learning natural scene categories. In: CVPR (2005)"},{"key":"49_CR18","unstructured":"Russell, B., Torralba, A., Liu, C., Fergus, R., Freeman, W.T.: Object recognition by scene alignment. In: NIPS (2007)"},{"key":"49_CR19","unstructured":"Boutell, M., Luo, J., Brown, C.M.: Improved semantic region labeling based on scene context. In: ICME (2005)"},{"key":"49_CR20","doi-asserted-by":"crossref","unstructured":"Li, L.J., Fei-Fei, L.: What, where and who? classifying events by scene and object recognition. In: ICCV (2007)","DOI":"10.1109\/ICCV.2007.4408872"},{"key":"49_CR21","doi-asserted-by":"crossref","unstructured":"Marszalek, M., Laptev, I., Schmid, C.: Actions in context. In: CVPR (2009)","DOI":"10.1109\/CVPRW.2009.5206557"},{"key":"49_CR22","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1016\/0022-247X(71)90184-3","volume":"33","author":"G. Kimeldorf","year":"1971","unstructured":"Kimeldorf, G., Wahba, G.: Some results on tchebychefan spline functions. Journal of Mathematical Analysis and Applications\u00a033, 82\u201395 (1971)","journal-title":"Journal of Mathematical Analysis and Applications"},{"key":"49_CR23","doi-asserted-by":"crossref","unstructured":"Yu, H., Hsieh, C., Chang, K., Lin, C.: Large linear classification when data cannot fit in memory. In: Proceedings of ACM SIGKDD, pp. 833\u2013842. ACM (2010)","DOI":"10.1145\/1835804.1835910"},{"key":"49_CR24","doi-asserted-by":"crossref","unstructured":"Xiao, J., Haysy, J., Ehinger, K.A., Oliva, A., Torralba, A.: Sun database: Large-scale scene recognition from abbey to zoo. In: CVPR (2010)","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"49_CR25","unstructured":"Zhang, B., Hsu, M., Dayal, U.: K-harmonic means-a data clustering algorithm. Hewlett-Packard Labs Technical Report HPL-1999-124 (1999)"},{"key":"49_CR26","first-page":"1871","volume":"9","author":"R. Fan","year":"2008","unstructured":"Fan, R., Chang, K., Hsieh, C., Wang, X., Lin, C.: Liblinear: A library for large linear classification. The Journal of Machine Learning Research\u00a09, 1871\u20131874 (2008)","journal-title":"The Journal of Machine Learning Research"},{"key":"49_CR27","unstructured":"Cao, L., Chang, S.F., Codella, N., Cotton, C., Ellis, D., Gong, L., Hill, M., Huang, G., Kender, J., Merler, M., Mu, Y., Natseve, A., Smith, J.R.: Ibm research and columbia university trecvid-2011 multimedia event detection (med) systems. In: NIST TRECVID Workshop (2011)"},{"key":"49_CR28","doi-asserted-by":"crossref","unstructured":"Natsev, A., Naphade, M.R., Smith, J.R.: Semantic representation, search and mining of multimedia content. In: ACM KDD, pp. 641\u2013646 (2004)","DOI":"10.1145\/1014052.1014133"},{"key":"49_CR29","doi-asserted-by":"crossref","unstructured":"Merler, M., Bert Huang, L.X., Hua, G., Natsev, A.: Semantic model vectors for complex video event recognition. IEEE Transactions on Multimedia (2011)","DOI":"10.1109\/TMM.2011.2168948"},{"key":"49_CR30","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., Serre, T.: Hmdb: A large video database for human motion recognition. In: ICCV (2011)","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"49_CR31","doi-asserted-by":"crossref","unstructured":"Jhuang, H., Serre, T., Wolf, L., Poggio, T.: A biologically inspired system for action recognition. In: ICCV (2007)","DOI":"10.1109\/ICCV.2007.4408988"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2012"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-33709-3_49","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,9]],"date-time":"2025-04-09T13:58:55Z","timestamp":1744207135000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-33709-3_49"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012]]},"ISBN":["9783642337086","9783642337093"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-33709-3_49","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2012]]}}}