{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T10:38:43Z","timestamp":1776681523536,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":18,"publisher":"ACM","license":[{"start":{"date-parts":[[2013,10,21]],"date-time":"2013-10-21T00:00:00Z","timestamp":1382313600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2013,10,21]]},"DOI":"10.1145\/2502081.2502215","type":"proceedings-article","created":{"date-parts":[[2013,10,22]],"date-time":"2013-10-22T13:42:56Z","timestamp":1382449376000},"page":"1055-1058","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Learning representations for affective video understanding"],"prefix":"10.1145","author":[{"given":"Esra","family":"Acar","sequence":"first","affiliation":[{"name":"DAI Laboratory, Technische Universitat Berlin, Berlin, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2013,10,21]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.5555\/539444"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.5555\/1818719.1819229"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/1816041.1816074"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2012.68"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1631272.1631357"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2010.2051871"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.59"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2011.15"},{"key":"e_1_3_2_1_9_1","volume-title":"Proc. Int. Conf. Data Mining and Applications","author":"Li T.","year":"2010","unstructured":"T. Li , A. B. Chan , and A. Chun . Automatic musical pattern feature extraction using convolutional neural network . In Proc. Int. Conf. Data Mining and Applications , 2010 . T. Li, A. B. Chan, and A. Chun. Automatic musical pattern feature extraction using convolutional neural network. In Proc. Int. Conf. Data Mining and Applications, 2010."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5946961"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1511\/2001.4.344"},{"key":"e_1_3_2_1_12_1","first-page":"325","volume-title":"Feature learning in dynamic environments: Modeling the acoustic structure of musical emotion","author":"Schmidt E. M.","year":"2012","unstructured":"E. M. Schmidt , J. Scott , and Y. E. Kim . Feature learning in dynamic environments: Modeling the acoustic structure of musical emotion . In International Society for Music Information Retrieval , pages 325 -- 330 , 2012 . E. M. Schmidt, J. Scott, and Y. E. Kim. Feature learning in dynamic environments: Modeling the acoustic structure of musical emotion. In International Society for Music Information Retrieval, pages 325--330, 2012."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISM.2008.14"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6288052"},{"key":"e_1_3_2_1_15_1","first-page":"145","volume-title":"3rd International Conference on Computer Vision Theory and Applications. VISAPP","volume":"2","author":"Wimmer M.","year":"2008","unstructured":"M. Wimmer , B. Schuller , D. Arsic , G. Rigoll , and B. Radig . Low-level fusion of audio and video feature for multi-modal emotion recognition . In 3rd International Conference on Computer Vision Theory and Applications. VISAPP , volume 2 , pages 145 -- 151 , 2008 . M. Wimmer, B. Schuller, D. Arsic, G. Rigoll, and B. Radig. Low-level fusion of audio and video feature for multi-modal emotion recognition. In 3rd International Conference on Computer Vision Theory and Applications. VISAPP, volume 2, pages 145--151, 2008."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1459359.1459457"},{"key":"e_1_3_2_1_17_1","volume-title":"Multimedia Tools and Applications","author":"Xu M.","year":"2012","unstructured":"M. Xu , J. Wang , X. He , J. S. Jin , S. Luo , and H. Lu . A three-level framework for affective content analysis and its case studies . Multimedia Tools and Applications , 2012 . M. Xu, J. Wang, X. He, J. S. Jin, S. Luo, and H. Lu. A three-level framework for affective content analysis and its case studies. Multimedia Tools and Applications, 2012."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2072529.2072532"}],"event":{"name":"MM '13: ACM Multimedia Conference","location":"Barcelona Spain","acronym":"MM '13","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 21st ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2502081.2502215","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2502081.2502215","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T07:28:22Z","timestamp":1750231702000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2502081.2502215"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,10,21]]},"references-count":18,"alternative-id":["10.1145\/2502081.2502215","10.1145\/2502081"],"URL":"https:\/\/doi.org\/10.1145\/2502081.2502215","relation":{},"subject":[],"published":{"date-parts":[[2013,10,21]]},"assertion":[{"value":"2013-10-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}