{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T16:22:09Z","timestamp":1761582129486,"version":"3.37.3"},"reference-count":40,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61631016,61901421"],"award-info":[{"award-number":["61631016,61901421"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,7,18]]},"DOI":"10.1109\/ijcnn52387.2021.9533473","type":"proceedings-article","created":{"date-parts":[[2021,9,20]],"date-time":"2021-09-20T21:27:41Z","timestamp":1632173261000},"page":"1-8","source":"Crossref","is-referenced-by-count":5,"title":["Context-Aware Based Visual-Audio Feature Fusion for Emotion Recognition"],"prefix":"10.1109","author":[{"given":"Huijie","family":"Cheng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yun","family":"Tie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Qi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cong","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"SentiBank large-scale ontology and classifiers for detecting sentiment and emotions in visual content","year":"2013","author":"borth","key":"ref39"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/2911996.2912006"},{"key":"ref33","article-title":"Faster R-CNN: towards real-time object detection with region proposal networks","author":"ren","year":"2015","journal-title":"Neural Information Processing Systems"},{"key":"ref32","first-page":"4489","article-title":"Learning spatiotemporal features with 3D convolutional networks","author":"tran","year":"2015","journal-title":"Proceedings of the IEEE International Conference on Computer Vision"},{"key":"ref31","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"2015","journal-title":"ICLRE"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123328"},{"journal-title":"Emotion in the Human Face","year":"1982","author":"ekman","key":"ref37"},{"key":"ref36","volume":"3","author":"plutchik","year":"1986","journal-title":"Emotion Theory Research and Experience"},{"key":"ref35","article-title":"Gated multimodal units for information fusion","author":"john","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"ref10","first-page":"4502","article-title":"Interaction networks for learning about objects, relations and physics","author":"battaglia","year":"2016","journal-title":"NIPS"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2482228"},{"journal-title":"Affective Computing","year":"1997","author":"picard","key":"ref11"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.212"},{"key":"ref13","article-title":"Hybrid attention based multimodal network for spoken language classification","author":"gu","year":"0","journal-title":"Proceedings of the Conference Association for Computational Linguistics"},{"key":"ref14","article-title":"Predicting emotions in user-generated videos","author":"jiang","year":"0","journal-title":"Proceedings of the Twenty-Eighth AAAI Conference on Artificial Intelligence"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.344"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"ref17","article-title":"Relational inductive biases, deep learning, and graph networks","author":"battaglia","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref18","article-title":"Semi-supervised classification with graph convolutional networks","author":"kipf","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref19","article-title":"Graph attention networks","author":"velickovic","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46478-7_47"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2922129"},{"key":"ref27","first-page":"1388","article-title":"Sparse modeling for topic-oriented video summarization","author":"rameswar","year":"0","journal-title":"Acoustics Speech and Signal Processing (ICASSP) 2017 IEEE International Conference on"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.282"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i01.5364"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00773"},{"journal-title":"Visual and audio aware bi-modal video emotion recognition","year":"2016","author":"xiang","key":"ref5"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967196"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1081"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.01024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2016.2622690"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/2857546.2857642"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00730"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2019.00034"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00288"},{"key":"ref24","article-title":"Discovering important people and objects for egocentric video summarization","author":"lee","year":"2012","journal-title":"In CVPR"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10584-0_33"},{"key":"ref25","article-title":"Automatic video summarization by graph modeling","author":"ngo","year":"2003","journal-title":"In ICCV"}],"event":{"name":"2021 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2021,7,18]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,7,22]]}},"container-title":["2021 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9533266\/9533267\/09533473.pdf?arnumber=9533473","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:46:00Z","timestamp":1652197560000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9533473\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,18]]},"references-count":40,"URL":"https:\/\/doi.org\/10.1109\/ijcnn52387.2021.9533473","relation":{},"subject":[],"published":{"date-parts":[[2021,7,18]]}}}