{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T01:07:15Z","timestamp":1740100035887,"version":"3.37.3"},"reference-count":18,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,7,5]],"date-time":"2021-07-05T00:00:00Z","timestamp":1625443200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,7,5]],"date-time":"2021-07-05T00:00:00Z","timestamp":1625443200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,7,5]]},"DOI":"10.1109\/icme51207.2021.9428225","type":"proceedings-article","created":{"date-parts":[[2021,6,9]],"date-time":"2021-06-09T21:14:21Z","timestamp":1623273261000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["Fusing Temporally Distributed Multi-Modal Semantic Clues for Video Question Answering"],"prefix":"10.1109","author":[{"given":"Fuwei","family":"Zhang","sequence":"first","affiliation":[{"name":"Sun Yat-sen University,School of Computer Science and Engineering, National Engineering Research Center of Digital Life,Guangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruomei","family":"Wang","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University,School of Computer Science and Engineering, National Engineering Research Center of Digital Life,Guangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Songhua","family":"Xu","sequence":"additional","affiliation":[{"name":"University of South Carolina,College of Engineering and Computing,Columbia,SC,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fan","family":"Zhou","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University,School of Computer Science and Engineering, National Engineering Research Center of Digital Life,Guangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.69"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"ref12","first-page":"1","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"2014","journal-title":"Computer Science"},{"key":"ref13","first-page":"1","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2018","journal-title":"Comput Lang"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1167"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3366710"},{"key":"ref16","first-page":"2758","article-title":"Tgifqa: Toward spatio-temporal reasoning in visual question answering","author":"jang","year":"2017","journal-title":"CVPR"},{"key":"ref17","first-page":"1","article-title":"Leveraging video descriptions to learn video question answering","author":"zeng","year":"2016","journal-title":"AAAI"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00210"},{"key":"ref4","first-page":"998","article-title":"Feature augmented memory with global attention network for videoqa","author":"jiayincai","year":"2020","journal-title":"IJCAI"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018658"},{"key":"ref6","first-page":"1204","article-title":"Compositional attention networks with two-stream fusion for video question answering","author":"yu","year":"2019","journal-title":"IEEE Transactions on Image Processing"},{"key":"ref5","first-page":"1","article-title":"Two-stream spatiotemporal compositional attention network for videoqa","author":"taiki","year":"2020","journal-title":"BMVC"},{"key":"ref8","first-page":"339","article-title":"Videos as space-time region graphs","author":"wang","year":"2018","journal-title":"ECCV"},{"key":"ref7","first-page":"1","article-title":"Mcqa: Multi-modal co-attention based network for question answering","author":"kumar","year":"2020","journal-title":"Comput Lang"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2963950"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6737"},{"key":"ref9","first-page":"1","article-title":"Graph&#x00B4; attention networks","author":"velickovi?c","year":"2018","journal-title":"ICLRE"}],"event":{"name":"2021 IEEE International Conference on Multimedia and Expo (ICME)","start":{"date-parts":[[2021,7,5]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,7,9]]}},"container-title":["2021 IEEE International Conference on Multimedia and Expo (ICME)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9428049\/9428068\/09428225.pdf?arnumber=9428225","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,27]],"date-time":"2022-06-27T21:27:32Z","timestamp":1656365252000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9428225\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,5]]},"references-count":18,"URL":"https:\/\/doi.org\/10.1109\/icme51207.2021.9428225","relation":{},"subject":[],"published":{"date-parts":[[2021,7,5]]}}}