{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:34:59Z","timestamp":1750221299233,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":21,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,10,15]],"date-time":"2018-10-15T00:00:00Z","timestamp":1539561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Research Foundation of Korea (NRF) funded by the Ministry of Science ICT & Future Planning","award":["2018R1C1B 6001223"],"award-info":[{"award-number":["2018R1C1B 6001223"]}]},{"name":"National Research Foundation of Korea (NRF) funded by the Ministry of Science ICT","award":["NRF-2017M3C4A7069369"],"award-info":[{"award-number":["NRF-2017M3C4A7069369"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,10,15]]},"DOI":"10.1145\/3265987.3265991","type":"proceedings-article","created":{"date-parts":[[2018,10,17]],"date-time":"2018-10-17T12:18:31Z","timestamp":1539778711000},"page":"35-39","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Video Understanding via Convolutional Temporal Pooling Network and Multimodal Feature Fusion"],"prefix":"10.1145","author":[{"given":"Heeseung","family":"Kwon","sequence":"first","affiliation":[{"name":"POSTECH, Pohang, Rebublic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Suha","family":"Kwak","sequence":"additional","affiliation":[{"name":"POSTECH, Pohang, Rebublic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minsu","family":"Cho","sequence":"additional","affiliation":[{"name":"POSTECH, Pohang, Rebublic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,10,15]]},"reference":[{"unstructured":"Sami Abu-El-Haija Nisarg Kothari Joonseok Lee Paul Natsev George Toderici Balakrishnan Varadarajan and Sudheendra Vijayanarasimhan. 2016. Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675 (2016).  Sami Abu-El-Haija Nisarg Kothari Joonseok Lee Paul Natsev George Toderici Balakrishnan Varadarajan and Sudheendra Vijayanarasimhan. 2016. Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675 (2016).","key":"e_1_3_2_1_1_1"},{"volume-title":"NetVLAD: CNN Architecture for Weakly Supervised Place Recognition. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","year":"2016","author":"Arandjelovic Relja","key":"e_1_3_2_1_2_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_3_1","DOI":"10.1145\/1390156.1390177"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_4_1","DOI":"10.1007\/11744047_33"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_5_1","DOI":"10.1109\/CVPR.2009.5206848"},{"unstructured":"Ali Diba Vivek Sharma and Luc Van Gool. 2016. Deep temporal linear encoding networks. arXiv preprint arXiv:1611.06678 (2016).  Ali Diba Vivek Sharma and Luc Van Gool. 2016. Deep temporal linear encoding networks. arXiv preprint arXiv:1611.06678 (2016).","key":"e_1_3_2_1_6_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_7_1","DOI":"10.1109\/CVPR.2015.7298878"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_8_1","DOI":"10.1162\/neco.1997.9.8.1735"},{"volume-title":"International Conference on Machine Learning. 448--456","year":"2015","author":"Ioffe Sergey","key":"e_1_3_2_1_9_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_10_1","DOI":"10.1109\/TPAMI.2012.59"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_11_1","DOI":"10.1162\/neco.1994.6.2.181"},{"volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","year":"2014","author":"Kingma Diederik P","key":"e_1_3_2_1_12_1"},{"volume-title":"Deep Local Video Feature for Action Recognition. In Computer Vision and Pattern Recognition Workshops (CVPRW), 2017 IEEE Conference on. IEEE, 1219--1225","year":"2017","author":"Lan Zhenzhong","key":"e_1_3_2_1_13_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_14_1","DOI":"10.1109\/CVPR.2008.4587756"},{"unstructured":"Antoine Miech Ivan Laptev and Josef Sivic. 2017. Learnable pooling with Context Gating for video classification. arXiv preprint arXiv:1706.06905 (2017).  Antoine Miech Ivan Laptev and Josef Sivic. 2017. Learnable pooling with Context Gating for video classification. arXiv preprint arXiv:1706.06905 (2017).","key":"e_1_3_2_1_15_1"},{"unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems. 568--576.   Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems. 568--576.","key":"e_1_3_2_1_16_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_17_1","DOI":"10.5555\/2627435.2670313"},{"volume-title":"Going Deeper With Convolutions. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","year":"2015","author":"Szegedy Christian","key":"e_1_3_2_1_18_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_19_1","DOI":"10.1109\/ICCV.2015.510"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_20_1","DOI":"10.1109\/ICCV.2013.441"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_21_1","DOI":"10.1007\/978-3-319-46484-8_2"}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"MM '18","name":"MM '18: ACM Multimedia Conference","location":"Seoul Republic of Korea"},"container-title":["Proceedings of the 1st Workshop and Challenge on Comprehensive Video Understanding in the Wild"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3265987.3265991","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3265987.3265991","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:13:15Z","timestamp":1750212795000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3265987.3265991"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,15]]},"references-count":21,"alternative-id":["10.1145\/3265987.3265991","10.1145\/3265987"],"URL":"https:\/\/doi.org\/10.1145\/3265987.3265991","relation":{},"subject":[],"published":{"date-parts":[[2018,10,15]]},"assertion":[{"value":"2018-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}