{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,21]],"date-time":"2026-04-21T15:34:01Z","timestamp":1776785641222,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,10,15]],"date-time":"2018-10-15T00:00:00Z","timestamp":1539561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,10,15]]},"DOI":"10.1145\/3240508.3240651","type":"proceedings-article","created":{"date-parts":[[2018,10,18]],"date-time":"2018-10-18T17:52:08Z","timestamp":1539885128000},"page":"976-983","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":55,"title":["Extractive Video Summarizer with Memory Augmented Neural Networks"],"prefix":"10.1145","author":[{"given":"Litong","family":"Feng","sequence":"first","affiliation":[{"name":"SenseTime Research, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziyin","family":"Li","sequence":"additional","affiliation":[{"name":"SenseTime Research, Shen Zhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhanghui","family":"Kuang","sequence":"additional","affiliation":[{"name":"SenseTime Research, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Zhang","sequence":"additional","affiliation":[{"name":"SenseTime Research, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Antoine Bordes Nicolas Usunier Sumit Chopra and Jason Weston. 2015. Large-scale simple question answering with memory networks. arXiv preprint arXiv:1506.02075.  Antoine Bordes Nicolas Usunier Sumit Chopra and Jason Weston. 2015. Large-scale simple question answering with memory networks. arXiv preprint arXiv:1506.02075."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298981"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"e_1_3_2_1_5_1","volume-title":"Advances in Neural Information Processing Systems","author":"Gong Boqing","year":"2014"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10584-0_33"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298928"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1141277.1141601"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_10_1","unstructured":"Po-Yao Huang Ye Yuan Zhenzhong Lan Lu Jiang and Alexander G Hauptmann. 2017. Video representation learning and latent concept mining for large-scale multi-label video classification. arXiv preprint arXiv:1707.01408.  Po-Yao Huang Ye Yuan Zhenzhong Lan Lu Jiang and Alexander G Hauptmann. 2017. Video representation learning and latent concept mining for large-scale multi-label video classification. arXiv preprint arXiv:1707.01408."},{"key":"e_1_3_2_1_11_1","unstructured":"Forrest N Iandola Song Han MatthewWMoskewicz Khalid Ashraf William J Dally and Kurt Keutzer. 2016. Squeezenet: alexnet-level accuracy with 50x fewer parameters and< 0.5 mb model size. arXiv preprint arXiv:1602.07360.  Forrest N Iandola Song Han MatthewWMoskewicz Khalid Ashraf William J Dally and Kurt Keutzer. 2016. Squeezenet: alexnet-level accuracy with 50x fewer parameters and< 0.5 mb model size. arXiv preprint arXiv:1602.07360."},{"key":"e_1_3_2_1_12_1","unstructured":"A. Bordes J. Weston S. Chopra. {n. d.} Memory networks. In 2015 ICLR.  A. Bordes J. Weston S. Chopra. {n. d.} Memory networks. In 2015 ICLR."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.image.2012.11.008"},{"key":"e_1_3_2_1_14_1","unstructured":"Zhong Ji Kailin Xiong Yanwei Pang and Xuelong Li. 2017. Video summarization with attention-based encoder-decoder networks. arXiv preprint arXiv:1708.09545.  Zhong Ji Kailin Xiong Yanwei Pang and Xuelong Li. 2017. Video summarization with attention-based encoder-decoder networks. arXiv preprint arXiv:1708.09545."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46604-0_48"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.20"},{"key":"e_1_3_2_1_17_1","unstructured":"Will Kay et al. 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950.  Will Kay et al. 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950."},{"key":"e_1_3_2_1_18_1","unstructured":"Jan Koutnik Klaus Greff Faustino Gomez and Juergen Schmidhuber. 2014. A clockwork rnn. arXiv preprint arXiv:1402.3511.  Jan Koutnik Klaus Greff Faustino Gomez and Juergen Schmidhuber. 2014. A clockwork rnn. arXiv preprint arXiv:1402.3511."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1023\/B:VISI.0000029664.99615.94"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.318"},{"key":"e_1_3_2_1_21_1","unstructured":"Robert Marich. 2013. Marketing to moviegoers: a handbook of strategies and tactics. SIU Press.  Robert Marich. 2013. Marketing to moviegoers: a handbook of strategies and tactics. SIU Press."},{"key":"e_1_3_2_1_22_1","unstructured":"Tomas Mikolov Armand Joulin Sumit Chopra Michael Mathieu and Marc'Aurelio Ranzato. 2014. Learning longer memory in recurrent neural networks. arXiv preprint arXiv:1412.7753.  Tomas Mikolov Armand Joulin Sumit Chopra Michael Mathieu and Marc'Aurelio Ranzato. 2014. Learning longer memory in recurrent neural networks. arXiv preprint arXiv:1412.7753."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00799-005-0129-9"},{"key":"e_1_3_2_1_24_1","unstructured":"Seil Na Sangho Lee Jisung Kim and Gunhee Kim. 2017. A read-write memory network for movie story understanding. arXiv preprint arXiv:1709.09345.  Seil Na Sangho Lee Jisung Kim and Gunhee Kim. 2017. A read-write memory network for movie story understanding. arXiv preprint arXiv:1709.09345."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2004.841694"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_35"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.229"},{"key":"e_1_3_2_1_28_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems 568--576.   Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems 568--576."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/1178677.1178722"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2983323.2983349"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 5179--5187","author":"Song Yale","year":"2015"},{"key":"e_1_3_2_1_32_1","unstructured":"Sainbayar Sukhbaatar Jason Weston Rob Fergus etal 2015. End-to-end memory networks. In Advances in neural information processing systems 2440-- 2448.   Sainbayar Sukhbaatar Jason Weston Rob Fergus et al. 2015. End-to-end memory networks. In Advances in neural information processing systems 2440-- 2448."},{"key":"e_1_3_2_1_33_1","unstructured":"Shuyang Sun Zhanghui Kuang Lu Sheng Wanli Ouyang and Wei Zhang. 2018. Optical flow guided feature: a fast and robust motion representation for video action recognition.  Shuyang Sun Zhanghui Kuang Lu Sheng Wanli Ouyang and Wei Zhang. 2018. Optical flow guided feature: a fast and robust motion representation for video action recognition."},{"key":"e_1_3_2_1_34_1","unstructured":"Ilya Sutskever Oriol Vinyals and Quoc V Le. 2014. Sequence to sequence learning with neural networks. In Advances in neural information processing systems 3104--3112.   Ilya Sutskever Oriol Vinyals and Quoc V Le. 2014. Sequence to sequence learning with neural networks. In Advances in neural information processing systems 3104--3112."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.501"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2013.2269186"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00126"},{"key":"e_1_3_2_1_39_1","unstructured":"2018. Youtube statistics. https:\/\/fortunelords.com\/youtube-statistics\/. (2018).  2018. Youtube statistics. https:\/\/fortunelords.com\/youtube-statistics\/. (2018)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2006.888023"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Yusseri Yusoff William J Christmas and Josef Kittler. 2000. Video shot cut detection using adaptive thresholding. In BMVC 1--10.  Yusseri Yusoff William J Christmas and Josef Kittler. 2000. Video shot cut detection using adaptive thresholding. In BMVC 1--10.","DOI":"10.5244\/C.14.37"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.120"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46478-7_47"},{"key":"e_1_3_2_1_44_1","unstructured":"Yujia Zhang Xiaodan Liang Dingwen Zhang Min Tan and Eric P Xing. 2018. Unsupervised object-level video summarization with online motion autoencoder. arXiv preprint arXiv:1801.00543.  Yujia Zhang Xiaodan Liang Dingwen Zhang Min Tan and Eric P Xing. 2018. Unsupervised object-level video summarization with online motion autoencoder. arXiv preprint arXiv:1801.00543."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123328"},{"key":"e_1_3_2_1_46_1","volume-title":"The Thirty-Second AAAI Conference on Artificial Intelligence.","author":"Zhou Kaiyang","year":"2018"}],"event":{"name":"MM '18: ACM Multimedia Conference","location":"Seoul Republic of Korea","acronym":"MM '18","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 26th ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240651","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3240508.3240651","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:43:31Z","timestamp":1750207411000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240651"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,15]]},"references-count":46,"alternative-id":["10.1145\/3240508.3240651","10.1145\/3240508"],"URL":"https:\/\/doi.org\/10.1145\/3240508.3240651","relation":{},"subject":[],"published":{"date-parts":[[2018,10,15]]},"assertion":[{"value":"2018-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}