{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:23:04Z","timestamp":1750220584645,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,7,25]],"date-time":"2020-07-25T00:00:00Z","timestamp":1595635200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,7,25]]},"DOI":"10.1145\/3397271.3401122","type":"proceedings-article","created":{"date-parts":[[2020,7,25]],"date-time":"2020-07-25T07:50:08Z","timestamp":1595663408000},"page":"1061-1070","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["3D Self-Attention for Unsupervised Video Quantization"],"prefix":"10.1145","author":[{"given":"Jingkuan","family":"Song","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, ChengDu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruimin","family":"Lang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, ChengDu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaosu","family":"Zhu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, ChengDu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, ChengDu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lianli","family":"Gao","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, ChengDu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Heng Tao","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, ChengDu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,7,25]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.572"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.124"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-25446-8_4"},{"key":"e_1_3_2_2_4_1","volume-title":"Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473","author":"Bahdanau Dzmitry","year":"2014","unstructured":"Dzmitry Bahdanau , Kyunghyun Cho , and Yoshua Bengio . 2014. Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473 ( 2014 ). Dzmitry Bahdanau, Kyunghyun Cho, and Yoshua Bengio. 2014. Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473 (2014)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01196"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298862"},{"key":"e_1_3_2_2_9_1","unstructured":"Christoph Feichtenhofer Axel Pinz and Richard Wildes. 2016. Spatiotemporal residual networks for video action recognition. In Advances in neural information processing systems. 3468--3476.  Christoph Feichtenhofer Axel Pinz and Richard Wildes. 2016. Spatiotemporal residual networks for video action recognition. In Advances in neural information processing systems. 3468--3476."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/102"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.379"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.193"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/MASSP.1984.1162229"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_15_1","volume-title":"Product quantization for nearest neighbor search","author":"Jegou Herve","year":"2010","unstructured":"Herve Jegou , Matthijs Douze , and Cordelia Schmid . 2010. Product quantization for nearest neighbor search . IEEE transactions on pattern analysis and machine intelligence, Vol. 33 , 1 ( 2010 ), 117--128. Herve Jegou, Matthijs Douze, and Cordelia Schmid. 2010. Product quantization for nearest neighbor search. IEEE transactions on pattern analysis and machine intelligence, Vol. 33, 1 (2010), 117--128."},{"key":"e_1_3_2_2_16_1","volume-title":"3D convolutional neural networks for human action recognition","author":"Ji Shuiwang","year":"2012","unstructured":"Shuiwang Ji , Wei Xu , Ming Yang , and Kai Yu. 2012. 3D convolutional neural networks for human action recognition . IEEE transactions on pattern analysis and machine intelligence, Vol. 35 , 1 ( 2012 ), 221--231. Shuiwang Ji, Wei Xu, Ming Yang, and Kai Yu. 2012. 3D convolutional neural networks for human action recognition. IEEE transactions on pattern analysis and machine intelligence, Vol. 35, 1 (2012), 221--231."},{"key":"e_1_3_2_2_17_1","volume-title":"Exploiting feature and class relationships in video categorization with regularized deep neural networks","author":"Jiang Yu-Gang","year":"2017","unstructured":"Yu-Gang Jiang , Zuxuan Wu , Jun Wang , Xiangyang Xue , and Shih-Fu Chang . 2017. Exploiting feature and class relationships in video categorization with regularized deep neural networks . IEEE transactions on pattern analysis and machine intelligence, Vol. 40 , 2 ( 2017 ), 352--364. Yu-Gang Jiang, Zuxuan Wu, Jun Wang, Xiangyang Xue, and Shih-Fu Chang. 2017. Exploiting feature and class relationships in video categorization with regularized deep neural networks. IEEE transactions on pattern analysis and machine intelligence, Vol. 40, 2 (2017), 352--364."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.223"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00518"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3132847.3133030"},{"key":"e_1_3_2_2_21_1","volume-title":"2019 a. Unsupervised Variational Video Hashing with 1D-CNN-LSTM Networks","author":"Li Shuyan","year":"2019","unstructured":"Shuyan Li , Zhixiang Chen , Xiu Li , Jiwen Lu , and Jie Zhou . 2019 a. Unsupervised Variational Video Hashing with 1D-CNN-LSTM Networks . IEEE Transactions on Multimedia ( 2019 ). Shuyan Li, Zhixiang Chen, Xiu Li, Jiwen Lu, and Jie Zhou. 2019 a. Unsupervised Variational Video Hashing with 1D-CNN-LSTM Networks. IEEE Transactions on Multimedia (2019)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350971"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018658"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.02.034"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_9"},{"key":"e_1_3_2_2_26_1","volume-title":"Stacked quantizers for compositional vector compression. arXiv preprint arXiv:1411.2173","author":"Martinez Julieta","year":"2014","unstructured":"Julieta Martinez , Holger H Hoos , and James J Little . 2014. Stacked quantizers for compositional vector compression. arXiv preprint arXiv:1411.2173 ( 2014 ). Julieta Martinez, Holger H Hoos, and James J Little. 2014. Stacked quantizers for compositional vector compression. arXiv preprint arXiv:1411.2173 (2014)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00313"},{"key":"e_1_3_2_2_28_1","volume-title":"TRECVID 2012-an overview of the goals, tasks, data, evaluation mechanisms and metrics.","author":"Over Paul","year":"2013","unstructured":"Paul Over , George Awad , Martial Michel , Jonathan Fiscus , Greg Sanders , Barbara Shaw , Wessel Kraaij , Alan F Smeaton , and Georges Qu\u00e9ot . 2013 . TRECVID 2012-an overview of the goals, tasks, data, evaluation mechanisms and metrics. (2013). Paul Over, George Awad, Martial Michel, Jonathan Fiscus, Greg Sanders, Barbara Shaw, Wessel Kraaij, Alan F Smeaton, and Georges Qu\u00e9ot. 2013. TRECVID 2012-an overview of the goals, tasks, data, evaluation mechanisms and metrics. (2013)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.5201\/ipol.2013.26"},{"key":"e_1_3_2_2_30_1","volume-title":"et almbox","author":"Russakovsky Olga","year":"2015","unstructured":"Olga Russakovsky , Jia Deng , Hao Su , Jonathan Krause , Sanjeev Satheesh , Sean Ma , Zhiheng Huang , Andrej Karpathy , Aditya Khosla , Michael Bernstein , et almbox . 2015 . Imagenet large scale visual recognition challenge. International journal of computer vision, Vol. 115 , 3 (2015), 211--252. Olga Russakovsky, Jia Deng, Hao Su, Jonathan Krause, Sanjeev Satheesh, Sean Ma, Zhiheng Huang, Andrej Karpathy, Aditya Khosla, Michael Bernstein, et almbox. 2015. Imagenet large scale visual recognition challenge. International journal of computer vision, Vol. 115, 3 (2015), 211--252."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2015.2405340"},{"key":"e_1_3_2_2_32_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014a. Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems. 568--576.  Karen Simonyan and Andrew Zisserman. 2014a. Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems. 568--576."},{"key":"e_1_3_2_2_33_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014b. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 ( 2014 ). Karen Simonyan and Andrew Zisserman. 2014b. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2072298.2072354"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2814344"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/128"},{"key":"e_1_3_2_2_37_1","volume-title":"The new data and new challenges in multimedia research. arXiv preprint arXiv:1503.01817","author":"Thomee Bart","year":"2015","unstructured":"Bart Thomee , David A Shamma , Gerald Friedland , Benjamin Elizalde , Karl Ni , Douglas Poland , Damian Borth , and Li-Jia Li. 2015. The new data and new challenges in multimedia research. arXiv preprint arXiv:1503.01817 , Vol. 1 , 8 ( 2015 ). Bart Thomee, David A Shamma, Gerald Friedland, Benjamin Elizalde, Karl Ni, Douglas Poland, Damian Borth, and Li-Jia Li. 2015. The new data and new challenges in multimedia research. arXiv preprint arXiv:1503.01817, Vol. 1, 8 (2015)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_2_39_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299059"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_2_42_1","volume-title":"International conference on machine learning. 2048--2057","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu , Jimmy Ba , Ryan Kiros , Kyunghyun Cho , Aaron Courville , Ruslan Salakhudinov , Rich Zemel , and Yoshua Bengio . 2015 . Show, attend and tell: Neural image caption generation with visual attention . In International conference on machine learning. 2048--2057 . Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron Courville, Ruslan Salakhudinov, Rich Zemel, and Yoshua Bengio. 2015. Show, attend and tell: Neural image caption generation with visual attention. In International conference on machine learning. 2048--2057."},{"key":"e_1_3_2_2_43_1","volume-title":"Generative Adversarial Product Quantisation. In 2018 ACM Multimedia Conference on Multimedia Conference. ACM, 861--869","author":"Yu Litao","year":"2018","unstructured":"Litao Yu , Yongsheng Gao , and Jun Zhou . 2018 a. Generative Adversarial Product Quantisation. In 2018 ACM Multimedia Conference on Multimedia Conference. ACM, 861--869 . Litao Yu, Yongsheng Gao, and Jun Zhou. 2018a. Generative Adversarial Product Quantisation. In 2018 ACM Multimedia Conference on Multimedia Conference. ACM, 861--869."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_12"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964308"},{"key":"e_1_3_2_2_46_1","first-page":"3","article-title":"Composite Quantization for Approximate Nearest Neighbor Search","volume":"2","author":"Zhang Ting","year":"2014","unstructured":"Ting Zhang , Chao Du , and Jingdong Wang . 2014 . Composite Quantization for Approximate Nearest Neighbor Search .. In ICML , Vol. 2. 3 . Ting Zhang, Chao Du, and Jingdong Wang. 2014. Composite Quantization for Approximate Nearest Neighbor Search.. In ICML, Vol. 2. 3.","journal-title":"ICML"},{"key":"e_1_3_2_2_47_1","volume-title":"Robust joint graph sparse coding for unsupervised spectral feature selection","author":"Zhu Xiaofeng","year":"2016","unstructured":"Xiaofeng Zhu , Xuelong Li , Shichao Zhang , Chunhua Ju , and Xindong Wu. 2016. Robust joint graph sparse coding for unsupervised spectral feature selection . IEEE transactions on neural networks and learning systems, Vol. 28 , 6 ( 2016 ), 1263--1275. Xiaofeng Zhu, Xuelong Li, Shichao Zhang, Chunhua Ju, and Xindong Wu. 2016. Robust joint graph sparse coding for unsupervised spectral feature selection. IEEE transactions on neural networks and learning systems, Vol. 28, 6 (2016), 1263--1275."}],"event":{"name":"SIGIR '20: The 43rd International ACM SIGIR conference on research and development in Information Retrieval","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Virtual Event China","acronym":"SIGIR '20"},"container-title":["Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3397271.3401122","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3397271.3401122","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:31:38Z","timestamp":1750195898000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3397271.3401122"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,7,25]]},"references-count":47,"alternative-id":["10.1145\/3397271.3401122","10.1145\/3397271"],"URL":"https:\/\/doi.org\/10.1145\/3397271.3401122","relation":{},"subject":[],"published":{"date-parts":[[2020,7,25]]},"assertion":[{"value":"2020-07-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}