{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T17:10:10Z","timestamp":1755882610632,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,26]],"date-time":"2024-04-26T00:00:00Z","timestamp":1714089600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,26]]},"DOI":"10.1145\/3663976.3664228","type":"proceedings-article","created":{"date-parts":[[2024,6,27]],"date-time":"2024-06-27T18:25:58Z","timestamp":1719512758000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["The effect of using video title in attention-based video summarization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2767-3336","authenticated-orcid":false,"given":"Changwei","family":"Li","sequence":"first","affiliation":[{"name":"University of Cincinnati, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8522-6937","authenticated-orcid":false,"given":"Zhiting","family":"Yeh","sequence":"additional","affiliation":[{"name":"University of Cincinnati, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1725-6398","authenticated-orcid":false,"given":"Jeshmitha","family":"Gunuganti","sequence":"additional","affiliation":[{"name":"University of Cincinnati, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7269-4975","authenticated-orcid":false,"given":"Jiabin","family":"Chang","sequence":"additional","affiliation":[{"name":"University of Cincinnati, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1701-2668","authenticated-orcid":false,"given":"Mehdi","family":"Norouzi","sequence":"additional","affiliation":[{"name":"University of Cincinnati, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,6,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2021.3117472"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_3_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, and Sylvain Gelly. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_4_1","volume-title":"Asian Conference on Computer Vision. Springer, 39\u201354","author":"Fajtl Jiri","year":"2018","unstructured":"Jiri Fajtl, Hajar\u00a0Sadeghi Sokeh, Vasileios Argyriou, Dorothy Monekosso, and Paolo Remagnino. 2018. Summarizing videos with attention. In Asian Conference on Computer Vision. Springer, 39\u201354."},{"key":"e_1_3_2_1_5_1","volume-title":"Supervised Video Summarization Via Multiple Feature Sets with Parallel Attention. In 2021 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1\u20136s.","author":"Ghauri Junaid\u00a0Ahmed","year":"2021","unstructured":"Junaid\u00a0Ahmed Ghauri, Sherzod Hakimov, and Ralph Ewerth. 2021. Supervised Video Summarization Via Multiple Feature Sets with Parallel Attention. In 2021 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1\u20136s."},{"key":"e_1_3_2_1_6_1","volume-title":"Diverse sequential subset selection for supervised video summarization. Advances in neural information processing systems 27","author":"Gong Boqing","year":"2014","unstructured":"Boqing Gong, Wei-Lun Chao, Kristen Grauman, and Fei Sha. 2014. Diverse sequential subset selection for supervised video summarization. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_2_1_7_1","volume-title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. arXiv preprint arXiv:2203.15086","author":"Gorti Satya\u00a0Krishna","year":"2022","unstructured":"Satya\u00a0Krishna Gorti, No\u00ebl Vouitsis, Junwei Ma, Keyvan Golestan, Maksims Volkovs, Animesh Garg, and Guangwei Yu. 2022. X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. arXiv preprint arXiv:2203.15086 (2022)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10584-0_33"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372278.3390695"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2019.2929228"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2904996"},{"key":"e_1_3_2_1_13_1","volume-title":"Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey\u00a0E. Hinton. 2012. Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25 (2012)."},{"key":"e_1_3_2_1_14_1","volume-title":"Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860","author":"Luo Huaishao","year":"2021","unstructured":"Huaishao Luo, Lei Ji, Ming Zhong, Yang Chen, Wen Lei, Nan Duan, and Tianrui Li. 2021. Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860 (2021)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.318"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00778"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_35"},{"key":"e_1_3_2_1_18_1","volume-title":"International Conference on Machine Learning. PMLR, 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, and Jack Clark. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_2_1_19_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. 5179\u20135187","author":"Song Yale","year":"2015","unstructured":"Yale Song, Jordi Vallmitjana, Amanda Stent, and Alejandro Jaimes. 2015. Tvsum: Summarizing web videos using titles. In Proceedings of the IEEE conference on computer vision and pattern recognition. 5179\u20135187."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","unstructured":"C. Sujatha and U. Mudenagudi. 2011. A Study on Keyframe Extraction Methods for Video Summary. In - 2011 International Conference on Computational Intelligence and Communication Networks. 73\u201377. https:\/\/doi.org\/10.1109\/CICN.2011.15 ID: 1.","DOI":"10.1109\/CICN.2011.15"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_23_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N. Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46478-7_47"}],"event":{"name":"CVIPPR 2024: 2024 2nd Asia Conference on Computer Vision, Image Processing and Pattern Recognition","acronym":"CVIPPR 2024","location":"Xiamen China"},"container-title":["Proceedings of the 2024 2nd Asia Conference on Computer Vision, Image Processing and Pattern Recognition"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3663976.3664228","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3663976.3664228","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T16:29:10Z","timestamp":1755880150000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3663976.3664228"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,26]]},"references-count":24,"alternative-id":["10.1145\/3663976.3664228","10.1145\/3663976"],"URL":"https:\/\/doi.org\/10.1145\/3663976.3664228","relation":{},"subject":[],"published":{"date-parts":[[2024,4,26]]},"assertion":[{"value":"2024-06-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}