{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:57:16Z","timestamp":1784231836533,"version":"3.55.0"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030585945","type":"print"},{"value":"9783030585952","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-58595-2_11","type":"book-chapter","created":{"date-parts":[[2020,11,20]],"date-time":"2020-11-20T03:29:28Z","timestamp":1605842968000},"page":"167-183","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":54,"title":["Global-and-Local Relative Position Embedding for Unsupervised Video Summarization"],"prefix":"10.1007","author":[{"given":"Yunjae","family":"Jung","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Donghyeon","family":"Cho","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sanghyun","family":"Woo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"In So","family":"Kweon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,11,20]]},"reference":[{"key":"11_CR1","doi-asserted-by":"crossref","unstructured":"Dai, Z., Yang, Z., Yang, Y., Carbonell, J., Le, Q.V., Salakhutdinov, R.: Transformer-XL: attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860 (2019)","DOI":"10.18653\/v1\/P19-1285"},{"key":"11_CR2","doi-asserted-by":"crossref","unstructured":"De Avila, S.E.F., Lopes, A.P.B., da Luz Jr, A., de Albuquerque Ara\u00fajo, A.: VSUMM: a mechanism designed to produce static video summaries and a novel evaluation method. Pattern Recogn. Lett. 32(1), 56\u201368 (2011)","DOI":"10.1016\/j.patrec.2010.08.004"},{"key":"11_CR3","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"11_CR4","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"11_CR5","unstructured":"Gong, B., Chao, W.L., Grauman, K., Sha, F.: Diverse sequential subset selection for supervised video summarization. In: Proceedings of Neural Information Processing Systems (NeurIPS), pp. 2069\u20132077 (2014)"},{"key":"11_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"505","DOI":"10.1007\/978-3-319-10584-0_33","volume-title":"Computer Vision \u2013 ECCV 2014","author":"M Gygli","year":"2014","unstructured":"Gygli, M., Grabner, H., Riemenschneider, H., Van Gool, L.: Creating summaries from user videos. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8695, pp. 505\u2013520. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10584-0_33"},{"key":"11_CR7","doi-asserted-by":"crossref","unstructured":"Gygli, M., Grabner, H., Van Gool, L.: Video summarization by learning submodular mixtures of objectives. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 3090\u20133098 (2015)","DOI":"10.1109\/CVPR.2015.7298928"},{"issue":"4","key":"11_CR8","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1145\/2766954","volume":"34","author":"N Joshi","year":"2015","unstructured":"Joshi, N., Kienzle, W., Toelle, M., Uyttendaele, M., Cohen, M.F.: Real-time hyperlapse creation via optimal frame selection. ACM Trans. Graph. (TOG) 34(4), 63 (2015)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"11_CR9","doi-asserted-by":"crossref","unstructured":"Jung, Y., Cho, D., Kim, D., Woo, S., Kweon, I.S.: Discriminative feature learning for unsupervised video summarization. In: Proceedings of Association for the Advancement of Artificial Intelligence (AAAI), vol. 33, pp. 8537\u20138544 (2019)","DOI":"10.1609\/aaai.v33i01.33018537"},{"key":"11_CR10","doi-asserted-by":"crossref","unstructured":"Kang, H.W., Matsushita, Y., Tang, X., Chen, X.Q.: Space-time video montage. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), vol. 2, pp. 1331\u20131338. IEEE (2006)","DOI":"10.1109\/CVPR.2006.284"},{"issue":"3","key":"11_CR11","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1093\/biomet\/33.3.239","volume":"33","author":"MG Kendall","year":"1945","unstructured":"Kendall, M.G.: The treatment of ties in ranking problems. Biometrika 33(3), 239\u2013251 (1945)","journal-title":"Biometrika"},{"key":"11_CR12","doi-asserted-by":"crossref","unstructured":"Khosla, A., Hamid, R., Lin, C.J., Sundaresan, N.: Large-scale video summarization using web-image priors. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 2698\u20132705 (2013)","DOI":"10.1109\/CVPR.2013.348"},{"key":"11_CR13","doi-asserted-by":"crossref","unstructured":"Kim, G., Xing, E.P.: Reconstructing storyline graphs for image recommendation from web community photos. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 3882\u20133889 (2014)","DOI":"10.1109\/CVPR.2014.496"},{"key":"11_CR14","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. In: Proceedings of International Conference on Learning Representations (ICLR) (2015)"},{"issue":"4","key":"11_CR15","doi-asserted-by":"publisher","first-page":"78","DOI":"10.1145\/2601097.2601195","volume":"33","author":"J Kopf","year":"2014","unstructured":"Kopf, J., Cohen, M.F., Szeliski, R.: First-person hyper-lapse videos. ACM Trans. Graph. (TOG) 33(4), 78 (2014)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"11_CR16","doi-asserted-by":"crossref","unstructured":"Lee, Y.J., Ghosh, J., Grauman, K.: Discovering important people and objects for egocentric video summarization. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 1346\u20131353. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6247820"},{"issue":"12","key":"11_CR17","doi-asserted-by":"publisher","first-page":"2178","DOI":"10.1109\/TPAMI.2010.31","volume":"32","author":"D Liu","year":"2010","unstructured":"Liu, D., Hua, G., Chen, T.: A hierarchical visual model for video object summarization. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) 32(12), 2178\u20132190 (2010)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI)"},{"key":"11_CR18","doi-asserted-by":"crossref","unstructured":"Lu, Z., Grauman, K.: Story-driven summarization for egocentric video. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 2714\u20132721 (2013)","DOI":"10.1109\/CVPR.2013.350"},{"key":"11_CR19","doi-asserted-by":"crossref","unstructured":"Mahasseni, B., Lam, M., Todorovic, S.: Unsupervised video summarization with adversarial LSTM networks. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), vol. 1 (2017)","DOI":"10.1109\/CVPR.2017.318"},{"key":"11_CR20","doi-asserted-by":"crossref","unstructured":"Ngo, C.W., Ma, Y.F., Zhang, H.J.: Automatic video summarization by graph modeling. In: Ninth IEEE International Conference on Computer Vision 2003, Proceedings, pp. 104\u2013109. IEEE (2003)","DOI":"10.1109\/ICCV.2003.1238320"},{"key":"11_CR21","doi-asserted-by":"crossref","unstructured":"Otani, M., Nakashima, Y., Rahtu, E., Heikkila, J.: Rethinking the evaluation of video summaries. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 7596\u20137604 (2019)","DOI":"10.1109\/CVPR.2019.00778"},{"key":"11_CR22","unstructured":"Paszke, A., et al.: Automatic differentiation in PyTorch. In: Proceedings of Neural Information Processing Systems Workshop (NIPS-W) (2017)"},{"key":"11_CR23","doi-asserted-by":"crossref","unstructured":"Poleg, Y., Halperin, T., Arora, C., Peleg, S.: EgoSampling: fast-forward and stereo for egocentric videos. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 4768\u20134776 (2015)","DOI":"10.1109\/CVPR.2015.7299109"},{"key":"11_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"540","DOI":"10.1007\/978-3-319-10599-4_35","volume-title":"Computer Vision \u2013 ECCV 2014","author":"D Potapov","year":"2014","unstructured":"Potapov, D., Douze, M., Harchaoui, Z., Schmid, C.: Category-specific video summarization. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8694, pp. 540\u2013555. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10599-4_35"},{"issue":"11","key":"11_CR25","doi-asserted-by":"publisher","first-page":"1971","DOI":"10.1109\/TPAMI.2008.29","volume":"30","author":"Y Pritch","year":"2008","unstructured":"Pritch, Y., Rav-Acha, A., Peleg, S.: Nonchronological video synopsis and indexing. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) 30(11), 1971\u20131984 (2008)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI)"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"Rochan, M., Wang, Y.: Video summarization by learning from unpaired data. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 7902\u20137911 (2019)","DOI":"10.1109\/CVPR.2019.00809"},{"key":"11_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"358","DOI":"10.1007\/978-3-030-01258-8_22","volume-title":"Computer Vision \u2013 ECCV 2018","author":"M Rochan","year":"2018","unstructured":"Rochan, M., Ye, L., Wang, Y.: Video summarization using fully convolutional sequence networks. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11216, pp. 358\u2013374. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01258-8_22"},{"key":"11_CR28","doi-asserted-by":"crossref","unstructured":"Sharghi, A., Laurel, J.S., Gong, B.: Query-focused video summarization: dataset, evaluation, and a memory network based approach. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 2127\u20132136 (2017)","DOI":"10.1109\/CVPR.2017.229"},{"key":"11_CR29","doi-asserted-by":"crossref","unstructured":"Shaw, P., Uszkoreit, J., Vaswani, A.: Self-attention with relative position representations. Proceedings of North American Chapter of the Association for Computational Linguistics (2018)","DOI":"10.18653\/v1\/N18-2074"},{"key":"11_CR30","doi-asserted-by":"crossref","unstructured":"Song, Y., Vallmitjana, J., Stent, A., Jaimes, A.: TVSum: summarizing web videos using titles. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 5179\u20135187 (2015)","DOI":"10.1109\/CVPR.2015.7299154"},{"key":"11_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"472","DOI":"10.1007\/978-3-319-10584-0_31","volume-title":"Computer Vision \u2013 ECCV 2014","author":"M Sun","year":"2014","unstructured":"Sun, M., Farhadi, A., Taskar, B., Seitz, S.: Salient montages from unconstrained videos. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8695, pp. 472\u2013488. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10584-0_31"},{"key":"11_CR32","doi-asserted-by":"crossref","unstructured":"Szegedy, C., et al.: Going deeper with convolutions. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 1\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"11_CR33","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of Neural Information Processing Systems (NeurIPS), pp. 5998\u20136008 (2017)"},{"key":"11_CR34","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., He, K.: Non-local neural networks. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 7794\u20137803 (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"11_CR35","doi-asserted-by":"crossref","unstructured":"Wei, H., Ni, B., Yan, Y., Yu, H., Yang, X., Yao, C.: Video summarization via semantic attended networks. In: Proceedings of Association for the Advancement of Artificial Intelligence (AAAI) (2018)","DOI":"10.1609\/aaai.v32i1.11297"},{"key":"11_CR36","doi-asserted-by":"crossref","unstructured":"Yang, H., Wang, B., Lin, S., Wipf, D., Guo, M., Guo, B.: Unsupervised extraction of video highlights via robust recurrent auto-encoders. In: Proceedings of International Conference on Computer Vision (ICCV), pp. 4633\u20134641 (2015)","DOI":"10.1109\/ICCV.2015.526"},{"key":"11_CR37","unstructured":"Yang, Z., Dai, Z., Yang, Y., Carbonell, J., Salakhutdinov, R.R., Le, Q.V.: XLNet: generalized autoregressive pretraining for language understanding. In: Advances in Neural Information Processing Systems, pp. 5754\u20135764 (2019)"},{"key":"11_CR38","unstructured":"Zhang, H., Goodfellow, I., Metaxas, D., Odena, A.: Self-attention generative adversarial networks. arXiv preprint arXiv:1805.08318 (2018)"},{"key":"11_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, K., Chao, W.L., Sha, F., Grauman, K.: Summary transfer: exemplar-based subset selection for video summarization. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 1059\u20131067 (2016)","DOI":"10.1109\/CVPR.2016.120"},{"key":"11_CR40","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"766","DOI":"10.1007\/978-3-319-46478-7_47","volume-title":"Computer Vision \u2013 ECCV 2016","author":"K Zhang","year":"2016","unstructured":"Zhang, K., Chao, W.-L., Sha, F., Grauman, K.: Video summarization with long short-term memory. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9911, pp. 766\u2013782. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46478-7_47"},{"key":"11_CR41","unstructured":"Zhang, Y., Li, K., Li, K., Zhong, B., Fu, Y.: Residual non-local attention networks for image restoration. In: Proceedings of International Conference on Learning Representations (ICLR) (2019)"},{"key":"11_CR42","doi-asserted-by":"crossref","unstructured":"Zhao, B., Li, X., Lu, X.: Hierarchical recurrent neural network for video summarization. In: Proceedings of Multimedia Conference (MM), pp. 863\u2013871. ACM (2017)","DOI":"10.1145\/3123266.3123328"},{"key":"11_CR43","doi-asserted-by":"crossref","unstructured":"Zhao, B., Li, X., Lu, X.: HSA-RNN: hierarchical structure-adaptive RNN for video summarization. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), pp. 7405\u20137414 (2018)","DOI":"10.1109\/CVPR.2018.00773"},{"key":"11_CR44","doi-asserted-by":"crossref","unstructured":"Zhou, K., Qiao, Y.: Deep reinforcement learning for unsupervised video summarization with diversity-representativeness reward. In: Proceedings of Association for the Advancement of Artificial Intelligence (AAAI) (2018)","DOI":"10.1609\/aaai.v32i1.12255"},{"key":"11_CR45","doi-asserted-by":"publisher","DOI":"10.1201\/9780367802417","volume-title":"CRC Standard Probability and Statistics Tables and Formulae","author":"D Zwillinger","year":"1999","unstructured":"Zwillinger, D., Kokoska, S.: CRC Standard Probability and Statistics Tables and Formulae. CRC Press, Boca Raton (1999)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2020"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-58595-2_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T00:05:20Z","timestamp":1732061120000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-58595-2_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030585945","9783030585952"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-58595-2_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"20 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Glasgow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2020.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OpenReview","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5025","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1360","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"27% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic. From the ECCV Workshops 249 full papers, 18 short papers, and 21 further contributions were published out of a total of 467 submissions.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}