{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:09:07Z","timestamp":1781586547802,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,30]],"date-time":"2024-05-30T00:00:00Z","timestamp":1717027200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"BMBF","award":["16KIS1517"],"award-info":[{"award-number":["16KIS1517"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,30]]},"DOI":"10.1145\/3652583.3658101","type":"proceedings-article","created":{"date-parts":[[2024,6,7]],"date-time":"2024-06-07T06:30:40Z","timestamp":1717741840000},"page":"506-514","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Identification of Speaker Roles and Situation Types in News Videos"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4354-9629","authenticated-orcid":false,"given":"Gullal S.","family":"Cheema","sequence":"first","affiliation":[{"name":"L3S Research Center, Leibniz University Hannover, Hannover, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7777-8914","authenticated-orcid":false,"given":"Judi","family":"Arafat","sequence":"additional","affiliation":[{"name":"Leibniz University Hannover, Hannover, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6902-7322","authenticated-orcid":false,"given":"Chiao-I","family":"Tseng","sequence":"additional","affiliation":[{"name":"University of Gothenburg, G\u00f6teborg, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7209-9295","authenticated-orcid":false,"given":"John A.","family":"Bateman","sequence":"additional","affiliation":[{"name":"University of Bremen, Bremen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0918-6297","authenticated-orcid":false,"given":"Ralph","family":"Ewerth","sequence":"additional","affiliation":[{"name":"TIB - Leibniz Information Centre for Science and Technology, and L3S Research Center, Leibniz University Hannover, Hannover, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6802-1241","authenticated-orcid":false,"given":"Eric","family":"M\u00fcller-Budack","sequence":"additional","affiliation":[{"name":"TIB - Leibniz Information Centre for Science and Technology, and L3S Research Center, Leibniz University Hannover, Hannover, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,6,7]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846277"},{"key":"e_1_3_2_1_2_1","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conf. on Neural Information Processing Systems 2020","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. In Advances in Neural Information Processing Systems 33: Annual Conf. on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the Seventeenth National Conference on Artificial Intelligence and Twelfth Conference on on Innovative Applications of Artificial Intelligence","author":"Barzilay Regina","year":"2000","unstructured":"Regina Barzilay, Michael Collins, Julia Hirschberg, and Steve Whittaker. 2000. The Rules Behind Roles: Identifying Speaker Role in Radio Broadcasts. In Proceedings of the Seventeenth National Conference on Artificial Intelligence and Twelfth Conference on on Innovative Applications of Artificial Intelligence, July 30 - August 3, 2000, Austin, Texas, USA. AAAI Press \/ The MIT Press, 679--684."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1515\/mc-2023-0029"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/AICCSA50499.2020.9316511"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Mohamed Lazhar Bellagha and Mounir Zrigui. 2020. Speaker Naming in TV programs Based on Speaker Role Recognition. https:\/\/github.com\/ MohamedBellagha\/Speaker-Role-Recognition.","DOI":"10.1109\/AICCSA50499.2020.9316511"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1162\/TACL_"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of SiKDD 472","author":"Brank Janez","year":"2017","unstructured":"Janez Brank, Gregor Leban, and Marko Grobelnik. 2017. Annotating documents with relevant wikipedia concepts. Proceedings of SiKDD 472 (2017)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1010933404324"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939785"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947650"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1423"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2011-370"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the Eighth International Conf. on Language Resources and Evaluation, LREC 2012","author":"Giraudel Aude","year":"2012","unstructured":"Aude Giraudel, Matthieu Carr\u00e9, Val\u00e9rie Mapelli, Juliette Kahn, Olivier Galibert, and Ludovic Quintard. 2012. The REPERE Corpus: a multimodal corpus for person recognition. In Proceedings of the Eighth International Conf. on Language Resources and Evaluation, LREC 2012, Istanbul, Turkey, May 23-25, 2012. European Language Resources Association (ELRA), 1102--1107."},{"key":"e_1_3_2_1_15_1","first-page":"121","article-title":"Looking at TV news: Strategies for research","volume":"13","author":"Griffin Michael","year":"1992","unstructured":"Michael Griffin. 1992. Looking at TV news: Strategies for research. Communication 13 (1992), 121--141.","journal-title":"Communication"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of The 12th Language Resources and Evaluation Conference, LREC 2020","author":"Guhr Oliver","year":"2020","unstructured":"Oliver Guhr, Anne-Kathrin Schumann, Frank Bahrmann, and Hans-Joachim B\u00f6hme. 2020. Training a Broad-Coverage German Sentiment Classification Model for Dialog Systems. In Proceedings of The 12th Language Resources and Evaluation Conference, LREC 2020, Marseille, France, May 11-16, 2020. European Language Resources Association, 1627--1632."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.1212303"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5494958"},{"key":"e_1_3_2_1_20_1","unstructured":"Klaus Krippendorff. 2011. Computing Krippendorff's Alpha-Reliability. (2011)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1126\/science"},{"key":"e_1_3_2_1_22_1","volume-title":"The handbook of brain theory and neural networks","author":"Lecun Yann","unstructured":"Yann Lecun and Yoshua Bengio. 1995. Convolutional networks for images, speech, and time-series. In The handbook of brain theory and neural networks. MIT Press."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.3115\/1614049.1614070"},{"key":"e_1_3_2_1_24_1","unstructured":"Simon L\u00fcdke Josephine Grau and Martin Drawitsch. 2021. News sentiment development on the example of ?Migration\". https:\/\/github.com\/text-analytics-20\/news-sentiment-development."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-demos.14"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 38th International Conf. on Machine Learning, ICML 2021","volume":"8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conf. on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event (Proceedings of Machine Learning Research, Vol. 139). PMLR, 8748--8763."},{"key":"e_1_3_2_1_27_1","volume-title":"International Conf. on Machine Learning, ICML 2023","volume":"28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust Speech Recognition via Large-Scale Weak Supervision. In International Conf. on Machine Learning, ICML 2023, 23-29 July 2023, Honolulu, Hawaii, USA (Proceedings of Machine Learning Research, Vol. 202). PMLR, 28492--28518. https:\/\/proceedings.mlr.press\/v202\/radford23a.html"},{"key":"e_1_3_2_1_28_1","volume-title":"Analysing motion picture cutting rates. Wide Screen 9, 1","author":"Redfern Nick","year":"2022","unstructured":"Nick Redfern. 2022. Analysing motion picture cutting rates. Wide Screen 9, 1 (2022)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404820"},{"key":"e_1_3_2_1_30_1","volume-title":"TransNet V2: An effective deep network architecture for fast shot transition detection. CoRR abs\/2008.04838","author":"Soucek Tom\u00e1s","year":"2020","unstructured":"Tom\u00e1s Soucek and Jakub Lokoc. 2020. TransNet V2: An effective deep network architecture for fast shot transition detection. CoRR abs\/2008.04838 (2020). arXiv:2008.04838"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.2307\/2331554"},{"key":"e_1_3_2_1_32_1","volume-title":"The MAIN model: A heuristic approach to understanding technology effects on credibility","author":"Sundar S Shyam","unstructured":"S Shyam Sundar. 2008. The MAIN model: A heuristic approach to understanding technology effects on credibility. MacArthur Foundation Digital Media and Learning Initiative Cambridge, MA."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1093\/jcmc\/zmab010"},{"key":"e_1_3_2_1_34_1","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conf. on Neural Information Processing Systems 2022","author":"Tong Zhan","year":"2022","unstructured":"Zhan Tong, Yibing Song, Jue Wang, and Limin Wang. 2022. VideoMAE: Masked Autoencoders are Data-Efficient Learners for Self-Supervised Video Pre-Training. In Advances in Neural Information Processing Systems 35: Annual Conf. on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/ 416f9cb3276121c42eebb86352a4354a-Abstract-Conference.html"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1126\/science"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947618"},{"key":"e_1_3_2_1_37_1","volume-title":"Human Language Technologies: Conference of the North American","author":"Zhang Bin","year":"2010","unstructured":"Bin Zhang, Brian Hutchinson, Wei Wu, and Mari Ostendorf. 2010. Extracting Phrase Patterns with Minimum Redundancy for Unsupervised Speaker Role Classification. In Human Language Technologies: Conference of the North American Chapter of the Association of Computational Linguistics, Proceedings, June 2-4, 2010, LA, California, USA. The Association for Computational Linguistics, 717--720."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2723009"}],"event":{"name":"ICMR '24: International Conference on Multimedia Retrieval","location":"Phuket Thailand","acronym":"ICMR '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia","SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2024 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3652583.3658101","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3652583.3658101","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T08:51:33Z","timestamp":1755766293000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3652583.3658101"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,30]]},"references-count":38,"alternative-id":["10.1145\/3652583.3658101","10.1145\/3652583"],"URL":"https:\/\/doi.org\/10.1145\/3652583.3658101","relation":{},"subject":[],"published":{"date-parts":[[2024,5,30]]},"assertion":[{"value":"2024-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}