{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T23:48:46Z","timestamp":1771026526021,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":44,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819658084","type":"print"},{"value":"9789819658091","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-5809-1_10","type":"book-chapter","created":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T18:34:04Z","timestamp":1745606044000},"page":"173-191","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Consensus-Aware Balance Learning for\u00a0Sexually Suggestive Video Classification"],"prefix":"10.1007","author":[{"given":"Di","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiahui","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haiying","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Matthew","family":"Burns","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,26]]},"reference":[{"key":"10_CR1","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: Proceeding of International Conference on Machine Learning, vol.\u00a02, pp.\u00a0813\u2013824 (2021)"},{"key":"10_CR2","unstructured":"Carreira, J., Noland, E., Banki-Horvath, A., Hillier, C., Zisserman, A.: A short note about kinetics-600. arXiv: Computer Vision and Pattern Recognition (2018)"},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Collins, R.L., Strasburger, V.C., Brown, J.D., Donnerstein, E., Lenhart, A., Ward, L.M.: Sexual media and childhood well-being and health. Pediatrics 140(Supplement_2), S162\u2013S166 (2017)","DOI":"10.1542\/peds.2016-1758X"},{"issue":"1","key":"10_CR5","first-page":"1052344","volume":"2024","author":"E Dastbaravardeh","year":"2024","unstructured":"Dastbaravardeh, E., Askarpour, S., Saberi Anari, M., Rezaee, K.: Channel attention-based approach with autoencoder network for human action recognition in low-resolution frames. Int. J. Intell. Syst. 2024(1), 1052344 (2024)","journal-title":"Int. J. Intell. Syst."},{"key":"10_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Deselaers, T., Pimenidis, L., Ney, H.: Bag-of-visual-words models for adult image classification and filtering. In: 2008 19th International Conference on Pattern Recognition, pp. 1\u20134 (2008)","DOI":"10.1109\/ICPR.2008.4761366"},{"key":"10_CR8","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North, pp. 4171\u20134186 (2019)"},{"key":"10_CR9","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv: Computer Vision and Pattern Recognition, pp. 1\u201322 (2020)"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Fleck, M.M., Forsyth, D.A., Bregler, C.: Finding naked people. In: The European Conference on Computer Vision, pp. 593\u2013602 (1996)","DOI":"10.1007\/3-540-61123-1_173"},{"issue":"5","key":"10_CR12","doi-asserted-by":"publisher","first-page":"739","DOI":"10.1177\/15248399211031536","volume":"23","author":"LR Fowler","year":"2022","unstructured":"Fowler, L.R., Schoen, L., Smith, H.S., Morain, S.R.: Sex education on tiktok: a content analysis of themes. Health Promot. Pract. 23(5), 739\u2013742 (2022)","journal-title":"Health Promot. Pract."},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Ganguly, D., Mofrad, M.H., Kovashka, A.: Detecting sexually provocative images. In: 2017 IEEE Winter Conference on Applications of Computer Vision, vol.\u00a03, pp. 660\u2013668 (2017)","DOI":"10.1109\/WACV.2017.79"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Garcia, M.B., Revano, T.F., Habal, B.G.M., Contreras, J.O., Enriquez, J.B.R.: A pornographic image and video filtering application using optimized nudity recognition and detection algorithm. In: 2018 IEEE 10th International Conference on Humanoid, Nanotechnology, Information Technology, Communication and Control, Environment and Management, vol.\u00a0521, pp. 1\u20135 (2018)","DOI":"10.1109\/HNICEM.2018.8666227"},{"key":"10_CR15","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., et al.: Audio set: an ontology and human-labeled dataset for audio events. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 776\u2013780. IEEE (2017)","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"10_CR16","doi-asserted-by":"crossref","unstructured":"George, E., Surdeanu, M.: It\u2019s not sexually suggestive; it\u2019s educative| separating sex education from suggestive content on tiktok videos. In: Findings of the Association for Computational Linguistics: ACL 2023, pp. 5904\u20135915 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.365"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Gong, Y., Chung, Y.A., Glass, J.: AST: audio spectrogram transformer. arXiv preprint arXiv:2104.01778 (2021)","DOI":"10.21437\/Interspeech.2021-698"},{"key":"10_CR18","doi-asserted-by":"crossref","unstructured":"Goyal, R., et al.: The \u201csomething something\u201d video database for learning and evaluating visual common sense. In: International Conference on Computer Vision (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"Graves, A., Graves, A.: Long short-term memory. Supervised sequence labelling with recurrent neural networks, pp. 37\u201345 (2012)","DOI":"10.1007\/978-3-642-24797-2_4"},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Hasan, F., Raza, D.M., Moon, H., Nahid, M.A.H.: Sentiment analysis from YouTube video using bi-LSTM-GRU classification check for updates. In: Proceedings of the 2nd International Conference on Big Data, IoT and Machine Learning: BIM 2023. vol.\u00a0867, p.\u00a0303. Springer (2024)","DOI":"10.1007\/978-981-99-8937-9_21"},{"key":"10_CR21","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Lee, H., Lee, S., Nam, T.: Implementation of high performance objectionable video classification system. In: 2006 8th International Conference Advanced Communication Technology, vol.\u00a04, p. 4 pp. - 962 (2006)","DOI":"10.1109\/ICACT.2006.206131"},{"key":"10_CR23","unstructured":"Li, K., et al.: Uniformer: unified transformer for efficient spatial-temporal representation learning. In: International Conference on Learning Representations, pp. 1\u201319"},{"key":"10_CR24","doi-asserted-by":"crossref","unstructured":"Li, Y., Li, Y., Vasconcelos, N.: RESOUND: Towards Action Recognition Without Representation Bias. In: The European Conference on Computer Vision, pp. 520\u2013535 (2018)","DOI":"10.1007\/978-3-030-01231-1_32"},{"key":"10_CR25","doi-asserted-by":"crossref","unstructured":"Liu, M., Nie, L., Wang, M., Chen, B.: Towards micro-video understanding by joint sequential-sparse modeling. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 970\u2013978 (2017)","DOI":"10.1145\/3123266.3123341"},{"issue":"3","key":"10_CR26","doi-asserted-by":"publisher","first-page":"1235","DOI":"10.1109\/TIP.2018.2875363","volume":"28","author":"M Liu","year":"2019","unstructured":"Liu, M., Nie, L., Wang, X., Tian, Q., Chen, B.: Online data organizer: micro-video categorization by structure-guided multimodal dictionary learning. IEEE Trans. Image Process. 28(3), 1235\u20131247 (2019). https:\/\/doi.org\/10.1109\/TIP.2018.2875363","journal-title":"IEEE Trans. Image Process."},{"key":"10_CR27","unstructured":"Lopes, A., Avila, S., Peixoto, A., Oliveira, R., Ara\u00fajo, A.: A bag-of-features approach based on hue-sift descriptor for nude detection. European Signal Processing Conference, European Signal Processing Conference, pp. 1552\u20131556 (2009)"},{"key":"10_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109905","volume":"145","author":"Y Ma","year":"2024","unstructured":"Ma, Y., Wang, R.: Relative-position embedding based spatially and temporally decoupled transformer for action recognition. Pattern Recogn. 145, 109905 (2024)","journal-title":"Pattern Recogn."},{"key":"10_CR29","doi-asserted-by":"crossref","unstructured":"Platzer, C., Stuetz, M., Lindorfer, M.: Skin sheriff. In: Proceedings of the 2nd International Workshop on Security and Forensics in Communication Systems, pp. 45\u201356 (2014)","DOI":"10.1145\/2598918.2598920"},{"key":"10_CR30","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"10_CR31","doi-asserted-by":"crossref","unstructured":"Rea, R., Lacey, L., Dahyotit, D., Dahyot, D.: Multimodal periodicity analysis for illicit content detection in videos. In: Conference on Visual Media Production,Conference on Visual Media Production, pp. 106\u2013116 (2006)","DOI":"10.1049\/cp:20061978"},{"key":"10_CR32","doi-asserted-by":"crossref","unstructured":"Summers, R.W.: Social Psychology: How Other People Influence Our Thoughts and Actions [2 volumes]. Bloomsbury Publishing USA (2016)","DOI":"10.5040\/9798216015956"},{"key":"10_CR33","unstructured":"Tan, W., Yao, Q., Liu, J.: Overlooked video classification in weakly supervised video anomaly detection. In: Winter Conference on Applications of Computer Vision, pp. 202\u2013210"},{"key":"10_CR34","first-page":"10078","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., Wang, L.: Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Adv. Neural. Inf. Process. Syst. 35, 10078\u201310093 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR35","doi-asserted-by":"crossref","unstructured":"Ulges, A., Stahl, A.: Automatic detection of child pornography using color visual words. In: 2011 IEEE International Conference on Multimedia and Expo. vol.\u00a04, pp. 1\u20136 (2011)","DOI":"10.1109\/ICME.2011.6011977"},{"key":"10_CR36","doi-asserted-by":"crossref","unstructured":"Wang, D., Zhu, M., Yuan, X., Qian, H.: Identification and annotation of erotic film based on content analysis. In: SPIE Proceedings,Electronic Imaging and Multimedia Technology IV. vol.\u00a05637, p.\u00a088 (2005)","DOI":"10.1117\/12.577235"},{"issue":"6","key":"10_CR37","doi-asserted-by":"publisher","first-page":"1899","DOI":"10.1007\/s11263-023-01917-4","volume":"132","author":"X Wang","year":"2024","unstructured":"Wang, X., et al.: Clip-guided prototype modulating for few-shot action recognition. Int. J. Comput. Vision 132(6), 1899\u20131912 (2024)","journal-title":"Int. J. Comput. Vision"},{"key":"10_CR38","doi-asserted-by":"crossref","unstructured":"Wang, X.: HyRSM++: hybrid relation guided temporal set matching for few-shot action recognition. Pattern Recogn. 147, 110110 (2024)","DOI":"10.1016\/j.patcog.2023.110110"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Wu, P., et al.: Open-vocabulary video anomaly detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18297\u201318307 (2024)","DOI":"10.1109\/CVPR52733.2024.01732"},{"key":"10_CR40","doi-asserted-by":"crossref","unstructured":"Wu, P., et al.: VadCLIP: adapting vision-language models for weakly supervised video anomaly detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 6074\u20136082 (2024)","DOI":"10.1609\/aaai.v38i6.28423"},{"key":"10_CR41","doi-asserted-by":"crossref","unstructured":"Yang, Z., Radke, R.J.: Context-aware video anomaly detection in long-term datasets. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4002\u20134011 (2024)","DOI":"10.1109\/CVPRW63382.2024.00404"},{"key":"10_CR42","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1016\/j.neucom.2012.11.029","volume":"110","author":"J Zhang","year":"2013","unstructured":"Zhang, J., Sui, L., Zhuo, L., Li, Z., Yang, Y.: An approach of bag-of-words based on visual attention model for pornographic images recognition in compressed domain. Neurocomputing 110, 145\u2013152 (2013)","journal-title":"Neurocomputing"},{"issue":"1","key":"10_CR43","doi-asserted-by":"publisher","first-page":"5827","DOI":"10.1038\/s41598-024-56518-z","volume":"14","author":"J Zhao","year":"2024","unstructured":"Zhao, J., et al.: Sentiment analysis of video danmakus based on MIBE-RoBERTa-FF-BiLSTM. Sci. Rep. 14(1), 5827 (2024)","journal-title":"Sci. Rep."},{"key":"10_CR44","doi-asserted-by":"crossref","unstructured":"Zuo, H., Wu, O., Hu, W., Xu, B.: Recognition of blue movies by fusion of audio and video. In: 2008 IEEE International Conference on Multimedia and Expo, pp. 37\u201340 (2008)","DOI":"10.1109\/ICME.2008.4607365"}],"container-title":["Lecture Notes in Computer Science","Computational Visual Media"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-5809-1_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T18:34:30Z","timestamp":1745606070000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-5809-1_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819658084","9789819658091"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-5809-1_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"26 April 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CVM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computational Visual Media","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong SAR","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 April 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 April 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cvm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iccvm.org\/2025\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}