{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,20]],"date-time":"2026-03-20T13:49:15Z","timestamp":1774014555439,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612085","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:12Z","timestamp":1698391632000},"page":"3337-3345","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Incorporating Domain Knowledge Graph into Multimodal Movie Genre Classification with Self-Supervised Attention and Contrastive Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4559-9868","authenticated-orcid":false,"given":"Jiaqi","family":"Li","sequence":"first","affiliation":[{"name":"Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0150-7236","authenticated-orcid":false,"given":"Guilin","family":"Qi","sequence":"additional","affiliation":[{"name":"Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8724-5796","authenticated-orcid":false,"given":"Chuanyi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Hohai University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8934-3920","authenticated-orcid":false,"given":"Yongrui","family":"Chen","sequence":"additional","affiliation":[{"name":"Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2431-7809","authenticated-orcid":false,"given":"Yiming","family":"Tan","sequence":"additional","affiliation":[{"name":"Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2540-1559","authenticated-orcid":false,"given":"Chenlong","family":"Xia","sequence":"additional","affiliation":[{"name":"Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0459-7567","authenticated-orcid":false,"given":"Ye","family":"Tian","sequence":"additional","affiliation":[{"name":"Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Manuel Montes-y G\u00f3mez, and Fabio A Gonz\u00e1lez","author":"Arevalo John","year":"2017","unstructured":"John Arevalo, Thamar Solorio, Manuel Montes-y G\u00f3mez, and Fabio A Gonz\u00e1lez. 2017. Gated multimodal units for information fusion. In ICLR."},{"key":"e_1_3_2_1_2_1","unstructured":"Alexei Baevski Wei-Ning Hsu Qiantong Xu Arun Babu Jiatao Gu and Michael Auli. 2022. Data2vec: A general framework for self-supervised learning in speech vision and language. In ICML."},{"key":"e_1_3_2_1_3_1","unstructured":"Junwen Bai Shufeng Kong and Carla P Gomes. 2022. Gaussian Mixture Variational Autoencoder with Contrastive Learning for Multi-Label Classification. In PMLR."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Max Bain Arsha Nagrani Andrew Brown and Andrew Zisserman. 2020. Condensed movies: Story based retrieval with contextual embeddings. In ACCV.","DOI":"10.1007\/978-3-030-69541-5_28"},{"key":"e_1_3_2_1_5_1","volume-title":"Multimodal movie genre classification using recurrent neural network","author":"Behrouzi Tina","year":"2022","unstructured":"Tina Behrouzi, Ramin Toosi, and Mohammad Ali Akhaee. 2022. Multimodal movie genre classification using recurrent neural network. Springer MULTIMED TOOLS APPL (2022)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Olfa Ben-Ahmed and Benoit Huet. 2018. Deep multimodal features for movie genre and interestingness prediction. In CBMI.","DOI":"10.1109\/CBMI.2018.8516504"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Michele Bevilacqua and Roberto Navigli. 2020. Breaking through the 80% glass ceiling: Raising the state of the art in word sense disambiguation by incorporating knowledge graph information. In ACL.","DOI":"10.18653\/v1\/2020.acl-main.255"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Leod\u00e9cio Braz Vin\u00edcius Teixeira Helio Pedrini and Zanoni Dias. 2021. Image-Text Integration Using a Multimodal Fusion Network Module for Movie Genre Classification. In ICPRS.","DOI":"10.1049\/icp.2021.1456"},{"key":"e_1_3_2_1_9_1","volume-title":"Where to look at the movies: Analyzing visual attention to understand movie editing. BEHAV RES METHODS","author":"Bruckert Alexandre","year":"2022","unstructured":"Alexandre Bruckert, Marc Christie, and Olivier Le Meur. 2022. Where to look at the movies: Analyzing visual attention to understand movie editing. BEHAV RES METHODS (2022)."},{"key":"e_1_3_2_1_10_1","volume-title":"Moviescope: Large-scale analysis of movies using multiple modalities. arXiv preprint arXiv:1908.03180","author":"Cascante Paola","year":"2019","unstructured":"Paola Cascante, Kalpathy Sitaraman, Mengjia Luo, and Vicente Ordonez. 2019. Moviescope: Large-scale analysis of movies using multiple modalities. arXiv preprint arXiv:1908.03180 (2019)."},{"key":"e_1_3_2_1_11_1","unstructured":"Richard J Chen Chengkuan Chen Yicong Li Tiffany Y Chen Andrew D Trister Rahul G Krishnan and Faisal Mahmood. 2022. Scaling vision transformers to gigapixel images via hierarchical self-supervised learning. In CVPR."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2012.01.132"},{"key":"e_1_3_2_1_13_1","volume-title":"Contrast learning visual attention for multi label classification. arXiv preprint arXiv:2107.11626","author":"Dao Son D","year":"2021","unstructured":"Son D Dao, Zhao Ethan, Phung Dinh, and Cai Jianfei. 2021. Contrast learning visual attention for multi label classification. arXiv preprint arXiv:2107.11626 (2021)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Tim Dettmers Pasquale Minervini Pontus Stenetorp and Sebastian Riedel. 2018. Convolutional 2d knowledge graph embeddings. In AAAI.","DOI":"10.1609\/aaai.v32i1.11573"},{"key":"e_1_3_2_1_15_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Ali Mert Ertugrul and Pinar Karagoz. 2018. Movie genre classification from plot summaries using bidirectional LSTM. In ICSC.","DOI":"10.1109\/ICSC.2018.00043"},{"key":"e_1_3_2_1_17_1","volume-title":"Rethinking movie genre classification with fine-grained semantic clustering. arXiv preprint arXiv:2012.02639","author":"Fish Edward","year":"2020","unstructured":"Edward Fish, Jon Weinbren, and Andrew Gilbert. 2020. Rethinking movie genre classification with fine-grained semantic clustering. arXiv preprint arXiv:2012.02639 (2020)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"John Giorgi Osvald Nitski Bo Wang and Gary Bader. 2021. DeCLUTR: Deep Contrastive Learning for Unsupervised Textual Representations. In ACL.","DOI":"10.18653\/v1\/2021.acl-long.72"},{"key":"e_1_3_2_1_19_1","unstructured":"Beliz Gunel Jingfei Du Alexis Conneau and Veselin Stoyanov. 2021. Supervised Contrastive Learning for Pre-trained Language Model Fine-tuning. In ICLR."},{"key":"e_1_3_2_1_20_1","unstructured":"Junlin Han Mehrdad Shoeiby Lars Petersson and Mohammad Ali Armin. 2021. Dual contrastive learning for unsupervised image-to-image translation. In CVPR."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Xu Han Shulin Cao Xin Lv Yankai Lin Zhiyuan Liu Maosong Sun and Juanzi Li. 2018. OpenKE: An Open Toolkit for Knowledge Embedding. In EMNLP.","DOI":"10.18653\/v1\/D18-2024"},{"key":"e_1_3_2_1_22_1","volume-title":"Learning discriminative representations for multi-label image recognition. JVCIR","author":"Hassanin Mohammed","year":"2022","unstructured":"Mohammed Hassanin, Ibrahim Radwan, Salman Khan, and Murat Tahtali. 2022. Learning discriminative representations for multi-label image recognition. JVCIR (2022)."},{"key":"e_1_3_2_1_23_1","volume-title":"Using self-supervised learning can improve model robustness and uncertainty. NIPS","author":"Hendrycks Dan","year":"2019","unstructured":"Dan Hendrycks, Mantas Mazeika, Saurav Kadavath, and Dawn Song. 2019. Using self-supervised learning can improve model robustness and uncertainty. NIPS (2019)."},{"key":"e_1_3_2_1_24_1","volume-title":"Movienet: A holistic dataset for movie understanding. In ECCV.","author":"Huang Qingqiu","year":"2020","unstructured":"Qingqiu Huang, Yu Xiong, Anyi Rao, Jiaze Wang, and Dahua Lin. 2020. Movienet: A holistic dataset for movie understanding. In ECCV."},{"key":"e_1_3_2_1_25_1","volume-title":"Long movie clip classification with state-space video models. arXiv preprint arXiv:2204.01692","author":"Islam Md Mohaiminul","year":"2022","unstructured":"Md Mohaiminul Islam and Gedas Bertasius. 2022. Long movie clip classification with state-space video models. arXiv preprint arXiv:2204.01692 (2022)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Dan Iter Kelvin Guu Larry Lansing and Dan Jurafsky. 2020. Pretraining with Contrastive Sentence Objectives Improves Discourse Performance of Language Models. In ACL.","DOI":"10.18653\/v1\/2020.acl-main.439"},{"key":"e_1_3_2_1_27_1","unstructured":"Guoliang Ji Shizhu He Liheng Xu Kang Liu and Jun Zhao. 2015. Knowledge graph embedding via dynamic mapping matrix. In IJCNLP."},{"key":"e_1_3_2_1_28_1","volume-title":"Supervised contrastive learning. NIPS","author":"Khosla Prannay","year":"2020","unstructured":"Prannay Khosla, Piotr Teterwak, Chen Wang, Aaron Sarna, Yonglong Tian, Phillip Isola, Aaron Maschinot, Ce Liu, and Dilip Krishnan. 2020. Supervised contrastive learning. NIPS (2020)."},{"key":"e_1_3_2_1_29_1","volume-title":"Supervised multimodal bitransformers for classifying images and text. arXiv preprint arXiv:1909.02950","author":"Kiela Douwe","year":"2019","unstructured":"Douwe Kiela, Suvrat Bhooshan, Hamed Firooz, Ethan Perez, and Davide Testuggine. 2019. Supervised multimodal bitransformers for classifying images and text. arXiv preprint arXiv:1909.02950 (2019)."},{"key":"e_1_3_2_1_30_1","volume-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML.","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML."},{"key":"e_1_3_2_1_31_1","unstructured":"Yankai Lin Zhiyuan Liu Maosong Sun Yang Liu and Xuan Zhu. 2015. Learning entity and relation embeddings for knowledge graph completion. In AAAI."},{"key":"e_1_3_2_1_32_1","unstructured":"Ilya Loshchilov and Frank Hutter. 2019. Decoupled weight decay regularization. In ICLR."},{"key":"e_1_3_2_1_33_1","volume-title":"Coco-lm: Correcting and contrasting text sequences for language model pretraining. NIPS","author":"Meng Yu","year":"2021","unstructured":"Yu Meng, Chenyan Xiong, Payal Bajaj, Paul Bennett, Jiawei Han, Xia Song, et al. 2021. Coco-lm: Correcting and contrasting text sequences for language model pretraining. NIPS (2021)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Ishan Misra and Laurens van der Maaten. 2020. Self-supervised learning of pretext-invariant representations. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00674"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Ishan Misra C Lawrence Zitnick and Martial Hebert. 2016. Shuffle and learn: unsupervised learning using temporal order verification. In ECCV.","DOI":"10.1007\/978-3-319-46448-0_32"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Mehdi Noroozi and Paolo Favaro. 2016. Unsupervised learning of visual representations by solving jigsaw puzzles. In ECCV.","DOI":"10.1007\/978-3-319-46466-4_5"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Sungho Park Jewook Lee Pilhyeon Lee Sunhee Hwang Dohyung Kim and Hyeran Byun. 2022. Fair contrastive learning for facial attribute classification. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01014"},{"key":"e_1_3_2_1_38_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In ICML."},{"key":"e_1_3_2_1_39_1","volume-title":"Movie description. IJCV","author":"Rohrbach Anna","year":"2017","unstructured":"Anna Rohrbach, Atousa Torabi, Marcus Rohrbach, Niket Tandon, Christopher Pal, Hugo Larochelle, Aaron Courville, and Bernt Schiele. 2017. Movie description. IJCV (2017)."},{"key":"e_1_3_2_1_40_1","unstructured":"Sethuraman Sankaran David Yang and Ser-Nam Lim. 2021. Refining Multimodal Representations using a modality-centric self-supervised module. (2021)."},{"key":"e_1_3_2_1_41_1","unstructured":"Seung Byum Seo Hyoungwook Nam and Payam Delgosha. 2022. MM-GATBT: Enriching Multimodal Representation Using Graph Attention Network. In ACL(Workshop)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"Gabriel S Sim\u00f5es J\u00f4natas Wehrmann Rodrigo C Barros and Duncan D Ruiz. 2016. Movie genre classification with convolutional neural networks. In IJCNN.","DOI":"10.1109\/IJCNN.2016.7727207"},{"key":"e_1_3_2_1_43_1","volume-title":"Rotate: Knowledge graph embedding by relational rotation in complex space. ICLR","author":"Sun Zhiqing","year":"2019","unstructured":"Zhiqing Sun, Zhi-Hong Deng, Jian-Yun Nie, and Jian Tang. 2019. Rotate: Knowledge graph embedding by relational rotation in complex space. ICLR (2019)."},{"key":"e_1_3_2_1_44_1","unstructured":"Th\u00e9o Trouillon Johannes Welbl Sebastian Riedel \u00c9ric Gaussier and Guillaume Bouchard. 2016. Complex embeddings for simple link prediction. In PMLR."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Valentin Vielzeuf Alexis Lechervy St\u00e9phane Pateux and Fr\u00e9d\u00e9ric Jurie. 2018. Centralnet: a multilayer approach for multimodal fusion. In ECCV(Workshop).","DOI":"10.1007\/978-3-030-11024-6_44"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Carl Vondrick Abhinav Shrivastava Alireza Fathi Sergio Guadarrama and Kevin Murphy. 2018. Tracking emerges by colorizing videos. In ECCV.","DOI":"10.1007\/978-3-030-01261-8_24"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Feng Wang and Huaping Liu. 2021. Understanding the behaviour of contrastive loss. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00252"},{"key":"e_1_3_2_1_48_1","unstructured":"Ran Wang Xinyu Dai et al. 2022a. Contrastive learning-enhanced nearest neighbor mechanism for multi-label text classification. In ACL."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Xiting Wang Kunpeng Liu Dongjie Wang Le Wu Yanjie Fu and Xing Xie. 2022b. Multi-level recommendation reasoning over knowledge graphs with reinforcement learning. In WWW.","DOI":"10.1145\/3485447.3512083"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Zhen Wang Jianwen Zhang Jianlin Feng and Zheng Chen. 2014. Knowledge graph embedding by translating on hyperplanes. In AAAI.","DOI":"10.1609\/aaai.v28i1.8870"},{"key":"e_1_3_2_1_51_1","volume-title":"Poster-based multiple movie genre classification using inter-channel features. Access","author":"Wi Jeong A","year":"2020","unstructured":"Jeong A Wi, Soojin Jang, and Youngbin Kim. 2020. Poster-based multiple movie genre classification using inter-channel features. Access (2020)."},{"key":"e_1_3_2_1_52_1","unstructured":"Xiao Xu Chenfei Wu Shachar Rosenman Vasudev Lal and Nan Duan. 2023. Bridge-Tower: Building Bridges Between Encoders in Vision-Language Representation Learning. In AAAI."},{"key":"e_1_3_2_1_53_1","volume-title":"A unified framework of deep networks for genre classification using movie trailer. APPL SOFT COMPUT","author":"Yadav Ashima","year":"2020","unstructured":"Ashima Yadav and Dinesh Kumar Vishwakarma. 2020. A unified framework of deep networks for genre classification using movie trailer. APPL SOFT COMPUT (2020)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Liang Yao Yin Zhang Baogang Wei Zhe Jin Rui Zhang Yangyang Zhang and Qinfei Chen. 2017. Incorporating knowledge graph embeddings into topic modeling. In AAAI.","DOI":"10.1609\/aaai.v31i1.10951"},{"key":"e_1_3_2_1_55_1","volume-title":"Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Zirui Wang, Vijay Vasudevan, Legg Yeung, Mojtaba Seyedhosseini, and Yonghui Wu. 2022. Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917 (2022)."},{"key":"e_1_3_2_1_56_1","volume-title":"Colorful image colorization","author":"Zhang Richard","unstructured":"Richard Zhang, Phillip Isola, and Alexei A Efros. 2016. Colorful image colorization. In ECCV. Springer."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"crossref","unstructured":"Shu Zhang Ran Xu Caiming Xiong and Chetan Ramaiah. 2022b. Use all the labels: A hierarchical multi-label contrastive learning framework. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01616"},{"key":"e_1_3_2_1_58_1","volume-title":"Effectively leveraging Multi-modal Features for Movie Genre Classification. arXiv preprint arXiv:2203.13281","author":"Zhang Zhongping","year":"2022","unstructured":"Zhongping Zhang, Yiwen Gu, Bryan A Plummer, Xin Miao, Jiayi Liu, and Huayan Wang. 2022a. Effectively leveraging Multi-modal Features for Movie Genre Classification. arXiv preprint arXiv:2203.13281 (2022)."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612085","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612085","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:05:09Z","timestamp":1755821109000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612085"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":58,"alternative-id":["10.1145\/3581783.3612085","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612085","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}