{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:11:04Z","timestamp":1784178664077,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,8]],"date-time":"2024-10-08T00:00:00Z","timestamp":1728345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100002428","name":"Austrian Science Fund","doi-asserted-by":"publisher","award":["10.55776\/P33526, 10.55776\/DFH23, 10.55776\/P36413, 10.55776\/COE12"],"award-info":[{"award-number":["10.55776\/P33526, 10.55776\/DFH23, 10.55776\/P36413, 10.55776\/COE12"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100002428","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,8]]},"DOI":"10.1145\/3640457.3688138","type":"proceedings-article","created":{"date-parts":[[2024,10,8]],"date-time":"2024-10-08T15:39:28Z","timestamp":1728401968000},"page":"380-390","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":21,"title":["A Multimodal Single-Branch Embedding Network for Recommendation in Cold-Start and Missing Modality Scenarios"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1850-2626","authenticated-orcid":false,"given":"Christian","family":"Ganh\u00f6r","sequence":"first","affiliation":[{"name":"Institute of Computational Perception, Johannes Kepler University Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5541-4919","authenticated-orcid":false,"given":"Marta","family":"Moscati","sequence":"additional","affiliation":[{"name":"Institute of Computational Perception, Johannes Kepler University Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5562-681X","authenticated-orcid":false,"given":"Anna","family":"Hausberger","sequence":"additional","affiliation":[{"name":"Institute of Computational Perception, Johannes Kepler University Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7715-4409","authenticated-orcid":false,"given":"Shah","family":"Nawaz","sequence":"additional","affiliation":[{"name":"Institute of Computational Perception, Johannes Kepler University Linz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1706-3406","authenticated-orcid":false,"given":"Markus","family":"Schedl","sequence":"additional","affiliation":[{"name":"Institute of Computational Perception, Johannes Kepler University Linz, Austria and Human-centered AI Group, AI Lab, Linz Institute of Technology, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,8]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00061"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3298689.3347038"},{"key":"e_1_3_2_1_4_1","volume-title":"Proc. of HR Workshop at ACM RecSys","author":"Behar Eric","year":"2023","unstructured":"Eric Behar, Julien Romero, Amel Bouzeghoub, and Katarzyna Wegrzyn-Wolska. 2023. Tackling Cold Start for Job Recommendation with Heterogeneous Graphs. In Proc. of HR Workshop at ACM RecSys (Singapore, Singapore)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2019.8852100"},{"key":"e_1_3_2_1_6_1","volume-title":"Proc. of WOMRAD Workshop at ACM RecSys","author":"Bogdanov Dmitry","year":"2014","unstructured":"Dmitry Bogdanov, Martin Haro, Ferdinand Fuhrmann, Emilia G\u00f3mez, and Perfecto Herrera. 2014. Content-based Music Recommendation Based on User Preference Examples. In Proc. of WOMRAD Workshop at ACM RecSys (Foster City, CA, USA). 4\u20139."},{"key":"e_1_3_2_1_7_1","volume-title":"Proc. of LML Workshop at ECML PKDD (Prague, Czech Republic). 108\u2013122","author":"Buitinck Lars","year":"2013","unstructured":"Lars Buitinck, Gilles Louppe, Mathieu Blondel, Fabian Pedregosa, Andreas Mueller, Olivier Grisel, Vlad Niculae, Peter Prettenhofer, Alexandre Gramfort, Jaques Grobler, Robert Layton, Jake VanderPlas, Arnaud Joly, Brian Holt, and Ga\u00ebl Varoquaux. 2013. API Design for Machine Learning Software: Experiences from the scikit-learn Project. In Proc. of LML Workshop at ECML PKDD (Prague, Czech Republic). 108\u2013122."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3560487"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3604915.3608806"},{"key":"e_1_3_2_1_10_1","volume-title":"Recommender Systems Leveraging Multimedia Content. Comput. Surveys 53, 5","author":"Deldjoo Yashar","year":"2020","unstructured":"Yashar Deldjoo, Markus Schedl, Paolo Cremonesi, and Gabriella Pasi. 2020. Recommender Systems Leveraging Multimedia Content. Comput. Surveys 53, 5 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Proc. of IEEE CVPR","author":"Deng J.","unstructured":"J. Deng, W. Dong, R. Socher, L.-J. Li, K. Li, and L. Fei-Fei. 2009. ImageNet: A Large-Scale Hierarchical Image Database. In Proc. of IEEE CVPR (Miami, FL, USA). 248\u2013255."},{"key":"e_1_3_2_1_12_1","volume-title":"Jukebox: A Generative Model for Music. arxiv:2005.00341","author":"Dhariwal Prafulla","year":"2020","unstructured":"Prafulla Dhariwal, Heewoo Jun, Christine Payne, Jong\u00a0Wook Kim, Alec Radford, and Ilya Sutskever. 2020. Jukebox: A Generative Model for Music. arxiv:2005.00341"},{"key":"e_1_3_2_1_13_1","volume-title":"Proc. of ISMIR","author":"Eghbal-Zadeh Hamid","year":"2015","unstructured":"Hamid Eghbal-Zadeh, Bernhard Lehner, Markus Schedl, and Gerhard Widmer. 2015. I-Vectors for Timbre-Based Music Similarity and Music Artist Classification. In Proc. of ISMIR (M\u00e1laga, Spain). 554\u2013560."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-16079-1"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614657"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3604915.3608773"},{"key":"e_1_3_2_1_18_1","article-title":"The MovieLens Datasets: History and Context","volume":"5","author":"Harper Maxwell","year":"2015","unstructured":"F.\u00a0Maxwell Harper and Joseph\u00a0A. Konstan. 2015. The MovieLens Datasets: History and Context. ACM Transactions on Interactive Intelligent Systems 5, 4 (2015).","journal-title":"ACM Transactions on Interactive Intelligent Systems"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052569"},{"key":"e_1_3_2_1_21_1","unstructured":"Yupeng Hou Jiacheng Li Zhankui He An Yan Xiusi Chen and Julian McAuley. 2024. Bridging Language and Items for Retrieval and Recommendation. arxiv:2403.03952"},{"key":"e_1_3_2_1_22_1","unstructured":"Christina Humer Marc Streit and Hendrik Strobelt. 2023. Amumo (Analyze Multi-Modal Models). https:\/\/github.com\/ginihumer\/Amumo"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614965"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01435"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446911"},{"key":"e_1_3_2_1_26_1","volume-title":"Proc. of NeurIPS","author":"Liang Weixin","year":"2022","unstructured":"Weixin Liang, Yuhui Zhang, Yongchan Kwon, Serena Yeung, and James Zou. 2022. Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning. In Proc. of NeurIPS (New Orleans, LA, USA). 17612\u201317625."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00628"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615283"},{"key":"e_1_3_2_1_29_1","volume-title":"Proc. of ISMIR (Virtual). 157\u2013165","author":"Liu Meijun","year":"2020","unstructured":"Meijun Liu, Eva Zangerle, Xiao Hu, Alessandro Melchiorre, and Markus Schedl. 2020. Pandemics, Music, and Collective Sentiment: Evidence from the Outbreak of Covid-19. In Proc. of ISMIR (Virtual). 157\u2013165."},{"key":"e_1_3_2_1_30_1","volume-title":"ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. 32","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. 32 (2019)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i3.16330"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10618-022-00859-8"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1874254"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3523227.3546756"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2021.102666"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557656"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Cataldo Musto Marco de Gemmis Pasquale Lops Fedelucio Narducci and Giovanni Semeraro. 2022. Semantics and Content-Based Recommendations. In Recommender Systems Handbook Francesco Ricci Lior Rokach and Bracha Shapira (Eds.). 251\u2013298.","DOI":"10.1007\/978-1-0716-2197-4_7"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_5"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00879"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM51629.2021.00154"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1018"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10844-022-00698-5"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/IWSSIP48289.2020.9145170"},{"key":"e_1_3_2_1_44_1","volume-title":"Proc. of ISMIR","author":"Pons Jordi","year":"2018","unstructured":"Jordi Pons, Oriol Nieto, Matthew Prockup, Erik\u00a0M. Schmidt, Andreas\u00a0F. Ehmann, and Xavier Serra. 2018. End-to-end Learning for Music Audio Tagging at Scale. In Proc. of ISMIR (Paris, France). 37\u2013644."},{"key":"e_1_3_2_1_45_1","volume-title":"Proc. of ISMIR LBD","author":"Pons Jordi","year":"2019","unstructured":"Jordi Pons and Xavier Serra. 2019. musicnn: Pre-trained Convolutional Neural Networks for Music Audio Tagging. In Proc. of ISMIR LBD (Delft, Netherlands)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460231.3478847"},{"key":"e_1_3_2_1_47_1","volume-title":"C","author":"Kiran","year":"2020","unstructured":"Kiran R, Pradeep Kumar, and Bharat Bhasker. 2020. DNNRec: A Novel Deep Learning Based Hybrid Recommender System. Expert Systems with Applicationse 144, C (2020)."},{"key":"e_1_3_2_1_48_1","volume-title":"Proc. of ICML (Virtual). 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proc. of ICML (Virtual). 8748\u20138763."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460231.3474228"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.13"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_52_1","volume-title":"Proc. of UAI (Montreal","author":"Rendle Steffen","year":"2009","unstructured":"Steffen Rendle, Christoph Freudenthaler, Zeno Gantner, and Lars Schmidt-Thieme. 2009. BPR: Bayesian Personalized Ranking from Implicit Feedback. In Proc. of UAI (Montreal, Quebec, Canada). 452\u2013461."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Francesco Ricci Lior Rokach and Bracha Shapira (Eds.). 2022. Recommender Systems Handbook. Springer US New York NY.","DOI":"10.1007\/978-1-0716-2197-4"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747704"},{"key":"e_1_3_2_1_55_1","volume-title":"Proc","author":"Saeed Muhammad\u00a0Saad","unstructured":"Muhammad\u00a0Saad Saeed, Shah Nawaz, Muhammad\u00a0Haris Khan, Muhammad\u00a0Zaigham Zaheer, Karthik Nandakumar, Muhammad\u00a0Haroon Yousaf, and Arif Mahmood. 2023. Single-branch Network for Multimodal Training. In Proc. of ICASSP (Rhodes Island, Greece). 1\u20135."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3604915.3608845"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13735-018-0154-2"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3523227.3551477"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_2_1_60_1","unstructured":"Aaron van\u00a0den Oord Yazhe Li and Oriol Vinyals. 2019. Representation Learning with Contrastive Predictive Coding. arxiv:1807.03748"},{"key":"e_1_3_2_1_61_1","first-page":"2579","article-title":"Visualizing Data using t-SNE","volume":"9","author":"van\u00a0der Maaten Laurens","year":"2008","unstructured":"Laurens van\u00a0der Maaten and Geoffrey\u00a0E. Hinton. 2008. Visualizing Data using t-SNE. Journal of Machine Learning Research 9 (2008), 2579\u20132605.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_62_1","volume-title":"Proc. of NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141\u00a0ukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Proc. of NeurIPS (Long Beach, CA, US). 6000\u20136010."},{"key":"e_1_3_2_1_63_1","volume-title":"Proc. of NeurIPS","author":"Volkovs Maksims","year":"2017","unstructured":"Maksims Volkovs, Guangwei Yu, and Tomi Poutanen. 2017. DropoutNet: Addressing Cold Start in Recommender Systems. In Proc. of NeurIPS (Long Beach, CA, USA). 4964\u20134973."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1373"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/222"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612522"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475665"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614972"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2023.3237638"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3275156"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/447"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475259"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615074"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401178"}],"event":{"name":"RecSys '24: 18th ACM Conference on Recommender Systems","location":"Bari Italy","acronym":"RecSys '24","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGAI ACM Special Interest Group on Artificial Intelligence","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGIR ACM Special Interest Group on Information Retrieval","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["18th ACM Conference on Recommender Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640457.3688138","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3640457.3688138","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:58:32Z","timestamp":1750294712000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640457.3688138"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,8]]},"references-count":75,"alternative-id":["10.1145\/3640457.3688138","10.1145\/3640457"],"URL":"https:\/\/doi.org\/10.1145\/3640457.3688138","relation":{},"subject":[],"published":{"date-parts":[[2024,10,8]]},"assertion":[{"value":"2024-10-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}