{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:04:14Z","timestamp":1750309454570,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T00:00:00Z","timestamp":1730678400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"USC Amazon Center"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,4]]},"DOI":"10.1145\/3678957.3685725","type":"proceedings-article","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T04:35:53Z","timestamp":1730262953000},"page":"124-133","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Can Text-to-image Model Assist Multi-modal Learning for Visual Recognition with Visual Modality Missing?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2053-9068","authenticated-orcid":false,"given":"Tiantian","family":"Feng","sequence":"first","affiliation":[{"name":"Signal Analysis and Interpretation Laboratory, University of Southern California, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7684-9057","authenticated-orcid":false,"given":"Daniel","family":"Yang","sequence":"additional","affiliation":[{"name":"Signal Analysis and Interpretation Lab, University of Southern California, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5281-1695","authenticated-orcid":false,"given":"Digbalay","family":"Bose","sequence":"additional","affiliation":[{"name":"Signal Analysis and Interpretation Lab, University of Southern California, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1052-6204","authenticated-orcid":false,"given":"Shrikanth","family":"Narayanan","sequence":"additional","affiliation":[{"name":"Signal Analysis and Interpretation Lab, University of Southern California, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,4]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"[1] [n. d.]. https:\/\/openai.com\/."},{"key":"e_1_3_2_2_2_1","first-page":"24206","article-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text","volume":"34","author":"Akbari Hassan","year":"2021","unstructured":"Hassan Akbari, Liangzhe Yuan, Rui Qian, Wei-Hong Chuang, Shih-Fu Chang, Yin Cui, and Boqing Gong. 2021. Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. Advances in Neural Information Processing Systems 34 (2021), 24206\u201324221.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1609\/icwsm.v12i1.14983"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403182"},{"key":"e_1_3_2_2_6_1","volume-title":"International conference on machine learning. PMLR, 1597\u20131607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In International conference on machine learning. PMLR, 1597\u20131607."},{"key":"e_1_3_2_2_7_1","volume-title":"Unifying Vision-and-Language Tasks via Text Generation. ArXiv abs\/2102.02779","author":"Cho Jaemin","year":"2021","unstructured":"Jaemin Cho, Jie Lei, Hao Tan, and Mohit Bansal. 2021. Unifying Vision-and-Language Tasks via Text Generation. ArXiv abs\/2102.02779 (2021). https:\/\/api.semanticscholar.org\/CorpusID:231802355"},{"key":"e_1_3_2_2_8_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_2_9_1","volume-title":"FedMultimodal: A Benchmark For Multimodal Federated Learning. arXiv preprint arXiv:2306.09486","author":"Feng Tiantian","year":"2023","unstructured":"Tiantian Feng, Digbalay Bose, Tuo Zhang, Rajat Hebbar, Anil Ramakrishna, Rahul Gupta, Mi Zhang, 2023. FedMultimodal: A Benchmark For Multimodal Federated Learning. arXiv preprint arXiv:2306.09486 (2023)."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_2_11_1","volume-title":"Ast: Audio spectrogram transformer. arXiv preprint arXiv:2104.01778","author":"Gong Yuan","year":"2021","unstructured":"Yuan Gong, Yu-An Chung, and James Glass. 2021. Ast: Audio spectrogram transformer. arXiv preprint arXiv:2104.01778 (2021)."},{"key":"e_1_3_2_2_12_1","volume-title":"Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839","author":"Gong Yuan","year":"2022","unstructured":"Yuan Gong, Andrew Rouditchenko, Alexander\u00a0H Liu, David Harwath, Leonid Karlinsky, Hilde Kuehne, and James Glass. 2022. Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839 (2022)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_2_14_1","volume-title":"Is synthetic data from generative models ready for image recognition?arXiv preprint arXiv:2210.07574","author":"He Ruifei","year":"2022","unstructured":"Ruifei He, Shuyang Sun, Xin Yu, Chuhui Xue, Wenqing Zhang, Philip Torr, Song Bai, and Xiaojuan Qi. 2022. Is synthetic data from generative models ready for image recognition?arXiv preprint arXiv:2210.07574 (2022)."},{"key":"e_1_3_2_2_15_1","volume-title":"International Conference on Machine Learning. PMLR, 9226\u20139259","author":"Huang Yu","year":"2022","unstructured":"Yu Huang, Junyang Lin, Chang Zhou, Hongxia Yang, and Longbo Huang. 2022. Modality competition: What makes joint training of multi-modal network fail in deep learning?(provably). In International Conference on Machine Learning. PMLR, 9226\u20139259."},{"key":"e_1_3_2_2_16_1","volume-title":"Supervised Multimodal Bitransformers for Classifying Images and Text. ArXiv abs\/1909.02950","author":"Kiela Douwe","year":"2019","unstructured":"Douwe Kiela, Suvrat Bhooshan, Hamed Firooz, and Davide Testuggine. 2019. Supervised Multimodal Bitransformers for Classifying Images and Text. ArXiv abs\/1909.02950 (2019)."},{"key":"e_1_3_2_2_17_1","volume-title":"International Conf. on Machine Learning.","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision. In International Conf. on Machine Learning."},{"key":"e_1_3_2_2_18_1","volume-title":"Deep learning. nature 521, 7553","author":"LeCun Yann","year":"2015","unstructured":"Yann LeCun, Yoshua Bengio, and Geoffrey Hinton. 2015. Deep learning. nature 521, 7553 (2015), 436\u2013444."},{"key":"e_1_3_2_2_19_1","volume-title":"Multimodal Prompting with Missing Modalities for Visual Recognition. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Lee Yi-Lun","year":"2023","unstructured":"Yi-Lun Lee, Yi-Hsuan Tsai, Wei-Chen Chiu, and Chen-Yu Lee. 2023. Multimodal Prompting with Missing Modalities for Visual Recognition. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_20_1","volume-title":"\u00a0H. Hoi","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C.\u00a0H. Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:246411402"},{"key":"e_1_3_2_2_21_1","unstructured":"Junnan Li Ramprasaath\u00a0R. Selvaraju 2021. Align before Fuse: Vision and Language Representation Learning with Momentum Distillation. In Neural Information Processing Systems."},{"key":"e_1_3_2_2_22_1","volume-title":"VisualBERT: A Simple and Performant Baseline for Vision and Language. ArXiv abs\/1908.03557","author":"Li Liunian\u00a0Harold","year":"2019","unstructured":"Liunian\u00a0Harold Li, Mark Yatskar, 2019. VisualBERT: A Simple and Performant Baseline for Vision and Language. ArXiv abs\/1908.03557 (2019). https:\/\/api.semanticscholar.org\/CorpusID:199528533"},{"key":"e_1_3_2_2_23_1","volume-title":"Foundations and recent trends in multimodal machine learning: Principles, challenges, and open questions. arXiv preprint arXiv:2209.03430","author":"Liang Paul\u00a0Pu","year":"2022","unstructured":"Paul\u00a0Pu Liang, Amir Zadeh, and Louis-Philippe Morency. 2022. Foundations and recent trends in multimodal machine learning: Principles, challenges, and open questions. arXiv preprint arXiv:2209.03430 (2022)."},{"key":"e_1_3_2_2_24_1","volume-title":"AudioLDM 2: Learning holistic audio generation with self-supervised pretraining. arXiv preprint arXiv:2308.05734","author":"Liu Haohe","year":"2023","unstructured":"Haohe Liu, Qiao Tian, Yi Yuan, 2023. AudioLDM 2: Learning holistic audio generation with self-supervised pretraining. arXiv preprint arXiv:2308.05734 (2023)."},{"key":"e_1_3_2_2_25_1","unstructured":"Jiasen Lu Dhruv Batra Devi Parikh and Stefan Lee. 2019. ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. In Neural Information Processing Systems. https:\/\/api.semanticscholar.org\/CorpusID:199453025"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01764"},{"key":"e_1_3_2_2_27_1","volume-title":"SMIL: Multimodal Learning with Severely Missing Modality. ArXiv abs\/2103.05677","author":"Ma Mengmeng","year":"2021","unstructured":"Mengmeng Ma, Jian Ren, Long Zhao, S. Tulyakov, Cathy Wu, and Xi Peng. 2021. SMIL: Multimodal Learning with Severely Missing Modality. ArXiv abs\/2103.05677 (2021). https:\/\/api.semanticscholar.org\/CorpusID:232170317"},{"key":"e_1_3_2_2_28_1","volume-title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models. arXiv preprint arXiv:2306.05424","author":"Maaz Muhammad","year":"2023","unstructured":"Muhammad Maaz, Hanoona Rasheed, Salman Khan, and Fahad\u00a0Shahbaz Khan. 2023. Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models. arXiv preprint arXiv:2306.05424 (2023)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2901464"},{"key":"e_1_3_2_2_30_1","volume-title":"Modeling Heart Rate and Activity Data for Personalized Fitness Recommendation. The World Wide Web Conference","author":"Ni Jianmo","year":"2019","unstructured":"Jianmo Ni, Larry Muhlstein, and Julian McAuley. 2019. Modeling Heart Rate and Activity Data for Personalized Fitness Recommendation. The World Wide Web Conference (2019). https:\/\/api.semanticscholar.org\/CorpusID:86422346"},{"key":"e_1_3_2_2_31_1","volume-title":"AVLEN: Audio-Visual-Language Embodied Navigation in 3D Environments. ArXiv abs\/2210.07940","author":"Paul Sudipta","year":"2022","unstructured":"Sudipta Paul, Amit\u00a0K. Roy-Chowdhury, and Anoop Cherian. 2022. AVLEN: Audio-Visual-Language Embodied Navigation in 3D Environments. ArXiv abs\/2210.07940 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252907859"},{"key":"e_1_3_2_2_32_1","volume-title":"Found in Translation: Learning Robust Joint Representations by Cyclic Translations Between Modalities. ArXiv abs\/1812.07809","author":"Pham Hai","year":"2018","unstructured":"Hai Pham, Paul\u00a0Pu Liang, Thomas Manzini, Louis-Philippe Morency, and Barnab\u00e1s P\u00f3czos. 2018. Found in Translation: Learning Robust Joint Representations by Cyclic Translations Between Modalities. ArXiv abs\/1812.07809 (2018). https:\/\/api.semanticscholar.org\/CorpusID:53500027"},{"key":"e_1_3_2_2_33_1","volume-title":"International Conf. on Machine Learning. PMLR, 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, 2021. Learning transferable visual models from natural language supervision. In International Conf. on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_2_2_34_1","volume-title":"Zero-Shot Text-to-Image Generation. ArXiv abs\/2102.12092","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, 2021. Zero-Shot Text-to-Image Generation. ArXiv abs\/2102.12092 (2021). https:\/\/api.semanticscholar.org\/CorpusID:232035663"},{"key":"e_1_3_2_2_35_1","volume-title":"High-Resolution Image Synthesis with Latent Diffusion Models. 2022 IEEE\/CVF Conf. on Computer Vision and Pattern Recognition","author":"Rombach Robin","year":"2021","unstructured":"Robin Rombach, A. Blattmann, Dominik Lorenz, Patrick Esser, and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. 2022 IEEE\/CVF Conf. on Computer Vision and Pattern Recognition (2021), 10674\u201310685."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_2_2_37_1","volume-title":"LAION-5B: An open large-scale dataset for training next generation image-text models. ArXiv abs\/2210.08402","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, 2022. LAION-5B: An open large-scale dataset for training next generation image-text models. ArXiv abs\/2210.08402 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252917726"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigData.2017.8257992"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Jordan Shipard Arnold Wiliem Kien\u00a0Nguyen Thanh Wei Xiang and Clinton Fookes. 2023. Diversity is Definitely Needed: Improving Model-Agnostic Zero-shot Classification via Stable Diffusion.","DOI":"10.1109\/CVPRW59228.2023.00084"},{"key":"e_1_3_2_2_40_1","volume-title":"UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir\u00a0Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463257"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00127"},{"key":"e_1_3_2_2_43_1","volume-title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations. ArXiv abs\/1908.08530","author":"Su Weijie","year":"2019","unstructured":"Weijie Su, Xizhou Zhu, Yue Cao, Bin Li, Lewei Lu, Furu Wei, and Jifeng Dai. 2019. VL-BERT: Pre-training of Generic Visual-Linguistic Representations. ArXiv abs\/1908.08530 (2019). https:\/\/api.semanticscholar.org\/CorpusID:201317624"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_2_2_45_1","volume-title":"Any-to-Any Generation via Composable Diffusion. arXiv preprint arXiv:2305.11846","author":"Tang Zineng","year":"2023","unstructured":"Zineng Tang, Ziyi Yang, Chenguang Zhu, Michael Zeng, and Mohit Bansal. 2023. Any-to-Any Generation via Composable Diffusion. arXiv preprint arXiv:2305.11846 (2023)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.528"},{"key":"e_1_3_2_2_47_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141\u00a0ukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems Vol.\u00a030."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-57959-7"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01524"},{"key":"e_1_3_2_2_50_1","volume-title":"International Conf. on Machine Learning.","author":"Wang Peng","year":"2022","unstructured":"Peng Wang, An Yang, Rui Men, Junyang Lin, Shuai Bai, 2022. OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework. In International Conf. on Machine Learning."},{"key":"e_1_3_2_2_51_1","volume-title":"What Makes Training Multi-Modal Classification Networks Hard?2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Wang Weiyao","year":"2019","unstructured":"Weiyao Wang, Du Tran, and Matt Feiszli. 2019. What Makes Training Multi-Modal Classification Networks Hard?2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019), 12692\u201312702. https:\/\/api.semanticscholar.org\/CorpusID:213979682"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW.2015.7169757"},{"key":"e_1_3_2_2_53_1","volume-title":"Internvid: A large-scale video-text dataset for multimodal understanding and generation. arXiv preprint arXiv:2307.06942","author":"Wang Yi","year":"2023","unstructured":"Yi Wang, Yinan He, Yizhuo Li, Kunchang Li, Jiashuo Yu, 2023. Internvid: A large-scale video-text dataset for multimodal understanding and generation. arXiv preprint arXiv:2307.06942 (2023)."},{"key":"e_1_3_2_2_54_1","volume-title":"Image and Video. In International Conf. on Machine Learning.","author":"Xu Haiyang","year":"2023","unstructured":"Haiyang Xu, Qinghao Ye, Mingshi Yan, Yaya Shi, Jiabo Ye, Yuanhong Xu, 2023. mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video. In International Conf. on Machine Learning."},{"key":"e_1_3_2_2_55_1","volume-title":"Multimodal learning with transformers: A survey","author":"Xu Peng","year":"2023","unstructured":"Peng Xu, Xiatian Zhu, and David\u00a0A Clifton. 2023. Multimodal learning with transformers: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)."},{"key":"e_1_3_2_2_56_1","volume-title":"Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Zirui Wang, Vijay Vasudevan, Legg Yeung, Mojtaba Seyedhosseini, and Yonghui Wu. 2022. Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917 (2022)."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475585"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3532064"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539388"},{"key":"e_1_3_2_2_60_1","volume-title":"Gpt-fl: Generative pre-trained model-assisted federated learning. arXiv preprint arXiv:2306.02210","author":"Zhang Tuo","year":"2023","unstructured":"Tuo Zhang, Tiantian Feng, Samiul Alam, Mi Zhang, Shrikanth\u00a0S Narayanan, and Salman Avestimehr. 2023. Gpt-fl: Generative pre-trained model-assisted federated learning. arXiv preprint arXiv:2306.02210 (2023)."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"crossref","unstructured":"Jinming Zhao Ruichen Li and Qin Jin. 2021. Missing Modality Imagination Network for Emotion Recognition with Uncertain Missing Modalities. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:236459819","DOI":"10.18653\/v1\/2021.acl-long.203"}],"event":{"name":"ICMI '24: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERACTION","acronym":"ICMI '24","location":"San Jose Costa Rica"},"container-title":["International Conference on Multimodel Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3678957.3685725","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3678957.3685725","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:12Z","timestamp":1750295412000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3678957.3685725"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,4]]},"references-count":61,"alternative-id":["10.1145\/3678957.3685725","10.1145\/3678957"],"URL":"https:\/\/doi.org\/10.1145\/3678957.3685725","relation":{},"subject":[],"published":{"date-parts":[[2024,11,4]]},"assertion":[{"value":"2024-11-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}