{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:04:40Z","timestamp":1750309480000,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":82,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3685509","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"11184-11193","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Sophia-in-Audition: Virtual Production with a Robot Performer"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9388-2000","authenticated-orcid":false,"given":"Taotao","family":"Zhou","sequence":"first","affiliation":[{"name":"ShanghaiTech University &amp; LumiAni Technology, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4747-2219","authenticated-orcid":false,"given":"Teng","family":"Xu","sequence":"additional","affiliation":[{"name":"ShanghaiTech University &amp; LumiAni Technology, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4827-4077","authenticated-orcid":false,"given":"Dong","family":"Zhang","sequence":"additional","affiliation":[{"name":"ShanghaiTech University &amp; LumiAni Technology, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5767-5808","authenticated-orcid":false,"given":"Yuyang","family":"Jiao","sequence":"additional","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3489-924X","authenticated-orcid":false,"given":"Peijun","family":"Xu","sequence":"additional","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2696-8865","authenticated-orcid":false,"given":"Yaoyu","family":"He","sequence":"additional","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8807-7787","authenticated-orcid":false,"given":"Lan","family":"Xu","sequence":"additional","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8580-0036","authenticated-orcid":false,"given":"Jingyi","family":"Yu","sequence":"additional","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Luma AI. 2023. Luma. https:\/\/lumalabs.ai\/interactive-scenes"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Wes Anderson. 2014. The Grand Budapest Hotel.","DOI":"10.5040\/9780571343355-div-00000007"},{"key":"e_1_3_2_2_3_1","unstructured":"Apple. 2024. Apple ARKit. https:\/\/developer.apple.com\/augmented-reality\/arkit\/."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/116873.116880"},{"key":"e_1_3_2_2_5_1","volume-title":"Srinivasan","author":"Barron Jonathan T.","year":"2021","unstructured":"Jonathan T. Barron, Ben Mildenhall, Matthew Tancik, Peter Hedman, Ricardo Martin-Brualla, and Pratul P. Srinivasan. 2021. Mip-NeRF: A Multiscale Representation for Anti-Aliasing Neural Radiance Fields. arxiv: 2103.13415 [cs.CV]"},{"key":"e_1_3_2_2_6_1","volume-title":"Vision, Modeling, and Visualization","author":"Berger Kai","year":"2011","unstructured":"Kai Berger, Kai Ruhl, Yannic Schroeder, Christian Bruemmer, Alexander Scholz, and Marcus Magnor. 2011. Markerless Motion Capture using multiple Color-Depth Sensors. In Vision, Modeling, and Visualization (2011), Peter Eisert, Joachim Hornegger, and Konrad Polthier (Eds.). The Eurographics Association, 317--324."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3450626.3459829"},{"key":"e_1_3_2_2_8_1","volume-title":"Multi-scale capture of facial geometry and motion. ACM transactions on graphics (TOG)","author":"Bickel Bernd","year":"2007","unstructured":"Bernd Bickel, Mario Botsch, Roland Angst, Wojciech Matusik, Miguel Otaduy, Hanspeter Pfister, and Markus Gross. 2007. Multi-scale capture of facial geometry and motion. ACM transactions on graphics (TOG), Vol. 26, 3 (2007), 33--es."},{"key":"e_1_3_2_2_9_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts Varun Jampani and Robin Rombach. 2023. Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets. arxiv: 2311.15127 [cs.CV] https:\/\/arxiv.org\/abs\/2311.15127"},{"key":"e_1_3_2_2_10_1","first-page":"1","article-title":"Human digital doubles with technological cognitive thinking and adaptive behaviour","volume":"7","author":"Bryndin Evgeniy","year":"2019","unstructured":"Evgeniy Bryndin. 2019. Human digital doubles with technological cognitive thinking and adaptive behaviour. Software Engineering, Vol. 7, 1 (2019), 1--9.","journal-title":"Software Engineering"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3-031--19824--3_20"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02045"},{"key":"e_1_3_2_2_13_1","unstructured":"Alfonso Cuar\u00f3n. 2013. Gravity."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICICI-BME.2013.6698465"},{"key":"e_1_3_2_2_15_1","first-page":"1","article-title":"The light stages and their applications to photoreal digital actors","volume":"2","author":"Debevec Paul","year":"2012","unstructured":"Paul Debevec. 2012. The light stages and their applications to photoreal digital actors. SIGGRAPH Asia, Vol. 2, 4 (2012), 1--6.","journal-title":"SIGGRAPH Asia"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/344779.344855"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Paul Debevec and Chloe LeGendre. 2022. HDR Lighting Dilation for Dynamic Range Reduction on Virtual Production Stages. arxiv: 2205.07873 [cs.GR]","DOI":"10.1145\/3532719.3543243"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/566654.566614"},{"key":"e_1_3_2_2_19_1","unstructured":"Deepfakes. 2017. Deepfakes. https:\/\/github.com\/deepfakes\/faceswap."},{"volume-title":"New Visions In Performance","author":"Dixon Steve","key":"e_1_3_2_2_20_1","unstructured":"Steve Dixon. 2005. The digital double. In New Visions In Performance. Routledge, 13--30."},{"key":"e_1_3_2_2_21_1","unstructured":"Geminoid DK. 2011. Geminoid DK. https:\/\/robots.ieee.org\/robots\/geminoiddk\/."},{"key":"e_1_3_2_2_22_1","volume-title":"Relighting Human Locomotion with Flowed Reflectance Fields. In Symposium on Rendering, Tomas Akenine-Moeller and Wolfgang Heidrich (Eds.). The Eurographics Association, 76--es.","author":"Einarsson Per","year":"2006","unstructured":"Per Einarsson, Charles-Felix Chabert, Andrew Jones, Wan-Chun Ma, Bruce Lamond, Tim Hawkins, Mark Bolas, Sebastian Sylwan, and Paul Debevec. 2006. Relighting Human Locomotion with Flowed Reflectance Fields. In Symposium on Rendering, Tomas Akenine-Moeller and Wolfgang Heidrich (Eds.). The Eurographics Association, 76--es."},{"key":"e_1_3_2_2_23_1","unstructured":"Erica. 2015. Erica. https:\/\/robotsguide.com\/robots\/erica\/."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3450626.3459936","article-title":"Learning an animatable detailed 3D face model from in-the-wild images","volume":"40","author":"Feng Yao","year":"2021","unstructured":"Yao Feng, Haiwen Feng, Michael J Black, and Timo Bolkart. 2021. Learning an animatable detailed 3D face model from in-the-wild images. ACM Transactions on Graphics (ToG), Vol. 40, 4 (2021), 1--13.","journal-title":"ACM Transactions on Graphics (ToG)"},{"key":"e_1_3_2_2_25_1","unstructured":"David Fincher. 2008. The Curious Case of Benjamin Button."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00542"},{"key":"e_1_3_2_2_27_1","volume-title":"Bracketing in research: A typology. Qualitative health research","author":"Gearing Robin Edward","year":"2004","unstructured":"Robin Edward Gearing. 2004. Bracketing in research: A typology. Qualitative health research, Vol. 14, 10 (2004), 1429--1452."},{"key":"e_1_3_2_2_28_1","unstructured":"Gourieff. 2023. ReActor for Stable Diffusion. https:\/\/github.com\/Gourieff\/sd-webui-reactor."},{"key":"e_1_3_2_2_29_1","volume-title":"AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning. arXiv preprint arXiv:2307.04725","author":"Guo Yuwei","year":"2023","unstructured":"Yuwei Guo, Ceyuan Yang, Anyi Rao, Yaohui Wang, Yu Qiao, Dahua Lin, and Bo Dai. 2023. AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning. arXiv preprint arXiv:2307.04725 (2023)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/584993.585053"},{"key":"e_1_3_2_2_31_1","volume-title":"VIDIT: Virtual image dataset for illumination transfer. arXiv preprint arXiv:2005.05460","author":"Helou Majed El","year":"2020","unstructured":"Majed El Helou, Ruofan Zhou, Johan Barthas, and Sabine S\u00fcsstrunk. 2020. VIDIT: Virtual image dataset for illumination transfer. arXiv preprint arXiv:2005.05460 (2020)."},{"key":"e_1_3_2_2_32_1","unstructured":"HRP-4C. 2009. HRP-4C. https:\/\/robots.ieee.org\/robots\/hrp4c\/."},{"key":"e_1_3_2_2_33_1","volume-title":"Humannorm: Learning normal diffusion model for high-quality and realistic 3d human generation. arXiv preprint arXiv:2310.01406","author":"Huang Xin","year":"2023","unstructured":"Xin Huang, Ruizhi Shao, Qi Zhang, Hongwen Zhang, Ying Feng, Yebin Liu, and Qing Wang. 2023. Humannorm: Learning normal diffusion model for high-quality and realistic 3d human generation. arXiv preprint arXiv:2310.01406 (2023)."},{"key":"e_1_3_2_2_34_1","volume-title":"Ebsynth: Fast Example-based Image Synthesis and Style Transfer. https:\/\/github.com\/jamriska\/ebsynth.","author":"Jamriska Ondrej","year":"2018","unstructured":"Ondrej Jamriska. 2018. Ebsynth: Fast Example-based Image Synthesis and Style Transfer. https:\/\/github.com\/jamriska\/ebsynth."},{"key":"e_1_3_2_2_35_1","unstructured":"Joe Johnston and George Lucas. 2019. The Mandalorians."},{"key":"e_1_3_2_2_36_1","unstructured":"Gerard Johnstone. 2022. M3GAN."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/1185657.1185857"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1186\/s12984-017-0270-x"},{"key":"e_1_3_2_2_40_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Kuang Zhengfei","year":"2024","unstructured":"Zhengfei Kuang, Yunzhi Zhang, Hong-Xing Yu, Samir Agarwala, Elliott Wu, Jiajun Wu, et al. 2024. Stanford-ORB: A Real-World 3D Object Inverse Rendering Benchmark. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_41_1","unstructured":"Yorgos Lanthimos. 2023. Poor Things."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925934"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-005-0291-5"},{"key":"e_1_3_2_2_44_1","volume-title":"Pavel Tokmakov, Sergey Zakharov, and Carl Vondrick.","author":"Liu Ruoshi","year":"2023","unstructured":"Ruoshi Liu, Rundi Wu, Basile Van Hoorick, Pavel Tokmakov, Sergey Zakharov, and Carl Vondrick. 2023. Zero-1-to-3: Zero-shot One Image to 3D Object. arxiv: 2303.11328 [cs.CV]"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3528223.3530127"},{"key":"e_1_3_2_2_47_1","volume-title":"Evaluation of 3D markerless motion capture accuracy using OpenPose with multiple video cameras. Frontiers in sports and active living","author":"Nakano Nobuyasu","year":"2020","unstructured":"Nobuyasu Nakano, Tetsuro Sakura, Kazuhiro Ueda, Leon Omura, Arata Kimura, Yoichi Iino, Senshi Fukashiro, and Shinsuke Yoshioka. 2020. Evaluation of 3D markerless motion capture accuracy using OpenPose with multiple video cameras. Frontiers in sports and active living, Vol. 2 (2020), 50."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2006.281935"},{"key":"e_1_3_2_2_49_1","unstructured":"OpenAI. 2023. ChatGPT: Optimizing Language Models for Dialogue. https:\/\/openai.com\/chatgpt\/. Accessed: 2024-01--23."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591500"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3450626.3459872"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/1276377.1276442"},{"key":"e_1_3_2_2_53_1","volume-title":"Luis RP, Jian Jiang, Sheng Zhang, Pingyu Wu, Bo Zhou, and Weiming Zhang.","author":"Perov Ivan","year":"2021","unstructured":"Ivan Perov, Daiheng Gao, Nikolay Chervoniy, Kunlin Liu, Sugasa Marangonda, Chris Um\u00e9, Mr. Dpfks, Carl Shift Facenheim, Luis RP, Jian Jiang, Sheng Zhang, Pingyu Wu, Bo Zhou, and Weiming Zhang. 2021. DeepFaceLab: Integrated, flexible and extensible face-swapping framework. arxiv: 2005.05535 [cs.CV]"},{"volume-title":"Physically based rendering: From theory to implementation","author":"Pharr Matt","key":"e_1_3_2_2_54_1","unstructured":"Matt Pharr, Wenzel Jakob, and Greg Humphreys. 2023. Physically based rendering: From theory to implementation. MIT Press."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Pakkapon Phongthawee Worameth Chinchuthakun Nontaphat Sinsunthithet Amit Raj Varun Jampani Pramook Khungurn and Supasorn Suwajanakorn. 2023. DiffusionLight: Light Probes for Free by Painting a Chrome Ball. In ArXiv.","DOI":"10.1109\/CVPR52733.2024.00018"},{"key":"e_1_3_2_2_56_1","volume-title":"DiFaReli: Diffusion Face Relighting. arXiv preprint arXiv:2304.09479","author":"Ponglertnapakorn Puntawat","year":"2023","unstructured":"Puntawat Ponglertnapakorn, Nontawat Tritrong, and Supasorn Suwajanakorn. 2023. DiFaReli: Diffusion Face Relighting. arXiv preprint arXiv:2304.09479 (2023)."},{"key":"e_1_3_2_2_57_1","volume-title":"Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer","author":"Ranftl Ren\u00e9","year":"2020","unstructured":"Ren\u00e9 Ranftl, Katrin Lasinger, David Hafner, Konrad Schindler, and Vladlen Koltun. 2020. Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer. IEEE transactions on pattern analysis and machine intelligence, Vol. 44, 3 (2020), 1623--1637."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01129"},{"key":"e_1_3_2_2_59_1","unstructured":"Hanson Robotics. 2016. Sophia. https:\/\/www.hansonrobotics.com\/sophia-2020\/."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"crossref","unstructured":"Nataniel Ruiz Yuanzhen Li Varun Jampani Yael Pritch Michael Rubinstein and Kfir Aberman. 2022. DreamBooth: Fine Tuning Text-to-image Diffusion Models for Subject-Driven Generation. (2022).","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00239"},{"key":"e_1_3_2_2_63_1","unstructured":"Martin Scorsese. 2019. The Irishman."},{"key":"e_1_3_2_2_64_1","unstructured":"Virtual Production Studios. 2021. Virtual Production Stage. https:\/\/www.virtualproductionstudios.com\/virtual-production-stage\/."},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00618"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01989"},{"key":"e_1_3_2_2_67_1","volume-title":"Bracketing in qualitative research. Qualitative social work","author":"Tufford Lea","year":"2012","unstructured":"Lea Tufford and Peter Newman. 2012. Bracketing in qualitative research. Qualitative social work, Vol. 11, 1 (2012), 80--96."},{"key":"e_1_3_2_2_68_1","volume-title":"Accuracy of human motion capture systems for sport applications","author":"der Kruk Eline Van","year":"2018","unstructured":"Eline Van der Kruk and Marco M Reijne. 2018. Accuracy of human motion capture systems for sport applications; state-of-the-art review. European journal of sport science, Vol. 18, 6 (2018), 806--819."},{"key":"e_1_3_2_2_69_1","unstructured":"Vicon. 2023. Vicon. https:\/\/www.vicon.com\/"},{"key":"e_1_3_2_2_70_1","unstructured":"Vicon. 2023. Vicon Carapost. https:\/\/www.vicon.com\/software\/carapost\/"},{"key":"e_1_3_2_2_71_1","volume-title":"Neus: Learning neural implicit surfaces by","author":"Wang Peng","year":"2021","unstructured":"Peng Wang, Lingjie Liu, Yuan Liu, Christian Theobalt, Taku Komura, and Wenping Wang. 2021. Neus: Learning neural implicit surfaces by volume rendering for multi-view reconstruction. arXiv preprint arXiv:2106.10689 (2021)."},{"key":"e_1_3_2_2_72_1","volume-title":"Zero-shot image restoration using denoising diffusion null-space model. arXiv preprint arXiv:2212.00490","author":"Wang Yinhuai","year":"2022","unstructured":"Yinhuai Wang, Jiwen Yu, and Jian Zhang. 2022. Zero-shot image restoration using denoising diffusion null-space model. arXiv preprint arXiv:2212.00490 (2022)."},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/1073204.1073258"},{"key":"e_1_3_2_2_74_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. (2023)."},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612145"},{"key":"e_1_3_2_2_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00570"},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00455"},{"key":"e_1_3_2_2_78_1","volume-title":"Resshift: Efficient diffusion model for image super-resolution by residual shifting. arXiv preprint arXiv:2307.12348","author":"Yue Zongsheng","year":"2023","unstructured":"Zongsheng Yue, Jianyi Wang, and Chen Change Loy. 2023. Resshift: Efficient diffusion model for image super-resolution by residual shifting. arXiv preprint arXiv:2307.12348 (2023)."},{"volume-title":"Anatomy of Facial Expressions. Anatomy Next","author":"Zarins Uldis","key":"e_1_3_2_2_79_1","unstructured":"Uldis Zarins. 2017. Anatomy of Facial Expressions. Anatomy Next, Inc."},{"key":"e_1_3_2_2_80_1","doi-asserted-by":"crossref","unstructured":"Lvmin Zhang Anyi Rao and Maneesh Agrawala. 2023. Adding Conditional Control to Text-to-Image Diffusion Models.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00084"},{"key":"e_1_3_2_2_82_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00420"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3685509","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3685509","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:28Z","timestamp":1750295848000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3685509"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":82,"alternative-id":["10.1145\/3664647.3685509","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3685509","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}