{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:53:50Z","timestamp":1781538830614,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62222606"],"award-info":[{"award-number":["62222606"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810659","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"1899-1907","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Dual-Pathway Diffusion for Hand Correction in Synthetic Portraits: Global Context Aware and Local Structure Refinement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-3357-6319","authenticated-orcid":false,"given":"Yushe","family":"Cao","sequence":"first","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1540-5815","authenticated-orcid":false,"given":"Luoxi","family":"Jing","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6917-3502","authenticated-orcid":false,"given":"Yuanze","family":"Wang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8112-371X","authenticated-orcid":false,"given":"Dianxi","family":"Shi","sequence":"additional","affiliation":[{"name":"Intelligent Game and Decision Lab, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2591-7993","authenticated-orcid":false,"given":"Chun","family":"Yu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6801-0510","authenticated-orcid":false,"given":"Junliang","family":"Xing","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Omri Avrahami Ohad Fried and Dani Lischinski. 2023. Blended latent diffusion. ACM Transactions on Graphics 42 4 (2023) 1\u201311.","DOI":"10.1145\/3592450"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Ziyi Chang George\u00a0A Koulieris Hyung\u00a0Jin Chang and Hubert\u00a0PH Shum. 2025. On the design fundamentals of diffusion models: A survey. Pattern Recognition (2025) 111934.","DOI":"10.1016\/j.patcog.2025.111934"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","unstructured":"Zhuo Chen Yichao Yan Sehngqi Liu Yuhao Cheng Weiming Zhao Lincheng Li Mengxiao Bi and Xiaokang Yang. 2025. Revealing Directions for Text-guided 3D Face Editing. IEEE Transactions on Multimedia (2025) 1\u201314. 10.1109\/TMM.2025.3604978","DOI":"10.1109\/TMM.2025.3604978"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Florinel-Alin Croitoru Vlad Hondru Radu\u00a0Tudor Ionescu and Mubarak Shah. 2023. Diffusion models in vision: A survey. IEEE transactions on pattern analysis and machine intelligence 45 9 (2023) 10850\u201310869.","DOI":"10.1109\/TPAMI.2023.3261988"},{"key":"e_1_3_3_2_7_2","unstructured":"Rinon Gal Yuval Alaluf Yuval Atzmon Or Patashnik Amit\u00a0H Bermano Gal Chechik and Daniel Cohen-Or. 2022. An image is worth one word: Personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.01618 (2022)."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73661-2_10"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.265"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2020. Generative adversarial networks. Commun. ACM 63 11 (2020) 139\u2013144.","DOI":"10.1145\/3422622"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00995"},{"key":"e_1_3_3_2_12_2","unstructured":"Martin Heusel Hubert Ramsauer Thomas Unterthiner Bernhard Nessler and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in Neural Information Processing Systems 30 (2017)."},{"key":"e_1_3_3_2_13_2","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems 33 (2020) 6840\u20136851."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Yingli Hou Wei Zhang Zhiliang Zhu and Hai Yu. 2025. CLIP-GAN: Stacking CLIPs and GAN for Efficient and Controllable Text-to-Image Synthesis. IEEE Transactions on Multimedia 27 (2025) 3702\u20133715. 10.1109\/TMM.2025.3535304","DOI":"10.1109\/TMM.2025.3535304"},{"key":"e_1_3_3_2_15_2","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.09685 (2021)."},{"key":"e_1_3_3_2_16_2","first-page":"8153","volume-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Hu Li","year":"2024","unstructured":"Li Hu. 2024. Animate anyone: Consistent and controllable image-to-video synthesis for character animation. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 8153\u20138163."},{"key":"e_1_3_3_2_17_2","unstructured":"Lianghua Huang Wei Wang Zhi-Fan Wu Yupeng Shi Huanzhang Dou Chen Liang Yutong Feng Yu Liu and Jingren Zhou. 2024. In-context lora for diffusion transformers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.23775 (2024)."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Catalin Ionescu Dragos Papava Vlad Olaru and Cristian Sminchisescu. 2013. Human3. 6m: Large scale datasets and predictive methods for 3d human sensing in natural environments. IEEE Transactions on Pattern Analysis and Machine Intelligence 36 7 (2013) 1325\u20131339.","DOI":"10.1109\/TPAMI.2013.248"},{"key":"e_1_3_3_2_19_2","first-page":"4572","volume-title":"IEEE\/CVF Winter Conference on Applications of Computer Vision","author":"Kapitanov Alexander","year":"2024","unstructured":"Alexander Kapitanov, Karina Kvanchiani, Alexander Nagaev, Roman Kraynov, and Andrei Makhliarchuk. 2024. HaGRID\u2013HAnd Gesture Recognition Image Dataset. In IEEE\/CVF Winter Conference on Applications of Computer Vision. 4572\u20134581."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00213"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01270"},{"key":"e_1_3_3_2_22_2","unstructured":"Ilya Loshchilov Frank Hutter et\u00a0al. 2017. Fixing weight decay regularization in adam. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1711.05101 5 (2017)."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680693"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","unstructured":"Yuwu Lu Wai\u00a0Keung Wong Chun Yuan Zhihui Lai and Xuelong Li. 2024. Low-Rank Correlation Learning for Unsupervised Domain Adaptation. IEEE Transactions on Multimedia 26 (2024) 4153\u20134167. 10.1109\/TMM.2023.3321430","DOI":"10.1109\/TMM.2023.3321430"},{"key":"e_1_3_3_2_25_2","unstructured":"Camillo Lugaresi Jiuqiang Tang Hadon Nash Chris McClanahan Esha Uboweja Michael Hays Fan Zhang Chuo-Ling Chang Ming\u00a0Guang Yong Juhyun Lee et\u00a0al. 2019. Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1906.08172 (2019)."},{"key":"e_1_3_3_2_26_2","unstructured":"Yuchen Mao Hongwei Li Wei Pang Giorgos Papanastasiou Guang Yang and Chengjia Wang. 2024. SeLoRA: Self-Expanding Low-Rank Adaptation of Latent Diffusion Model for Medical Image Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.07196 (2024)."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"e_1_3_3_2_28_2","volume-title":"IEEE\/CVF International Conference on Computer Vision (ICCV), International Workshop on Observing and Understanding Hands in Action","volume":"2","author":"Narasimhaswamy Supreeth","year":"2023","unstructured":"Supreeth Narasimhaswamy, Uttaran Bhattacharya, Xiang Chen, Ishita Dasgupta, and Saayan Mitra. 2023. Text-to-handimage generation using pose-and mesh-guided diffusion. In IEEE\/CVF International Conference on Computer Vision (ICCV), International Workshop on Observing and Understanding Hands in Action , Vol.\u00a02."},{"key":"e_1_3_3_2_29_2","unstructured":"Dustin Podell Zion English Kyle Lacey Andreas Blattmann Tim Dockhorn Jonas M\u00fcller Joe Penna and Robin Rombach. 2023. Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.01952 (2023)."},{"key":"e_1_3_3_2_30_2","unstructured":"Can Qin Shu Zhang Ning Yu Yihao Feng Xinyi Yang Yingbo Zhou Huan Wang Juan\u00a0Carlos Niebles Caiming Xiong Silvio Savarese et\u00a0al. 2023. Unicontrol: A unified diffusion model for controllable visual generation in the wild. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.11147 (2023)."},{"key":"e_1_3_3_2_31_2","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_32_2","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06125 1 2 (2022) 3."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_2_35_2","unstructured":"James\u00a0Seale Smith Yen-Chang Hsu Lingyu Zhang Ting Hua Zsolt Kira Yilin Shen and Hongxia Jin. 2023. Continual diffusion: Continual customization of text-to-image diffusion with c-lora. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.06027 (2023)."},{"key":"e_1_3_3_2_36_2","unstructured":"Yang Song Jascha Sohl-Dickstein Diederik\u00a0P Kingma Abhishek Kumar Stefano Ermon and Ben Poole. 2020. Score-based generative modeling through stochastic differential equations. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2011.13456 (2020)."},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","unstructured":"Ashkan Taghipour Morteza Ghahremani Mohammed Bennamoun Aref\u00a0Miri Rekavandi Hamid Laga and Farid Boussaid. 2025. Box It to Bind It: Unified Layout Control and Attribute Binding in Text-to-Image Diffusion Models. IEEE Transactions on Multimedia (2025) 1\u201315. 10.1109\/TMM.2025.3607759","DOI":"10.1109\/TMM.2025.3607759"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i7.32815"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/3DV62453.2024.00043"},{"key":"e_1_3_3_2_40_2","unstructured":"Han Yang Sotiris Anagnostidis Enis Simsar and Thomas Hofmann. 2024. MegaPortrait: Revisiting Diffusion Control for High-fidelity Portrait Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.04357 (2024)."},{"key":"e_1_3_3_2_41_2","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.06721 (2023)."},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","unstructured":"Hua Yu Yaqing Hou Wenbin Pei Yew-Soon Ong and Qiang Zhang. 2025. DivDiff: A Conditional Diffusion Model for Diverse Human Motion Prediction. IEEE Transactions on Multimedia 27 (2025) 1848\u20131859. 10.1109\/TMM.2024.3521821","DOI":"10.1109\/TMM.2024.3521821"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00708"},{"key":"e_1_3_3_2_44_2","unstructured":"Fan Zhang Valentin Bazarevsky Andrey Vakunov Andrei Tkachenka George Sung Chuo-Ling Chang and Matthias Grundmann. 2020. Mediapipe hands: On-device real-time hand tracking. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2006.10214 (2020)."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","unstructured":"Huaiwen Zhang Tianci Wu and Yinwei Wei. 2025. Multi-View User Preference Modeling for Personalized Text-to-Image Generation. IEEE Transactions on Multimedia 27 (2025) 3082\u20133091. 10.1109\/TMM.2025.3557683","DOI":"10.1109\/TMM.2025.3557683"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_47_2","unstructured":"Wenliang Zhao Lujia Bai Yongming Rao Jie Zhou and Jiwen Lu. 2024. Unipc: A unified predictor-corrector framework for fast sampling of diffusion models. Advances in Neural Information Processing Systems 36 (2024)."}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:55:36Z","timestamp":1781535336000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810659"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":46,"alternative-id":["10.1145\/3805622.3810659","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810659","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}