{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:58:03Z","timestamp":1785488283781,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":64,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774576","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Scalable Sign Production: Leveraging Co-Articulated Gloss Dictionary for Fluid Sign Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-7862-2428","authenticated-orcid":false,"given":"Aparna","family":"Agrawal","sequence":"first","affiliation":[{"name":"CVIT, International Institute of Information Technology - Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0764-7322","authenticated-orcid":false,"given":"Seshadri","family":"Mazumder","sequence":"additional","affiliation":[{"name":"CVIT, International Institute of Information Technology - Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6767-7057","authenticated-orcid":false,"given":"C V","family":"Jawahar","sequence":"additional","affiliation":[{"name":"International Institute of Information Technology - Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5262-9722","authenticated-orcid":false,"given":"Vinay","family":"Namboodiri","sequence":"additional","affiliation":[{"name":"University of Bath, Claverton Down, Bath, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"crossref","unstructured":"Francisca\u00a0Adoma Acheampong Henry Nunoo-Mensah and Wenyu Chen. 2021. Transformer models for text-based emotion detection: a review of BERT-based approaches. Artificial Intelligence Review 54 8 (2021) 5789\u20135829.","DOI":"10.1007\/s10462-021-09958-2"},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_3"},{"key":"e_1_3_3_3_4_2","unstructured":"Samuel Albanie G\u00fcl Varol Liliane Momeni Hannah Bull Triantafyllos Afouras Himel Chowdhury Neil Fox Bencie Woll Rob Cooper Andrew McParland et\u00a0al. 2021. Bbc-oxford british sign language dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.03635 (2021)."},{"key":"e_1_3_3_3_5_2","unstructured":"Zhaoyi An and Rei Kawakami. 2025. Teach Me Sign: Stepwise Prompting LLM for Sign Language Production. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.10972 (2025)."},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"crossref","unstructured":"Safaeid\u00a0Hossain Arib Rabeya Akter Sejuti Rahman and Shafin Rahman. 2025. SignFormer-GCN: Continuous sign language translation using spatio-temporal graph convolutional networks. PloS one 20 2 (2025) e0316298.","DOI":"10.1371\/journal.pone.0316298"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02016"},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"crossref","unstructured":"Maryam Aziz and Achraf Othman. 2023. Evolution and trends in sign Language Avatar systems: Unveiling a 40-Year journey via systematic review. Multimodal Technologies and Interaction 7 10 (2023) 97.","DOI":"10.3390\/mti7100097"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"crossref","unstructured":"Eyal Betzalel Coby Penso and Ethan Fetaya. 2024. Evaluation metrics for generative models: An empirical study. Machine Learning and Knowledge Extraction 6 3 (2024) 1531\u20131544.","DOI":"10.3390\/make6030073"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00812"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/FG52635.2021.9667087"},{"key":"e_1_3_3_3_12_2","unstructured":"Matthijs Douze Alexandr Guzhva Chengqi Deng Jeff Johnson Gergely Szilvasy Pierre-Emmanuel Mazar\u00e9 Maria Lomeli Lucas Hosseini and Herv\u00e9 J\u00e9gou. 2024. The Faiss library. (2024). arxiv:https:\/\/arXiv.org\/abs\/2401.08281"},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"publisher","DOI":"10.63317\/4wzouow58zat"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00276"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350535"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"publisher","DOI":"10.1515\/9781400868339"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","DOI":"10.63317\/27pdd6xpkah6"},{"key":"e_1_3_3_3_18_2","first-page":"123","volume-title":"Proceedings of the 19th International Conference on Natural Language Processing (ICON)","author":"Ghosh Abhigyan","year":"2022","unstructured":"Abhigyan Ghosh and Radhika Mamidi. 2022. English to Indian Sign Language: Rule-Based Translation System Along With Multi-Word Expressions and Synonym Substitution. In Proceedings of the 19th International Conference on Natural Language Processing (ICON). New Delhi, India, 123\u2013127."},{"key":"e_1_3_3_3_19_2","unstructured":"Jianyuan Guo Peike Li and Trevor Cohn. 2025. Bridging Sign and Spoken Languages: Pseudo Gloss Generation for Sign Language Translation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.15438 (2025)."},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"crossref","unstructured":"Mert Inan Katherine Atwell Anthony Sicilia Lorna Quandt and Malihe Alikhani. 2024. Generating Signed Language Instructions in Large-Scale Dialogue Systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.14026 (2024).","DOI":"10.18653\/v1\/2024.naacl-industry.13"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00817"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","DOI":"10.63317\/2ptokbmctz25"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"crossref","unstructured":"Ronan Johnson. 2021. Towards enhanced visual clarity of sign language avatars through recreation of fine facial detail. Machine Translation 35 3 (2021) 431\u2013445.","DOI":"10.1007\/s10590-021-09269-x"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"crossref","unstructured":"Abhinav Joshi Susmit Agrawal and Ashutosh Modi. 2023. ISLTranslate: Dataset for translating Indian sign language. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.05440 (2023).","DOI":"10.18653\/v1\/2023.findings-acl.665"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"crossref","unstructured":"Abhinav Joshi Romit Mohanty Mounika Kanakanti Andesha Mangla Sudeep Choudhary Monali Barbate and Ashutosh Modi. 2024. isign: A benchmark for indian sign language processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.05404 (2024).","DOI":"10.18653\/v1\/2024.findings-acl.643"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"publisher","unstructured":"Sang-Ki Ko Chang\u00a0Jo Kim Hyedong Jung and Choongsang Cho. 2019. Neural Sign Language Translation Based on Human Keypoint Estimation. Applied Sciences 9 13 (2019) 2683. 10.3390\/app9132683","DOI":"10.3390\/app9132683"},{"key":"e_1_3_3_3_27_2","unstructured":"Zhimin Li Jianwei Zhang Qin Lin Jiangfeng Xiong Yanxin Long Xinchi Deng Yingfang Zhang Xingchao Liu Minbin Huang Zedong Xiao et\u00a0al. 2024. Hunyuan-dit: A powerful multi-resolution diffusion transformer with fine-grained chinese understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.08748 (2024)."},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i6.32612"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3596711.3596800"},{"key":"e_1_3_3_3_30_2","unstructured":"Camillo Lugaresi Jiuqiang Tang Hadon Nash Chris McClanahan Esha Uboweja Michael Hays Fan Zhang Chuo-Ling Chang Ming\u00a0Guang Yong Juhyun Lee et\u00a0al. 2019. Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1906.08172 (2019)."},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICMI.2002.1166987"},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/IOTSMS62296.2024.10710270"},{"key":"e_1_3_3_3_33_2","first-page":"22","volume-title":"Proceedings of the Second International Workshop on Automatic Translation for Signed and Spoken Languages","author":"Moryossef Amit","year":"2023","unstructured":"Amit Moryossef, Mathias M\u00fcller, Anne G\u00f6hring, Zifan Jiang, Yoav Goldberg, and Sarah Ebling. 2023. An open-source gloss-based baseline for spoken to signed language translation. In Proceedings of the Second International Workshop on Automatic Translation for Signed and Spoken Languages. 22\u201333."},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"publisher","unstructured":"B.\u00a0O. Olusanya A.\u00a0C. Davis and H.\u00a0J. Hoffman. 2019. Hearing loss: rising prevalence and impact. Bulletin of the World Health Organization 97 10 (2019) 646\u2013646A. 10.2471\/BLT.19.224683","DOI":"10.2471\/BLT.19.224683"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3347319.3356837"},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19797-0_3"},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"crossref","unstructured":"Lorna\u00a0C Quandt Athena Willis Melody Schwenk Kaitlyn Weeks and Ruthie Ferster. 2022. Attitudes toward signing avatars vary depending on hearing status age of signed language acquisition and avatar type. Frontiers in psychology 13 (2022) 730917.","DOI":"10.3389\/fpsyg.2022.730917"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"publisher","DOI":"10.17632\/kcmpdxky7p.1"},{"key":"e_1_3_3_3_39_2","unstructured":"Charles Raude KR Prajwal Liliane Momeni Hannah Bull Samuel Albanie Andrew Zisserman and G\u00fcl Varol. 2024. A tale of two languages: Large-vocabulary continuous sign language recognition from spoken language supervision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.10266 (2024)."},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.1117\/12.768060"},{"key":"e_1_3_3_3_41_2","unstructured":"Ben Saunders Necati\u00a0Cihan Camgoz and Richard Bowden. 2020. Everybody sign now: Translating spoken language to photo realistic sign language video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2011.09846 (2020)."},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00508"},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00679"},{"key":"e_1_3_3_3_44_2","unstructured":"Abhimanyu Sharma. 2025. India\u2019s language policy for deaf and hard-of-hearing people. Language Policy (2025) 1\u201323."},{"key":"e_1_3_3_3_45_2","first-page":"336","volume-title":"European Conference on Computer Vision","author":"Shen Liao","year":"2024","unstructured":"Liao Shen, Tianqi Liu, Huiqiang Sun, Xinyi Ye, Baopu Li, Jianming Zhang, and Zhiguo Cao. 2024. Dreammover: Leveraging the prior of diffusion models for image interpolation with large motion. In European Conference on Computer Vision. Springer, 336\u2013353."},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"crossref","unstructured":"Bowen Shi Diane Brentari Greg Shakhnarovich and Karen Livescu. 2022. Open-domain sign language translation learned from online video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.12870 (2022).","DOI":"10.18653\/v1\/2022.emnlp-main.427"},{"key":"e_1_3_3_3_47_2","volume-title":"Conference on Neural Information Processing Systems (NeurIPS)","author":"Siarohin Aliaksandr","year":"2019","unstructured":"Aliaksandr Siarohin, St\u00e9phane Lathuili\u00e8re, Sergey Tulyakov, Elisa Ricci, and Nicu Sebe. 2019. First Order Motion Model for Image Animation. In Conference on Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"crossref","unstructured":"Dave Uthus Garrett Tanzer and Manfred Georg. 2023. Youtube-asl: A large-scale open-domain american sign language-english parallel corpus. Advances in Neural Information Processing Systems 36 (2023) 29029\u201329047.","DOI":"10.52202\/075280-1264"},{"key":"e_1_3_3_3_49_2","unstructured":"Harry Walsh Ben Saunders and Richard Bowden. 2024. Sign stitching: a novel approach to sign language production. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.07663 (2024)."},{"key":"e_1_3_3_3_50_2","unstructured":"Xiang Wang Shiwei Zhang Longxiang Tang Yingya Zhang Changxin Gao Yuehuan Wang and Nong Sang. 2025. Unianimate-dit: Human image animation with large-scale video diffusion transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.11289 (2025)."},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"crossref","unstructured":"Wufeng Xue Lei Zhang Xuanqin Mou and Alan\u00a0C Bovik. 2013. Gradient magnitude similarity deviation: A highly efficient perceptual image quality index. IEEE transactions on image processing 23 2 (2013) 684\u2013695.","DOI":"10.1109\/TIP.2013.2293423"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00684"},{"key":"e_1_3_3_3_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00455"},{"key":"e_1_3_3_3_54_2","unstructured":"Yu Yu Weibin Zhang and Yun Deng. 2021. Frechet inception distance (fid) for evaluating gans. China University of Mining Technology Beijing Graduate School 3 11 (2021)."},{"key":"e_1_3_3_3_55_2","doi-asserted-by":"crossref","unstructured":"Guozhen Zhang Chuxnu Liu Yutao Cui Xiaotong Zhao Kai Ma and Limin Wang. 2024. Vfimamba: Video frame interpolation with state space models. Advances in Neural Information Processing Systems 37 (2024) 107225\u2013107248.","DOI":"10.52202\/079017-3405"},{"key":"e_1_3_3_3_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713855"},{"key":"e_1_3_3_3_57_2","unstructured":"Lvmin Zhang and Maneesh Agrawala. 2025. Packing Input Frame Contexts in Next-Frame Prediction Models for Video Generation. Arxiv (2025)."},{"key":"e_1_3_3_3_58_2","doi-asserted-by":"crossref","unstructured":"Lin Zhang Lei Zhang and Alan\u00a0C Bovik. 2015. A feature-enriched completely blind image quality evaluator. IEEE Transactions on Image Processing 24 8 (2015) 2579\u20132591.","DOI":"10.1109\/TIP.2015.2426416"},{"key":"e_1_3_3_3_59_2","first-page":"335","volume-title":"European Conference on Computer Vision","author":"Zhang Pengyu","year":"2024","unstructured":"Pengyu Zhang, Hao Yin, Zeren Wang, Wenyue Chen, Shengming Li, Dong Wang, Huchuan Lu, and Xu Jia. 2024. Evsign: Sign language recognition and translation with streaming events. In European Conference on Computer Vision. Springer, 335\u2013351."},{"key":"e_1_3_3_3_60_2","doi-asserted-by":"crossref","unstructured":"Yanqiong Zhang and Xianwei Jiang. 2024. Recent Advances on Deep Learning for Sign Language Recognition. Computer Modeling in Engineering & Sciences (CMES) 139 3 (2024).","DOI":"10.32604\/cmes.2023.045731"},{"key":"e_1_3_3_3_61_2","doi-asserted-by":"crossref","unstructured":"Weichao Zhao Hezhen Hu Wengang Zhou Yunyao Mao Min Wang and Houqiang Li. 2024. Masa: Motion-aware masked autoencoder with semantic alignment for sign language recognition. IEEE Transactions on Circuits and Systems for Video Technology 34 11 (2024) 10793\u201310804.","DOI":"10.1109\/TCSVT.2024.3409728"},{"key":"e_1_3_3_3_62_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01908"},{"key":"e_1_3_3_3_63_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00137"},{"key":"e_1_3_3_3_64_2","doi-asserted-by":"publisher","DOI":"10.1109\/SMC54092.2024.10831194"},{"key":"e_1_3_3_3_65_2","first-page":"36","volume-title":"European Conference on Computer Vision","author":"Zuo Ronglai","year":"2024","unstructured":"Ronglai Zuo, Fangyun Wei, Zenggui Chen, Brian Mak, Jiaolong Yang, and Xin Tong. 2024. A simple baseline for spoken language to sign language translation with 3d avatars. In European Conference on Computer Vision. Springer, 36\u201354."}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774576","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:06:03Z","timestamp":1785485163000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774576"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":64,"alternative-id":["10.1145\/3774521.3774576","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774576","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}