{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,28]],"date-time":"2026-08-28T05:21:13Z","timestamp":1787894473014,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,9]],"date-time":"2023-10-09T00:00:00Z","timestamp":1696809600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,9]]},"DOI":"10.1145\/3610661.3617514","type":"proceedings-article","created":{"date-parts":[[2023,10,9]],"date-time":"2023-10-09T16:51:22Z","timestamp":1696870282000},"page":"259-271","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["Design of Generative Multimodal AI Agents to Enable Persons with Learning Disability"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5602-6416","authenticated-orcid":false,"given":"Rajagopal","family":"A","sequence":"first","affiliation":[{"name":"Indian Institute of Technology, Madras, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7370-2490","authenticated-orcid":false,"given":"Nirmala","family":"V","sequence":"additional","affiliation":[{"name":"Queen Mary's College, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8548-3333","authenticated-orcid":false,"given":"Immanuel Johnraja","family":"Jebadurai","sequence":"additional","affiliation":[{"name":"Karunya Institute of Technology &amp; Sciences, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5361-9438","authenticated-orcid":false,"given":"Arun Muthuraj","family":"Vedamanickam","sequence":"additional","affiliation":[{"name":"National Institute of Technology, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3821-1403","authenticated-orcid":false,"given":"Prajakta Uthaya","family":"Kumar","sequence":"additional","affiliation":[{"name":"National Institute of Design, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,9]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"National Center for Education Statistics. (2023). Students With Disabilities. nces.ed.gov. Published","author":"National Center for Education Statistics.","year":"2023","unstructured":"National Center for Education Statistics. National Center for Education Statistics. (2023). Students With Disabilities. nces.ed.gov. Published May 2023. https:\/\/nces.ed.gov\/programs\/coe\/pdf\/2023\/cgg_508.pdf"},{"key":"e_1_3_2_1_2_1","unstructured":"National Center for Learning Disabilities (NCLD). Visual and Auditory Processing Disorders | LD OnLine. www.ldonline.org. https:\/\/www.ldonline.org\/ld-topics\/processing-deficits\/visual-and-auditory-processing-disorders"},{"key":"e_1_3_2_1_3_1","unstructured":"Learning Disabilities Association of America. Learning Disabilities - Getting Help. https:\/\/ldaamerica.org\/support\/new-to-ld\/"},{"key":"e_1_3_2_1_4_1","volume-title":"The State of Learning Disabilities. Https:\/\/Www.ncld.org\/Research\/State-of-Learning-Disabilities\/Executive-Summary","author":"Horowitz S. H.","year":"2017","unstructured":"Horowitz, S. H., Rawe, J., & Whittaker, M. C. The State of Learning Disabilities. Https:\/\/Www.ncld.org\/Research\/State-of-Learning-Disabilities\/Executive-Summary. National Center for Learning Disabilities.; 2017."},{"key":"e_1_3_2_1_5_1","volume-title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision. arXiv:210203334 [cs, stat]. Published online","author":"Kim W","year":"2021","unstructured":"Kim W, Son B, Kim I. ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision. arXiv:210203334 [cs, stat]. Published online June 10, 2021. https:\/\/arxiv.org\/abs\/2102.03334"},{"key":"e_1_3_2_1_6_1","volume-title":"GIT: A Generative Image-to-text Transformer for Vision and Language. arXiv:220514100 [cs]. Published online","author":"Wang J","year":"2022","unstructured":"Wang J, Yang Z, Hu X, GIT: A Generative Image-to-text Transformer for Vision and Language. arXiv:220514100 [cs]. Published online December 15, 2022. https:\/\/arxiv.org\/abs\/2205.14100"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2305.03726"},{"key":"e_1_3_2_1_8_1","volume-title":"Language Models are Few-Shot Learners. arxivorg. Published online","author":"Brown TB","year":"2020","unstructured":"Brown TB, Mann B, Ryder N, Language Models are Few-Shot Learners. arxivorg. Published online May 28, 2020. https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_2_1_9_1","volume-title":"Zero-Shot Text-to-Image Generation. arXiv:210212092 [cs]. Published online","author":"Ramesh A","year":"2021","unstructured":"Ramesh A, Pavlov M, Goh G, Zero-Shot Text-to-Image Generation. arXiv:210212092 [cs]. Published online February 26, 2021. https:\/\/arxiv.org\/abs\/2102.12092"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","unstructured":"Girdhar R El-Nouby A Liu Z ImageBind: One Embedding Space To Bind Them All. arXiv.org. doi:https:\/\/doi.org\/10.48550\/arXiv.2305.05665","DOI":"10.48550\/arXiv.2305.05665"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","unstructured":"Peng Z Wang W Dong L Kosmos-2: Grounding Multimodal Large Language Models to the World. arXiv.org. doi:https:\/\/doi.org\/10.48550\/arXiv.2306.14824","DOI":"10.48550\/arXiv.2306.14824"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","unstructured":"Tang Z Yang Z Zhu C Zeng M Bansal M. Any-to-Any Generation via Composable Diffusion. arXiv.org. doi:https:\/\/doi.org\/10.48550\/arXiv.2305.11846","DOI":"10.48550\/arXiv.2305.11846"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544549.3577043"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","unstructured":"Gozalo-Brizuela R Garrido-Merch\u00e1n EC. A survey of Generative AI Applications. arXiv.org. doi:https:\/\/doi.org\/10.48550\/arXiv.2306.02781","DOI":"10.48550\/arXiv.2306.02781"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2305.07605"},{"key":"e_1_3_2_1_16_1","volume-title":"Accessed","author":"Assistive Technology","year":"2023","unstructured":"Assistive Technology and Universal Design: A Toolkit for Interagency Collaboration | Interagency Committee on Disability Research. icdr.acl.gov. Accessed July 31, 2023. https:\/\/icdr.acl.gov\/resources\/reports\/assistive-technology-and-universal-design-toolkit-interagency-collaboration"},{"key":"e_1_3_2_1_17_1","first-page":"31","article-title":"Multimodal Neural Machine Translation Using Synthetic Images Transformed by Latent Diffusion Model","author":"Yuasa R","year":"2023","unstructured":"Yuasa R, Tamura A, Kajiwara T, Ninomiya T, Kato T. Multimodal Neural Machine Translation Using Synthetic Images Transformed by Latent Diffusion Model. ACLWeb. Published July 1, 2023. Accessed July 31, 2023. https:\/\/aclanthology.org\/2023.acl-srw.12","journal-title":"ACLWeb. Published"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","unstructured":"Kuo W Piergiovanni AJ Kim D MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks. arXiv.org. doi:https:\/\/doi.org\/10.48550\/arXiv.2303.16839","DOI":"10.48550\/arXiv.2303.16839"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","unstructured":"Nir Kshetri. ChatGPT in Developing Economies. 2023;25(2):16-19. doi:https:\/\/doi.org\/10.1109\/mitp.2023.3254639","DOI":"10.1109\/mitp.2023.3254639"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.nlp4convai-1.12"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10639-023-11834-1"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","unstructured":"Lobentanzer S Saez-Rodriguez J. A Platform for the Biomedical Application of Large Language Models. arXiv.org. doi:https:\/\/doi.org\/10.48550\/arXiv.2305.06488","DOI":"10.48550\/arXiv.2305.06488"},{"key":"e_1_3_2_1_23_1","first-page":"1","author":"Semantic Kernel","year":"2023","unstructured":"Semantic Kernel. GitHub. Published August 1, 2023. Accessed August 1, 2023. https:\/\/github.com\/microsoft\/semantic-kernel","journal-title":"GitHub. Published"},{"key":"e_1_3_2_1_24_1","first-page":"1","author":"Langchain","year":"2023","unstructured":"Langchain, GitHub. Published August 1, 2023. Accessed August 1, 2023. https:\/\/github.com\/langchain-ai\/langchain","journal-title":"GitHub. Published"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","unstructured":"Li B Zhang Y Chen L MIMIC-IT: Multi-Modal In-Context Instruction Tuning. arXiv.org. https:\/\/doi.org\/10.48550\/arXiv.2306.05425","DOI":"10.48550\/arXiv.2306.05425"},{"key":"e_1_3_2_1_26_1","unstructured":"Valerie L. Frazer Visual Skills for Reading and Learning: Vision Perception Webinar https:\/\/www.newhorizonsvisiontherapy.com\/category\/vision-and-learning\/"},{"key":"e_1_3_2_1_27_1","unstructured":"Strengths of Students with Learning Disabilities The National Center for Learning Disabilities https:\/\/www.youtube.com\/watch?v=CYHzJGTA6KM"},{"key":"e_1_3_2_1_28_1","volume-title":"ISBN 1944883428","author":"Deborah Ross-Swain S.","unstructured":"Deborah Ross-Swain, Donna S. Geffner. Auditory Processing Disorders: Assessment, Management, and Treatment,\u00a0Plural Publishing, ISBN 1944883428"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1044\/jslhr.4002.432"},{"key":"e_1_3_2_1_30_1","first-page":"319","article-title":"central auditory processing deficiencies in the real world. what teachers and parents want to know. Ferre, 2002","volume":"23","author":"Managing Children's","unstructured":"Managing Children's central auditory processing deficiencies in the real world. what teachers and parents want to know. Ferre, 2002, Seminars in Hearing, 23, 319-326","journal-title":"Seminars in Hearing"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1044\/leader.FTR2.12102007.20"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1044\/jshr.2801.96"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Jeanneret Medina Maximiliano \u201cIt Deserves to Be Further Developed: A Study of Mainstream Web Interface Adaptability for People with Low Vision.\" CHI Conference on Human Factors in Computing Systems Extended Abstracts. 2022.","DOI":"10.1145\/3491101.3519622"}],"event":{"name":"ICMI '23: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERACTION","location":"Paris France","acronym":"ICMI '23","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["International Cconference on Multimodal Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610661.3617514","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610661.3617514","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T19:31:05Z","timestamp":1755891065000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610661.3617514"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,9]]},"references-count":33,"alternative-id":["10.1145\/3610661.3617514","10.1145\/3610661"],"URL":"https:\/\/doi.org\/10.1145\/3610661.3617514","relation":{},"subject":[],"published":{"date-parts":[[2023,10,9]]},"assertion":[{"value":"2023-10-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}