{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T23:26:02Z","timestamp":1785367562229,"version":"3.55.0"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,2,19]],"date-time":"2024-02-19T00:00:00Z","timestamp":1708300800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,2,19]],"date-time":"2024-02-19T00:00:00Z","timestamp":1708300800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,2,19]]},"DOI":"10.1109\/icaiic60209.2024.10463430","type":"proceedings-article","created":{"date-parts":[[2024,3,20]],"date-time":"2024-03-20T18:12:10Z","timestamp":1710958330000},"page":"722-727","source":"Crossref","is-referenced-by-count":28,"title":["ChatGPT for Visually Impaired and Blind"],"prefix":"10.1109","author":[{"given":"Askat","family":"Kuzdeuov","sequence":"first","affiliation":[{"name":"Institute of Smart Systems and AI, Nazarbayev University,Astana,Kazakhstan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Olzhas","family":"Mukayev","sequence":"additional","affiliation":[{"name":"Institute of Smart Systems and AI, Nazarbayev University,Astana,Kazakhstan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shakhizat","family":"Nurgaliyev","sequence":"additional","affiliation":[{"name":"Institute of Smart Systems and AI, Nazarbayev University,Astana,Kazakhstan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alisher","family":"Kunbolsyn","sequence":"additional","affiliation":[{"name":"Institute of Smart Systems and AI, Nazarbayev University,Astana,Kazakhstan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huseyin Atakan","family":"Varol","sequence":"additional","affiliation":[{"name":"Institute of Smart Systems and AI, Nazarbayev University,Astana,Kazakhstan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"9","key":"ref1","doi-asserted-by":"crossref","first-page":"e888","DOI":"10.1016\/S2214-109X(17)30293-0","article-title":"Magnitude, temporal trends, and projections of the global prevalence of blindness and distance and near vision impairment: A systematic review and meta-analysis","volume":"5","author":"Bourne","year":"2017","journal-title":"The Lancet Global Health"},{"key":"ref2","article-title":"The Braille Literacy Crisis in America","year":"2009","journal-title":"National Federation of the Blind Jernigan Institute, Tech. Rep."},{"issue":"6","key":"ref3","doi-asserted-by":"crossref","first-page":"481","DOI":"10.1177\/0145482X19887620","article-title":"Employment and unemployment rates of people who are blind or visually impaired: Estimates from multiple sources","volume":"113","author":"McDonnall","year":"2019","journal-title":"Journal of Visual Impairment & Blindness"},{"issue":"4","key":"ref4","doi-asserted-by":"crossref","first-page":"526","DOI":"10.1177\/0145482X221121830","article-title":"Beyond employment rates: Earnings of people with visual impairments","volume":"116","author":"McDonnall","year":"2022","journal-title":"Journal of Visual Impairment & Blindness"},{"issue":"1-2","key":"ref5","first-page":"9","article-title":"Readers experiences of Braille in an evolving technological world","volume":"54","author":"Marshall","year":"2020","journal-title":"Visible Language"},{"issue":"2","key":"ref6","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1016\/j.tics.2012.12.002","article-title":"Evolution, brain, and the nature of language","volume":"17","author":"Berwick","year":"2013","journal-title":"Trends in Cognitive Sciences"},{"key":"ref7","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Brown","year":"2020"},{"key":"ref8","volume-title":"ChatGPT: Optimizing language models for dialogue","year":"2022"},{"key":"ref9","year":"2023","journal-title":"Gpt-4 technical report"},{"key":"ref10","volume-title":"Gpt-4v(ision) system card","year":"2023"},{"key":"ref11","author":"Radford","year":"2022","journal-title":"Robust speech recognition via large-scale weak supervision"},{"key":"ref12","author":"Touvron","year":"2023","journal-title":"LLaMA: Open and efficient foundation language models"},{"key":"ref13","author":"Touvron","year":"2023","journal-title":"Llama 2: Open foundation and fine-tuned chat models"},{"key":"ref14","volume-title":"llama.cpp","author":"Gerganov","year":"2023"},{"key":"ref15","volume":"abs\/2310.03744","author":"Liu","year":"2023","journal-title":"Improved baselines with visual instruction tuning"},{"key":"ref16","volume-title":"CogVLM: Visual expert for pretrained language models","volume":"abs\/2311.03079","author":"Wang","year":"2023"},{"key":"ref17","volume-title":"Piper: A fast, local neural text to speech system","author":"Hansen","year":"2023"},{"key":"ref18","article-title":"Speech command recognition: Text-to-speech and speech corpus scraping are all you need","author":"Kuzdeuov","year":"2023","journal-title":"TechRxiv"},{"key":"ref19","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2019-2441","article-title":"LibriTTS: A corpus derived from librispeech for text-to-speech","volume-title":"Proc. Interspeech","author":"Zen","year":"2019"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1584"},{"key":"ref21","author":"Yamagishi","year":"2019","journal-title":"CSTR VCTK Corpus: English multi-speaker corpus for CSTR voice cloning toolkit"},{"key":"ref22","volume-title":"Silero VAD: pre-trained enterprise-grade voice activity detec-tor (vad), number detector and language classifier","author":"Team","year":"2021"},{"key":"ref23","volume-title":"whisper.cpp","author":"Gerganov","year":"2023"},{"key":"ref24","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ser. Proceedings of Machine Learning Research","volume":"139","author":"Kim","year":"2021"},{"key":"ref25","first-page":"1015","article-title":"ESC: Dataset for Environmental Sound Classification","volume-title":"Proc. of the Annual ACM Conference on Multimedia","author":"Piczak","year":"2015"},{"key":"ref26","author":"Morshed","year":"2022","journal-title":"Attention-free keyword spotting"},{"key":"ref27","author":"Warden","year":"2018","journal-title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1286"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-698"}],"event":{"name":"2024 International Conference on Artificial Intelligence in Information and Communication (ICAIIC)","location":"Osaka, Japan","start":{"date-parts":[[2024,2,19]]},"end":{"date-parts":[[2024,2,22]]}},"container-title":["2024 International Conference on Artificial Intelligence in Information and Communication (ICAIIC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10463165\/10463194\/10463430.pdf?arnumber=10463430","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,26]],"date-time":"2024-03-26T19:35:30Z","timestamp":1711481730000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10463430\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,19]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icaiic60209.2024.10463430","relation":{},"subject":[],"published":{"date-parts":[[2024,2,19]]}}}