{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T15:53:57Z","timestamp":1780502037704,"version":"3.54.1"},"reference-count":27,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,11,28]],"date-time":"2023-11-28T00:00:00Z","timestamp":1701129600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,28]],"date-time":"2023-11-28T00:00:00Z","timestamp":1701129600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62076225"],"award-info":[{"award-number":["62076225"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Educ Inf Technol"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s10639-023-12310-6","type":"journal-article","created":{"date-parts":[[2023,11,28]],"date-time":"2023-11-28T11:02:49Z","timestamp":1701169369000},"page":"1033-1055","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Optimizing image captioning algorithm to facilitate english writing"],"prefix":"10.1007","volume":"29","author":[{"given":"Xiaxia","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1140-5014","authenticated-orcid":false,"given":"Yao","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,11,28]]},"reference":[{"key":"12310_CR1","unstructured":"Bahdanau, D., Cho, K., & Bengio, Y. (2014). Neural machine translation by jointly learning to align and translate. arXiv preprint. arXiv:1409.0473"},{"key":"12310_CR2","unstructured":"Bengio, S., Vinyals, O., Jaitly, N., & Shazeer, N. (2015). Scheduled sampling for sequence prediction with recurrent neural networks. Advances In Neural Information Processing Systems, 28"},{"issue":"4","key":"12310_CR3","doi-asserted-by":"publisher","first-page":"222","DOI":"10.1177\/00224669211008256","volume":"55","author":"KK Brady","year":"2022","unstructured":"Brady, K. K., Evmenova, A. S., Regan, K. S., Ainsworth, M. K., & Gafurov, B. S. (2022). Using a technology-based graphic organizer to improve the planning and persuasive paragraph writing by adolescents with disabilities and writing difficulties. The Journal of Special Education, 55(4), 222\u2013233.","journal-title":"The Journal of Special Education"},{"key":"12310_CR4","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., others (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint. arXiv:2010.11929"},{"key":"12310_CR5","unstructured":"Fan, A., Grave, E., & Joulin, A. (2019). Reducing transformer depth on demand with structured dropout. arXiv preprint. arXiv:1909.11556"},{"key":"12310_CR6","doi-asserted-by":"crossref","unstructured":"Gowda, T., & May, J. (2020). Finding the optimal vocabulary size for neural machine translation. arXiv preprint. arXiv:2004.02334","DOI":"10.18653\/v1\/2020.findings-emnlp.352"},{"key":"12310_CR7","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1016\/j.neucom.2018.02.106","volume":"328","author":"X He","year":"2019","unstructured":"He, X., Yang, Y., Shi, B., & Bai, X. (2019). Vd-san: Visual-densely semantic attention network for image caption generation. Neurocomputing, 328, 48\u201355.","journal-title":"Neurocomputing"},{"key":"12310_CR8","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh, M., Young, P., & Hockenmaier, J. (2013). Framing image description as a ranking task: Data, models and evaluation metrics. Journal of Artificial Intelligence Research, 47, 853\u2013899.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"12310_CR9","doi-asserted-by":"crossref","unstructured":"Huang, G., Sun, Y., Liu, Z., Sedra, D., & Weinberger, K.Q. (2016). Deep networks with stochastic depth. European conference on computer vision (pp. 646\u2013661)","DOI":"10.1007\/978-3-319-46493-0_39"},{"key":"12310_CR10","doi-asserted-by":"crossref","unstructured":"Hwang, W.-Y., Nguyen, V.-G., & Purba, S.W.D. (2022). Systematic survey of anything-to-text recognition and constructing its framework in language learning. Education and Information Technologies, 1\u201327","DOI":"10.1007\/s10639-022-11112-6"},{"key":"12310_CR11","unstructured":"Kiros, R., Salakhutdinov, R., & Zemel, R. (2014). Multimodal neural language models. International conference on machine learning (pp. 595\u2013603)"},{"key":"12310_CR12","doi-asserted-by":"crossref","unstructured":"Lewis, M., Liu, Y., Goyal, N., Ghazvininejad, M., Mohamed, A., Levy, O., & Zettlemoyer, L. (2019). Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint. arXiv:1910.13461","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"12310_CR13","doi-asserted-by":"crossref","unstructured":"Liu, C., Hou, J., Tu, Y.-F., Wang, Y., & Hwang, G.-J. (2021). Incorporating a reflective thinking promoting mechanism into artificial intelligencesupported english writing environments. Interactive Learning Environments, 1-19","DOI":"10.1080\/10494820.2021.2012812"},{"key":"12310_CR14","doi-asserted-by":"crossref","unstructured":"Lu, J., Yang, J., Batra, D., & Parikh, D. (2018). Neural baby talk. Proceedings of the ieee conference on computer vision and pattern recognition (pp. 7219\u20137228)","DOI":"10.1109\/CVPR.2018.00754"},{"issue":"4","key":"12310_CR15","doi-asserted-by":"publisher","first-page":"666","DOI":"10.1080\/09588221.2020.1744665","volume":"35","author":"T-H Nguyen","year":"2022","unstructured":"Nguyen, T.-H., Hwang, W.-Y., Pham, X.-L., & Pham, T. (2022). Self-experienced storytelling in an authentic context to facilitate efl writing. Computer Assisted Language Learning, 35(4), 666\u2013695.","journal-title":"Computer Assisted Language Learning"},{"key":"12310_CR16","doi-asserted-by":"crossref","unstructured":"Pedersoli, M., Lucas, T., Schmid, C., & Verbeek, J. (2017). Areas of attention for image captioning. Proceedings of the ieee international conference on computer vision (pp. 1242\u20131250)","DOI":"10.1109\/ICCV.2017.140"},{"key":"12310_CR17","doi-asserted-by":"crossref","unstructured":"Sammani, F., & Melas-Kyriazi, L. (2020). Show, edit and tell: a framework for editing image captions. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 4808\u20134816).","DOI":"10.1109\/CVPR42600.2020.00486"},{"key":"12310_CR18","doi-asserted-by":"crossref","unstructured":"Sennrich, R., Haddow, B., & Birch, A. (2015). Neural machine translation of rare words with subword units. arXiv preprint. arXiv:1508.07909","DOI":"10.18653\/v1\/P16-1162"},{"issue":"2","key":"12310_CR19","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1017\/S0958344020000038","volume":"32","author":"R Shadiev","year":"2020","unstructured":"Shadiev, R., Wu, T.-T., & Huang, Y.-M. (2020). Using image-to-text recognition technology to facilitate vocabulary acquisition in authentic contexts. ReCALL, 32(2), 195\u2013212.","journal-title":"ReCALL"},{"issue":"1","key":"12310_CR20","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: a simple way to prevent neural networks from overfitting. The journal of machine learning research, 15(1), 1929\u20131958.","journal-title":"The journal of machine learning research"},{"key":"12310_CR21","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., & Polosukhin, I. (2017). Attention is all you need. Advances In Neural Information Processing Systems, 30"},{"key":"12310_CR22","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., & Erhan, D. (2015). Show and tell: A neural image caption generator. Proceedings of the ieee conference on computer vision and pattern recognition (pp. 3156\u20133164)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"12310_CR23","unstructured":"Wang, P., Yang, A., Men, R., Lin, J., Bai, S., Li, Z., & Yang, H. (2022). Unifying architectures, tasks, and modalities through a simple sequenceto-sequence learning framework. arXiv preprint. arXiv:2202.03052"},{"key":"12310_CR24","doi-asserted-by":"crossref","unstructured":"Wu, Z., Nagarajan, T., Kumar, A., Rennie, S., Davis, L.S., Grauman, K., & Feris, R. (2018). Blockdrop: Dynamic inference paths in residual networks. Proceedings of the ieee conference on computer vision and pattern recognition (pp. 8817\u20138826)","DOI":"10.1109\/CVPR.2018.00919"},{"key":"12310_CR25","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., & Bengio, Y. (2015). Show, attend and tell: Neural image caption generation with visual attention. International conference on machine learning (pp. 2048\u20132057)"},{"key":"12310_CR26","unstructured":"Yang, Z., Yuan, Y., Wu, Y., Cohen, W.W., & Salakhutdinov, R.R. (2016). Review networks for caption generation. Advances in neural information processing systems, 29"},{"key":"12310_CR27","doi-asserted-by":"crossref","unstructured":"Zheng, G., Mukherjee, S., Dong, X.L., & Li, F. (2018). Opentag: Open attribute value extraction from product profiles. Proceedings of the 24th acm sigkdd international conference on knowledge discovery & data mining (pp. 1049\u20131058)","DOI":"10.1145\/3219819.3219839"}],"container-title":["Education and Information Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10639-023-12310-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10639-023-12310-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10639-023-12310-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,4]],"date-time":"2024-01-04T10:28:37Z","timestamp":1704364117000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10639-023-12310-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,28]]},"references-count":27,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["12310"],"URL":"https:\/\/doi.org\/10.1007\/s10639-023-12310-6","relation":{},"ISSN":["1360-2357","1573-7608"],"issn-type":[{"value":"1360-2357","type":"print"},{"value":"1573-7608","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,28]]},"assertion":[{"value":"29 December 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 November 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}