{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,21]],"date-time":"2026-08-21T12:30:52Z","timestamp":1787315452792,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":44,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030535513","type":"print"},{"value":"9783030535520","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-53552-0_14","type":"book-chapter","created":{"date-parts":[[2020,7,17]],"date-time":"2020-07-17T12:50:23Z","timestamp":1594990223000},"page":"128-142","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Active Learning Based Framework for Image Captioning Corpus Creation"],"prefix":"10.1007","author":[{"given":"Moustapha","family":"Cheikh","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mounir","family":"Zrigui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,7,18]]},"reference":[{"issue":"1","key":"14_CR1","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1007\/s11263-016-0966-6","volume":"123","author":"A Agrawal","year":"2017","unstructured":"Agrawal, A., et al.: VQA: visual question answering. Int. J. Comput. Vision 123(1), 4\u201331 (2017)","journal-title":"Int. J. Comput. Vision"},{"key":"14_CR2","doi-asserted-by":"crossref","unstructured":"Al-Muzaini, H.A., Al-Yahya, T.N., Benhidour, H.: Automatic Arabic image captioning using RNN-LSTM-based language model and CNN. Database 9(6) (2018)","DOI":"10.14569\/IJACSA.2018.090610"},{"key":"14_CR3","series-title":"Communications in Computer and Information Science","doi-asserted-by":"publisher","first-page":"491","DOI":"10.1007\/978-3-319-24770-0_42","volume-title":"Information and Software Technologies","author":"R Ayadi","year":"2015","unstructured":"Ayadi, R., Maraoui, M., Zrigui, M.: LDA and LSI as a dimensionality reduction method in Arabic document classification. In: Dregvaite, G., Damasevicius, R. (eds.) ICIST 2015. CCIS, vol. 538, pp. 491\u2013502. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24770-0_42"},{"key":"14_CR4","unstructured":"Bacha, K., Zrigui, M.: Machine translation system on the pair of Arabic\/English. In: KEOD, pp. 347\u2013351 (2012)"},{"key":"14_CR5","unstructured":"Brants, T.: Web 1t 5-gram version 1. http:\/\/www.ldc.upenn.edu\/Catalog\/CatalogEntry.jsp?catalogId=LDC2006T13 (2006)"},{"key":"14_CR6","doi-asserted-by":"crossref","unstructured":"Cho, K., et al.: Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078 (2014)","DOI":"10.3115\/v1\/D14-1179"},{"key":"14_CR7","unstructured":"Chung, J., Gulcehre, C., Cho, K., Bengio, Y.: Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv preprint arXiv:1412.3555 (2014)"},{"key":"14_CR8","doi-asserted-by":"crossref","unstructured":"Donahue, J., et al.: Long-term recurrent convolutional networks for visual recognition and description. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2625\u20132634 (2015)","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"14_CR9","doi-asserted-by":"crossref","unstructured":"Elliott, D., Frank, S., Sima\u2019an, K., Specia, L.: Multi30k: multilingual English-German image descriptions. arXiv preprint arXiv:1605.00459 (2016)","DOI":"10.18653\/v1\/W16-3210"},{"key":"14_CR10","doi-asserted-by":"crossref","unstructured":"Farhani, N., Terbeh, N., Zrigui, M.: Image to text conversion: state of the art and extended work. In: 2017 IEEE\/ACS 14th International Conference on Computer Systems and Applications (AICCSA), pp. 937\u2013943. IEEE (2017)","DOI":"10.1109\/AICCSA.2017.159"},{"issue":"9","key":"14_CR11","doi-asserted-by":"publisher","first-page":"1627","DOI":"10.1109\/TPAMI.2009.167","volume":"32","author":"PF Felzenszwalb","year":"2010","unstructured":"Felzenszwalb, P.F., Girshick, R.B., McAllester, D., Ramanan, D.: Object detection with discriminatively trained part-based models. IEEE Trans. Pattern Anal. Mach. Intell. 32(9), 1627\u20131645 (2010)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"14_CR12","unstructured":"Filipe, J., Fred, A.L.N. (eds.): ICAART 2013 - Proceedings of the 5th International Conference on Agents and Artificial Intelligence, Barcelona, Spain, 15\u201318 February 2013, vol. 2. SciTePress (2013)"},{"key":"14_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1007\/978-3-319-10593-2_35","volume-title":"Computer Vision \u2013 ECCV 2014","author":"Y Gong","year":"2014","unstructured":"Gong, Y., Wang, L., Hodosh, M., Hockenmaier, J., Lazebnik, S.: Improving image-sentence embeddings using large weakly annotated photo collections. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8692, pp. 529\u2013545. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10593-2_35"},{"key":"14_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"14_CR15","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh, M., Young, P., Hockenmaier, J.: Framing image description as a ranking task: data, models and evaluation metrics. J. Artif. Intell. Res. 47, 853\u2013899 (2013)","journal-title":"J. Artif. Intell. Res."},{"key":"14_CR16","doi-asserted-by":"crossref","unstructured":"Jia, Y., et al.: Caffe: convolutional architecture for fast feature embedding. In: Proceedings of the 22nd ACM International Conference on Multimedia, pp. 675\u2013678. ACM (2014)","DOI":"10.1145\/2647868.2654889"},{"key":"14_CR17","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"14_CR18","unstructured":"Kiros, R., Salakhutdinov, R., Zemel, R.: Multimodal neural language models. In: International Conference on Machine Learning, pp. 595\u2013603 (2014)"},{"key":"14_CR19","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: ImageNet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"key":"14_CR20","doi-asserted-by":"crossref","unstructured":"Kulkarni, G., et al.: Baby talk: understanding and generating simple image descriptions. In: 2011 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2011, pp. 1601\u20131608 (2011)","DOI":"10.1109\/CVPR.2011.5995466"},{"issue":"1","key":"14_CR21","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1162\/tacl_a_00188","volume":"2","author":"P Kuznetsova","year":"2014","unstructured":"Kuznetsova, P., Ordonez, V., Berg, T., Choi, Y.: TREETALK: composition and compression of trees for image descriptions. Trans. Assoc. Comput. Linguist. 2(1), 351\u2013362 (2014)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"14_CR22","unstructured":"Li, S., Kulkarni, G., Berg, T.L., Berg, A.C., Choi, Y.: Composing simple image descriptions using web-scale N-grams. In: Proceedings of the Fifteenth Conference on Computational Natural Language Learning, pp. 220\u2013228. Association for Computational Linguistics (2011)"},{"key":"14_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"14_CR24","doi-asserted-by":"crossref","unstructured":"Lu, J., Xiong, C., Parikh, D., Socher, R.: Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), vol. 6, p. 2 (2017)","DOI":"10.1109\/CVPR.2017.345"},{"key":"14_CR25","doi-asserted-by":"crossref","unstructured":"Mahmoud, A., Zrigui, M.: Artificial method for building monolingual plagiarized Arabic corpus. Computaci\u00f3n Sistemas 22(3), 767\u2013776 (2018)","DOI":"10.13053\/cys-22-3-3019"},{"issue":"1","key":"14_CR26","first-page":"75","volume":"22","author":"S Mansouri","year":"2018","unstructured":"Mansouri, S., Charhad, M., Zrigui, M.: A heuristic approach to detect and localize text in Arabic news video. Computaci\u00f3n Sistemas 22(1), 75\u201382 (2018)","journal-title":"Computaci\u00f3n Sistemas"},{"key":"14_CR27","unstructured":"Mikolov, T., Sutskever, I., Chen, K., Corrado, G.S., Dean, J.: Distributed representations of words and phrases and their compositionality. In: Advances in Neural Information Processing Systems, pp. 3111\u20133119 (2013)"},{"key":"14_CR28","unstructured":"Mitchell, M., et al.: Midge: generating image descriptions from computer vision detections. In: Proceedings of the 13th Conference of the European Chapter of the Association for Computational Linguistics, pp. 747\u2013756. Association for Computational Linguistics (2012)"},{"key":"14_CR29","doi-asserted-by":"crossref","unstructured":"Mun, J., Cho, M., Han, B.: Text-guided attention model for image captioning. In: AAAI, pp. 4233\u20134239 (2017)","DOI":"10.1609\/aaai.v31i1.11237"},{"key":"14_CR30","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.L.: Im2Text: describing images using 1 million captioned photographs. In: Advances in Neural Information Processing Systems, pp. 1143\u20131151 (2011)"},{"key":"14_CR31","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: BLEU: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting on Association for computational Linguistics, pp. 311\u2013318. Association for Computational Linguistics (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"14_CR32","unstructured":"Raffel, C., Ellis, D.P.: Feed-forward networks with attention can solve some long-term memory problems. arXiv preprint arXiv:1512.08756 (2015)"},{"key":"14_CR33","unstructured":"\u0158eh\u016f\u0159ek, R., Sojka, P.: Software framework for topic modelling with large corpora. In: Proceedings of the LREC 2010 Workshop on New Challenges for NLP Frameworks, pp. 45\u201350. ELRA, Valletta, Malta, May 2010. http:\/\/is.muni.cz\/publication\/884893\/en"},{"key":"14_CR34","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"issue":"1","key":"14_CR35","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1162\/tacl_a_00177","volume":"2","author":"R Socher","year":"2014","unstructured":"Socher, R., Karpathy, A., Le, Q.V., Manning, C.D., Ng, A.Y.: Grounded compositional semantics for finding and describing images with sentences. Trans. Assoc. Comput. Linguis. 2(1), 207\u2013218 (2014)","journal-title":"Trans. Assoc. Comput. Linguis."},{"key":"14_CR36","unstructured":"Sutskever, I., Vinyals, O., Le, Q.V.: Sequence to sequence learning with neural networks. In: Advances in Neural Information Processing Systems, pp. 3104\u20133112 (2014)"},{"key":"14_CR37","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"14_CR38","unstructured":"Xu, K., et al.: Show, attend and tell: neural image caption generation with visual attention. In: International Conference on Machine Learning, pp. 2048\u20132057 (2015)"},{"key":"14_CR39","unstructured":"Yang, Y., Teo, C.L., Daum\u00e9 III, H., Aloimonos, Y.: Corpus-guided sentence generation of natural images. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 444\u2013454. Association for Computational Linguistics (2011)"},{"key":"14_CR40","doi-asserted-by":"crossref","unstructured":"Yoshikawa, Y., Shigeto, Y., Takeuchi, A.: Stair captions: constructing a large-scale Japanese image caption dataset. arXiv preprint arXiv:1705.00823 (2017)","DOI":"10.18653\/v1\/P17-2066"},{"key":"14_CR41","doi-asserted-by":"crossref","unstructured":"You, Q., Jin, H., Wang, Z., Fang, C., Luo, J.: Image captioning with semantic attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4651\u20134659 (2016)","DOI":"10.1109\/CVPR.2016.503"},{"key":"14_CR42","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguist. 2, 67\u201378 (2014)","journal-title":"Trans. Assoc. Comput. Linguist."},{"issue":"2","key":"14_CR43","doi-asserted-by":"publisher","first-page":"125","DOI":"10.2498\/cit.1001770","volume":"20","author":"M Zrigui","year":"2012","unstructured":"Zrigui, M., Ayadi, R., Mars, M., Maraoui, M.: Arabic text classification framework based on latent dirichlet allocation. J. Comput. Inf. Technol. 20(2), 125\u2013140 (2012)","journal-title":"J. Comput. Inf. Technol."},{"issue":"3","key":"14_CR44","doi-asserted-by":"publisher","first-page":"245","DOI":"10.2498\/cit.1001478","volume":"18","author":"M Zrigui","year":"2010","unstructured":"Zrigui, M., Charhad, M., Zouaghi, A.: A framework of indexation and document video retrieval based on the conceptual graphs. J. Comput. Inf. Technol. 18(3), 245\u2013256 (2010)","journal-title":"J. Comput. Inf. Technol."}],"container-title":["Lecture Notes in Computer Science","Learning and Intelligent Optimization"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-53552-0_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,3]],"date-time":"2022-11-03T01:00:11Z","timestamp":1667437211000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-53552-0_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030535513","9783030535520"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-53552-0_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"18 July 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"LION","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Learning and Intelligent Optimization","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Athens","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 May 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 May 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"lion2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.caopt.com\/LION14\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}