{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,24]],"date-time":"2025-11-24T12:43:41Z","timestamp":1763988221565,"version":"3.37.3"},"reference-count":68,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2021,6,30]],"date-time":"2021-06-30T00:00:00Z","timestamp":1625011200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,6,30]],"date-time":"2021-06-30T00:00:00Z","timestamp":1625011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100008982","name":"National Science Foundation","doi-asserted-by":"publisher","award":["1838193"],"award-info":[{"award-number":["1838193"]}],"id":[{"id":"10.13039\/501100008982","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["IJDAR"],"published-print":{"date-parts":[[2021,12]]},"DOI":"10.1007\/s10032-021-00382-4","type":"journal-article","created":{"date-parts":[[2021,6,30]],"date-time":"2021-06-30T20:02:30Z","timestamp":1625083350000},"page":"349-362","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Extracting text from scanned Arabic books: a large-scale benchmark dataset and a fine-tuned Faster-R-CNN model"],"prefix":"10.1007","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3690-7141","authenticated-orcid":false,"given":"Randa","family":"Elanwar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenda","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Margrit","family":"Betke","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Derry","family":"Wijaya","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,6,30]]},"reference":[{"key":"382_CR1","unstructured":"Abdelaziz, I., Abdou, S.: Altecondb: a large-vocabulary arabic online handwriting recognition database. arXiv:1412.7626 (2014)"},{"key":"382_CR2","unstructured":"Dobais, M.A.A, Alrasheed, F.A.G., Latif, G., Alzubaidi, L.: Adoptive thresholding and geometric features based physical layout analysis of scanned arabic books. In: 2018 IEEE 2nd international workshop on arabic and derived script analysis and recognition (ASAR), pp. 171\u2013176. IEEE (2018)"},{"key":"382_CR3","doi-asserted-by":"crossref","unstructured":"Albadi, N., Kurdi, M., Mishra, S.: Are they our brothers? Analysis and detection of religious hate speech in the arabic twittersphere. In: IEEE\/ACM International Conference on Advances in Social Networks Analysis and Mining (ASONAM), pp. 69\u201376 (2018)","DOI":"10.1109\/ASONAM.2018.8508247"},{"key":"382_CR4","unstructured":"Alexey, B., Yao, W.C., Yuan, L.H.: Yolov4: optimal speed and accuracy of object detection. In arXiv:2004.10934 (2020)"},{"key":"382_CR5","doi-asserted-by":"crossref","unstructured":"Almutairi, A., Almashan, M.: Instance segmentation of newspaper elements using mask R-CNN. In: 2019 18th IEEE International Conference On Machine Learning And Applications (ICMLA), pp. 1371\u20131375. IEEE (2019)","DOI":"10.1109\/ICMLA.2019.00223"},{"key":"382_CR6","doi-asserted-by":"crossref","unstructured":"Alshameri, A., Abdou, S., Mostafa, K.: A combined algorithm for layout analysis of Arabic document images and text lines extraction. Int. J. Comput. Appl. 49(23), 30\u201337 (2012)","DOI":"10.5120\/7945-1282"},{"key":"382_CR7","unstructured":"ALTEC dataset. http:\/\/www.altec-center.org\/conference\/?page_id=87"},{"key":"382_CR8","unstructured":"Amazon Mechanical Turk. https:\/\/www.mturk.com\/mturk\/welcome"},{"key":"382_CR9","unstructured":"The ASAR Physical Layout Analysis Challenge at the 2018 IEEE 2nd International Workshop on Arabic and Derived Script Analysis and Recognition, London, U.K., March 2018. https:\/\/asar.ieee.tn\/competition\/"},{"key":"382_CR10","unstructured":"2018 IEEE 2nd International Workshop on Arabic and Derived Script Analysis and Recognition, London, U.K., March 2018"},{"key":"382_CR11","doi-asserted-by":"crossref","unstructured":"Asi, A.,\u00a0Cohen, R.,\u00a0Kedem, K.,\u00a0El-Sana, J.,\u00a0Dinstein,I.: A coarse-to-fine approach for layout analysis of ancient manuscripts. In: 14th International Conference on Frontiers in Handwriting Recognition, pp. 140\u2013145 (2014)","DOI":"10.1109\/ICFHR.2014.31"},{"key":"382_CR12","doi-asserted-by":"crossref","unstructured":"Barakat, B.,\u00a0Droby, A.,\u00a0Kassis, M., El-Sana, J.: Text line segmentation for challenging handwritten document images using fully convolutional network. In: 16th International Conference on Frontiers in Handwriting Recognition (ICFHR), pp. 374\u2013379 (2018)","DOI":"10.1109\/ICFHR-2018.2018.00072"},{"key":"382_CR13","doi-asserted-by":"crossref","unstructured":"Barakat, B.K., El-Sana, J.: Binarization free layout analysis for arabic historical documents using fully convolutional networks. In: 2018 IEEE 2nd International Workshop on Arabic and Derived Script Analysis and Recognition (ASAR), pp. 151\u2013155. IEEE (2018)","DOI":"10.1109\/ASAR.2018.8480333"},{"key":"382_CR14","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1007\/978-1-4471-4072-6_5","volume-title":"Guide to OCR for Arabic Scripts","author":"A Bela\u00efd","year":"2012","unstructured":"Bela\u00efd, A., Ouwayed, N.: Segmentation of ancient Arabic documents. In: M\u00e4rgner, V., El Abed, H. (eds.) Guide to OCR for Arabic Scripts, pp. 103\u2013122. Springer, London (2012)"},{"key":"382_CR15","doi-asserted-by":"crossref","unstructured":"Boussellaa, W., Zahour, A., Taconet, B., Alimi, A., Benabdelhafid, A.: PRAAD: preprocessing and analysis tool for Arabic ancient documents. In: 9th International Conference on Document Analysis and Recognition (ICDAR), vol.\u00a02, pp. 1058\u20131062 (2007)","DOI":"10.1109\/ICDAR.2007.4377077"},{"key":"382_CR16","doi-asserted-by":"crossref","unstructured":"Bukhari, S.S., Azawi, A., Ali, M.I., Shafait, F., Breuel, T.M.: Document image segmentation using discriminative learning over connected components. In: Proceedings of the 9th IAPR International Workshop on Document Analysis Systems, Boston, pp. 183\u2013190 (2010)","DOI":"10.1145\/1815330.1815354"},{"key":"382_CR17","doi-asserted-by":"crossref","unstructured":"Bukhari, S.S., Breuel, T.M., Asi, A., El Sana, J.: Layout analysis for arabic historical document images using machine learning. In: 2012 International Conference on Frontiers in Handwriting Recognition, pp. 639\u2013644. IEEE (2012)","DOI":"10.1109\/ICFHR.2012.227"},{"issue":"2","key":"382_CR18","doi-asserted-by":"publisher","first-page":"125","DOI":"10.3390\/info11020125","volume":"11","author":"A Buslaev","year":"2020","unstructured":"Buslaev, A., Iglovikov, V.I., Khvedchenya, E., Parinov, A., Druzhinin, M., Kalinin, A.A.: Albumentations: fast and flexible image augmentations. Information 11(2), 125 (2020)","journal-title":"Information"},{"key":"382_CR19","doi-asserted-by":"crossref","unstructured":"Chen, K., Liu, C.L., Seuret, M., Liwicki, M., Hennebert, J., Ingold, R.: Page segmentation for historical document images based on superpixel classification with unsupervised feature learning. In: 12th IAPR workshop on document analysis systems (DAS), pp. 299\u2013304 (2016)","DOI":"10.1109\/DAS.2016.13"},{"key":"382_CR20","doi-asserted-by":"crossref","unstructured":"Chen, K., Seuret, M., Liwicki, M., Hennebert, J., Ingold, R.: Page segmentation of historical document images with convolutional autoencoders. In: 13th International Conference on Document Analysis and Recognition (ICDAR), pp. 1011\u20131015 (2015)","DOI":"10.1109\/ICDAR.2015.7333914"},{"key":"382_CR21","unstructured":"Cotterell, R., Callison-Burch, C.: A multi-dialect, multi-genre corpus of informal written arabic. In: LREC, pp. 241\u2013245 (2014)"},{"key":"382_CR22","unstructured":"Cotterell., Ryan, B., Chris, C..: A multi-dialect, multi-genre corpus of informal written Arabic. In: LREC, pp. 241\u2013245 (2014)"},{"key":"382_CR23","first-page":"248","volume":"2009","author":"J Deng","year":"2009","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. Comput. Vis. Pattern Recognit. CVPR 2009, 248\u2013255 (2009)","journal-title":"Comput. Vis. Pattern Recognit. CVPR"},{"key":"382_CR24","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., Fei, L.F.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"382_CR25","doi-asserted-by":"crossref","unstructured":"Abed, H.E., M\u00e4rgner, V., Kherallah, M., Alimi, A.M.: ICDAR 2009 online arabic handwriting recognition competition. In: 2009 10th International Conference on Document Analysis and Recognition, pp. 1388\u20131392. IEEE (2009)","DOI":"10.1109\/ICDAR.2009.284"},{"key":"382_CR26","doi-asserted-by":"crossref","unstructured":"El-Mawass, N., Alaboodi, S.: Detecting arabic spammers and content polluters on twitter. In: Sixth International Conference on Digital Information Processing and Communications (ICDIPC), pp. 53\u201358 (2016)","DOI":"10.1109\/ICDIPC.2016.7470791"},{"key":"382_CR27","doi-asserted-by":"crossref","unstructured":"Elanwar, R., Betke, M.: The ASAR 2018 competition on physical layout analysis of scanned arabic books (PLA-SAB 2018). In: 2018 IEEE 2nd International Workshop on Arabic and Derived Script Analysis and Recognition (ASAR), pp. 177\u2013182. IEEE (2018)","DOI":"10.1109\/ASAR.2018.8480194"},{"issue":"1\u20132","key":"382_CR28","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1007\/s10032-018-0298-x","volume":"21","author":"R Elanwar","year":"2018","unstructured":"Elanwar, R., Qin, W., Betke, M.: Making scanned arabic documents machine accessible using an ensemble of SVM classifiers. Int. J. Doc. Anal. Recognit. (IJDAR) 21(1\u20132), 59\u201375 (2018)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"382_CR29","doi-asserted-by":"crossref","unstructured":"Farra, N., McKeown, K., Habash, N.: Annotating targets of opinions in Arabic using crowdsourcing. In: Second workshop on Arabic natural language processing, pp. 89\u201398 (2015)","DOI":"10.18653\/v1\/W15-3210"},{"key":"382_CR30","doi-asserted-by":"crossref","unstructured":"Girshick, Ross.: Fast R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"382_CR31","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 580\u2013587 (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"382_CR32","doi-asserted-by":"crossref","unstructured":"Hadjar, K., Ingold, R.: Arabic newspaper page segmentation. In: 7th International Conference on Document Analysis and Recognition, pp. 895\u2014899 (2003)","DOI":"10.1109\/ICDAR.2003.1227789"},{"key":"382_CR33","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"382_CR34","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"4","key":"382_CR35","doi-asserted-by":"publisher","first-page":"1275","DOI":"10.1007\/s10044-017-0595-x","volume":"20","author":"AM Hesham","year":"2017","unstructured":"Hesham, A.M., Rashwan, M.A.A., Barhamtoshy, H.M.A., Abdou, S.M., Badr, A.A., Farag, I.: Arabic document layout analysis. Pattern Anal. Appl. 20(4), 1275\u20131287 (2017)","journal-title":"Pattern Anal. Appl."},{"key":"382_CR36","doi-asserted-by":"crossref","unstructured":"Kassis, M., El-Sana, J.: Scribble based interactive page layout segmentation using gabor filter. In: 15th International Conference on Frontiers in Handwriting Recognition (ICFHR), pp. 13\u201318 (2016)","DOI":"10.1109\/ICFHR.2016.0016"},{"key":"382_CR37","doi-asserted-by":"publisher","first-page":"107144","DOI":"10.1016\/j.patcog.2019.107144","volume":"100","author":"M Ibn Khedher","year":"2020","unstructured":"Ibn Khedher, M., Jmila, H., El-Yacoubi, M.A.: Automatic processing of historical arabic documents: a comprehensive survey. Pattern Recognit. 100, 107144 (2020)","journal-title":"Pattern Recognit."},{"key":"382_CR38","unstructured":"Lawson, N., Eustice, K., Perkowitz, M., Yetisgen-Yildiz, M.: Annotating large email datasets for named entity recognition with Mechanical Turk. In: Proceedings of the NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon\u2019s Mechanical Turk, pp. 71\u201379 (2010)"},{"key":"382_CR39","unstructured":"LabelMe tool. http:\/\/labelme.csail.mit.edu\/Release3.0\/"},{"key":"382_CR40","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440 (2015)","DOI":"10.1109\/CVPR.2015.7298965"},{"issue":"3","key":"382_CR41","doi-asserted-by":"publisher","first-page":"1096","DOI":"10.1016\/j.patcog.2013.08.009","volume":"47","author":"SA Mahmoud","year":"2014","unstructured":"Mahmoud, S.A., Ahmad, I., Khatib, W.G.A., Alshayeb, M., Parvez, M.T., M\u00e4rgner, V., Fink, G.A.: KHATT: an open arabic offline handwritten text database. Pattern Recognit. 47(3), 1096\u20131112 (2014)","journal-title":"Pattern Recognit."},{"key":"382_CR42","doi-asserted-by":"crossref","unstructured":"Mahmoud, S.A., Luqman, H., Al-Helali, B.M., BinMakhashen, G., Parvez, M.T.: Online-khatt: an open-vocabulary database for arabic online-text processing. Open Cybern. Syst. J. 12(1), 42\u201359 (2018)","DOI":"10.2174\/1874110X01812010042"},{"key":"382_CR43","unstructured":"Minghao, L., Yiheng, X., Lei, C., Shaohan, H., Furu, W., Zhoujun, L., Ming, Z.: Docbank: a benchmark dataset for document layout analysis. arXiv:2006.01038 (2020)"},{"key":"382_CR44","doi-asserted-by":"crossref","unstructured":"Neche, C., Belaid, A., Kacem-Echi, A.: Arabic handwritten documents segmentation into text-lines and words using deep learning. In: International Conference on Document Analysis and Recognition Workshops (ICDARW), pp. 19\u201324 (2019)","DOI":"10.1109\/ICDARW.2019.50110"},{"issue":"4","key":"382_CR45","doi-asserted-by":"publisher","first-page":"590","DOI":"10.1016\/j.imavis.2009.09.013","volume":"28","author":"N Nikolaou","year":"2010","unstructured":"Nikolaou, N., Makridis, M., Gatos, B., Stamatopoulos, N., Papamarkos, N.: Segmentation of historical machine-printed documents using adaptive run length smoothing and skeleton segmentation paths. Image Vis. Comput. 28(4), 590\u2013604 (2010)","journal-title":"Image Vis. Comput."},{"key":"382_CR46","doi-asserted-by":"crossref","unstructured":"Pastor-Pellicer, J., Afzal, M.Z., Liwicki, M., Castro-Bleda, M.J.: Complete system for text line extraction using convolutional neural networks and water-shed transform. In: 12th IAPR Workshop on Document Analysis Systems (DAS), pp. 30-35 (2016)","DOI":"10.1109\/DAS.2016.58"},{"key":"382_CR47","unstructured":"Pechwitz, M., Maddouri, S.S., M\u00e4rgner, V., Ellouze, N., Amiri, H.: IFN\/ENIT-database of handwritten arabic words. In: Proceedings of CIFED, volume\u00a02, pp. 127\u2013136. Citeseer (2002)"},{"key":"382_CR48","doi-asserted-by":"crossref","unstructured":"Pletschacher S., Antonacopoulos, A.: The PAGE (page analysis and ground-truth elements) format framework. In: 20th International Conference on Pattern Recognition (ICPR), pp. 257\u2013260 (2010)","DOI":"10.1109\/ICPR.2010.72"},{"key":"382_CR49","unstructured":"PyTorch sytem of libraries and tools for machine learning. https:\/\/pytorch.org\/ (2020)"},{"key":"382_CR50","unstructured":"Rashtchian, C., Youngand, P., Hodosh, M., Hockenmaier, J.: Collecting image annotations using Amazon\u2019s mechanical turk. In: Proceedings of the NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon\u2019s Mechanical Turk, pp. 139\u2013147 (2010)"},{"key":"382_CR51","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in neural information processing systems, pp. 91\u201399 (2015)"},{"key":"382_CR52","doi-asserted-by":"crossref","unstructured":"Saad, R.S.M., Elanwar, R., Abdel Kader, N.S., Mashali, S., Betke, M., Asar 2018 layout analysis challenge: using random forests to analyze scanned Arabic books. In: 2nd IEEE International Workshop on Arabic and derived Script Analysis and Recognition (ASAR 2018), London, March 2018, 2018. p. 6","DOI":"10.1109\/ASAR.2018.8480330"},{"key":"382_CR53","unstructured":"Rana\u00a0S.M.S., Randa\u00a0I.E., Abdel Kader, N.S., Samia, M., Margrit, B.: BCE-Arabic-v1 dataset: towards interpreting arabic document images for people with visual impairments. In: Proceedings of the 9th ACM International Conference on Pervasive Technologies Related to Assistive Environments, pp. 1\u20138 (2016)"},{"issue":"6","key":"382_CR54","doi-asserted-by":"publisher","first-page":"941","DOI":"10.1109\/TPAMI.2007.70837","volume":"30","author":"Faisal Shafait","year":"2008","unstructured":"Shafait, Faisal, Keysers, D., Breuel, T.: Performance evaluation and benchmarking of six-page segmentation algorithms. IEEE Trans. Pattern Anal. Mach. Intell. 30(6), 941\u2013954 (2008)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"382_CR55","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556 (2014)"},{"key":"382_CR56","doi-asserted-by":"crossref","unstructured":"Slimane, F., Ingold, R., Kanoun, S., Alimi, A.M., Hennebert, J.: A new arabic printed text image database and evaluation protocols. In: 2009 10th International Conference on Document Analysis and Recognition, pp. 946\u2013950. IEEE (2009)","DOI":"10.1109\/ICDAR.2009.155"},{"key":"382_CR57","unstructured":"Strassel, S.: Linguistic resources for arabic handwriting recognition. In: Proceedings of the Second International Conference on Arabic Language Resources and Tools, Cairo, Egypt (2009)"},{"key":"382_CR58","doi-asserted-by":"crossref","unstructured":"Studer, L., Alberti, M., Pondenkandath, V., Goktepey, P., Kolonko, T., Fischeryz, A., Liwicki, M., Ingold, R.: A comprehensive study of imagenet pre-training for historical document image analysis. In: 15th International Conference on Document Analysis and Recognition (ICDAR), pp. 720\u2013725 (2019)","DOI":"10.1109\/ICDAR.2019.00120"},{"issue":"11","key":"382_CR59","doi-asserted-by":"publisher","first-page":"1958","DOI":"10.1109\/TPAMI.2008.128","volume":"30","author":"A Torralba","year":"2008","unstructured":"Torralba, A., Fergus, R., Freeman, W.T.: 80 million tiny images: a large data set for nonparametric object and scene recognition. IEEE Trans. Pattern Anal. Mach. Intell. 30(11), 1958\u20131970 (2008)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"382_CR60","doi-asserted-by":"crossref","unstructured":"Wei, H., Seuret, M.,\u00a0Chen, K.,\u00a0Fischer, A.,\u00a0Liwicki, M.,\u00a0Ingold, R.: Selecting autoencoder features for layout analysis of historical documents. In: ACM 3rd International Workshop on Historical Document Imaging and Processing, pp. 55\u201362 (2015)","DOI":"10.1145\/2809544.2809548"},{"key":"382_CR61","doi-asserted-by":"crossref","unstructured":"Wick, C., Puppe, F.: Fully convolutional neural networks for page segmentation of historical document images. In: 2018 13th IAPR International Workshop on Document Analysis Systems (DAS), pp. 287\u2013292. IEEE (2018)","DOI":"10.1109\/DAS.2018.39"},{"key":"382_CR62","doi-asserted-by":"crossref","unstructured":"Wray, S., Mubarak, H., Ali,A.: Best practices for crowdsourcing dialectal arabic speech transcription. In: ANLP Workshop, p.\u00a099 (2015)","DOI":"10.18653\/v1\/W15-3211"},{"key":"382_CR63","doi-asserted-by":"crossref","unstructured":"Wray, S., Mubarak, H., Ali, A.: Best practices for crowdsourcing dialectal arabic speech transcription. In: Proceedings of the Second Workshop on Arabic Natural Language Processing, pp. 99\u2013107 (2015)","DOI":"10.18653\/v1\/W15-3211"},{"key":"382_CR64","unstructured":"Zaidan, O.F., Callison-Burch, C.: The Arabic online commentary dataset: an annotated dataset of informal Arabic with high dialectal content. In: Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies: short papers, 2:37\u201341 (2011)"},{"issue":"1","key":"382_CR65","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1162\/COLI_a_00169","volume":"40","author":"OF Zaidan","year":"2014","unstructured":"Zaidan, O.F., Callison-Burch, C.: Arabic dialect identification. Comput. Linguist. 40(1), 171\u2013202 (2014)","journal-title":"Comput. Linguist."},{"key":"382_CR66","unstructured":"Zaidan, O.F., Burch, C.C..: The arabic online commentary dataset: an annotated dataset of informal arabic with high dialectal content. In: Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies: short papers-volume 2, pp. 37\u201341. Association for Computational Linguistics (2011)"},{"key":"382_CR67","doi-asserted-by":"crossref","unstructured":"Zaidan, O.F., Burch, C.C.: Arabic dialect identification. Comput. Linguist. 40(1), 171\u2013202 (2014)","DOI":"10.1162\/COLI_a_00169"},{"key":"382_CR68","doi-asserted-by":"crossref","unstructured":"Zhong, X., Jianbin, T., Jimeno, Y.A.: Publaynet: largest dataset ever for document layout analysis. In: 15th International Conference on Document Analysis and Recognition (ICDAR) (2019)","DOI":"10.1109\/ICDAR.2019.00166"}],"container-title":["International Journal on Document Analysis and Recognition (IJDAR)"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-021-00382-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10032-021-00382-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-021-00382-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,3]],"date-time":"2024-09-03T01:17:48Z","timestamp":1725326268000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10032-021-00382-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,30]]},"references-count":68,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2021,12]]}},"alternative-id":["382"],"URL":"https:\/\/doi.org\/10.1007\/s10032-021-00382-4","relation":{},"ISSN":["1433-2833","1433-2825"],"issn-type":[{"type":"print","value":"1433-2833"},{"type":"electronic","value":"1433-2825"}],"subject":[],"published":{"date-parts":[[2021,6,30]]},"assertion":[{"value":"3 July 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 April 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 June 2021","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 June 2021","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}