{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T14:38:37Z","timestamp":1742913517460,"version":"3.40.3"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031250682"},{"type":"electronic","value":"9783031250699"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-25069-9_21","type":"book-chapter","created":{"date-parts":[[2023,2,14]],"date-time":"2023-02-14T00:15:46Z","timestamp":1676333746000},"page":"314-328","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Incorporating Self-attention Mechanism and\u00a0Multi-task Learning into\u00a0Scene Text Detection"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5808-8795","authenticated-orcid":false,"given":"Ning","family":"Ding","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7793-1039","authenticated-orcid":false,"given":"Liangrui","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1009-1367","authenticated-orcid":false,"given":"Changsong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9549-2962","authenticated-orcid":false,"given":"Yuqi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3672-1331","authenticated-orcid":false,"given":"Ruixue","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4529-528X","authenticated-orcid":false,"given":"Jie","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,2,14]]},"reference":[{"key":"21_CR1","doi-asserted-by":"crossref","unstructured":"Bu\u0161ta, M., Neumann, L., Matas, J.: Deep TextSpotter: an End-to-End Trainable Scene Text Localization and Recognition Framework. In: ICCV, pp. 2223\u20132231 (2017)","DOI":"10.1109\/ICCV.2017.242"},{"key":"21_CR2","doi-asserted-by":"crossref","unstructured":"Chen, K., Pang, J., Wang, J., et al.: Hybrid task cascade for instance segmentation. In: CVPR, pp. 4974\u20134983 (2019)","DOI":"10.1109\/CVPR.2019.00511"},{"key":"21_CR3","unstructured":"Chen, K., Wang, J., Pang, J., et al.: MMDetection: open MMLab detection toolbox and benchmark. arXiv preprint arXiv:1906.07155 (2019)"},{"key":"21_CR4","doi-asserted-by":"crossref","unstructured":"Dasgupta, K., Das, S., Bhattacharya, U.: Stratified multi-task learning for robust spotting of scene texts. In: ICPR, pp. 3130\u20133137 (2021)","DOI":"10.1109\/ICPR48806.2021.9411951"},{"key":"21_CR5","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., et al.: ImageNet: a large-scale hierarchical image database. In: CVPR, pp. 248\u2013255 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"21_CR6","doi-asserted-by":"crossref","unstructured":"Gao, Y., Huang, Z., Dai, Y., et al.: DSAN: double supervised network with attention mechanism for scene text recognition. In: VCIP, pp. 1\u20134 (2019)","DOI":"10.1109\/VCIP47243.2019.8965779"},{"key":"21_CR7","doi-asserted-by":"crossref","unstructured":"Guo, M.H., Xu, T.X., Liu, J.J., et al.: Attention mechanisms in computer vision: a survey. Comput. Vis. Media 8(3), 331\u2013368 (2022)","DOI":"10.1007\/s41095-022-0271-y"},{"key":"21_CR8","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., et al.: Mask R-CNN. In: ICCV, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"21_CR9","doi-asserted-by":"crossref","unstructured":"Hu, Y., Zhang, Y., Yu, W., et al.: Transformer-convolution network for arbitrary shape text detection. In: ICMLSC, pp. 120\u2013126 (2022)","DOI":"10.1145\/3523150.3523169"},{"key":"21_CR10","first-page":"1","volume":"60","author":"Z Huang","year":"2021","unstructured":"Huang, Z., Li, W., Xia, X.G., et al.: A novel nonlocal-aware pyramid and multiscale multitask refinement detector for object detection in remote sensing images. IEEE Trans. Geosci. Remote Sens. 60, 1\u201320 (2021)","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Kittenplon, Y., Lavi, I., Fogel, S., et al.: Towards weakly-supervised text spotting using a multi-task transformer. In: CVPR, pp. 4604\u20134613 (2022)","DOI":"10.1109\/CVPR52688.2022.00456"},{"key":"21_CR12","unstructured":"Liang, T., Chu, X., Liu, Y., et al.: CBNetV2: a composite backbone network architecture for object detection. arXiv preprint arXiv:2107.00420 (2021)"},{"key":"21_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wang, Y., Wang, S., et al.: CBNet: a novel composite backbone network architecture for object detection. In: AAAI, pp. 11653\u201311660 (2020)","DOI":"10.1609\/aaai.v34i07.6834"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: ICCV, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"21_CR15","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (2019)"},{"key":"21_CR16","doi-asserted-by":"crossref","unstructured":"Nayef, N., Yin, F., Bizid, I., et al.: ICDAR2017 robust reading challenge on multi-lingual scene text detection and script identification-RRC-MLT. In: ICDAR, pp. 1454\u20131459 (2017)","DOI":"10.1109\/ICDAR.2017.237"},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"Nayef, N., Patel, Y., Busta, M., et al.: ICDAR2019 robust reading challenge on multi-lingual scene text detection and recognition-RRC-MLT-2019. In: ICDAR, pp. 1582\u20131587 (2019)","DOI":"10.1109\/ICDAR.2019.00254"},{"key":"21_CR18","first-page":"8024","volume":"32","author":"A Paszke","year":"2019","unstructured":"Paszke, A., Gross, S., Massa, F., et al.: PyTorch: an imperative style, high-performance deep learning library. Adv. Neural. Inf. Process. Syst. 32, 8024\u20138035 (2019)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Petsiuk, V., Jain, R., Manjunatha, V., et al.: Black-box explanation of object detectors via saliency maps. In: CVPR, pp. 11443\u201311452 (2021)","DOI":"10.1109\/CVPR46437.2021.01128"},{"key":"21_CR20","doi-asserted-by":"crossref","unstructured":"Qin, X., Zhou, Y., Guo, Y., et al.: Mask is all you need: rethinking mask R-CNN for dense and arbitrary-shaped scene text detection. In: ACM MM, pp. 414\u2013423 (2021)","DOI":"10.1145\/3474085.3475178"},{"key":"21_CR21","doi-asserted-by":"crossref","unstructured":"Raisi, Z., Naiel, M.A., Younes, G., et al.: Transformer-based text detection in the wild. In: CVPR Workshop, pp. 3162\u20133171 (2021)","DOI":"10.1109\/CVPRW53098.2021.00353"},{"key":"21_CR22","doi-asserted-by":"crossref","unstructured":"Sarshogh, M.R., Hines, K.: A multi-task network for localization and recognition of text in images. In: ICDAR, pp. 494\u2013501 (2019)","DOI":"10.1109\/ICDAR.2019.00085"},{"key":"21_CR23","doi-asserted-by":"crossref","unstructured":"Tang, J., Zhang, W., Liu, H., et al.: few could be better than all: feature sampling and grouping for scene text detection. In: CVPR, pp. 4563\u20134572 (2022)","DOI":"10.1109\/CVPR52688.2022.00452"},{"key":"21_CR24","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al.: Attention is all you need. In: NIPS (2017)"},{"issue":"12","key":"21_CR25","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.3390\/electronics9122125","volume":"9","author":"X Wu","year":"2020","unstructured":"Wu, X., Wang, T., Wang, S.: Cross-modal learning based on semantic correlation and multi-task learning for text-video retrieval. Electronics 9(12), 2125 (2020)","journal-title":"Electronics"},{"key":"21_CR26","doi-asserted-by":"crossref","unstructured":"Yan, R., Peng, L., Xiao, S., et al.: MEAN: multi-element attention network for scene text recognition. In: ICPR, pp. 6850\u20136857 (2021)","DOI":"10.1109\/ICPR48806.2021.9413166"},{"key":"21_CR27","doi-asserted-by":"crossref","unstructured":"Yan, R., Peng, L., Xiao, S., et al.: Primitive representation learning for scene text recognition. In: CVPR, pp. 284\u2013293 (2021)","DOI":"10.1109\/CVPR46437.2021.00035"},{"key":"21_CR28","doi-asserted-by":"crossref","unstructured":"Zhang, C., Liang, B., Huang, Z., et al.: Look more than once: an accurate detector for text of arbitrary shapes. In: CVPR, pp. 10552\u201310561 (2019)","DOI":"10.1109\/CVPR.2019.01080"},{"key":"21_CR29","unstructured":"Zhou, X., Koltun, V., Kr\u00e4henb\u00fchl, P.: Probabilistic two-stage detection. arXiv preprint arXiv:2103.07461 (2021)"},{"key":"21_CR30","unstructured":"Zhou, X., Wang, D., Kr\u00e4henb\u00fchl, P.: Objects as points. arXiv preprint arXiv:1904.07850 (2019)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-25069-9_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T12:52:15Z","timestamp":1709815935000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-25069-9_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031250682","9783031250699"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-25069-9_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"14 February 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"From the workshops, 367 reviewed full papers have been selected for publication","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}