{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T03:03:25Z","timestamp":1743131005808,"version":"3.40.3"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031235849"},{"type":"electronic","value":"9783031235856"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-23585-6_3","type":"book-chapter","created":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:29:01Z","timestamp":1672532941000},"page":"25-35","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Scheduled Mask Method for\u00a0TextVQA"],"prefix":"10.1007","author":[{"given":"Mingjie","family":"Han","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wancong","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,1,1]]},"reference":[{"issue":"7553","key":"3_CR1","first-page":"436","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun, Y., Bengio, Y., Hinton, G.: Deep learning. Nature 521(7553), 436\u2013444 (2015)","journal-title":"Deep learning. Nature"},{"key":"3_CR2","doi-asserted-by":"crossref","unstructured":"Zhao, Z.Q., Zheng, P., Xu, S.t., Wu, X.: Object detection with deep learning: a review. IEEE Trans. Neural Networks Learn. Syst. 30(11), 3212\u20133232 (2019)","DOI":"10.1109\/TNNLS.2018.2876865"},{"key":"3_CR3","doi-asserted-by":"crossref","unstructured":"Amit, Y., Felzenszwalb, P., Girshick, R.: Object detection. Computer Vision : A Reference Guide, pp. 1\u20139 (2020)","DOI":"10.1007\/978-3-030-03243-2_660-1"},{"key":"3_CR4","doi-asserted-by":"crossref","unstructured":"Wang, P., Chen, P., Yuan, Y., Liu, D., Huang, Z., Hou, X., Cottrell, G.: Understanding convolution for semantic segmentation. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1451\u20131460. IEEE (2018)","DOI":"10.1109\/WACV.2018.00163"},{"issue":"2","key":"3_CR5","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1007\/s13735-017-0141-z","volume":"7","author":"Y Guo","year":"2018","unstructured":"Guo, Y., Liu, Y., Georgiou, T., Lew, M.S.: A review of semantic segmentation using deep neural networks. Int. J. Multimed. Inf. Retrieval 7(2), 87\u201393 (2018)","journal-title":"Int. J. Multimed. Inf. Retrieval"},{"key":"3_CR6","unstructured":"Gregor, K., Danihelka, I., Graves, A., Rezende, D., Wierstra, D.: Draw: a recurrent neural network for image generation. In: International Conference on Machine Learning, PMLR, pp. 1462\u20131471 (2015)"},{"key":"3_CR7","doi-asserted-by":"crossref","unstructured":"Qiao, T., Zhang, J., Xu, D., Tao, D.: Mirrorgan: learning text-to-image generation by redescription. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1505\u20131514 (2019)","DOI":"10.1109\/CVPR.2019.00160"},{"issue":"4","key":"3_CR8","doi-asserted-by":"publisher","first-page":"150","DOI":"10.3390\/info10040150","volume":"10","author":"K Kowsari","year":"2019","unstructured":"Kowsari, K., Jafari Meimandi, K., Heidarysafa, M., Mendu, S., Barnes, L., Brown, D.: Text classification algorithms: a survey. Information 10(4), 150 (2019)","journal-title":"Information"},{"key":"3_CR9","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1016\/j.eswa.2018.03.058","volume":"106","author":"MM Miro\u0144czuk","year":"2018","unstructured":"Miro\u0144czuk, M.M., Protasiewicz, J.: A recent overview of the state-of-the-art elements of text classification. Expert Syst. Appl. 106, 36\u201354 (2018)","journal-title":"Expert Syst. Appl."},{"key":"3_CR10","doi-asserted-by":"crossref","unstructured":"Piskorski, J., Yangarber, R.: Information extraction: past, present and future. In: Multi-source, Multilingual Information Extraction and Summarization, pp. 23\u201349. Springer (2013)","DOI":"10.1007\/978-3-642-28569-1_2"},{"issue":"4","key":"3_CR11","doi-asserted-by":"publisher","DOI":"10.1063\/5.0021106","volume":"7","author":"EA Olivetti","year":"2020","unstructured":"Olivetti, E.A., Cole, J.M., Kim, E., Kononova, O., Ceder, G., Han, T.Y.J., Hiszpanski, A.M.: Data-driven materials research enabled by natural language processing and information extraction. Appl. Phys. Rev. 7(4), 041317 (2020)","journal-title":"Appl. Phys. Rev."},{"key":"3_CR12","doi-asserted-by":"crossref","unstructured":"Haque, S., LeClair, A., Wu, L., McMillan, C.: Improved automatic summarization of subroutines via attention to file context. In: Proceedings of the 17th International Conference on Mining Software Repositories, pp. 300\u2013310 (2020)","DOI":"10.1145\/3379597.3387449"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Moreno, L., Marcus, A.: Automatic software summarization: the state of the art. In: Proceedings of the 40th International Conference on Software Engineering: Companion Proceeedings, pp. 530\u2013531 (2018)","DOI":"10.1145\/3183440.3183464"},{"key":"3_CR14","doi-asserted-by":"crossref","unstructured":"Hu, R., Singh, A., Darrell, T., Rohrbach, M.: Iterative answer prediction with pointer-augmented multimodal transformers for textvqa. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9992\u201310002 (2020)","DOI":"10.1109\/CVPR42600.2020.01001"},{"key":"3_CR15","doi-asserted-by":"crossref","unstructured":"Singh, A., et al.: Towards vqa models that can read. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8317\u20138326 (2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"3_CR16","doi-asserted-by":"crossref","unstructured":"Gao, C., et al.: Structured multimodal attentions for textvqa. IEEE Trans. Pattern Anal. Mach. Intell. (2021)","DOI":"10.1109\/TPAMI.2021.3132034"},{"key":"3_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108455","volume":"124","author":"X Li","year":"2022","unstructured":"Li, X., Wu, B., Song, J., Gao, L., Zeng, P., Gan, C.: Text-instance graph: exploring the relational semantics for text-based visual question answering. Pattern Recogn. 124, 108455 (2022)","journal-title":"Pattern Recogn."},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Liu, F., Xu, G., Wu, Q., Du, Q., Jia, W., Tan, M.: Cascade reasoning network for text-based visual question answering. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4060\u20134069 (2020)","DOI":"10.1145\/3394171.3413924"},{"key":"3_CR19","doi-asserted-by":"crossref","unstructured":"Mikolov, T., Karafi\u00e1t, M., Burget, L., Cernock\u1ef3, J., Khudanpur, S.: Recurrent neural network based language model. In: Interspeech. Volume 2, Makuhari, pp. 1045\u20131048 (2010)","DOI":"10.21437\/Interspeech.2010-343"},{"key":"3_CR20","unstructured":"Bengio, S., Vinyals, O., Jaitly, N., Shazeer, N.: Scheduled sampling for sequence prediction with recurrent neural networks. In: Advances in Neural Information Processing Systems 28 (2015)"},{"key":"3_CR21","doi-asserted-by":"crossref","unstructured":"Yu, L., Zhang, W., Wang, J., Yu, Y.: Seqgan: sequence generative adversarial nets with policy gradient. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 31 (2017)","DOI":"10.1609\/aaai.v31i1.10804"},{"key":"3_CR22","unstructured":"Finn, C., Goodfellow, I., Levine, S.: Unsupervised learning for physical interaction through video prediction. In: Advances in Neural Information Processing Systems 29 (2016)"},{"key":"3_CR23","doi-asserted-by":"crossref","unstructured":"Martinez, J., Black, M.J., Romero, J.: On human motion prediction using recurrent neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2891\u20132900 (2017)","DOI":"10.1109\/CVPR.2017.497"},{"key":"3_CR24","unstructured":"Ranzato, M., Chopra, S., Auli, M., Zaremba, W.: Sequence level training with recurrent neural networks. arXiv preprint arXiv:1511.06732 (2015)"},{"key":"3_CR25","unstructured":"Lamb, A.M., Alias Parth Goyal, A.G., Zhang, Y., Zhang, S., Courville, A.C., Bengio, Y.: Professor forcing: A new algorithm for training recurrent networks. Advances in neural information processing systems 29 (2016)"},{"key":"3_CR26","doi-asserted-by":"crossref","unstructured":"Schmidt, F.: Generalization in generation: A closer look at exposure bias. arXiv preprint arXiv:1910.00292 (2019)","DOI":"10.18653\/v1\/D19-5616"},{"key":"3_CR27","unstructured":"Krasin, I., et al.: Openimages: a public dataset for large-scale multi-label and multi-class image classification. Dataset available from https:\/\/github.com\/openimages 2(3) (2017) 18"},{"key":"3_CR28","doi-asserted-by":"crossref","unstructured":"Antol, S., Agrawal, A., Lu, J., Mitchell, M., Batra, D., Zitnick, C.L., Parikh, D.: Vqa: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"3_CR29","doi-asserted-by":"crossref","unstructured":"Borisyuk, F., Gordo, A., Sivakumar, V.: Rosetta: large scale system for text detection and recognition in images. In: Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 71\u201379 (2018)","DOI":"10.1145\/3219819.3219861"},{"key":"3_CR30","doi-asserted-by":"crossref","unstructured":"Zeng, G., Zhang, Y., Zhou, Y., Yang, X.: Beyond ocr+ vqa: involving ocr into the flow for robust and accurate textvqa. In: Proceedings of the 29th ACM International Conference on Multimedia (2021) 376\u2013385","DOI":"10.1145\/3474085.3475606"},{"key":"3_CR31","unstructured":"Zhu, Q., Gao, C., Wang, P., Wu, Q.: Simple is not easy: A simple strong baseline for textvqa and textcaps. arXiv preprint arXiv:2012.05153 2 (2020)"},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Han, W., Huang, H., Han, T.: Finding the evidence: Localization-aware answer prediction for text visual question answering. arXiv preprint arXiv:2010.02582 (2020)","DOI":"10.18653\/v1\/2020.coling-main.278"}],"container-title":["Lecture Notes in Computer Science","Cognitive Computing \u2013 ICCC 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-23585-6_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:33:43Z","timestamp":1672533223000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-23585-6_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031235849","9783031235856"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-23585-6_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"1 January 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Cognitive Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Honolulu, HI","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 December 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 December 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iccc2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.servicessociety.org\/iccc","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EDAS","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"17","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"6","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"35% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}