{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T18:29:45Z","timestamp":1743013785476,"version":"3.40.3"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031705458"},{"type":"electronic","value":"9783031705465"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-70546-5_16","type":"book-chapter","created":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T05:02:47Z","timestamp":1725944567000},"page":"270-286","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multimodal Adaptive Inference for\u00a0Document Image Classification with\u00a0Anytime Early Exiting"],"prefix":"10.1007","author":[{"given":"Omar","family":"Hamed","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9383-3842","authenticated-orcid":false,"given":"Souhail","family":"Bakkali","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2640-181X","authenticated-orcid":false,"given":"Matthew","family":"Blaschko","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3732-9323","authenticated-orcid":false,"given":"Sien","family":"Moens","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9838-3024","authenticated-orcid":false,"given":"Jordy","family":"Van Landeghem","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,11]]},"reference":[{"key":"16_CR1","doi-asserted-by":"crossref","unstructured":"Appalaraju, S., Tang, P., Dong, Q., Sankaran, N., Zhou, Y., Manmatha, R.: DocFormerv2: local features for document understanding. arXiv preprint arXiv:2306.01733 (2023)","DOI":"10.1609\/aaai.v38i2.27828"},{"key":"16_CR2","doi-asserted-by":"crossref","unstructured":"Araujo, V., Hurtado, J., Soto, A., Moens, M.F.: Entropy-based stability-plasticity for lifelong learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3721\u20133728 (2022)","DOI":"10.1109\/CVPRW56347.2022.00416"},{"key":"16_CR3","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1016\/j.ins.2020.02.041","volume":"521","author":"E Baccarelli","year":"2020","unstructured":"Baccarelli, E., Scardapane, S., Scarpiniti, M., Momenzadeh, A., Uncini, A.: Optimized training and scalable implementation of conditional deep neural networks with early exits for fog-supported iot applications. Inf. Sci. 521, 107\u2013143 (2020)","journal-title":"Inf. Sci."},{"key":"16_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"306","DOI":"10.1007\/978-3-030-30484-3_26","volume-title":"Artificial Neural Networks and Machine Learning \u2013 ICANN 2019: Deep Learning","author":"K Berestizshevsky","year":"2019","unstructured":"Berestizshevsky, K., Even, G.: Dynamically sacrificing accuracy for reduced computation: cascaded inference based on softmax confidence. In: Tetko, I.V., K\u016frkov\u00e1, V., Karpov, P., Theis, F. (eds.) ICANN 2019. LNCS, vol. 11728, pp. 306\u2013320. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-30484-3_26"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Blau, T., et al.: GRAM: Global reasoning for multi-page VQA. arXiv preprint arXiv:2401.03411 (2024)","DOI":"10.1109\/CVPR52733.2024.01477"},{"key":"16_CR6","unstructured":"Geng, S., Gao, P., Fu, Z., Zhang, Y.: RomeBERt: robust training of multi-exit BERT (2021)"},{"key":"16_CR7","first-page":"39","volume":"34","author":"J Gu","year":"2021","unstructured":"Gu, J., et al.: UniDoc: unified pretraining framework for document understanding. Adv. Neural. Inf. Process. Syst. 34, 39\u201350 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Harley, A.W., Ufkes, A., Derpanis, K.G.: Evaluation of deep convolutional nets for document image classification and retrieval. In: 2015 13th International Conference on Document Analysis and Recognition (ICDAR), pp. 991\u2013995. IEEE (2015)","DOI":"10.1109\/ICDAR.2015.7333910"},{"key":"16_CR9","unstructured":"Huang, G., Chen, D., Li, T., Wu, F., Van Der\u00a0Maaten, L., Weinberger, K.Q.: Multi-scale dense networks for resource efficient image classification. arXiv preprint arXiv:1703.09844 (2017)"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Huang, Y., Lv, T., Cui, L., Lu, Y., Wei, F.: LayoutLMv3: pre-training for document AI with unified text and image masking. In: ACM International Conference on Multimedia, pp. 4083\u20134091 (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"16_CR11","doi-asserted-by":"crossref","unstructured":"Ilhan, F., et\u00a0al.: Adaptive deep neural network inference optimization with EENet. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1373\u20131382 (2024)","DOI":"10.1109\/WACV57701.2024.00140"},{"issue":"11","key":"16_CR12","doi-asserted-by":"publisher","first-page":"2881","DOI":"10.1109\/TCAD.2018.2857338","volume":"37","author":"NK Jayakodi","year":"2018","unstructured":"Jayakodi, N.K., Chatterjee, A., Choi, W., Doppa, J.R., Pande, P.P.: Trading-off accuracy and energy of deep inference on embedded systems: a co-design approach. IEEE Trans. Comput. Aided Des. Integr. Circuits Syst. 37(11), 2881\u20132893 (2018)","journal-title":"IEEE Trans. Comput. Aided Des. Integr. Circuits Syst."},{"key":"16_CR13","doi-asserted-by":"crossref","unstructured":"Laskaridis, S., Kouris, A., Lane, N.D.: Adaptive inference through early-exit networks: design, challenges and directions. In: Proceedings of the 5th International Workshop on Embedded and Mobile Deep Learning, pp.\u00a01\u20136 (2021)","DOI":"10.1145\/3469116.3470012"},{"key":"16_CR14","doi-asserted-by":"crossref","unstructured":"Li, H., Zhang, H., Qi, X., Yang, R., Huang, G.: Improved techniques for training adaptive deep networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1891\u20131900 (2019)","DOI":"10.1109\/ICCV.2019.00198"},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Li, P., et al.: SelfDoc: self-supervised document representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5652\u20135660 (2021)","DOI":"10.1109\/CVPR46437.2021.00560"},{"key":"16_CR16","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: Towards efficient NLP: a standard evaluation and a strong baseline. In: Carpuat, M., de\u00a0Marneffe, M.C., Meza\u00a0Ruiz, I.V. (eds.) Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Seattle, United States, pp. 3288\u20133303. Association for Computational Linguistics, July 2022","DOI":"10.18653\/v1\/2022.naacl-main.240"},{"issue":"5","key":"16_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3527155","volume":"55","author":"Y Matsubara","year":"2022","unstructured":"Matsubara, Y., Levorato, M., Restuccia, F.: Split computing and early exiting for deep learning applications: survey and research challenges. ACM Comput. Surv. 55(5), 1\u201330 (2022)","journal-title":"ACM Comput. Surv."},{"key":"16_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"732","DOI":"10.1007\/978-3-030-86331-9_47","volume-title":"Document Analysis and Recognition \u2013 ICDAR 2021","author":"R Powalski","year":"2021","unstructured":"Powalski, R., Borchmann, \u0141, Jurkiewicz, D., Dwojak, T., Pietruszka, M., Pa\u0142ka, G.: Going Full-TILT Boogie on document understanding with text-image-layout transformer. In: Llad\u00f3s, J., Lopresti, D., Uchida, S. (eds.) ICDAR 2021. LNCS, vol. 12822, pp. 732\u2013747. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-86331-9_47"},{"key":"16_CR19","doi-asserted-by":"crossref","unstructured":"Tang, S., et al.: You need multiple exiting: dynamic early exiting for accelerating unified vision language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10781\u201310791 (2023)","DOI":"10.1109\/CVPR52729.2023.01038"},{"key":"16_CR20","doi-asserted-by":"crossref","unstructured":"Teerapittayanon, S., McDanel, B., Kung, H.T.: BranchyNet: fast inference via early exiting from deep neural networks. In: 2016 23rd International Conference on Pattern Recognition (ICPR), pp. 2464\u20132469. IEEE (2016)","DOI":"10.1109\/ICPR.2016.7900006"},{"key":"16_CR21","doi-asserted-by":"crossref","unstructured":"Tishby, N., Zaslavsky, N.: Deep learning and the information bottleneck principle. In: 2015 IEEE information theory workshop (ITW), pp.\u00a01\u20135. IEEE (2015)","DOI":"10.1109\/ITW.2015.7133169"},{"key":"16_CR22","unstructured":"Wang, D., et al.: DocLLM: a layout-aware generative language model for multimodal document understanding (2023)"},{"key":"16_CR23","first-page":"11960","volume":"34","author":"Y Wang","year":"2021","unstructured":"Wang, Y., Huang, R., Song, S., Huang, Z., Huang, G.: Not all images are worth 16x16 words: dynamic transformers for efficient image recognition. Adv. Neural. Inf. Process. Syst. 34, 11960\u201311973 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR24","doi-asserted-by":"crossref","unstructured":"Xin, J., Tang, R., Yu, Y., Lin, J.: BERxiT: early exiting for BERT with better fine-tuning and extension to regression. In: Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume, pp. 91\u2013104 (2021)","DOI":"10.18653\/v1\/2021.eacl-main.8"},{"key":"16_CR25","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"275","DOI":"10.1007\/978-3-030-58517-4_17","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Q Xing","year":"2020","unstructured":"Xing, Q., Xu, M., Li, T., Guan, Z.: Early exit or not: resource-efficient blind quality enhancement for compressed images. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 275\u2013292. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_17"},{"key":"16_CR26","doi-asserted-by":"crossref","unstructured":"Xu, C., McAuley, J.: A survey on dynamic neural networks for natural language processing. arXiv preprint arXiv:2202.07101 (2022)","DOI":"10.18653\/v1\/2023.findings-eacl.180"},{"key":"16_CR27","doi-asserted-by":"crossref","unstructured":"Xue, Z., Marculescu, R.: Dynamic multimodal fusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2574\u20132583 (2023)","DOI":"10.1109\/CVPRW59228.2023.00256"},{"key":"16_CR28","unstructured":"Zhang, L., Tan, Z., Song, J., Chen, J., Bao, C., Ma, K.: Scan: a scalable neural networks framework towards compact and efficient models. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"16_CR29","first-page":"18330","volume":"33","author":"W Zhou","year":"2020","unstructured":"Zhou, W., Xu, C., Ge, T., McAuley, J., Xu, K., Wei, F.: BERT loses patience: fast and robust inference with early exit. Adv. Neural. Inf. Process. Syst. 33, 18330\u201318341 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition - ICDAR 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-70546-5_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T23:06:05Z","timestamp":1732748765000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-70546-5_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031705458","9783031705465"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-70546-5_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"11 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Athens","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icdar2024.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}