{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T14:58:57Z","timestamp":1781621937813,"version":"3.54.5"},"publisher-location":"Cham","reference-count":118,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031705458","type":"print"},{"value":"9783031705465","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-70546-5_12","type":"book-chapter","created":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T05:02:47Z","timestamp":1725944567000},"page":"195-217","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["DistilDoc: Knowledge Distillation for\u00a0Visually-Rich Document Applications"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9838-3024","authenticated-orcid":false,"given":"Jordy","family":"Van Landeghem","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0735-8406","authenticated-orcid":false,"given":"Subhajit","family":"Maity","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0269-2202","authenticated-orcid":false,"given":"Ayan","family":"Banerjee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2640-181X","authenticated-orcid":false,"given":"Matthew","family":"Blaschko","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3732-9323","authenticated-orcid":false,"given":"Marie-Francine","family":"Moens","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4533-4739","authenticated-orcid":false,"given":"Josep","family":"Llad\u00f3s","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6648-8270","authenticated-orcid":false,"given":"Sanket","family":"Biswas","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,11]]},"reference":[{"key":"12_CR1","doi-asserted-by":"crossref","unstructured":"Aditya, S., Saha, R., Yang, Y., Baral, C.: Spatial knowledge distillation to aid visual reasoning. In: 2019 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 227\u2013235 (2019)","DOI":"10.1109\/WACV.2019.00030"},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Ahn, S., Hu, S.X., Damianou, A., Lawrence, N.D., Dai, Z.: Variational information distillation for knowledge transfer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9163\u20139171 (2019)","DOI":"10.1109\/CVPR.2019.00938"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Antonacopoulos, A., Bridson, D., Papadopoulos, C., Pletschacher, S.: A realistic dataset for performance evaluation of document layout analysis. In: 2009 10th International Conference on Document Analysis and Recognition, pp. 296\u2013300. IEEE (2009)","DOI":"10.1109\/ICDAR.2009.271"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Appalaraju, S., Jasani, B., Kota, B.U., Xie, Y., Manmatha, R.: Docformer: end-to-end transformer for document understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 993\u20131003 (2021)","DOI":"10.1109\/ICCV48922.2021.00103"},{"key":"12_CR5","unstructured":"Ba, J., Caruana, R.: Do deep nets really need to be deep? Adv. Neural Inf. Process. Syst. (2014)"},{"key":"12_CR6","unstructured":"Bagherinezhad, H., Horton, M., Rastegari, M., Farhadi, A.: Label refinery: improving imagenet classification through label progression. arXiv preprint arXiv:1805.02641 (2018)"},{"key":"12_CR7","doi-asserted-by":"publisher","unstructured":"Banerjee, A., Biswas, S., Llad\u00f3s, J., Pal, U.: Swindocsegmenter: an end-to-end unified domain adaptive transformer for document instance segmentation. In: International Conference on Document Analysis and Recognition. pp. 307\u2013325. Springer, Heidelberg (2023). https:\/\/doi.org\/10.1007\/978-3-031-41676-7_18","DOI":"10.1007\/978-3-031-41676-7_18"},{"key":"12_CR8","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: BEiT: BERT pre-training of image transformers. In: International Conference on Learning Representations (2022)"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Bhojanapalli, S., Chakrabarti, A., Glasner, D., Li, D., Unterthiner, T., Veit, A.: Understanding robustness of transformers for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10231\u201310241 (2021)","DOI":"10.1109\/ICCV48922.2021.01007"},{"issue":"6","key":"12_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3355610","volume":"52","author":"GM Binmakhashen","year":"2019","unstructured":"Binmakhashen, G.M., Mahmoud, S.A.: Document layout analysis: a comprehensive survey. ACM Comput. Surv. (CSUR) 52(6), 1\u201336 (2019)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"12_CR11","unstructured":"Biswas, S., Banerjee, A., Llad\u00f3s, J., Pal, U.: Docsegtr: an instance-level end-to-end document image segmentation transformer. arXiv preprint arXiv:2201.11438 (2022)"},{"issue":"3","key":"12_CR12","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1007\/s10032-021-00380-6","volume":"24","author":"S Biswas","year":"2021","unstructured":"Biswas, S., Riba, P., Llad\u00f3s, J., Pal, U.: Beyond document object detection: instance-level segmentation of complex layouts. Int. J. Doc. Anal, Recogn. (IJDAR) 24(3), 269\u2013281 (2021)","journal-title":"Int. J. Doc. Anal, Recogn. (IJDAR)"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Biten, A.F., et al.: Scene text visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2019)","DOI":"10.1109\/ICCV.2019.00439"},{"key":"12_CR14","unstructured":"Borchmann, \u0141., et al.: DUE: end-to-end document understanding benchmark. In: Thirty-Fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2) (2021)"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Cai, H., Chen, T., Zhang, W., Yu, Y., Wang, J.: Efficient architecture search by network transformation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a032 (2018)","DOI":"10.1609\/aaai.v32i1.11709"},{"key":"12_CR16","doi-asserted-by":"crossref","unstructured":"Cao, Y., Long, M., Wang, J., Liu, S.: Deep visual-semantic quantization for efficient image retrieval. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1328\u20131337 (2017)","DOI":"10.1109\/CVPR.2017.104"},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Chen, D., Mei, J., Zhang, H., Wang, C., Feng, Y., Chen, C.: Knowledge distillation with the reused teacher classifier. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE Computer Society (2022)","DOI":"10.1109\/CVPR52688.2022.01163"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Chen, D., Mei, J.P., Wang, C., Feng, Y., Chen, C.: Online knowledge distillation with diverse peers. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 3430\u20133437 (2020)","DOI":"10.1609\/aaai.v34i04.5746"},{"key":"12_CR19","doi-asserted-by":"crossref","unstructured":"Chen, D., Mei, J.P., Zhang, H., Wang, C., Feng, Y., Chen, C.: Knowledge distillation with the reused teacher classifier. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.01163"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Chen, D., et al.: Cross-layer distillation with semantic calibration. In: Proceedings of the AAAI Conference on Artificial Intelligence (2021)","DOI":"10.1609\/aaai.v35i8.16865"},{"key":"12_CR21","unstructured":"Chen, G., Choi, W., Yu, X., Han, T., Chandraker, M.: Learning efficient object detection models with knowledge distillation. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Chen, P., Liu, S., Zhao, H., Jia, J.: Distilling knowledge via knowledge review. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2021)","DOI":"10.1109\/CVPR46437.2021.00497"},{"key":"12_CR23","unstructured":"Cordonnier, J.B., Loukas, A., Jaggi, M.: On the relationship between self-attention and convolutional layers. arXiv preprint arXiv:1911.03584 (2019)"},{"key":"12_CR24","unstructured":"Cui, L., Xu, Y., Lv, T., Wei, F.: Document AI: benchmarks, models and applications. arXiv preprint arXiv:2111.08609 (2021)"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Da, C., Luo, C., Zheng, Q., Yao, C.: Vision grid transformer for document layout analysis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19462\u201319472 (2023)","DOI":"10.1109\/ICCV51070.2023.01783"},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"12_CR27","unstructured":"Dettmers, T., Pagnoni, A., Holtzman, A., Zettlemoyer, L.: Qlora: efficient finetuning of quantized llms. arXiv preprint arXiv:2305.14314 (2023)"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Ding, Y., et al.: V-Doc: visual questions answers with Documents. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21492\u201321498 (2022)","DOI":"10.1109\/CVPR52688.2022.02083"},{"key":"12_CR29","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16$$\\times $$16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"12_CR30","unstructured":"Galil, I., Dabbah, M., El-Yaniv, R.: What can we learn from the selective prediction and uncertainty estimation performance of 523 imagenet classifiers. arXiv preprint arXiv:2302.11874 (2023)"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Gao, S., Huang, F., Cai, W., Huang, H.: Network pruning via performance maximization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9270\u20139280 (2021)","DOI":"10.1109\/CVPR46437.2021.00915"},{"key":"12_CR32","unstructured":"Geifman, Y., El-Yaniv, R.: Selective classification for deep neural networks. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"12_CR33","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S.J., Tao, D.: Knowledge distillation: a survey. Int. J. Comput. Vision 129, 1789\u20131819 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"12_CR34","first-page":"39","volume":"34","author":"J Gu","year":"2021","unstructured":"Gu, J., et al.: Unidoc: unified pretraining framework for document understanding. Adv. Neural. Inf. Process. Syst. 34, 39\u201350 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR35","unstructured":"Guo, C., Pleiss, G., Sun, Y., Weinberger, K.Q.: On calibration of modern neural networks. In: Proceedings of the 34th International Conference on Machine Learning, Icml 2017, vol. 70, pp. 1321\u20131330 (2017)"},{"key":"12_CR36","doi-asserted-by":"crossref","unstructured":"Haralick: Document image understanding: geometric and logical layout. In: 1994 Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 385\u2013390. IEEE (1994)","DOI":"10.1109\/CVPR.1994.323855"},{"key":"12_CR37","doi-asserted-by":"crossref","unstructured":"Harley, A.W., Ufkes, A., Derpanis, K.G.: Evaluation of deep convolutional nets for document image classification and retrieval. In: 2015 13th International Conference on Document Analysis and Recognition (ICDAR), pp. 991\u2013995. IEEE (2015)","DOI":"10.1109\/ICDAR.2015.7333910"},{"key":"12_CR38","doi-asserted-by":"crossref","unstructured":"He, J., Hu, Y., Wang, L., Xu, X., Liu, N., Liu, H.: Do-GOOD: towards distribution shift evaluation for pre-trained visual document understanding models. In: SIGIR, vol.\u00a023, pp. 23\u201327 (2023)","DOI":"10.1145\/3539618.3591670"},{"key":"12_CR39","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"12_CR40","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"12_CR41","doi-asserted-by":"crossref","unstructured":"He, Y.Y., Wu, J., Wei, X.S.: Distilling virtual examples for long-tailed recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 235\u2013244 (2021)","DOI":"10.1109\/ICCV48922.2021.00030"},{"key":"12_CR42","doi-asserted-by":"crossref","unstructured":"Heo, B., Lee, M., Yun, S., Choi, J.Y.: Knowledge transfer via distillation of activation boundaries formed by hidden neurons. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 3779\u20133787 (2019)","DOI":"10.1609\/aaai.v33i01.33013779"},{"key":"12_CR43","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)"},{"key":"12_CR44","doi-asserted-by":"crossref","unstructured":"Hsieh, C.Y., et al.: Distilling step-by-step! outperforming larger language models with less training data and smaller model sizes. arXiv preprint arXiv:2305.02301 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.507"},{"key":"12_CR45","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)"},{"key":"12_CR46","doi-asserted-by":"crossref","unstructured":"Huang, Y., Lv, T., Cui, L., Lu, Y., Wei, F.: LayoutLMv3: pre-training for document AI with unified text and image masking. In: ACM International Conference on Multimedia, pp. 4083\u20134091 (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"12_CR47","unstructured":"Jaeger, P.F., L\u00fcth, C.T., Klein, L., Bungert, T.J.: A call to reflect on evaluation practices for failure detection in image classification. In: International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=YnkGMIh0gvX"},{"key":"12_CR48","doi-asserted-by":"crossref","unstructured":"Jain, R., Wigington, C.: Multimodal document image classification. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 71\u201377. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00021"},{"key":"12_CR49","doi-asserted-by":"crossref","unstructured":"Jaume, G., Ekenel, H.K., Thiran, J.P.: Funsd: a dataset for form understanding in noisy scanned documents. In: 2019 International Conference on Document Analysis and Recognition Workshops (ICDARW), vol.\u00a02, pp.\u00a01\u20136. IEEE (2019)","DOI":"10.1109\/ICDARW.2019.10029"},{"key":"12_CR50","doi-asserted-by":"crossref","unstructured":"Kang, L., Kumar, J., Ye, P., Li, Y., Doermann, D.: Convolutional neural networks for document image classification. In: 2014 22nd International Conference on Pattern Recognition, pp. 3168\u20133172. IEEE (2014)","DOI":"10.1109\/ICPR.2014.546"},{"key":"12_CR51","doi-asserted-by":"crossref","unstructured":"Kim, T., Oh, J., Kim, N., Cho, S., Yun, S.Y.: Comparing kullback-leibler divergence and mean squared error loss in knowledge distillation. arXiv preprint arXiv:2105.08919 (2021)","DOI":"10.24963\/ijcai.2021\/362"},{"key":"12_CR52","unstructured":"Komodakis, N., Zagoruyko, S.: Paying more attention to attention: improving the performance of convolutional neural networks via attention transfer. In: ICLR (2017)"},{"key":"12_CR53","doi-asserted-by":"crossref","unstructured":"Kumar, J., Doermann, D.: Unsupervised classification of structurally similar document images. In: 2013 12th International Conference on Document Analysis and Recognition, pp. 1225\u20131229. IEEE (2013)","DOI":"10.1109\/ICDAR.2013.248"},{"key":"12_CR54","unstructured":"Larson, S., Lim, G., Ai, Y., Kuang, D., Leach, K.: Evaluating out-of-distribution performance on document image classifiers. In: Thirty-Sixth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (2022)"},{"key":"12_CR55","doi-asserted-by":"crossref","unstructured":"Larson, S., Lim, G., Leach, K.: On evaluation of document classification with RVL-CDIP. In: Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, pp. 2665\u20132678. Association for Computational Linguistics, Dubrovnik (2023)","DOI":"10.18653\/v1\/2023.eacl-main.195"},{"key":"12_CR56","doi-asserted-by":"crossref","unstructured":"Lewis, D., Agam, G., Argamon, S., Frieder, O., Grossman, D., Heard, J.: Building a test collection for complex document information processing. In: Proceedings of the 29th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 665\u2013666 (2006)","DOI":"10.1145\/1148170.1148307"},{"key":"12_CR57","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, Y., Lv, T., Cui, L., Zhang, C., Wei, F.: Dit: self-supervised pre-training for document image transformer. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 3530\u20133539 (2022)","DOI":"10.1145\/3503161.3547911"},{"key":"12_CR58","doi-asserted-by":"crossref","unstructured":"Li, P., Gu, J., Kuen, J., Morariu, V.I., Zhao, H., Jain, R., Manjunatha, V., Liu, H.: Selfdoc: self-supervised document representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5652\u20135660 (2021)","DOI":"10.1109\/CVPR46437.2021.00560"},{"key":"12_CR59","doi-asserted-by":"publisher","unstructured":"Li, Y., Mao, H., Girshick, R., He, K.: Exploring plain vision transformer backbones for object detection. In: European Conference on Computer Vision, pp. 280\u2013296. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_17","DOI":"10.1007\/978-3-031-20077-9_17"},{"key":"12_CR60","unstructured":"Li, Y., Xie, S., Chen, X., Dollar, P., He, K., Girshick, R.: Benchmarking detection transfer learning with vision transformers. arXiv preprint arXiv:2111.11429 (2021)"},{"key":"12_CR61","doi-asserted-by":"crossref","unstructured":"Li, Z., Gu, Q.: I-vit: integer-only quantization for efficient vision transformer inference. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17065\u201317075 (2023)","DOI":"10.1109\/ICCV51070.2023.01565"},{"key":"12_CR62","doi-asserted-by":"crossref","unstructured":"Liao, H., et al.: DocTr: document transformer for structured information extraction in documents. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19584\u201319594 (2023)","DOI":"10.1109\/ICCV51070.2023.01794"},{"key":"12_CR63","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"12_CR64","doi-asserted-by":"crossref","unstructured":"Liu, C., et al.: Progressive neural architecture search. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 19\u201334 (2018)","DOI":"10.1007\/978-3-030-01246-5_2"},{"key":"12_CR65","unstructured":"Liu, H., Simonyan, K., Vinyals, O., Fernando, C., Kavukcuoglu, K.: Hierarchical representations for efficient architecture search. arXiv preprint arXiv:1711.00436 (2017)"},{"key":"12_CR66","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1016\/j.neucom.2021.04.114","volume":"453","author":"L Liu","year":"2021","unstructured":"Liu, L., Wang, Z., Qiu, T., Chen, Q., Lu, Y., Suen, C.Y.: Document image classification: progress over two decades. Neurocomputing 453, 223\u2013240 (2021)","journal-title":"Neurocomputing"},{"key":"12_CR67","unstructured":"Liu, Z., Sun, M., Zhou, T., Huang, G., Darrell, T.: Rethinking the value of network pruning. arXiv preprint arXiv:1810.05270 (2018)"},{"key":"12_CR68","doi-asserted-by":"crossref","unstructured":"Luo, C., Cheng, C., Zheng, Q., Yao, C.: GeoLayoutLM: geometric pre-training for visual information extraction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7092\u20137101 (2023)","DOI":"10.1109\/CVPR52729.2023.00685"},{"key":"12_CR69","doi-asserted-by":"publisher","unstructured":"Maity, S., et al.: Selfdocseg: a self-supervised vision-based approach towards document segmentation. In: International Conference on Document Analysis and Recognition, pp. 342\u2013360. Springer, Heidelberg (2023). https:\/\/doi.org\/10.1007\/978-3-031-41676-7_20","DOI":"10.1007\/978-3-031-41676-7_20"},{"key":"12_CR70","doi-asserted-by":"crossref","unstructured":"Mathew, M., Bagal, V., Tito, R., Karatzas, D., Valveny, E., Jawahar, C.: InfographicVQA. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1697\u20131706 (2022)","DOI":"10.1109\/WACV51458.2022.00264"},{"key":"12_CR71","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., Jawahar, C.: Docvqa: a dataset for vqa on document images. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2200\u20132209 (2021)","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"12_CR72","doi-asserted-by":"crossref","unstructured":"Mirzadeh, S.I., Farajtabar, M., Li, A., Levine, N., Matsukawa, A., Ghasemzadeh, H.: Improved knowledge distillation via teacher assistant. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 5191\u20135198 (2020)","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"12_CR73","doi-asserted-by":"crossref","unstructured":"Naeini, M.P., Cooper, G., Hauskrecht, M.: Obtaining well calibrated probabilities using Bayesian binning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a029 (2015)","DOI":"10.1609\/aaai.v29i1.9602"},{"key":"12_CR74","doi-asserted-by":"crossref","unstructured":"Niculescu-Mizil, A., Caruana, R.: Predicting good probabilities with supervised learning. In: Proceedings of the 22nd International Conference on Machine Learning, pp. 625\u2013632 (2005)","DOI":"10.1145\/1102351.1102430"},{"key":"12_CR75","doi-asserted-by":"crossref","unstructured":"Park, W., Kim, D., Lu, Y., Cho, M.: Relational knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2019)","DOI":"10.1109\/CVPR.2019.00409"},{"key":"12_CR76","doi-asserted-by":"crossref","unstructured":"Passalis, N., Tzelepi, M., Tefas, A.: Heterogeneous knowledge distillation using information flow modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2339\u20132348 (2020)","DOI":"10.1109\/CVPR42600.2020.00241"},{"key":"12_CR77","doi-asserted-by":"crossref","unstructured":"Pfitzmann, B., Auer, C., Dolfi, M., Nassar, A.S., Staar, P.: DocLayNet: a large human-annotated dataset for document-layout segmentation. In: Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 3743\u20133751 (2022)","DOI":"10.1145\/3534678.3539043"},{"key":"12_CR78","unstructured":"Pham, H., Guan, M., Zoph, B., Le, Q., Dean, J.: Efficient neural architecture search via parameters sharing. In: International Conference on Machine Learning, pp. 4095\u20134104. PMLR (2018)"},{"key":"12_CR79","doi-asserted-by":"crossref","unstructured":"Phuong, M., Lampert, C.H.: Distillation-based training for multi-exit architectures. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1355\u20131364 (2019)","DOI":"10.1109\/ICCV.2019.00144"},{"key":"12_CR80","doi-asserted-by":"crossref","unstructured":"Pistone, G., Sempi, C.: An infinite-dimensional geometric structure on the space of all the probability measures equivalent to a given one. Ann. Stat. 1543\u20131561 (1995)","DOI":"10.1214\/aos\/1176324311"},{"key":"12_CR81","unstructured":"Romero, A., Ballas, N., Kahou, S.E., Chassang, A., Gatta, C., Bengio, Y.: Fitnets: hints for thin deep nets. arXiv preprint arXiv:1412.6550 (2014)"},{"key":"12_CR82","unstructured":"Saad-Falcon, J., Barrow, J., Siu, A., Nenkova, A., Rossi, R.A., Dernoncourt, F.: PDFTriage: question answering over long, structured documents. arXiv preprint arXiv:2309.08872 (2023)"},{"key":"12_CR83","doi-asserted-by":"publisher","first-page":"376","DOI":"10.1162\/tacl_a_00466","volume":"10","author":"Z Shen","year":"2022","unstructured":"Shen, Z., Lo, K., Wang, L.L., Kuehl, B., Weld, D.S., Downey, D.: VILA: improving structured content extraction from scientific PDFs using visual layout groups. Trans. Assoc. Comput. Linguist. 10, 376\u2013392 (2022)","journal-title":"Trans. Assoc. Comput. Linguist."},{"issue":"2","key":"12_CR84","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1016\/S0378-3758(00)00115-4","volume":"90","author":"H Shimodaira","year":"2000","unstructured":"Shimodaira, H.: Improving predictive inference under covariate shift by weighting the log-likelihood function. J. Stat. Plan. Inference 90(2), 227\u2013244 (2000)","journal-title":"J. Stat. Plan. Inference"},{"key":"12_CR85","doi-asserted-by":"crossref","unstructured":"\u0160imsa, \u0160., et\u00a0al.: DocILE benchmark for document information localization and extraction. arXiv preprint arXiv:2302.05658 (2023)","DOI":"10.1007\/978-3-031-41679-8_9"},{"key":"12_CR86","doi-asserted-by":"publisher","unstructured":"Stanis\u0142awek, T., et al.: Kleister: key information extraction datasets involving long documents with complex layouts. In: International Conference on Document Analysis and Recognition, pp. 564\u2013579. Springer, Heidelberg (2021). https:\/\/doi.org\/10.1007\/978-3-030-86549-8_36","DOI":"10.1007\/978-3-030-86549-8_36"},{"key":"12_CR87","first-page":"6906","volume":"34","author":"S Stanton","year":"2021","unstructured":"Stanton, S., Izmailov, P., Kirichenko, P., Alemi, A.A., Wilson, A.G.: Does knowledge distillation really work? Adv. Neural. Inf. Process. Syst. 34, 6906\u20136919 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR88","doi-asserted-by":"crossref","unstructured":"Tang, Z., et al.: Unifying vision, text, and layout for universal document processing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19254\u201319264 (2023)","DOI":"10.1109\/CVPR52729.2023.01845"},{"key":"12_CR89","unstructured":"Tian, Y., Krishnan, D., Isola, P.: Contrastive representation distillation. In: International Conference on Learning Representations (ICLR) (2019)"},{"key":"12_CR90","doi-asserted-by":"publisher","unstructured":"Tito, R., Mathew, M., Jawahar, C., Valveny, E., Karatzas, D.: ICDAR 2021 competition on document visual question answering. In: International Conference on Document Analysis and Recognition, pp. 635\u2013649. Springer, Heidelberg (2021). DOI: https:\/\/doi.org\/10.1007\/978-3-030-86337-1_42","DOI":"10.1007\/978-3-030-86337-1_42"},{"key":"12_CR91","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"12_CR92","unstructured":"Van\u00a0Landeghem, J.: Intelligent Automation for AI-driven Document Understanding. Ph.D. thesis, KU Leuven (2024)"},{"key":"12_CR93","doi-asserted-by":"crossref","unstructured":"Van\u00a0Landeghem, J., Biswas, S., Blaschko, M., Moens, M.F.: Beyond document page classification: design, datasets, and challenges. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2962\u20132972 (2024)","DOI":"10.1109\/WACV57701.2024.00294"},{"key":"12_CR94","doi-asserted-by":"crossref","unstructured":"Van\u00a0Landeghem, J., Biswas, S., Blaschko, M.B., Moens, M.F.: Beyond document page classification: design, datasets, and challenges. arXiv preprint arXiv:2308.12896 (2023)","DOI":"10.1109\/WACV57701.2024.00294"},{"key":"12_CR95","doi-asserted-by":"crossref","unstructured":"Van\u00a0Landeghem, J., et al.: Document understanding dataset and evaluation (DUDE). In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19528\u201319540 (2023)","DOI":"10.1109\/ICCV51070.2023.01789"},{"key":"12_CR96","doi-asserted-by":"publisher","unstructured":"Van\u00a0Landeghem, J., et al.: ICDAR 2023 competition on document understanding of everything (DUDE). In: International Conference on Document Analysis and Recognition, pp. 420\u2013434. Springer, Heidelberg (2023). https:\/\/doi.org\/10.1007\/978-3-031-41679-8_24","DOI":"10.1007\/978-3-031-41679-8_24"},{"key":"12_CR97","unstructured":"Vapnik, V.: Principles of risk minimization for learning theory. Adv. Neural Inf. Process. Syst. 831\u2013838 (1992)"},{"key":"12_CR98","first-page":"607","volume":"35","author":"C Wang","year":"2022","unstructured":"Wang, C., Yang, Q., Huang, R., Song, S., Huang, G.: Efficient knowledge distillation from model checkpoints. Adv. Neural. Inf. Process. Syst. 35, 607\u2013619 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR99","unstructured":"Wang, W., Li, Y., Ou, Y., Zhang, Y.: Layout and task aware instruction prompt for zero-shot document image question answering. arXiv preprint arXiv:2306.00526 (2023)"},{"key":"12_CR100","doi-asserted-by":"crossref","unstructured":"Wu, X., et al.: A region-based document VQA. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4909\u20134920 (2022)","DOI":"10.1145\/3503161.3548172"},{"key":"12_CR101","unstructured":"Wu, Y., Kirillov, A., Massa, F., Lo, W.Y., Girshick, R.: Detectron2 (2019). https:\/\/github.com\/facebookresearch\/detectron2"},{"key":"12_CR102","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"275","DOI":"10.1007\/978-3-030-58517-4_17","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Q Xing","year":"2020","unstructured":"Xing, Q., Xu, M., Li, T., Guan, Z.: Early exit or not: resource-efficient blind quality enhancement for compressed images. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 275\u2013292. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_17"},{"key":"12_CR103","doi-asserted-by":"crossref","unstructured":"Xu, Y., et\u00a0al.: Layoutlmv2: multi-modal pre-training for visually-rich document understanding. arXiv preprint arXiv:2012.14740 (2020)","DOI":"10.18653\/v1\/2021.acl-long.201"},{"key":"12_CR104","doi-asserted-by":"crossref","unstructured":"Xu, Y., Li, M., Cui, L., Huang, S., Wei, F., Zhou, M.: Layoutlm: pre-training of text and layout for document image understanding. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 1192\u20131200 (2020)","DOI":"10.1145\/3394486.3403172"},{"key":"12_CR105","doi-asserted-by":"crossref","unstructured":"Yang, Z., Zeng, A., Li, Z., Zhang, T., Yuan, C., Li, Y.: From knowledge distillation to self-knowledge distillation: a unified approach with normalized loss and customized soft labels. arXiv preprint arXiv:2303.13005 (2023)","DOI":"10.1109\/ICCV51070.2023.01576"},{"key":"12_CR106","doi-asserted-by":"crossref","unstructured":"Yim, J., Joo, D., Bae, J., Kim, J.: A gift from knowledge distillation: fast optimization, network minimization and transfer learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4133\u20134141 (2017)","DOI":"10.1109\/CVPR.2017.754"},{"key":"12_CR107","doi-asserted-by":"crossref","unstructured":"You, S., Xu, C., Xu, C., Tao, D.: Learning from multiple teacher networks. In: Proceedings of the 23rd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, pp. 1285\u20131294 (2017)","DOI":"10.1145\/3097983.3098135"},{"key":"12_CR108","doi-asserted-by":"crossref","unstructured":"Yuan, L., et al.: Central similarity quantization for efficient image and video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3083\u20133092 (2020)","DOI":"10.1109\/CVPR42600.2020.00315"},{"key":"12_CR109","doi-asserted-by":"crossref","unstructured":"Zhang, L., Song, J., Gao, A., Chen, J., Bao, C., Ma, K.: Be your own teacher: improve the performance of convolutional neural networks via self distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2019)","DOI":"10.1109\/ICCV.2019.00381"},{"key":"12_CR110","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Xiang, T., Hospedales, T.M., Lu, H.: Deep mutual learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4320\u20134328 (2018)","DOI":"10.1109\/CVPR.2018.00454"},{"key":"12_CR111","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhang, H., Arik, S.O., Lee, H., Pfister, T.: Distilling effective supervision from severe label noise. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9294\u20139303 (2020)","DOI":"10.1109\/CVPR42600.2020.00931"},{"key":"12_CR112","doi-asserted-by":"crossref","unstructured":"Zhao, B., Cui, Q., Song, R., Qiu, Y., Liang, J.: Decoupled knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11953\u201311962 (2022)","DOI":"10.1109\/CVPR52688.2022.01165"},{"key":"12_CR113","unstructured":"Zhao, W.X., et\u00a0al.: A survey of large language models. arXiv preprint arXiv:2303.18223 (2023)"},{"key":"12_CR114","doi-asserted-by":"crossref","unstructured":"Zhong, X., Tang, J., Yepes, A.J.: Publaynet: largest dataset ever for document layout analysis. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1015\u20131022. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00166"},{"key":"12_CR115","first-page":"18330","volume":"33","author":"W Zhou","year":"2020","unstructured":"Zhou, W., Xu, C., Ge, T., McAuley, J., Xu, K., Wei, F.: Bert loses patience: fast and robust inference with early exit. Adv. Neural. Inf. Process. Syst. 33, 18330\u201318341 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR116","unstructured":"Zhu, M., Gupta, S.: To prune, or not to prune: exploring the efficacy of pruning for model compression. arXiv preprint arXiv:1710.01878 (2017)"},{"key":"12_CR117","doi-asserted-by":"publisher","unstructured":"Zhu, X., Han, X., Peng, S., Lei, S., Deng, C., Feng, J.: Beyond layout embedding: layout attention with gaussian biases for structured document understanding. In: Bouamor, H., Pino, J., Bali, K. (eds.) Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 7773\u20137784. Association for Computational Linguistics, Singapore (2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-emnlp.521. https:\/\/aclanthology.org\/2023.findings-emnlp.521","DOI":"10.18653\/v1\/2023.findings-emnlp.521"},{"key":"12_CR118","unstructured":"Zhu, X., Li, J., Liu, Y., Ma, C., Wang, W.: A survey on model compression for large language models. arXiv preprint arXiv:2308.07633 (2023)"}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition - ICDAR 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-70546-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T23:07:02Z","timestamp":1732748822000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-70546-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031705458","9783031705465"],"references-count":118,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-70546-5_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"11 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Athens","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icdar2024.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}