{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T06:31:55Z","timestamp":1782801115872,"version":"3.54.5"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032180100","type":"print"},{"value":"9783032180117","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-18011-7_3","type":"book-chapter","created":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T06:07:00Z","timestamp":1770962820000},"page":"27-44","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An Agentic System with\u00a0Reinforcement-Learned Subsystem Improvements for\u00a0Parsing Form-Like Documents"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-4806-7207","authenticated-orcid":false,"given":"Ayesha","family":"Amjad","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7421-0479","authenticated-orcid":false,"given":"Saurav","family":"Sthapit","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0638-9689","authenticated-orcid":false,"given":"Tahir Qasim","family":"Syed","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,14]]},"reference":[{"key":"3_CR1","doi-asserted-by":"publisher","DOI":"10.1109\/access.2022.3192828","author":"H Arslan","year":"2022","unstructured":"Arslan, H., Arslan, H.: End to end invoice processing application based on key fields extraction. IEEE Access (2022). https:\/\/doi.org\/10.1109\/access.2022.3192828","journal-title":"IEEE Access"},{"key":"3_CR2","doi-asserted-by":"publisher","unstructured":"Cao, P., Wang, Y., Zhang, Q., Meng, Z.: GenKIE: robust generative multimodal document key information extraction. In: Conference on Empirical Methods in Natural Language Processing (2023). https:\/\/doi.org\/10.48550\/arxiv.2310.16131","DOI":"10.48550\/arxiv.2310.16131"},{"key":"3_CR3","doi-asserted-by":"publisher","unstructured":"Cao, R., Rongyu, C., Cao, R., Luo, P., Luo, P., Luo, P.: Extracting zero-shot structured information from form-like documents: pretraining with keys and triggers. In: AAAI Conference on Artificial Intelligence (2021). https:\/\/doi.org\/10.1609\/aaai.v35i14.17494","DOI":"10.1609\/aaai.v35i14.17494"},{"key":"3_CR4","unstructured":"contributors, P.: PaddleOCR: awesome multilingual OCR toolkits based on PaddlePaddle. https:\/\/github.com\/PaddlePaddle\/PaddleOCR. Accessed 8 Dec 2024"},{"key":"3_CR5","unstructured":"Contributors, V.: Vowpal Wabbit reinforcement learning (2024). Accessed 20 Apr 2025"},{"key":"3_CR6","unstructured":"Dubey, A., et\u00a0al.: The Llama 3 herd of models. arXiv preprint: arXiv:2407.21783 (2024)"},{"key":"3_CR7","doi-asserted-by":"publisher","unstructured":"Gemelli, A., Vivoli, E., Marinai, S.: Graph neural networks and representation embedding for table extraction in PDF documents. In: International Conference on Pattern Recognition (2022). https:\/\/doi.org\/10.1109\/icpr56361.2022.9956590","DOI":"10.1109\/icpr56361.2022.9956590"},{"key":"3_CR8","doi-asserted-by":"publisher","unstructured":"Gilani, A., et al.: Table detection using deep learning. In: IEEE International Conference on Document Analysis and Recognition (2017). https:\/\/doi.org\/10.1109\/icdar.2017.131","DOI":"10.1109\/icdar.2017.131"},{"key":"3_CR9","unstructured":"Goodman, N.: Meta-prompt: a simple self-improving ai design (2023). https:\/\/noahgoodman.substack.com\/p\/meta-prompt-a-simple-self-improving"},{"key":"3_CR10","unstructured":"Gu, J., et al.: A survey on LLM-as-a-judge. ArXiv abs\/2411.15594 (2024). https:\/\/api.semanticscholar.org\/CorpusID:274234014"},{"key":"3_CR11","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00454","author":"Z Gu","year":"2022","unstructured":"Gu, Z., et al.: XYLayoutLM: towards layout-aware multimodal networks for visually-rich document understanding. Comput. Vis. Pattern Recognit. (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.00454","journal-title":"Comput. Vis. Pattern Recognit."},{"key":"3_CR12","unstructured":"Guo, T., et al.: Large language model based multi-agents: a survey of progress and challenges (2024). https:\/\/arxiv.org\/abs\/2402.01680"},{"key":"3_CR13","doi-asserted-by":"publisher","unstructured":"Hong, T., Kim, D., Ji, M., Hwang, W., Nam, D., Park, S.: BROS: a pre-trained language model focusing on text and layout for better key information extraction from documents. In: AAAI Conference on Artificial Intelligence (2021). https:\/\/doi.org\/10.1609\/aaai.v36i10.21322","DOI":"10.1609\/aaai.v36i10.21322"},{"key":"3_CR14","doi-asserted-by":"publisher","unstructured":"Hu, K., Wu, Z., Zhong, Z., Lin, W., Sun, L., Huo, Q.: A question-answering approach to key value pair extraction from form-like document images. In: AAAI Conference on Artificial Intelligence (2023). https:\/\/doi.org\/10.48550\/arxiv.2304.07957","DOI":"10.48550\/arxiv.2304.07957"},{"key":"3_CR15","doi-asserted-by":"publisher","unstructured":"Huang, Z., et al.: ICDAR2019 competition on scanned receipt OCR and information extraction. In: IEEE International Conference on Document Analysis and Recognition (2019). https:\/\/doi.org\/10.1109\/icdar.2019.00244","DOI":"10.1109\/icdar.2019.00244"},{"key":"3_CR16","unstructured":"Hurst, A., et\u00a0al.: GPT-4o system card. arXiv preprint: arXiv:2410.21276 (2024)"},{"key":"3_CR17","doi-asserted-by":"publisher","unstructured":"Jiao, Y., et al.: Instruct and extract: instruction tuning for on-demand information extraction. In: Conference on Empirical Methods in Natural Language Processing (2023). https:\/\/doi.org\/10.48550\/arxiv.2310.16040","DOI":"10.48550\/arxiv.2310.16040"},{"key":"3_CR18","doi-asserted-by":"publisher","unstructured":"Josifoski, M., et al.: GenIE: generative information extraction. In: North American Chapter of the Association for Computational Linguistics (2022). https:\/\/doi.org\/10.18653\/v1\/2022.naacl-main.342","DOI":"10.18653\/v1\/2022.naacl-main.342"},{"key":"3_CR19","doi-asserted-by":"publisher","unstructured":"Katti, A.R., et al.: Chargrid: towards understanding 2D documents. In: Conference on Empirical Methods in Natural Language Processing (2018). https:\/\/doi.org\/10.18653\/v1\/d18-1476","DOI":"10.18653\/v1\/d18-1476"},{"key":"3_CR20","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. In: Advances in Neural Information Processing Systems, vol. 35, pp. 22199\u201322213 (2022)"},{"key":"3_CR21","doi-asserted-by":"publisher","unstructured":"Lee, C.Y., et al.: FormNetV2: multimodal graph contrastive learning for form document information extraction. In: Annual Meeting of the Association for Computational Linguistics (2023). https:\/\/doi.org\/10.48550\/arxiv.2305.02549","DOI":"10.48550\/arxiv.2305.02549"},{"key":"3_CR22","unstructured":"Madaan, A., et al.: Self-refine: iterative refinement with self-feedback (2023). https:\/\/arxiv.org\/abs\/2303.17651"},{"key":"3_CR23","doi-asserted-by":"publisher","unstructured":"Majumder, B.P., et al.: Representation learning for information extraction from form-like documents. In: Annual Meeting of the Association for Computational Linguistics (2020). https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.580","DOI":"10.18653\/v1\/2020.acl-main.580"},{"key":"3_CR24","doi-asserted-by":"crossref","unstructured":"Palm, R.B., Winther, O., Laws, F.: CloudScan - a configuration-free invoice analysis system using recurrent neural networks (2017). http:\/\/arxiv.org\/abs\/1708.07403 [cs]","DOI":"10.1109\/ICDAR.2017.74"},{"key":"3_CR25","unstructured":"Park, S.H., et al.: CORD: a consolidated receipt dataset for Post-OCR parsing. null (2019). https:\/\/doi.org\/null"},{"key":"3_CR26","doi-asserted-by":"publisher","DOI":"10.1002\/9780470316887","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"ML Puterman","year":"1994","unstructured":"Puterman, M.L.: Markov Decision Processes: Discrete Stochastic Dynamic Programming, 1st edn. John Wiley & Sons Inc, USA (1994)","edition":"1"},{"key":"3_CR27","doi-asserted-by":"publisher","unstructured":"Palm, R.B., Winther, O., Laws, F.: CloudScan - a configuration-free invoice analysis system using recurrent neural networks. In: IEEE International Conference on Document Analysis and Recognition (2017). https:\/\/doi.org\/10.1109\/icdar.2017.74","DOI":"10.1109\/icdar.2017.74"},{"key":"3_CR28","doi-asserted-by":"crossref","unstructured":"Powalski, R., Borchmann, L., Jurkiewicz, D., Dwojak, T., Pietruszka, M., Palka, G.: Going full-tilt boogie on document understanding with text-image-layout transformer. eprint 2102.09550 (2021)","DOI":"10.1007\/978-3-030-86331-9_47"},{"key":"3_CR29","doi-asserted-by":"publisher","unstructured":"Riba, P., et al.: Table detection in invoice documents by graph neural networks. In: IEEE International Conference on Document Analysis and Recognition (2019). https:\/\/doi.org\/10.1109\/icdar.2019.00028","DOI":"10.1109\/icdar.2019.00028"},{"key":"3_CR30","unstructured":"Towers, M., et\u00a0al.: Gymnasium: a standard interface for reinforcement learning environments. arXiv preprint: arXiv:2407.17032 (2024)"},{"key":"3_CR31","doi-asserted-by":"publisher","unstructured":"Wang, J., Jin, L., Ding, K.: LiLT: a simple yet effective language-independent layout transformer for structured document understanding. Ann. Meet. Assoc. Comput. Linguist. (2022). https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.534","DOI":"10.18653\/v1\/2022.acl-long.534"},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Woodridge, M., Jennings, N.: Intelligent agents: theory and practice the knowledge engineering review. Knowl. Eng. Rev. (1995)","DOI":"10.1017\/S0269888900008122"},{"key":"3_CR33","doi-asserted-by":"crossref","unstructured":"Xu, L., et al.: MAgIC: investigation of large language model powered multi-agent in cognition, adaptability, rationality and collaboration (2024). https:\/\/arxiv.org\/abs\/2311.08562","DOI":"10.18653\/v1\/2024.emnlp-main.416"}],"container-title":["Lecture Notes in Computer Science","Engineering Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-18011-7_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T06:07:10Z","timestamp":1770962830000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-18011-7_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032180100","9783032180117"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-18011-7_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"14 February 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"EMAS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Workshop on Engineering Multi-Agent Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Detroit, MI","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 May 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 May 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"emas2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/emas.in.tu-clausthal.de\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}