{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T09:06:48Z","timestamp":1780909608913,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":59,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819214679","type":"print"},{"value":"9789819214686","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-92-1468-6_29","type":"book-chapter","created":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T08:32:01Z","timestamp":1780907521000},"page":"424-436","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SELAUR: Self Evolving LLM Agent via\u00a0Uncertainty-Aware Rewards"],"prefix":"10.1007","author":[{"given":"Dengjia","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoou","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaqing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kenton","family":"Murray","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hua","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,9]]},"reference":[{"key":"29_CR1","doi-asserted-by":"crossref","unstructured":"Ahmadian, A., et al.: Back to basics: revisiting reinforce style optimization for learning from human feedback in LLMs (2024). arXiv:2402.14740 arXiv preprint","DOI":"10.18653\/v1\/2024.acl-long.662"},{"key":"29_CR2","doi-asserted-by":"crossref","unstructured":"Aissi, M.S., et al.: Reinforcement learning for aligning large language models agents with interactive environments: quantifying and mitigating prompt overfitting. In: Findings of the Association for Computational Linguistics: NAACL 2025, pp. 7030\u20137046 (2025)","DOI":"10.18653\/v1\/2025.findings-naacl.390"},{"key":"29_CR3","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: NeurIPS, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"29_CR4","unstructured":"Carta, T., Romac, C., Wolf, T., Lamprier, S., Sigaud, O., Oudeyer, P.Y.: Grounding large language models in interactive environments with online reinforcement learning. In: ICML, pp. 3676\u20133713. PMLR (2023)"},{"issue":"2","key":"29_CR5","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1016\/j.jbi.2011.11.003","volume":"45","author":"Y Chen","year":"2012","unstructured":"Chen, Y., Mani, S., Xu, H.: Applying active learning to assertion classification of concepts in clinical text. J. Biomed. Inf. 45(2), 265\u2013272 (2012)","journal-title":"J. Biomed. Inf."},{"key":"29_CR6","unstructured":"Christiano, P.F., Leike, J., Brown, T., Martic, M., Legg, S., Amodei, D.: Deep reinforcement learning from human preferences. In: NeurIPS (2017)"},{"key":"29_CR7","unstructured":"Chrysos, G., Li, Y., Angelopoulos, A.N., Bates, S., Plank, B., Khan, M.E.: Quantify uncertainty and hallucination in foundation models: the next frontier in reliable AI. In: ICLR 2025 Workshop Proposals (2025)"},{"key":"29_CR8","unstructured":"Feng, L., Xue, Z., Liu, T., An, B.: Group-in-group policy optimization for LLM agent training. arXiv preprint arXiv:2505.10978 (2025)"},{"key":"29_CR9","unstructured":"Fomicheva, M., et al.: Uncertainty estimation for quality estimation. In: ACL (2020)"},{"key":"29_CR10","unstructured":"Guo, D., et\u00a0al.: Deepseek-r1: Incentivizing reasoning capability in LLMs via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)"},{"key":"29_CR11","doi-asserted-by":"crossref","unstructured":"He, G., Demartini, G., Gadiraju, U.: Plan-then-execute: an empirical study of user trust and team performance when using LLM agents as a daily assistant. In: CHI, pp. 1\u201322 (2025)","DOI":"10.1145\/3706598.3713218"},{"key":"29_CR12","doi-asserted-by":"crossref","unstructured":"Huang, L., Li, D., Liu, H., Cheng, L.: Beyond accuracy: the role of calibration in self-improving large language models. arXiv preprint arXiv:2504.02902 (2025)","DOI":"10.1109\/BigData66926.2025.11400862"},{"key":"29_CR13","unstructured":"Huang, W., Abbeel, P., Pathak, D., Mordatch, I.: Inner monologue: embodied reasoning through planning with language models. In: NeurIPS (2022)"},{"key":"29_CR14","doi-asserted-by":"crossref","unstructured":"Huang, Y., Liu, Y., Thirukovalluru, R., Cohan, A., Dhingra, B.: Calibrating longform generations from large language models. arXiv preprint arXiv:2402.06544 (2024)","DOI":"10.18653\/v1\/2024.findings-emnlp.785"},{"key":"29_CR15","unstructured":"Jeon, H.J., Milli, S., Dragan, A.D.: Reward-rational (implicit) choice: a unifying formalism for reward learning. In: NeurIPS (2020)"},{"key":"29_CR16","unstructured":"Kadavath, S.: Language models (mostly) know what they know. arXiv preprint arXiv:2207.05221 (2022)"},{"key":"29_CR17","doi-asserted-by":"crossref","unstructured":"Kamath, A., Jia, R., Liang, P.: Selective question answering under domain shift. In: ACL (2020)","DOI":"10.18653\/v1\/2020.acl-main.503"},{"key":"29_CR18","doi-asserted-by":"crossref","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. In: NeurIPS, vol. 35, pp. 22199\u201322213 (2022)","DOI":"10.52202\/068431-1613"},{"key":"29_CR19","unstructured":"Kwon, M., ElSayed-Aly, I., Feng, L.: Adaptive reward design for reinforcement learning in complex robotic tasks. arXiv e-prints pp. arXiv\u20132412 (2024)"},{"key":"29_CR20","doi-asserted-by":"crossref","unstructured":"van Leeuwen, P.J., Chiu, J.C., Yang, C.K.: Uncertainty quantification for deep learning. arXiv preprint arXiv:2405.20550 (2024)","DOI":"10.5194\/egusphere-egu24-12205"},{"key":"29_CR21","doi-asserted-by":"crossref","unstructured":"Liu, J., et al.: OVD-explorer: optimism should not be the sole pursuit of exploration in noisy environments. In: AAAI, vol.\u00a038, pp. 13954\u201313962 (2024)","DOI":"10.1609\/aaai.v38i12.29303"},{"key":"29_CR22","doi-asserted-by":"crossref","unstructured":"Liu, Z., et\u00a0al.: A survey on the feedback mechanism of LLM-based ai agents. In: IJCAI, pp. 10582\u201310592 (2025)","DOI":"10.24963\/ijcai.2025\/1175"},{"key":"29_CR23","unstructured":"Lyu, C., Gao, S., Gu, Y., Zhang, W., Gao, J., Liu, K., Wang, Z., Li, S., Zhao, Q., Huang, H., et\u00a0al.: Exploring the limit of outcome reward for learning mathematical reasoning. arXiv preprint arXiv:2502.06781 (2025)"},{"key":"29_CR24","unstructured":"Malinin, A., Gales, M.: Uncertainty estimation and calibration in natural language processing. In: ICLR Workshop on Uncertainty and Robustness in Deep Learning (2020)"},{"key":"29_CR25","doi-asserted-by":"crossref","unstructured":"Manakul, P., Liusie, A., Gales, M.J.: Selfcheckgpt: zero-resource black-box hallucination detection for generative large language models. In: ACL (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.557"},{"key":"29_CR26","unstructured":"Min, D., Zhang, K., Wu, T., Cheng, L.: Quco-rag: quantifying uncertainty from the pre-training corpus for dynamic retrieval-augmented generation. arXiv preprint arXiv:2512.19134 (2025)"},{"key":"29_CR27","unstructured":"Ng, A.Y., Harada, D., Russell, S.J.: Policy invariance under reward transformations: theory and application to reward shaping. In: ICML (1999)"},{"key":"29_CR28","unstructured":"Ouyang, L., Wu, J., Jiang, X., Almeida, D.: Training language models to follow instructions with human feedback. In: NeurIPS (2022)"},{"key":"29_CR29","unstructured":"Plaat, A., Wong, A., Verberne, S., Broekens, J., Van Stein, N.: Multi-step reasoning with large language models, a survey. ACM Comput. Surv. (2025)"},{"key":"29_CR30","unstructured":"Qi, Z., et\u00a0al.: Webrl: training LLM web agents via self-evolving online curriculum reinforcement learning. In: ICLR (2025)"},{"key":"29_CR31","doi-asserted-by":"crossref","unstructured":"Roberts, A., Raffel, C., Shazeer, N.: How much knowledge can you pack into the parameters of a language model? In: EMNLP, pp. 5418\u20135426 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.437"},{"key":"29_CR32","first-page":"68539","volume":"36","author":"T Schick","year":"2023","unstructured":"Schick, T., et al.: Toolformer: language models can teach themselves to use tools. NeurIPS 36, 68539\u201368551 (2023)","journal-title":"NeurIPS"},{"key":"29_CR33","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"29_CR34","unstructured":"Shao, Z., et\u00a0al.: Deepseekmath: pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)"},{"key":"29_CR35","unstructured":"Shen, C., Chen, Z., Luo, D., Xu, D., Chen, H., Ni, J.: Exploring multi-modal integration with tool-augmented LLM agents for precise causal discovery, 1(3) (2024). arXiv preprint arXiv:2412.13667"},{"key":"29_CR36","doi-asserted-by":"crossref","unstructured":"Shinn, N., Cassano, F., Gopinath, A., Narasimhan, K., Yao, S.: Reflexion: language agents with verbal reinforcement learning. In: NeurIPS, vol. 36, pp. 8634\u20138652 (2023)","DOI":"10.52202\/075280-0377"},{"key":"29_CR37","doi-asserted-by":"crossref","unstructured":"Shorinwa, O., Mei, Z., Lidard, J., Ren, A.Z., Majumdar, A.: A survey on uncertainty quantification of large language models: taxonomy, open research challenges, and future directions. ACM Comput. Surv. (2025)","DOI":"10.1145\/3744238"},{"key":"29_CR38","unstructured":"Shridhar, M., Yuan, X., C\u00f4t\u00e9, M.A., Bisk, Y., Trischler, A., Hausknecht, M.: Alfworld: aligning text and embodied environments for interactive learning, (2020). arXiv:2010.03768 arXiv preprint"},{"key":"29_CR39","doi-asserted-by":"crossref","unstructured":"Song, Y., Yin, D., Yue, X., Huang, J., Li, S., Lin, B.Y.: Trial and error: exploration-based trajectory optimization of LLM agents. In: ACL, pp. 7584\u20137600 (2024)","DOI":"10.18653\/v1\/2024.acl-long.409"},{"key":"29_CR40","unstructured":"Su, J., et al.: Cp-router: an uncertainty-aware router between LLM and LRM. arXiv preprint arXiv:2505.19970 (2025)"},{"key":"29_CR41","unstructured":"Sukhija, B., Treven, L., Sferrazza, C., Dorfler, F., Abbeel, P., Krause, A.: Optimism via intrinsic rewards: scalable and principled exploration for model-based reinforcement learning. In: 7th Robot Learning Workshop: Towards Robots with Human-Level Abilities (2025)"},{"key":"29_CR42","doi-asserted-by":"crossref","unstructured":"Sun, S., Liu, Y., Wang, S., Iter, D., Zhu, C., Iyyer, M.: Pearl: prompting large language models to plan and execute actions over long documents. In: EACL, pp. 469\u2013486 (2024)","DOI":"10.18653\/v1\/2024.eacl-long.29"},{"key":"29_CR43","unstructured":"Suri, M., Mathur, P., Lipka, N., Dernoncourt, F., Rossi, R.A., Manocha, D.: Structured uncertainty guided clarification for LLM agents. arXiv preprint arXiv:2511.08798 (2025)"},{"key":"29_CR44","unstructured":"Wang, G., et al.: Voyager: an open-ended embodied agent with large language models. arXiv preprint arXiv:2305.16291 (2023)"},{"key":"29_CR45","doi-asserted-by":"crossref","unstructured":"Wang, H., Prasad, A., Stengel-Eskin, E., Bansal, M.: Soft self-consistency improves language model agents. arXiv preprint arXiv:2402.13212 (2024)","DOI":"10.18653\/v1\/2024.acl-short.28"},{"issue":"6","key":"29_CR46","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-024-40231-1","volume":"18","author":"L Wang","year":"2024","unstructured":"Wang, L., et al.: A survey on large language model based autonomous agents. Front. Comp. Sci. 18(6), 186345 (2024)","journal-title":"Front. Comp. Sci."},{"key":"29_CR47","doi-asserted-by":"crossref","unstructured":"Wei, J.: Chain-of-thought prompting elicits reasoning in large language models. In: NeurIPS, vol. 35, pp. 24824\u201324837 (2022)","DOI":"10.52202\/068431-1800"},{"issue":"2","key":"29_CR48","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4222-0","volume":"68","author":"Z Xi","year":"2025","unstructured":"Xi, Z., et al.: The rise and potential of large language model based agents: a survey. Sci. China Inf. Sci. 68(2), 121101 (2025)","journal-title":"Sci. China Inf. Sci."},{"key":"29_CR49","unstructured":"Xie, C., et al.: Unlocking exploration in RLVR: uncertainty-aware advantage shaping for deeper reasoning. arXiv preprint arXiv:2510.10649 (2025)"},{"key":"29_CR50","doi-asserted-by":"crossref","unstructured":"Yang, Y., Yoo, H., Lee, H.: Maqa: Evaluating uncertainty quantification in LLMs regarding data uncertainty. arXiv preprint arXiv:2408.06816 (2024)","DOI":"10.18653\/v1\/2025.findings-naacl.325"},{"key":"29_CR51","doi-asserted-by":"crossref","unstructured":"Yao, S., Chen, H., Yang, J., Narasimhan, K.: Webshop: towards scalable real-world web interaction with grounded language agents. In: NeurIPS, vol. 35, pp. 20744\u201320757 (2022)","DOI":"10.52202\/068431-1508"},{"key":"29_CR52","unstructured":"Yao, S., et al.: React: synergizing reasoning and acting in language models. In: ICLR (2022)"},{"key":"29_CR53","unstructured":"Ye, C., et al.: Beyond correctness: harmonizing process and outcome rewards through rl training. arXiv preprint arXiv:2509.03403 (2025)"},{"key":"29_CR54","unstructured":"Yehudai, A., et al.: Survey on evaluation of LLM-based agents. arXiv preprint arXiv:2503.16416 (2025)"},{"key":"29_CR55","unstructured":"Yu, R., et al.: Reward models in deep reinforcement learning: a survey. arXiv preprint arXiv:2506.15421 (2025)"},{"key":"29_CR56","unstructured":"Zhang, K., et al.: A survey of reinforcement learning for large reasoning models. arXiv preprint arXiv:2509.08827 (2025)"},{"key":"29_CR57","unstructured":"Zhao, Q., et al.: Saup: situation awareness uncertainty propagation on LLM agent. arXiv preprint arXiv:2412.01033 (2024)"},{"key":"29_CR58","unstructured":"Zhou, X., Cheng, L.: Robust uncertainty quantification for self-evolving large language models via continual domain pretraining. arXiv preprint arXiv:2510.22931 (2025)"},{"key":"29_CR59","unstructured":"Zhu, X., Xia, M., Wei, Z., Chen, W.L., Chen, D., Meng, Y.: The surprising effectiveness of negative reinforcement in LLM reasoning. arXiv preprint arXiv:2506.01347 (2025)"}],"container-title":["Lecture Notes in Computer Science","Advances in Knowledge Discovery and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-1468-6_29","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T08:34:00Z","timestamp":1780907640000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-1468-6_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819214679","9789819214686"],"references-count":59,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-1468-6_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"9 June 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 June 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 June 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.pakdd2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}