{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T23:03:17Z","timestamp":1784329397623,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":14,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819233939","type":"print"},{"value":"9789819233946","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T00:00:00Z","timestamp":1784332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T00:00:00Z","timestamp":1784332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3394-6_16","type":"book-chapter","created":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T22:21:24Z","timestamp":1784326884000},"page":"182-193","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["PSPO: Trainable Potential-Based Reward Shaping with Internal Model Signals for Post-Training Policy Optimization of Large Language Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6417-3280","authenticated-orcid":false,"given":"Miaobo","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5519-6222","authenticated-orcid":false,"given":"Bokun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0501-8652","authenticated-orcid":false,"given":"Shuhao","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9495-0389","authenticated-orcid":false,"given":"Ruohan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1069-4830","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2672-6650","authenticated-orcid":false,"given":"Xiaobo","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6042-3454","authenticated-orcid":false,"given":"Daren","family":"Zha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1799-3948","authenticated-orcid":false,"given":"Jun","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,18]]},"reference":[{"key":"16_CR1","unstructured":"Azerbayev, Z. et al.: Llemma: an open language model for mathematics. https:\/\/arxiv.org\/abs\/2310.10631 (2024)"},{"key":"16_CR2","unstructured":"Cao, M. et al.: Beyond sparse rewards: enhancing reinforcement learning with language model critique in text generation. https:\/\/arxiv.org\/abs\/2401.07382 (2024)"},{"key":"16_CR3","unstructured":"Chan, A.J., Sun, H., Holt, S., Schaar, M.V.D.: Dense reward for free in reinforcement learning from human feedback. https:\/\/arxiv.org\/abs\/2402.00782 (2024)"},{"key":"16_CR4","unstructured":"Cobbe, K. et al.: Training verifiers to solve math word problems. https:\/\/arxiv.org\/abs\/2110.14168 (2021)"},{"key":"16_CR5","unstructured":"Hendrycks, D. et al.: Measuring mathematical problem solving with the MATH dataset. https:\/\/arxiv.org\/abs\/2103.03874 (2021)"},{"key":"16_CR6","unstructured":"Lewkowycz, A. et al.: Solving quantitative reasoning problems with language models. https:\/\/arxiv.org\/abs\/2206.14858 (2022)"},{"key":"16_CR7","unstructured":"Ng, A.Y., Harada, D., Russell, S.: Policy invariance under reward transformations: theory and application to reward shaping. In: ICML, vol. 99, pp. 278\u2013287. Citeseer (1999). https:\/\/www.teach.cs.toronto.edu\/~csc2542h\/fall\/material\/csc2542f16_reward_shaping.pdf"},{"key":"16_CR8","unstructured":"Ouyang, L. et al.: Training language models to follow instructions with human feedback. https:\/\/arxiv.org\/abs\/2203.02155 (2022)"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Ermon, S., Manning, C.D., Finn, C.: Direct preference optimization: your language model is secretly a reward model. https:\/\/arxiv.org\/abs\/2305.18290 (2024)","DOI":"10.52202\/075280-2338"},{"key":"16_CR10","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. https:\/\/arxiv.org\/abs\/1707.06347 (2017)"},{"key":"16_CR11","unstructured":"Shao, Z. et al.: DeepSeekMath: pushing the limits of mathematical reasoning in open language models. https:\/\/arxiv.org\/abs\/2402.03300 (2024)"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Wang, Y. et al.: MMLU-pro: a more robust and challenging multi-task language understanding benchmark. https:\/\/arxiv.org\/abs\/2406.01574 (2024)","DOI":"10.52202\/079017-3018"},{"key":"16_CR13","unstructured":"Wei, T., Luan, J., Liu, W., Dong, S., Wang, B.: CMATH: can your language model pass Chinese elementary school math test?. https:\/\/arxiv.org\/abs\/2306.16636 (2023)"},{"key":"16_CR14","unstructured":"Zhong, W. et al.: AGIEval: a human-centric benchmark for evaluating foundation models. https:\/\/arxiv.org\/abs\/2304.06364 (2023)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3394-6_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T22:21:25Z","timestamp":1784326885000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3394-6_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,18]]},"ISBN":["9789819233939","9789819233946"],"references-count":14,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3394-6_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,18]]},"assertion":[{"value":"18 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}