{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T21:02:21Z","timestamp":1784408541791,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":28,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819233809","type":"print"},{"value":"9789819233816","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3381-6_21","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T20:14:14Z","timestamp":1784405654000},"page":"254-267","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Navigating Coverage Imbalance: Prospective Weighting for Offline Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Hongyi","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3601-9286","authenticated-orcid":false,"given":"Lianke","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liu","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nianbin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"21_CR1","first-page":"29304","volume":"34","author":"R Agarwal","year":"2021","unstructured":"Agarwal, R., Schwarzer, M., Castro, P.S., Courville, A., Bellemare, M.G.: Deep reinforcement learning at the edge of the statistical precipice. Adv. Neural Inf. Proces. Syst. 34, 29304\u201329320 (2021)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"21_CR2","unstructured":"Chen, L., et al.: Decision transformer: reinforcement learning via sequence modeling. arXiv preprint. https:\/\/arxiv.org\/abs\/2106.01345 (2021)"},{"issue":"1","key":"21_CR3","doi-asserted-by":"publisher","first-page":"146","DOI":"10.2307\/1924845","volume":"61","author":"R Dorfman","year":"1979","unstructured":"Dorfman, R.: A formula for the Gini coefficient. Rev. Econ. Stat. 61(1), 146\u2013149 (1979)","journal-title":"Rev. Econ. Stat."},{"key":"21_CR4","unstructured":"Fu, J., Kumar, A., Nachum, O., Tucker, G., Levine, S.: D4RL: datasets for deep data-driven reinforcement learning. arXiv preprint. https:\/\/arxiv.org\/abs\/2004.07219 (2020)"},{"key":"21_CR5","first-page":"20132","volume":"34","author":"S Fujimoto","year":"2021","unstructured":"Fujimoto, S., Gu, S.: A minimalist approach to offline reinforcement learning. Adv. Neural Inf. Proces. Syst. 34, 20132\u201320145 (2021)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"21_CR6","unstructured":"Fujimoto, S., Meger, D., Precup, D.: Off-policy deep reinforcement learning without exploration. In: Proceedings of the International Conference on Machine Learning (ICML), pp. 2052\u20132062 (2019)"},{"issue":"3","key":"21_CR7","doi-asserted-by":"publisher","first-page":"739","DOI":"10.1162\/003355399556133","volume":"114","author":"X Gabaix","year":"1999","unstructured":"Gabaix, X.: Zipf\u2019s law for cities: an explanation. Q. J. Econ. 114(3), 739\u2013767 (1999)","journal-title":"Q. J. Econ."},{"key":"21_CR8","unstructured":"Hong, Z.-W., Agrawal, P., Tachet des Combes, R., Laroche, R.: Harnessing mixed offline reinforcement learning datasets via trajectory weighting. In: Proceedings of the International Conference on Learning Representations (ICLR 2023), Kigali, Rwanda (2023)"},{"key":"21_CR9","unstructured":"Chen, L., et al.: Decision transformer: reinforcement learning via sequence modeling. In: Advances in Neural Information Processing Systems (NeurIPS) (2021)"},{"key":"21_CR10","doi-asserted-by":"publisher","first-page":"4985","DOI":"10.52202\/075280-0221","volume":"36","author":"Z-W Hong","year":"2023","unstructured":"Hong, Z.-W., et al.: Beyond uniform sampling: offline reinforcement learning with imbalanced datasets. Adv. Neural Inf. Proces. Syst. 36, 4985\u20135009 (2023)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"21_CR11","unstructured":"Kostrikov, I., Nair, A., Levine, S.: Offline reinforcement learning with implicit Q-learning. arXiv preprint. https:\/\/arxiv.org\/abs\/2110.06169 (2021)"},{"key":"21_CR12","first-page":"11784","volume":"32","author":"A Kumar","year":"2019","unstructured":"Kumar, A., Fu, J., Soh, M., Tucker, G., Levine, S.: Stabilizing off-policy Q-learning via bootstrapping error reduction. Adv. Neural Inf. Proces. Syst. 32, 11784\u201311794 (2019)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"21_CR13","first-page":"1179","volume":"33","author":"A Kumar","year":"2020","unstructured":"Kumar, A., Zhou, A., Tucker, G., Levine, S.: Conservative Q-learning for offline reinforcement learning. Adv. Neural Inf. Proces. Syst. 33, 1179\u20131191 (2020)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"21_CR14","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1007\/978-3-642-27645-3_2","volume-title":"Reinforcement Learning","author":"S Lange","year":"2012","unstructured":"Lange, S., Gabel, T., Riedmiller, M.: Batch reinforcement learning. In: Reinforcement Learning, pp. 45\u201373. Springer, Berlin (2012)"},{"key":"21_CR15","unstructured":"Laroche, R., Tachet des Combes, R.: On the occupancy measure of non-Markovian policies in continuous MDPs. In: Proceedings of the 40th International Conference on Machine Learning (ICML 2023) (2023)"},{"key":"21_CR16","unstructured":"Levine, S., Kumar, A., Tucker, G., Fu, J.: Offline reinforcement learning: tutorial, review, and perspectives on open problems. arXiv preprint. https:\/\/arxiv.org\/abs\/2005.01643 (2020)"},{"key":"21_CR17","doi-asserted-by":"publisher","first-page":"685","DOI":"10.1007\/s00778-023-00833-w","volume":"33","author":"A Liang","year":"2024","unstructured":"Liang, A., et al.: Sub-trajectory clustering with deep reinforcement learning. VLDB J. 33, 685\u2013702 (2024)","journal-title":"VLDB J."},{"issue":"5","key":"21_CR18","doi-asserted-by":"publisher","first-page":"4639","DOI":"10.1109\/LRA.2024.3379805","volume":"9","author":"H Lin","year":"2024","unstructured":"Lin, H., et al.: Safety-aware causal representation for trustworthy offline reinforcement learning in autonomous driving. IEEE Rob. Autom. Lett. 9(5), 4639\u20134646 (2024)","journal-title":"IEEE Rob. Autom. Lett."},{"key":"21_CR19","unstructured":"Ma, Y.J., Jayaraman, D., Bastani, O.: Conservative offline distributional reinforcement learning. In: Advances in Neural Information Processing Systems (NeurIPS) (2021)"},{"key":"21_CR20","unstructured":"Nachum, O., Chow, Y., Dai, B., Li, L.: DualDICE: behavior-agnostic estimation of discounted stationary distribution corrections. In: Advances in Neural Information Processing Systems, pp. 2315\u20132325 (2019)"},{"issue":"5","key":"21_CR21","doi-asserted-by":"publisher","first-page":"323","DOI":"10.1080\/00107510500052444","volume":"46","author":"M Newman","year":"2005","unstructured":"Newman, M.: Power laws, Pareto distributions and Zipf\u2019s law. Contemp. Phys. 46(5), 323\u2013351 (2005)","journal-title":"Contemp. Phys."},{"key":"21_CR22","unstructured":"Ng, A., Jordan, M., Weiss, Y.: On spectral clustering: analysis and an algorithm. In: Advances in Neural Information Processing Systems, vol. 14. MIT Press (2001)"},{"key":"21_CR23","first-page":"28954","volume":"34","author":"T Yu","year":"2021","unstructured":"Yu, T., Kumar, A., Rafailov, R., Rajeswaran, A., Levine, S., Finn, C.: COMBO: conservative offline model-based policy optimization. Adv. Neural Inf. Proces. Syst. 34, 28954\u201328967 (2021)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"21_CR24","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"ML Puterman","year":"2014","unstructured":"Puterman, M.L.: Markov Decision Processes: Discrete Stochastic Dynamic Programming. John Wiley & Sons, Hoboken (2014)"},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Tobin, J., Fong, R., Ray, A., Schneider, J., Zaremba, W., Abbeel, P.: Domain randomization for transferring deep neural networks from simulation to the real world. In: Proceedings of the 2017 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 23\u201330. IEEE (2017)","DOI":"10.1109\/IROS.2017.8202133"},{"key":"21_CR26","unstructured":"Wu, Y., Tucker, G., Nachum, O.: Behavior regularized offline reinforcement learning. arXiv preprint. https:\/\/arxiv.org\/abs\/1911.11361 (2019)"},{"key":"21_CR27","unstructured":"Xu, H., et al.: Offline RL with no OOD actions: in-sample learning via implicit value regularization. In: Proceedings of the International Conference on Learning Representations (ICLR) (2023)"},{"key":"21_CR28","unstructured":"Zhang, R., Dai, B., Li, L., Schuurmans, D.: GenDice: generalized offline estimation of stationary values. arXiv preprint. https:\/\/arxiv.org\/abs\/2002.09072 (2020)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3381-6_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T20:14:17Z","timestamp":1784405657000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3381-6_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819233809","9789819233816"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3381-6_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}