{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:16:10Z","timestamp":1740122170595,"version":"3.37.3"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,5,16]],"date-time":"2024-05-16T00:00:00Z","timestamp":1715817600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,16]],"date-time":"2024-05-16T00:00:00Z","timestamp":1715817600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072099","62076060","61806053","61932007","61932007"],"award-info":[{"award-number":["62072099","62076060","61806053","61932007","61932007"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s10458-024-09650-z","type":"journal-article","created":{"date-parts":[[2024,5,16]],"date-time":"2024-05-16T12:01:56Z","timestamp":1715860916000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Offline policy reuse-guided anytime online collective multiagent planning and its application to mobility-on-demand systems"],"prefix":"10.1007","volume":"38","author":[{"given":"Wanyuan","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qian","family":"Che","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifeng","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weiwei","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"An","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yichuan","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,5,16]]},"reference":[{"key":"9650_CR1","unstructured":"(2016). Taxi and limousine commission (tlc) trip record data. https:\/\/www1.nyc.gov\/site\/tlc\/about\/tlc-trip-record-data.page"},{"issue":"3","key":"9650_CR2","doi-asserted-by":"publisher","first-page":"462","DOI":"10.1073\/pnas.1611675114","volume":"114","author":"J Alonso-Mora","year":"2017","unstructured":"Alonso-Mora, J., Samaranayake, S., Wallar, A., Frazzoli, E., & Rus, D. (2017). On-demand high-capacity ride-sharing via dynamic trip-vehicle assignment. Proceedings of the National Academy of Sciences, 114(3), 462\u2013467.","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"9650_CR3","unstructured":"Bellman, R. E. (1957). Dynamic programming, adaptive computation and machine learning. Princeton University Press."},{"key":"9650_CR4","unstructured":"Bertsekas, D. P. (2005). Dynamic programming and optimal control (3rd ed.). Athena Scientific."},{"key":"9650_CR5","doi-asserted-by":"crossref","unstructured":"Bistaffa, F., Farinelli, A. & Ramchurn, S. D. (2015), Sharing rides with friends: A coalition formation algorithm for ridesharing. In Proceedings of the 29th AAAI conference on artificial intelligence, January 25\u201330, 2015, Austin, TX, USA (pp. 608\u2013614).","DOI":"10.1609\/aaai.v29i1.9242"},{"key":"9650_CR6","doi-asserted-by":"crossref","unstructured":"Boyd, S., & Vandenberghe, L. (2004). Convex optimization (1st Ed.). Cambridge University Press.","DOI":"10.1017\/CBO9780511804441"},{"key":"9650_CR7","doi-asserted-by":"crossref","unstructured":"Brown, M., Saisubramanian, S., Varakantham, P., & Tambe M. (2014). STREETS: Game-theoretic traffic patrolling with exploration and exploitation. In AAAI\u201914 (pp. 2966\u20132971).","DOI":"10.1609\/aaai.v28i2.19028"},{"key":"9650_CR8","doi-asserted-by":"crossref","unstructured":"Chaudhari, H. A., Byers, J. W. & Terzi, E. (2020). Learn to earn: Enabling coordination within a ride-hailing fleet. In IEEE international conference on big data, big data 2020, Atlanta, GA, USA, December 10\u201313, 2020 (pp. 1127\u20131136).","DOI":"10.1109\/BigData50022.2020.9378416"},{"key":"9650_CR9","unstructured":"Claes, D., Oliehoek, FA., Baier, H., & Tuyls, K. (2017). Decentralised online planning for multi-robot warehouse commissioning. In Proceedings of the 16th conference on autonomous agents and multiAgent systems, AAMAS 2017, S\u00e3o Paulo, Brazil, May 8\u201312 (pp. 492\u2013500)."},{"key":"9650_CR10","doi-asserted-by":"crossref","unstructured":"Dickerson, J. P., Sankararaman, K. A., Srinivasan, A., & Xu, P. (2018a). Allocation problems in ride-sharing platforms: Online matching with offline reusable resources. In Proceedings of the 32nd AAAI conference on artificial intelligence (AAAI\u201918), New Orleans, Louisiana, USA, February 2\u20137, 2018 (pp. 1007\u20131014).","DOI":"10.1609\/aaai.v32i1.11477"},{"key":"9650_CR11","unstructured":"Dickerson, J. P., Sankararaman, K. A., Srinivasan, A., & Xu, P. (2018b). Assigning tasks to workers based on historical data: Online task assignment with two-sided arrivals. In Proceedings of the 17th international conference on autonomous agents and multiAgent systems, AAMAS 2018, Stockholm, Sweden, July 10\u201315, 2018 (pp. 318\u2013326)."},{"key":"9650_CR12","doi-asserted-by":"crossref","unstructured":"Duchi, J., Shalev-Shwartz, S., Singer, Y., & Chandra, T. (2008). Efficient projections onto the l1-ball for learning in high dimensions. In ICML\u201908 (pp. 272\u2013279).","DOI":"10.1145\/1390156.1390191"},{"key":"9650_CR13","unstructured":"Flaxman, A., Kalai, A. T. & McMahan, H. B. (2005). Online convex optimization in the bandit setting: gradient descent without a gradient. In SODA\u201905 (pp. 385\u2013394)."},{"key":"9650_CR14","doi-asserted-by":"crossref","unstructured":"Gilpin, A., Hoda, S., Pe\u00f1a, J., & Sandholm, T. (2007). Gradient-based algorithms for finding Nash equilibria in extensive form games. In WINE\u201907 (pp. 57\u201369).","DOI":"10.1007\/978-3-540-77105-0_9"},{"key":"9650_CR15","doi-asserted-by":"crossref","unstructured":"He, S. & Shin, K. G. (2019). Spatio-temporal capsule-based reinforcement learning for mobility-on-demand network coordination. In The World Wide Web conference, WWW 2019, San Francisco, CA, USA, May 13\u201317, 2019 (pp. 2806\u20132813).","DOI":"10.1145\/3308558.3313401"},{"key":"9650_CR16","doi-asserted-by":"crossref","unstructured":"Jin, J., Zhou, M., Zhang, W., Li, M., Guo, Z., Qin, Z., Jiao, Y., Tang, X., Wang, C., Wang, J., & Wu, G. (2019). Coride: Joint order dispatching and fleet management for multi-scale ride-hailing platforms. In Proceedings of the 28th ACM international conference on information and knowledge management, CIKM 2019, Beijing, China, November 3\u20137, 2019 (pp. 1983\u20131992).","DOI":"10.1145\/3357384.3357978"},{"key":"9650_CR17","unstructured":"Kumar, R. R. & Varakantham, P. (2017). Exploiting anonymity and homogeneity in factored dec-mdps through precomputed binomial distributions. In Proceedings of the 16th conference on autonomous agents and multiAgent systems, AAMAS\u201917 (pp. 732\u2013740)."},{"key":"9650_CR18","doi-asserted-by":"crossref","unstructured":"Li, M., Qin, Z., Jiao, Y., Yang, Y., Wang, J., Wang, C., Wu, G., & Ye, J. (2019). Efficient ridesharing order dispatching with mean field multi-agent reinforcement learning. In The World Wide Web conference, WWW 2019, San Francisco, CA, USA, May 13\u201317, 2019 (pp. 983\u2013994).","DOI":"10.1145\/3308558.3313433"},{"key":"9650_CR19","doi-asserted-by":"crossref","unstructured":"Lin, K., Zhao, R., Xu, Z., & Zhou, J. (2018). Efficient large-scale fleet management via multi-agent deep reinforcement learning. In Proceedings of the 24th ACM SIGKDD international conference on knowledge discovery & data mining, KDD 2018, London, UK, August 19\u201323, 2018 (pp. 1774\u20131783).","DOI":"10.1145\/3219819.3219993"},{"key":"9650_CR20","unstructured":"Lowalekar, M., Varakantham, P. & Jaillet, P. (2020). Competitive ratios for online multi-capacity ridesharing. In Proceedings of the 19th international conference on autonomous agents and multiagent systems, AAMAS \u201920, Auckland, New Zealand, May 9\u201313, 2020 (pp. 771\u2013779)."},{"key":"9650_CR21","doi-asserted-by":"crossref","unstructured":"Lowalekar, M., Varakantham, P., Ghosh, S., & Jena, S., & Jaillet, P. (2017). Online repositioning in bike sharing systems. In Proceedings of the 27th international conference on automated planning and scheduling (ICAPS\u201917), Pittsburgh, Pennsylvania, USA, June 18\u201323, 2017 (pp. 200\u2013208).","DOI":"10.1609\/icaps.v27i1.13824"},{"key":"9650_CR22","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1016\/j.artint.2018.04.005","volume":"261","author":"M Lowalekar","year":"2018","unstructured":"Lowalekar, M., Varakantham, P., & Jaillet, P. (2018). Online spatio-temporal matching in stochastic and dynamic domains. Artificial Intelligence, 261, 71\u2013112.","journal-title":"Artificial Intelligence"},{"issue":"7","key":"9650_CR23","doi-asserted-by":"publisher","first-page":"1782","DOI":"10.1109\/TKDE.2014.2334313","volume":"27","author":"S Ma","year":"2015","unstructured":"Ma, S., Zheng, Y., & Wolfson, O. (2015). Real-time city-scale taxi ridesharing. IEEE Transactions on Knowledge and Data Engineering, 27(7), 1782\u20131795.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"9650_CR24","unstructured":"Mukhopadhyay, A., Vorobeychik, Y., & Dubey, A. (2017). Prioritized allocation of emergency responders based on a continuous-time incident prediction model. In Proceedings of the 16th conference on autonomous agents and multiAgent systems (AAMAS\u201917), S\u00e3o Paulo, Brazil, May 8\u201312, 2017 (pp. 168\u2013177)."},{"key":"9650_CR25","doi-asserted-by":"crossref","unstructured":"Nguyen, D. T., Kumar, A. & Lau, H. C. (2017a). Collective multiagent sequential decision making under uncertainty. In Proceedings of the 31st AAAI conference on artificial intelligence, February 4\u20139, 2017, San Francisco, California, USA (pp. 3036\u20133043).","DOI":"10.1609\/aaai.v31i1.10708"},{"key":"9650_CR26","unstructured":"Nguyen, D. T., Kumar, A. & Lau, H. C. (2017b). Policy gradient with value function approximation for collective multiagent planning. In Advances in neural information processing systems 30: Annual conference on neural information processing systems 2017, December 4\u20139, 2017, Long Beach, CA, USA (pp. 4319\u20134329)."},{"key":"9650_CR27","unstructured":"Nguyen, D. T., Kumar, A., & Lau, H. C. (2018). Credit assignment for collective multiagent rl with global rewards. In NeuIPS\u201918 (pp. 8102\u20138113)."},{"issue":"5","key":"9650_CR28","first-page":"272","volume":"50","author":"ZT Qin","year":"2020","unstructured":"Qin, Z. T., Tang, X., Jiao, Y., Zhang, F., Xu, Z., Zhu, H., & Ye, J. (2020). Ride-hailing order dispatching at didi via reinforcement learning. Interfaces, 50(5), 272\u2013286.","journal-title":"Interfaces"},{"key":"9650_CR29","doi-asserted-by":"crossref","unstructured":"Rosenfeld, A. & Kraus, S. (2017). When security games hit traffic: Optimal traffic enforcement under one sided uncertainty. In Sierra C (Ed.), IJCAI\u201917 (pp. 3814\u20133822).","DOI":"10.24963\/ijcai.2017\/533"},{"issue":"103","key":"9650_CR30","first-page":"381","volume":"289","author":"A Rosenfeld","year":"2020","unstructured":"Rosenfeld, A., Maksimov, O., & Kraus, S. (2020). When security games hit traffic: A deployed optimal traffic enforcement system. Artificial Intelligence, 289(103), 381.","journal-title":"Artificial Intelligence"},{"key":"9650_CR31","unstructured":"Sutton, R. S., McAllester, D. A., Singh, S. P., & Mansour, Y. (1999). Policy gradient methods for reinforcement learning with function approximation. In Advances in neural information processing systems 12, [NIPS conference, Denver, Colorado, USA, November 29\u2013December 4, 1999] (pp. 1057\u20131063)."},{"key":"9650_CR32","doi-asserted-by":"crossref","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning\u2014An introduction. Adaptive computation and machine learning. MIT Press.","DOI":"10.1109\/TNN.1998.712192"},{"key":"9650_CR33","doi-asserted-by":"crossref","unstructured":"Tang, X., Zhang, F., Qin, Z., Wang, Y., Shi, D., Song, B., Tong, Y., Zhu, H., & Ye, J. (2021). Value function is all you need: A unified learning framework for ride hailing platforms. In KDD\u201921.","DOI":"10.1145\/3447548.3467096"},{"key":"9650_CR34","first-page":"1633","volume":"10","author":"ME Taylor","year":"2009","unstructured":"Taylor, M. E., & Stone, P. (2009). Transfer learning for reinforcement learning domains: A survey. Journal of Machine Learning Research, 10, 1633\u20131685.","journal-title":"Journal of Machine Learning Research"},{"issue":"5","key":"9650_CR35","first-page":"2295","volume":"33","author":"Y Tong","year":"2021","unstructured":"Tong, Y., Zeng, Y., Ding, B., Wang, L., & Chen, L. (2021). Two-sided online micro-task assignment in spatial crowdsourcing. IIEEE Transactions on Knowledge and Data Engineering, 33(5), 2295\u20132309.","journal-title":"IIEEE Transactions on Knowledge and Data Engineering"},{"key":"9650_CR36","doi-asserted-by":"crossref","unstructured":"Varakantham, P., Adulyasak, Y. & Jaillet, P. (2014). Decentralized stochastic planning with anonymity in interactions. In AAAI\u201914 (pp. 2505\u20132512).","DOI":"10.1609\/aaai.v28i1.9069"},{"issue":"2","key":"9650_CR37","doi-asserted-by":"publisher","first-page":"747","DOI":"10.1109\/TITS.2019.2955989","volume":"22","author":"W Wang","year":"2021","unstructured":"Wang, W., Dong, Z., An, B., & Jiang, Y. (2021). Toward efficient city-scale patrol planning using decomposition and grafting. IEEE Transactions on Intelligent Transportation Systems, 22(2), 747\u2013757.","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"9650_CR38","doi-asserted-by":"crossref","unstructured":"Williams, R. J. (1992). Simple statistical gradient-following algorithms for connectionist reinforcement learning. In Machine learning (pp. 229\u2013256).","DOI":"10.1007\/BF00992696"},{"key":"9650_CR39","doi-asserted-by":"crossref","unstructured":"Xu, Z., Li, Z., Guan, Q., Zhang, D., Li, Q., Nan, J., Liu, C., Bian, W., & Ye, J. (2018). Large-scale order dispatch in on-demand ride-hailing platforms: A learning and planning approach. In KDD\u201918 (pp. 905\u2013913).","DOI":"10.1145\/3219819.3219824"},{"key":"9650_CR40","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2021.3054893","author":"Y Xu","year":"2021","unstructured":"Xu, Y., Wang, W., Xiong, G., Liu, X., Wu, W., & Liu, K. (2021). Network-flow-based efficient vehicle dispatch for city-scale ride-hailing systems. Transactions on Intelligent Transportation Systems. https:\/\/doi.org\/10.1109\/TITS.2021.3054893","journal-title":"Transactions on Intelligent Transportation Systems"},{"key":"9650_CR41","unstructured":"Yue, Y., Marla, L. & Krishnan, R. (2012). An efficient simulation-based approach to ambulance fleet allocation and dynamic redeployment. In Proceedings of the 26th AAAI conference on artificial intelligence, July 22\u201326, 2012, Toronto, ON, Canada."},{"key":"9650_CR42","doi-asserted-by":"crossref","unstructured":"Zhou, M., Jin, J., Zhang, W., Qin, Z., Jiao, Y., Wang, C., Wu, G., Yu, Y., & Ye, J. (2019). Multi-agent reinforcement learning for order-dispatching via order-vehicle distribution matching. In Proceedings of the 28th ACM international conference on information and knowledge management, CIKM 2019, Beijing, China, November 3\u20137, 2019 (pp. 2645\u20132653).","DOI":"10.1145\/3357384.3357799"}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-024-09650-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10458-024-09650-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-024-09650-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T23:07:45Z","timestamp":1719875265000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10458-024-09650-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,16]]},"references-count":42,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["9650"],"URL":"https:\/\/doi.org\/10.1007\/s10458-024-09650-z","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"type":"print","value":"1387-2532"},{"type":"electronic","value":"1573-7454"}],"subject":[],"published":{"date-parts":[[2024,5,16]]},"assertion":[{"value":"19 April 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 May 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declared that they have no Conflict of interest to this work.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"19"}}