{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T20:25:17Z","timestamp":1787689517213,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","funder":[{"name":"Fundamental and Interdisciplinary Disciplines Breakthrough Plan of the Ministry of Education of China","award":["no. JYB2025XDXM108"],"award-info":[{"award-number":["no. JYB2025XDXM108"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3783946","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"2184-2195","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Effective Online 3D Bin Packing with Lookahead Parcels Using Monte Carlo Tree Search"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1772-5454","authenticated-orcid":false,"given":"Jiangyi","family":"Fang","sequence":"first","affiliation":[{"name":"Key Lab of High Confidence Software Technologies (Peking University), Ministry of Education, Beijing, China and School of Computer Science, Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5050-6197","authenticated-orcid":false,"given":"Bowen","family":"Zhou","sequence":"additional","affiliation":[{"name":"Faculty of Computing, Harbin Institute of Technology, Harbin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9783-6389","authenticated-orcid":false,"given":"Haotian","family":"Wang","sequence":"additional","affiliation":[{"name":"JD Logistics, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8422-4777","authenticated-orcid":false,"given":"Xin","family":"Zhu","sequence":"additional","affiliation":[{"name":"JD Logistics, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7627-8485","authenticated-orcid":false,"given":"Leye","family":"Wang","sequence":"additional","affiliation":[{"name":"Key Lab of High Confidence Software Technologies (Peking University), Ministry of Education, Beijing, China and School of Computer Science, Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"https:\/\/www.amazon.com Accessed on","year":"2025","unstructured":"2025. Amazon. https:\/\/www.amazon.com Accessed on July 18, 2025."},{"key":"e_1_3_2_2_2_1","volume-title":"https:\/\/www.jdl.com Accessed on","author":"Logistics JD","year":"2025","unstructured":"2025. JD Logistics. https:\/\/www.jdl.com Accessed on July 18, 2025."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cie.2022.108122"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ejor.2018.10.056"},{"key":"e_1_3_2_2_5_1","volume-title":"Learning to act using real-time dynamic programming. Artificial intelligence 72, 1--2","author":"Barto Andrew G","year":"1995","unstructured":"Andrew G Barto, Steven J Bradtke, and Satinder P Singh. 1995. Learning to act using real-time dynamic programming. Artificial intelligence 72, 1--2 (1995), 81--138."},{"key":"e_1_3_2_2_6_1","volume-title":"On approximating total variation distance. arXiv preprint arXiv:2206.07209","author":"Bhattacharyya Arnab","year":"2022","unstructured":"Arnab Bhattacharyya, Sutanu Gayen, Kuldeep S Meel, Dimitrios Myrisiotis, Aduri Pavan, and NV Vinodchandran. 2022. On approximating total variation distance. arXiv preprint arXiv:2206.07209 (2022)."},{"key":"e_1_3_2_2_7_1","volume-title":"The Bottom-Left Bin-Packing Heuristic: An Efficient Implementation","author":"Chazelle Bernard","year":"1983","unstructured":"Bernard Chazelle. 1983. The Bottom-Left Bin-Packing Heuristic: An Efficient Implementation. IEEE Trans. Comput. (1983)."},{"key":"e_1_3_2_2_8_1","volume-title":"Extreme Point-Based Heuristics for Three-Dimensional Bin Packing. INFORMS Journal on Computing","author":"Crainic Teodor Gabriel","year":"2008","unstructured":"Teodor Gabriel Crainic, Guido Perboli, and Roberto Tadei. 2008. Extreme Point-Based Heuristics for Three-Dimensional Bin Packing. INFORMS Journal on Computing (2008)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013494"},{"key":"e_1_3_2_2_10_1","first-page":"14024","article-title":"Online planning with lookahead policies","volume":"33","author":"Efroni Yonathan","year":"2020","unstructured":"Yonathan Efroni, Mohammad Ghavamzadeh, and Shie Mannor. 2020. Online planning with lookahead policies. Advances in Neural Information Processing Systems 33 (2020), 14024--14033.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_11_1","volume-title":"Lam Thu Bui, and RanWang.","author":"Ha Chi Trung","year":"2017","unstructured":"Chi Trung Ha, Trung Thanh Nguyen, Lam Thu Bui, and RanWang. 2017. An Online Packing Heuristic for the Three-Dimensional Container Loading Problem in Dynamic Environments and the Physical Internet. In Applications of Evolutionary Computation."},{"key":"e_1_3_2_2_12_1","volume-title":"Temporal difference learning for model predictive control. arXiv preprint arXiv:2203.04955","author":"Hansen Nicklas","year":"2022","unstructured":"Nicklas Hansen, XiaolongWang, and Hao Su. 2022. Temporal difference learning for model predictive control. arXiv preprint arXiv:2203.04955 (2022)."},{"key":"e_1_3_2_2_13_1","volume-title":"Solving a new 3D bin packing problem with deep reinforcement learning method. arXiv preprint arXiv:1708.05930","author":"Hu Haoyuan","year":"2017","unstructured":"Haoyuan Hu, Xiaodong Zhang, Xiaowei Yan, Longfei Wang, and Yinghui Xu. 2017. Solving a new 3D bin packing problem with deep reinforcement learning method. arXiv preprint arXiv:1708.05930 (2017)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3414685.3417796"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40305-023-00493-1"},{"key":"e_1_3_2_2_16_1","volume-title":"A Review of Yolo algorithm developments. Procedia computer science 199","author":"Jiang Peiyuan","year":"2022","unstructured":"Peiyuan Jiang, Daji Ergu, Fangyao Liu, Ying Cai, and Bo Ma. 2022. A Review of Yolo algorithm developments. Procedia computer science 199 (2022), 1066--1073."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30198-1_45"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Korhan Karabulut and Mustafa Murat Inceoglu. 2004. A Hybrid Genetic Algorithm for Packing in 3D with Deepest Bottom Left with Fill Method. In Advances in Information Systems.","DOI":"10.1007\/978-3-540-30198-1_45"},{"key":"e_1_3_2_2_19_1","first-page":"13","article-title":"Model predictive control. Switzerland","volume":"38","author":"Kouvaritakis Basil","year":"2016","unstructured":"Basil Kouvaritakis and Mark Cannon. 2016. Model predictive control. Switzerland: Springer International Publishing 38, 13--56 (2016), 7.","journal-title":"Springer International Publishing"},{"key":"e_1_3_2_2_20_1","volume-title":"The three-dimensional bin packing problem. Operations research 48, 2","author":"Martello Silvano","year":"2000","unstructured":"Silvano Martello, David Pisinger, and Daniele Vigo. 2000. The three-dimensional bin packing problem. Operations research 48, 2 (2000), 256--267."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2022.03.037"},{"key":"e_1_3_2_2_22_1","volume-title":"Reinforcement Learning with Lookahead Information. arXiv preprint arXiv:2406.02258","author":"Merlis Nadav","year":"2024","unstructured":"Nadav Merlis. 2024. Reinforcement Learning with Lookahead Information. arXiv preprint arXiv:2406.02258 (2024)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","first-page":"83627","DOI":"10.52202\/079017-2660","article-title":"The value of reward lookahead in reinforcement learning","volume":"37","author":"Merlis Nadav","year":"2024","unstructured":"Nadav Merlis, Dorian Baudry, and Vianney Perchet. 2024. The value of reward lookahead in reinforcement learning. Advances in Neural Information Processing Systems 37 (2024), 83627--83664.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_24_1","volume-title":"Think too fast nor too slow: The computational tradeoff between planning and reinforcement learning. arXiv preprint arXiv:2005.07404","author":"Moerland Thomas M","year":"2020","unstructured":"Thomas M Moerland, Anna Deichler, Simone Baldi, Joost Broekens, and Catholijn M Jonker. 2020. Think too fast nor too slow: The computational tradeoff between planning and reinforcement learning. arXiv preprint arXiv:2005.07404 (2020)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICC54714.2021.9703142"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10796-022-10252-x"},{"key":"e_1_3_2_2_27_1","volume-title":"Adjustable robust reinforcement learning for online 3D bin packing. Advances in Neural Information Processing Systems","author":"Pan Yuxin","year":"2023","unstructured":"Yuxin Pan, Yize Chen, and Fangzhen Lin. 2023. Adjustable robust reinforcement learning for online 3D bin packing. Advances in Neural Information Processing Systems (2023)."},{"key":"e_1_3_2_2_28_1","volume-title":"Robust reinforcement learning using offline data. Advances in neural information processing systems 35","author":"Panaganti Kishan","year":"2022","unstructured":"Kishan Panaganti, Zaiyan Xu, Dileep Kalathil, and Mohammad Ghavamzadeh. 2022. Robust reinforcement learning using offline data. Advances in neural information processing systems 35 (2022), 32211--32224."},{"key":"e_1_3_2_2_29_1","volume-title":"Critical random walk in random environment on trees. The Annals of Probability","author":"Pemantle Robin","year":"1995","unstructured":"Robin Pemantle and Yuval Peres. 1995. Critical random walk in random environment on trees. The Annals of Probability (1995), 105--140."},{"key":"e_1_3_2_2_30_1","volume-title":"IEEE\/RSJ International Conference on Intelligent Robots and Systems.","author":"Puche Aaron Valero","year":"2022","unstructured":"Aaron Valero Puche and Sukhan Lee. 2022. Online 3D bin packing reinforcement learning solution with buffer. In IEEE\/RSJ International Conference on Intelligent Robots and Systems."},{"key":"e_1_3_2_2_31_1","volume-title":"Innovations & Forecast 2025--2033. https:\/\/www.researchandmarkets.com\/reports\/6101938 Accessed on","author":"Markets Research","year":"2025","unstructured":"Research and Markets. 2025. Logistics Market - Trends, Innovations & Forecast 2025--2033. https:\/\/www.researchandmarkets.com\/reports\/6101938 Accessed on June, 2025."},{"key":"e_1_3_2_2_32_1","first-page":"56","article-title":"A survey of intelligent sensing technologies in autonomous driving","volume":"19","author":"Hong SHAO","year":"2021","unstructured":"Hong SHAO, Daxiong XIE, and Yihua HUANG. 2021. A survey of intelligent sensing technologies in autonomous driving. ZTE Communications 19, 3 (2021), 56.","journal-title":"ZTE Communications"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"David Silver Thomas Hubert Julian Schrittwieser Ioannis Antonoglou Matthew Lai Arthur Guez Marc Lanctot Laurent Sifre Dharshan Kumaran Thore Graepel et al. 2018. A general reinforcement learning algorithm that masters chess shogi and Go through self-play. Science 362 6419 (2018) 1140--1144.","DOI":"10.1126\/science.aar6404"},{"key":"e_1_3_2_2_34_1","volume-title":"Mastering the game of go without human knowledge. Nature","author":"Silver David","year":"2017","unstructured":"David Silver, Julian Schrittwieser, Karen Simonyan, Ioannis Antonoglou, Aja Huang, Arthur Guez, Thomas Hubert, Lucas Baker, Matthew Lai, Adrian Bolton, Yutian Chen, Timothy P. Lillicrap, Fan Hui, Laurent Sifre, George van den Driessche, Thore Graepel, and Demis Hassabis. 2017. Mastering the game of go without human knowledge. Nature (2017)."},{"key":"e_1_3_2_2_35_1","volume-title":"Conference on Robot Learning. PMLR, 1136--1145","author":"Song Shuai","year":"2023","unstructured":"Shuai Song, Shuo Yang, Ran Song, Shilei Chu, Wei Zhang, et al. 2023. Towards online 3d bin packing: Learning synergies between packing and unpacking via drl. In Conference on Robot Learning. PMLR, 1136--1145."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160765"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-022-10228-y"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compind.2024.104202"},{"key":"e_1_3_2_2_39_1","volume-title":"Swagat Kumar, and Rajesh Sinha.","author":"Verma Richa","year":"2020","unstructured":"Richa Verma, Aniruddha Singhal, Harshad Khadilkar, Ansuma Basumatary, Siddharth Nayak, Harsh Vardhan Singh, Swagat Kumar, and Rajesh Sinha. 2020. A generalized reinforcement learning algorithm for online 3d bin-packing. arXiv preprint arXiv:2007.00463 (2020)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794049"},{"key":"e_1_3_2_2_41_1","volume-title":"Dense robotic packing of irregular and novel 3D objects","author":"Wang Fan","year":"2021","unstructured":"Fan Wang and Kris Hauser. 2021. Dense robotic packing of irregular and novel 3D objects. IEEE Transactions on Robotics (2021)."},{"key":"e_1_3_2_2_42_1","volume-title":"Machine learning for the multi-dimensional bin packing problem: Literature review and empirical evaluation. arXiv preprint arXiv:2312.08103","author":"Wu Wenjie","year":"2023","unstructured":"Wenjie Wu, Changjun Fan, Jincai Huang, Zhong Liu, and Junchi Yan. 2023. Machine learning for the multi-dimensional bin packing problem: Literature review and empirical evaluation. arXiv preprint arXiv:2312.08103 (2023)."},{"key":"e_1_3_2_2_43_1","volume-title":"GOPT: Generalizable online 3D bin packing via transformer-based deep reinforcement learning","author":"Xiong Heng","year":"2024","unstructured":"Heng Xiong, Changrong Guo, Jian Peng, Kai Ding, Wenjie Chen, Xuchong Qiu, Long Bai, and Jianfeng Xu. 2024. GOPT: Generalizable online 3D bin packing via transformer-based deep reinforcement learning. IEEE Robotics and Automation Letters (2024)."},{"key":"e_1_3_2_2_44_1","volume-title":"Heuristics integrated deep reinforcement learning for online 3D bin packing","author":"Yang Shuo","year":"2023","unstructured":"Shuo Yang, Shuai Song, Shilei Chu, Ran Song, Jiyu Cheng, Yibin Li, and Wei Zhang. 2023. Heuristics integrated deep reinforcement learning for online 3D bin packing. IEEE Transactions on Automation Science and Engineering (2023)."},{"key":"e_1_3_2_2_45_1","first-page":"22","article-title":"A practical reinforcement learning framework for automatic radar detection","volume":"21","author":"Junpeng","year":"2023","unstructured":"Junpeng YU and Yiyu CHEN. 2023. A practical reinforcement learning framework for automatic radar detection. ZTE Communications 21, 3 (2023), 22.","journal-title":"ZTE Communications"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i1.16155"},{"key":"e_1_3_2_2_47_1","volume-title":"International Conference on Learning Representations.","author":"Zhao Hang","year":"2022","unstructured":"Hang Zhao, Yang Yu, and Kai Xu. 2022. Learning Efficient Online 3D Bin Packing on Packing Configuration Trees. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_48_1","volume-title":"Learning practically feasible policies for online 3D bin packing. Science China Information Sciences","author":"Zhao Hang","year":"2022","unstructured":"Hang Zhao, Chenyang Zhu, Xin Xu, Hui Huang, and Kai Xu. 2022. Learning practically feasible policies for online 3D bin packing. Science China Information Sciences (2022)."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2005.12.002"},{"key":"e_1_3_2_2_50_1","volume-title":"Proceedings of the 30th ACM International Conference on Information & Knowledge Management. 4393--4402","author":"Zhu Qianwen","year":"2021","unstructured":"Qianwen Zhu, Xihan Li, Zihan Zhang, Zhixing Luo, Xialiang Tong, Mingxuan Yuan, and Jia Zeng. 2021. Learning to pack:Adata-driven tree search algorithm for large-scale 3d bin packing problem. In Proceedings of the 30th ACM International Conference on Information & Knowledge Management. 4393--4402."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1098\/rsta.2013.0313"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3783946","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T20:15:11Z","timestamp":1787688911000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3783946"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":51,"alternative-id":["10.1145\/3770854.3783946","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3783946","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}