{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,8]],"date-time":"2026-02-08T07:17:53Z","timestamp":1770535073869,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T00:00:00Z","timestamp":1531958400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Jiangsu Science Foundation","award":["BK20170013"],"award-info":[{"award-number":["BK20170013"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,7,19]]},"DOI":"10.1145\/3219819.3220122","type":"proceedings-article","created":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T13:05:12Z","timestamp":1532005512000},"page":"1187-1196","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":96,"title":["Stabilizing Reinforcement Learning in Dynamic Environment with Application to Online Recommendation"],"prefix":"10.1145","author":[{"given":"Shi-Yong","family":"Chen","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qing","family":"Da","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Tan","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hai-Kuan","family":"Huang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hai-Hong","family":"Tang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","article-title":"Addressing Environment Non-Stationarity by Repeating Q-learning Updates","volume":"17","author":"Abdallah Sherief","year":"2016","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1013689704352"},{"key":"e_1_3_2_1_3_1","volume-title":"Reinforcement Learning based Recommender System using Biclustering Technique. CoRR","author":"Choi Sungwoon","year":"2018"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.5120\/ijca2017913081"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btt662"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219846"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00184-007-0161-4"},{"key":"e_1_3_2_1_8_1","volume-title":"DJ-MC: A Reinforcement-Learning Agent for Music Playlist Recommendation Proceedings of the 14th International Conference on Autonomous Agents and Multiagent Systems","author":"Liebman Elad","year":"2015"},{"key":"e_1_3_2_1_9_1","volume-title":"Riedmiller","author":"Mnih Volodymyr","year":"2013"},{"key":"e_1_3_2_1_10_1","volume-title":"et almbox. . 2015 a. Human-level control through deep reinforcement learning. Nature","author":"Mnih Volodymyr","year":"2015"},{"key":"e_1_3_2_1_11_1","volume-title":"2015 b. Human-level control through deep reinforcement learning. Nature","author":"Mnih Volodymyr","year":"2015"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10015-013-0106-0"},{"key":"e_1_3_2_1_13_1","volume-title":"Deep Exploration via Bootstrapped DQN. CoRR","author":"Osband Ian","year":"2016"},{"key":"e_1_3_2_1_14_1","volume-title":"Play it again: reactivation of waking experience and memory. Trends in neurosciences","author":"O'Neill Joseph","year":"2010"},{"key":"e_1_3_2_1_15_1","volume-title":"Wiering","author":"Pieters Mathijs","year":"2016"},{"key":"e_1_3_2_1_16_1","volume-title":"Prioritized Experience Replay. CoRR","author":"Schaul Tom","year":"2015"},{"key":"e_1_3_2_1_17_1","volume-title":"Mastering the game of Go with deep neural networks and tree search. Nature","author":"Silver David","year":"2016"},{"key":"e_1_3_2_1_18_1","volume-title":"et almbox","author":"Silver David","year":"2017"},{"key":"e_1_3_2_1_19_1","volume-title":"Barto","author":"Sutton Richard S.","year":"1998"},{"key":"e_1_3_2_1_20_1","volume-title":"Advances in Neural Information Processing Systems 24.","author":"van Hasselt Hado"},{"key":"e_1_3_2_1_21_1","volume-title":"Deep Reinforcement Learning with Double Q-Learning Proceedings of the 30th AAAI Conference on Artificial Intelligence","author":"van Hasselt Hado","year":"2016"},{"key":"e_1_3_2_1_22_1","volume-title":"Dueling Network Architectures for Deep Reinforcement Learning. CoRR","author":"Wang Ziyu","year":"2015"},{"key":"e_1_3_2_1_23_1","volume-title":"Dueling Network Architectures for Deep Reinforcement Learning Proceedings of the 33th International Conference on Machine Learning","author":"Wang Ziyu","year":"2016"},{"key":"e_1_3_2_1_25_1","volume-title":"Reinforcement Learning in Dynamic Environments using Instantiated Information Proceedings of the 18th International Conference on Machine Learning","author":"Marco Wiering"}],"event":{"name":"KDD '18: The 24th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining","location":"London United Kingdom","acronym":"KDD '18","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery &amp; Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3219819.3220122","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3219819.3220122","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:07:30Z","timestamp":1750212450000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3219819.3220122"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,7,19]]},"references-count":24,"alternative-id":["10.1145\/3219819.3220122","10.1145\/3219819"],"URL":"https:\/\/doi.org\/10.1145\/3219819.3220122","relation":{},"subject":[],"published":{"date-parts":[[2018,7,19]]},"assertion":[{"value":"2018-07-19","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}