{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T19:31:12Z","timestamp":1771702272447,"version":"3.50.1"},"publisher-location":"New York, New York, USA","reference-count":21,"publisher":"ACM Press","license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1145\/3184558.3191630","type":"proceedings-article","created":{"date-parts":[[2018,4,18]],"date-time":"2018-04-18T18:04:25Z","timestamp":1524074665000},"page":"1707-1712","source":"Crossref","is-referenced-by-count":4,"title":["Stochastic Multi-path Routing Problem with Non-stationary Rewards"],"prefix":"10.1145","author":[{"given":"Pankaj","family":"Trivedi","sequence":"first","affiliation":[{"name":"PayU Payments Pvt Ltd, Gurgaon, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arvind","family":"Singh","sequence":"additional","affiliation":[{"name":"PayU Payments Pvt Ltd, Gurgaon, India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","reference":[{"key":"key-10.1145\/3184558.3191630-1","unstructured":"Omar Besbes, Yonatan Gur, Assaf Zeevi Optimal Exploration-Exploitation in a Multi-Armed-Bandit Problem with Non-stationary Rewards."},{"key":"key-10.1145\/3184558.3191630-2","doi-asserted-by":"crossref","unstructured":"P. Auer, N. Cesa-Bianchi, and P. Fischer. Finite-time analysis of the multiarmed bandit problem. Machine Learning, 47(2--3):235--256, 2002.","DOI":"10.1023\/A:1013689704352"},{"key":"key-10.1145\/3184558.3191630-3","doi-asserted-by":"crossref","unstructured":"M. Zelen. Play the winner rule and the controlled clinical trials. Journal of the American Statistical Association, 64:131--146, 1969.","DOI":"10.1080\/01621459.1969.10500959"},{"key":"key-10.1145\/3184558.3191630-4","doi-asserted-by":"crossref","unstructured":"D. A. Berry and B. Fristedt. Bandit problems: sequential allocation of experiments. Chapman and Hall, 1985.","DOI":"10.1007\/978-94-015-3711-7"},{"key":"key-10.1145\/3184558.3191630-5","unstructured":"O. Chapelle and L. Li. An empirical evaluation of thompson sampling. In NIPS, 2011."},{"key":"key-10.1145\/3184558.3191630-6","doi-asserted-by":"crossref","unstructured":"P. Auer, N. Cesa-Bianchi, and P. Fischer. Finite-time analysis of the multiarmed bandit problem. Machine Learning, 2002.","DOI":"10.1023\/A:1013689704352"},{"key":"key-10.1145\/3184558.3191630-7","doi-asserted-by":"crossref","unstructured":"R. D. Kleinberg and T. Leighton. The value of knowing a demand curve: Bounds on regret for online posted-price auctions. In Proceedings of the 44th Annual IEEE Symposium on Foundations of Computer Science (FOCS), pages 594--605, 2003.","DOI":"10.1109\/SFCS.2003.1238232"},{"key":"key-10.1145\/3184558.3191630-8","doi-asserted-by":"crossref","unstructured":"Sebastien Bubeck and Nicolo Cesa-Bianchi. Regret analysis of stochastic and nonstochastic multi-armed bandit problems. Foundations and Trends in Machine Learning, 5(1):1--122, 2012.","DOI":"10.1561\/2200000024"},{"key":"key-10.1145\/3184558.3191630-9","doi-asserted-by":"crossref","unstructured":"J. C. Gittins. Bandit processes and dynamic allocation indices (with discussion). Journal of the Royal Statistical Society, Series B, 41:148--177, 1979.","DOI":"10.1111\/j.2517-6161.1979.tb01068.x"},{"key":"key-10.1145\/3184558.3191630-10","doi-asserted-by":"crossref","unstructured":"O. Besbes, Y. Gur, and A. Zeevi. Non-stationary stochastic optimization. Working paper, 2014.","DOI":"10.2139\/ssrn.2296012"},{"key":"key-10.1145\/3184558.3191630-11","unstructured":"S. Agrawal and N. Goyal. Analysis of thompson sampling for the multi-armed bandit problem. CoRR, abs\/1111.1797, 2011."},{"key":"key-10.1145\/3184558.3191630-12","doi-asserted-by":"crossref","unstructured":"M. Babaioff, Y. Sharma, and A. Slivkins. Characterizing truthful multi-armed bandit mechanisms: extended abstract. In Tenth ACM Conference on Electronic Commerce, pages 79--88. ACM, 2009.","DOI":"10.1145\/1566374.1566386"},{"key":"key-10.1145\/3184558.3191630-13","unstructured":"O. Chapelle and L. Li. An empirical evaluation of thompson sampling. In J. Shawe- Taylor, R. Zemel, P. Bartlett, F. Pereira, and K. Weinberger, editors, Advances in Neural Information Processing Systems 24, pages 2249--2257. Curran Associates, Inc., 2011."},{"key":"key-10.1145\/3184558.3191630-14","doi-asserted-by":"crossref","unstructured":"N. Gatti, A. Lazaric, and F. Trov&#195;. A truthful learning mechanism for contextual multi-slot sponsored search auctions with externalities. In Thirteenth ACM Conference on Electronic Commerce, pages 605--622, 2012.","DOI":"10.1145\/2229012.2229057"},{"key":"key-10.1145\/3184558.3191630-15","doi-asserted-by":"crossref","unstructured":"O.-C. Granmo. Solving two-armed bernoulli bandit problems using a bayesian learning automaton. International Journal of Intelligent Computing and Cybernetics, 3(2):207--234, 2010.","DOI":"10.1108\/17563781011049179"},{"key":"key-10.1145\/3184558.3191630-16","doi-asserted-by":"crossref","unstructured":"S. Scott. A modern bayesian look at the multi-armed bandit. Applied Stochastic Models in Business and Industry, 26:639--658, 2010.","DOI":"10.1002\/asmb.874"},{"key":"key-10.1145\/3184558.3191630-17","doi-asserted-by":"crossref","unstructured":"D. Bergemann and J. Valimaki. Learning and strategic pricing. Econometrica, 64:1125--1149, 1996.","DOI":"10.2307\/2171959"},{"key":"key-10.1145\/3184558.3191630-18","unstructured":"D. Bergemann and U. Hege. The financing of innovation: Learning and stopping. RAND Journal of Economics, 36 (4):719--752, 2005."},{"key":"key-10.1145\/3184558.3191630-19","doi-asserted-by":"crossref","unstructured":"B. Awerbuch and R. D. Kleinberg. Addaptive routing with end-to-end feedback: distributed learning and geometric approaches. In Proceedings of the 36th ACM Symposiuim on Theory of Computing (STOC), pages 45--53, 2004.","DOI":"10.1145\/1007352.1007367"},{"key":"key-10.1145\/3184558.3191630-20","doi-asserted-by":"crossref","unstructured":"F. Caro and G. Gallien. Dynamic assortment with demand learning for seasonal consumer goods. Management Science, 53:276--292, 2007.","DOI":"10.1287\/mnsc.1060.0613"},{"key":"key-10.1145\/3184558.3191630-21","doi-asserted-by":"crossref","unstructured":"S. Pandey, D. Agarwal, D. Charkrabarti, and V. Josifovski. Bandits for taxonomies: A model-based approach. In SIAM International Conference on Data Mining, 2007","DOI":"10.1137\/1.9781611972771.20"}],"event":{"name":"Companion of the The Web Conference 2018","location":"Lyon, France","acronym":"WWW '18","number":"2018","sponsor":["IW3C2, International World Wide Web Conference Committee","SIGWEB, ACM Special Interest Group on Hypertext, Hypermedia, and Web"],"start":{"date-parts":[[2018,4,23]]},"end":{"date-parts":[[2018,4,27]]}},"container-title":["Companion of the The Web Conference 2018 on The Web Conference 2018 - WWW '18"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3184558.3191630","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/dl.acm.org\/ft_gateway.cfm?id=3191630&ftid=1958451&dwn=1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:39:26Z","timestamp":1750210766000},"score":1,"resource":{"primary":{"URL":"http:\/\/dl.acm.org\/citation.cfm?doid=3184558.3191630"}},"subtitle":["Building PayU's Dynamic Routing"],"proceedings-subject":"The Web Conference 2018","short-title":[],"issued":{"date-parts":[[2018]]},"references-count":21,"URL":"https:\/\/doi.org\/10.1145\/3184558.3191630","relation":{},"subject":[],"published":{"date-parts":[[2018]]}}}