{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:32:44Z","timestamp":1740123164532,"version":"3.37.3"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62106122"],"award-info":[{"award-number":["62106122"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2022,9]]},"DOI":"10.1007\/s10994-022-06214-8","type":"journal-article","created":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T21:02:37Z","timestamp":1658178157000},"page":"3279-3306","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["The pure exploration problem with general reward functions depending on full distributions"],"prefix":"10.1007","volume":"111","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0764-5592","authenticated-orcid":false,"given":"Siwei","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,7,18]]},"reference":[{"key":"6214_CR1","unstructured":"Audibert, J. Y., Bubeck, S., & Munos, R. (2010). Best arm identification in multi-armed bandits. In COLT 2010 - the Conference on Learning Theory (pp. 41\u201353), Haifa, Israel."},{"issue":"2\u20133","key":"6214_CR2","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1023\/A:1013689704352","volume":"47","author":"P Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., & Fischer, P. (2002). Finite-time analysis of the multiarmed bandit problem. Machine Learning, 47(2\u20133), 235\u2013256.","journal-title":"Machine Learning"},{"key":"6214_CR3","doi-asserted-by":"crossref","unstructured":"Avenhaus, R., & Canty, M. J. (1996). Compliance quantified: An introduction to data verification. Cambridge University Press.","DOI":"10.1017\/CBO9780511526510"},{"key":"6214_CR4","doi-asserted-by":"crossref","unstructured":"Bensefia, A., Paquet, T., & Heutte, L. (2004). Handwriting analysis for writer verification. In Ninth International Workshop on Frontiers in Handwriting Recognition (pp. 196\u2013201). IEEE.","DOI":"10.1109\/IWFHR.2004.49"},{"key":"6214_CR5","unstructured":"Berend, D., & Kontorovich, A. (2012). On the convergence of the empirical distribution. VI International Workshop \u201cApplied Problems in Theory of Probabilities and Mathematical Statistics Related to Modeling of Information Systems\", Autumn Session."},{"key":"6214_CR6","unstructured":"Berry, D. A., & Fristedt, B. (1985). Bandit problems: Sequential allocation of experiments (Monographs on statistics and applied probability). Springer."},{"key":"6214_CR7","unstructured":"Bessler, S. A. (1960). Theory and applications of the sequential design of experiments, k-actions and infinitely many experiments. Part I. Theory. Tech. rep., Stanford Univ CA Applied Mathematics and Statistics Labs."},{"key":"6214_CR8","unstructured":"Bubeck, S., Wang, T., & Viswanathan, N. (2013). Multiple identifications in multi-armed bandits. In International Conference on Machine Learning (pp. 258\u2013265)."},{"key":"6214_CR9","unstructured":"Carpentier, A., & Valko, M. (2014) Extreme bandits. In Neural Information Processing Systems."},{"issue":"9","key":"6214_CR10","doi-asserted-by":"publisher","first-page":"2353","DOI":"10.1109\/TAC.2014.2321951","volume":"59","author":"CD Charalambous","year":"2014","unstructured":"Charalambous, C. D., Tzortzis, I., Loyka, S., & Charalambous, T. (2014). Extremum problems with total variation distance and their applications. IEEE Transactions on Automatic Control, 59(9), 2353\u20132368.","journal-title":"IEEE Transactions on Automatic Control"},{"key":"6214_CR11","unstructured":"Chen, L., Li, J., & Qiao, M. (2017).Towards instance optimal bounds for best arm identification. In Conference on Learning Theory (pp. 535\u2013592). PMLR."},{"key":"6214_CR12","unstructured":"Chen, W., Hu, W., Li, F., Li, J., Liu, Y., & Lu, P. (2016). Combinatorial multi-armed bandit with general reward functions. In NIPS."},{"issue":"3","key":"6214_CR13","doi-asserted-by":"publisher","first-page":"755","DOI":"10.1214\/aoms\/1177706205","volume":"30","author":"H Chernoff","year":"1959","unstructured":"Chernoff, H. (1959). Sequential design of experiments. Annals of Mathematical Statistics, 30(3), 755\u2013770.","journal-title":"Annals of Mathematical Statistics"},{"key":"6214_CR14","doi-asserted-by":"crossref","unstructured":"David, Y., & Shimkin, N. (2016). Pure exploration for max-quantile bandits. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases (pp. 556\u2013571). Springer.","DOI":"10.1007\/978-3-319-46128-1_35"},{"key":"6214_CR15","first-page":"14492","volume":"32","author":"R Degenne","year":"2019","unstructured":"Degenne, R., Koolen, W. M., & M\u00e9nard, P. (2019). Non-asymptotic pure exploration by solving games. Advances in Neural Information Processing Systems, 32, 14492\u201314501.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"1","key":"6214_CR16","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1007\/BF02613905","volume":"43","author":"V Dragalin","year":"1996","unstructured":"Dragalin, V. (1996). A simple and effective scanning rule for a multi-channel system. Metrika, 43(1), 165\u2013182.","journal-title":"Metrika"},{"key":"6214_CR17","doi-asserted-by":"crossref","unstructured":"Dvoretzky, A., Kiefer, J., & Wolfowitz, J. (1956). Asymptotic minimax character of the sample distribution function and of the classical multinomial estimator. The Annals of Mathematical Statistics, 642\u2013669.","DOI":"10.1214\/aoms\/1177728174"},{"key":"6214_CR18","first-page":"1079","volume":"7","author":"E Evendar","year":"2006","unstructured":"Evendar, E., Mannor, S., & Mansour, Y. (2006). Action elimination and stopping conditions for the multi-armed bandit and reinforcement learning problems. Journal of Machine Learning Research, 7, 1079\u20131105.","journal-title":"Journal of Machine Learning Research"},{"key":"6214_CR19","unstructured":"Galichet, N., Sebag, M., & Teytaud, O. (2013). Exploration vs exploitation vs safety: Risk-aware multi-armed bandits. In Asian Conference on Machine Learning (pp. 245\u2013260). PMLR."},{"key":"6214_CR20","unstructured":"Garivier, A., & Kaufmann, E. (2016). Optimal best arm identification with fixed confidence. In Conference on Learning Theory (pp. 998\u20131027). PMLR."},{"issue":"3","key":"6214_CR21","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1111\/j.1751-5823.2002.tb00178.x","volume":"70","author":"AL Gibbs","year":"2002","unstructured":"Gibbs, A. L., & Su, F. E. (2002). On choosing and bounding probability metrics. International statistical review, 70(3), 419\u2013435.","journal-title":"International statistical review"},{"key":"6214_CR22","doi-asserted-by":"crossref","unstructured":"Gittins, J., Glazebrook, K., & Weber, R. (2011). Multi-armed bandit allocation indices. Wiley.","DOI":"10.1002\/9780470980033"},{"key":"6214_CR23","unstructured":"Jamieson, K., Malloy, M., Nowak, R., & Bubeck, S. (2014). lil\u2019ucb: An optimal exploration algorithm for multi-armed bandits. In Conference on Learning Theory (pp. 423\u2013439). PMLR."},{"key":"6214_CR24","unstructured":"Kalyanakrishnan, S., Tewari, A., Auer, P., & Stone, P. (2012). Pac subset selection in stochastic multi-armed bandits. In International Conference on Machine Learning."},{"key":"6214_CR25","unstructured":"Kaufmann, E., Capp\u00e9, O., & Garivier, A. (2014). On the complexity of a\/b testing. In Conference on Learning Theory (pp. 461\u2013481). PMLR."},{"issue":"1","key":"6214_CR26","first-page":"1","volume":"17","author":"E Kaufmann","year":"2016","unstructured":"Kaufmann, E., Capp\u00e9, O., & Garivier, A. (2016). On the complexity of best-arm identification in multi-armed bandit models. The Journal of Machine Learning Research, 17(1), 1\u201342.","journal-title":"The Journal of Machine Learning Research"},{"key":"6214_CR27","unstructured":"Kaufmann, E., & Kalyanakrishnan, S. (2013). Information complexity in bandit subset selection. In Conference on Learning Theory (pp. 228\u2013251)."},{"key":"6214_CR28","unstructured":"Larsen, R.J. (1976) Statistics in the real world: A book of examples. Tech. rep."},{"issue":"318","key":"6214_CR29","doi-asserted-by":"publisher","first-page":"399","DOI":"10.1080\/01621459.1967.10482916","volume":"62","author":"HW Lilliefors","year":"1967","unstructured":"Lilliefors, H. W. (1967). On the Kolmogorov\u2013Smirnov test for normality with mean and variance unknown. Journal of the American statistical Association, 62(318), 399\u2013402.","journal-title":"Journal of the American statistical Association"},{"issue":"1","key":"6214_CR30","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1023\/A:1006556606079","volume":"11","author":"O Maron","year":"1997","unstructured":"Maron, O., & Moore, A. W. (1997). The racing algorithm: Model selection for lazy learners. Artificial Intelligence Review, 11(1), 193\u2013225.","journal-title":"Artificial Intelligence Review"},{"key":"6214_CR31","doi-asserted-by":"crossref","unstructured":"Massart, P. (1990). The tight constant in the dvoretzky-kiefer-wolfowitz inequality. The Annals of Probability, 1269\u20131283.","DOI":"10.1214\/aop\/1176990746"},{"key":"6214_CR32","doi-asserted-by":"crossref","unstructured":"Pochampally, K.K., & Gupta, S. M. (2014). Six sigma case studies with Minitab\u00ae. CRC Press.","DOI":"10.1201\/b16371"},{"key":"6214_CR33","unstructured":"Sani, A., Lazaric, A., & Munos, R. (2012). Risk-aversion in multi-armed bandits. In Proceedings of the 25th International Conference on Neural Information Processing Systems-Volume 2 (pp. 3275\u20133283)."},{"key":"6214_CR34","doi-asserted-by":"crossref","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning: An introduction, Vol.\u00a01. MIT Press.","DOI":"10.1109\/TNN.1998.712192"},{"key":"6214_CR35","unstructured":"Szorenyi, B., Busa-Fekete, R., Weng, P., & H\u00fcllermeier, E. (2015) Qualitative multi-armed bandits: A quantile-based approach. In International Conference on Machine Learning (pp. 1660\u20131668)."},{"issue":"6","key":"6214_CR36","doi-asserted-by":"publisher","first-page":"1093","DOI":"10.1109\/JSTSP.2016.2592622","volume":"10","author":"S Vakili","year":"2016","unstructured":"Vakili, S., & Zhao, Q. (2016). Risk-averse multi-armed bandit problems under mean-variance measure. IEEE Journal of Selected Topics in Signal Processing, 10(6), 1093\u20131111.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"issue":"2","key":"6214_CR37","doi-asserted-by":"publisher","first-page":"294","DOI":"10.1137\/1111025","volume":"11","author":"KS Zigangirov","year":"1966","unstructured":"Zigangirov, K. S. (1966). On a problem in optimal scanning. Theory of Probability & Its Applications, 11(2), 294\u2013298.","journal-title":"Theory of Probability & Its Applications"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-022-06214-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-022-06214-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-022-06214-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,18]],"date-time":"2023-07-18T00:03:07Z","timestamp":1689638587000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-022-06214-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,18]]},"references-count":37,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2022,9]]}},"alternative-id":["6214"],"URL":"https:\/\/doi.org\/10.1007\/s10994-022-06214-8","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"type":"print","value":"0885-6125"},{"type":"electronic","value":"1573-0565"}],"subject":[],"published":{"date-parts":[[2022,7,18]]},"assertion":[{"value":"17 November 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 June 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 July 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"We hereby confirm that all the authors are aware of this submission and consent to its review by Machine Learning Journal.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"We hereby confirm that all the authors are aware of this submission and consent to its publication.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}