{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,5]],"date-time":"2025-10-05T20:02:18Z","timestamp":1759694538040},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2019,5,8]],"date-time":"2019-05-08T00:00:00Z","timestamp":1557273600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2019,5,8]],"date-time":"2019-05-08T00:00:00Z","timestamp":1557273600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Inf Retrieval J"],"published-print":{"date-parts":[[2020,2]]},"DOI":"10.1007\/s10791-019-09357-w","type":"journal-article","created":{"date-parts":[[2019,5,8]],"date-time":"2019-05-08T20:00:50Z","timestamp":1557345650000},"page":"49-85","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Fewer topics? A million topics? Both?! On topics subsets in test collections"],"prefix":"10.1007","volume":"23","author":[{"given":"Kevin","family":"Roitero","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J. Shane","family":"Culpepper","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mark","family":"Sanderson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Falk","family":"Scholer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefano","family":"Mizzaro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,8]]},"reference":[{"key":"9357_CR1","doi-asserted-by":"crossref","unstructured":"Allan, J., Carterette, B., Aslam, J. A., Pavlu, V., Dachev, B., & Kanoulas, E. (2007). Million query track 2007 overview. In Proceedings of TREC.","DOI":"10.21236\/ADA477388"},{"issue":"1","key":"9357_CR2","first-page":"43","volume":"19","author":"JE Bartlett","year":"2001","unstructured":"Bartlett, J. E., Kotrlik, J. W., & Higgins, C. C. (2001). Organizational research: Determining appropriate sample size in survey research. Information Technology, Learning, and Performance Journal, 19(1), 43\u201350.","journal-title":"Information Technology, Learning, and Performance Journal"},{"key":"9357_CR3","doi-asserted-by":"crossref","unstructured":"Berto, A., Mizzaro, S., & Robertson, S. (2013). On using fewer topics in information retrieval evaluations. In Proceedings of the ICTIR, (p.\u00a09).","DOI":"10.1145\/2499178.2499184"},{"key":"9357_CR4","doi-asserted-by":"crossref","unstructured":"Bodoff, D., & Li, P. (2007). Test theory for assessing ir test collections. In Proceedings of the 30th annual international ACM SIGIR conference on research and development in information retrieval, (pp. 367\u2013374). New York: ACM.","DOI":"10.1145\/1277741.1277805"},{"key":"9357_CR5","doi-asserted-by":"crossref","unstructured":"Buckley, C., & Voorhees, E. (2000). Evaluating evaluation measure stability. In Proceedings of the 23rd SIGIR, (pp. 33\u201340).","DOI":"10.1145\/345508.345543"},{"key":"9357_CR6","doi-asserted-by":"crossref","unstructured":"Carterette, B., Allan, J., & Sitaraman, R. (2006). Minimal test collections for retrieval evaluation. In Proceedings of the 29th SIGIR, (pp 268\u2013275).","DOI":"10.1145\/1148170.1148219"},{"key":"9357_CR7","doi-asserted-by":"crossref","unstructured":"Carterette, B., Pavlu, V., Fang, H., & Kanoulas, E. (2009a). Million query track 2009 overview. In Proceedings of TREC.","DOI":"10.6028\/NIST.SP.500-278.million-query-overview"},{"key":"9357_CR8","doi-asserted-by":"crossref","unstructured":"Carterette, B., Pavlu, V., Kanoulas, E., Aslam, J. A., & Allan, J. (2009b). If i had a million queries. In Proceedings of the 31th ECIR, ECIR \u201909, (pp. 288\u2013300).","DOI":"10.1007\/978-3-642-00958-7_27"},{"key":"9357_CR9","doi-asserted-by":"publisher","unstructured":"Carterette, B., & Smucker, M. D. (2007). Hypothesis testing with incomplete relevance judgments. In Proceedings of the sixteenth ACM conference on conference on information and knowledge management, (pp 643\u2013652). New York: ACM. CIKM \u201907. https:\/\/doi.org\/10.1145\/1321440.1321530.","DOI":"10.1145\/1321440.1321530"},{"issue":"1","key":"9357_CR10","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1145\/2094072.2094076","volume":"30","author":"BA Carterette","year":"2012","unstructured":"Carterette, B. A. (2012). Multiple testing in statistical analysis of systems-based information retrieval experiments. ACM Transactions on Information Systems (TOIS), 30(1), 4.","journal-title":"ACM Transactions on Information Systems (TOIS)"},{"key":"9357_CR11","doi-asserted-by":"crossref","unstructured":"Cattelan, M., & Mizzaro, S. (2009). IR evaluation without a common set of topics. In Proceedings of the ICTIR, (pp. 342\u2013345).","DOI":"10.1007\/978-3-642-04417-5_35"},{"key":"9357_CR12","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1186\/1471-2288-2-8","volume":"2","author":"R Feise","year":"2002","unstructured":"Feise, R. (2002). Do multiple outcome measures require $$p$$-value adjustment? BMC Medical Research Methodology, 2, 8.","journal-title":"BMC Medical Research Methodology"},{"issue":"1\u201321","key":"9357_CR13","first-page":"26","volume":"21","author":"J Guiver","year":"2009","unstructured":"Guiver, J., Mizzaro, S., & Robertson, S. (2009). A few good topics: Experiments in topic set reduction for retrieval evaluation. ACM Transactions on Information Systems, 21(1\u201321), 26.","journal-title":"ACM Transactions on Information Systems"},{"key":"9357_CR14","doi-asserted-by":"crossref","unstructured":"Hauff, C., Hiemstra, D., Azzopardi, L., & de\u00a0Jong, F. (2010). A case for automatic system evaluation. In Proceedings of the ECIR, (pp. 153\u2013165).","DOI":"10.1007\/978-3-642-12275-0_16"},{"key":"9357_CR15","doi-asserted-by":"crossref","unstructured":"Hauff, C., Hiemstra, D., de\u00a0Jong, F., & Azzopardi, L. (2009). Relying on topic subsets for system ranking estimation. In Proceedings of the 18th CIKM, (pp. 1859\u20131862).","DOI":"10.1145\/1645953.1646249"},{"key":"9357_CR16","doi-asserted-by":"publisher","unstructured":"Hosseini, M., Cox, I. J., Milic-Frayling, N., Shokouhi, M., & Yilmaz, E. (2012). An uncertainty-aware query selection model for evaluation of ir systems. In Proceedings of the 35th international ACM SIGIR conference on research and development in information retrieval, (pp. 901\u2013910). New York, NY, USA: ACM. SIGIR \u201912. https:\/\/doi.org\/10.1145\/2348283.2348403","DOI":"10.1145\/2348283.2348403"},{"key":"9357_CR17","doi-asserted-by":"crossref","unstructured":"Hosseini, M., Cox, I. J., Milic-Frayling, N., Sweeting, T., & Vinay, V. (2011a). Prioritizing relevance judgments to improve the construction of IR test collections. In Proceedings of the 20th CIKM 2011, (pp. 641\u2013646)","DOI":"10.1145\/2063576.2063671"},{"key":"9357_CR18","doi-asserted-by":"crossref","unstructured":"Hosseini, M., Cox, I. J., Milic-Frayling, N., Vinay, V., & Sweeting, T. (2011b). Selecting a subset of queries for acquisition of further relevance judgements. In Proceedings of the ICTIR, (pp. 113\u2013124). lNCS 6931.","DOI":"10.1007\/978-3-642-23318-0_12"},{"issue":"1","key":"9357_CR19","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.ipm.2017.09.002","volume":"54","author":"M Kutlu","year":"2018","unstructured":"Kutlu, M., Elsayed, T., & Lease, M. (2018). Intelligent topic selection for low-cost information retrieval evaluation: A new perspective on deep vs. shallow judging. Information Processing and Management, 54(1), 37\u201359. https:\/\/doi.org\/10.1016\/j.ipm.2017.09.002.","journal-title":"Information Processing and Management"},{"key":"9357_CR20","doi-asserted-by":"publisher","unstructured":"Mehrotra, R., & Yilmaz, E. (2015). Representative & informative query selection for learning to rank using submodular functions. In Proceedings of the of the 38th international ACM SIGIR conference on research and development in information retrieval, (pp. 545\u2013554). New York, NY, USA: ACM, SIGIR \u201915. https:\/\/doi.org\/10.1145\/2766462.2767753","DOI":"10.1145\/2766462.2767753"},{"key":"9357_CR21","doi-asserted-by":"crossref","unstructured":"Mizzaro, S., & Robertson, S. (2007). HITS hits TREC\u2014Exploring IR evaluation results with network analysis. In Proceedings of the 30th SIGIR, (pp. 479\u2013486).","DOI":"10.1145\/1277741.1277824"},{"key":"9357_CR22","doi-asserted-by":"crossref","unstructured":"Moffat, A., Scholer, F., & Thomas, P. (2012). Models and metrics: IR evaluation as a user process. In Proceedings of the Australasian document computing symposium, Dunedin, New Zealand, (pp. 47\u201354).","DOI":"10.1145\/2407085.2407092"},{"key":"9357_CR23","unstructured":"Pavlu, V., & Aslam, J. (2007). A practical sampling strategy for efficient retrieval evaluation. Tech. rep., technical report, college of computer and information science, Northeastern University."},{"key":"9357_CR24","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9781139058452","volume-title":"Mining of massive datasets","author":"A Rajaraman","year":"2011","unstructured":"Rajaraman, A., & Ullman, J. D. (2011). Mining of massive datasets (1st ed.). Cambridge: Cambridge University Press.","edition":"1"},{"key":"9357_CR25","first-page":"129","volume-title":"Lecture Notes in Computer Science","author":"Stephen Robertson","year":"2011","unstructured":"Robertson, S. (2011). On the contributions of topics to system evaluation. In Proceedings of the ECIR, lNCS 6611, (pp. 129\u2013140)."},{"key":"9357_CR26","doi-asserted-by":"publisher","first-page":"605","DOI":"10.1007\/978-3-319-56608-5_55","volume-title":"Advances in information retrieval","author":"K Roitero","year":"2017","unstructured":"Roitero, K., Maddalena, E., & Mizzaro, S. (2017). Do easy topics predict effectiveness better than difficult topics? In J. M. Jose, C. Hauff, I. S. Alt\u0131ngovde, D. Song, D. Albakour, S. Watt, & J. Tait (Eds.), Advances in information retrieval (pp. 605\u2013611). Cham: Springer International Publishing."},{"issue":"3","key":"9357_CR27","doi-asserted-by":"publisher","first-page":"12:1","DOI":"10.1145\/3239573","volume":"10","author":"K Roitero","year":"2018","unstructured":"Roitero, K., Soprano, M., Brunello, A., & Mizzarom, S. (2018a). Reproduce and improve: An evolutionary approach to select a few good topics for information retrieval evaluation. ACM Journal of Data and Information Quality, 10(3), 12:1\u201312:21. https:\/\/doi.org\/10.1145\/3239573.","journal-title":"ACM Journal of Data and Information Quality"},{"key":"9357_CR28","doi-asserted-by":"publisher","unstructured":"Roitero, K., Soprano, M., & Mizzaro, S. (2018b). Effectiveness evaluation with a subset of topics: A practical approach. In The 41st international ACM SIGIR conference on research and development in information retrieval, (pp. 1145\u20131148). New York, NY, USA:ACM, SIGIR \u201918. https:\/\/doi.org\/10.1145\/3209978.3210108","DOI":"10.1145\/3209978.3210108"},{"key":"9357_CR29","doi-asserted-by":"crossref","unstructured":"Rose, D. E., & Levinson, D. (2004). Understanding user goals in web search. In Proceedings of the 13th international conference on World Wide Web, (pp. 13\u201319). New York, NY:ACM Press.","DOI":"10.1145\/988672.988675"},{"key":"9357_CR30","doi-asserted-by":"publisher","unstructured":"Sakai, T. (2007), Alternatives to bpref. In Proceedings of the 30th annual international ACM SIGIR Conference on research and development in information retrieval, (pp. 71\u201378). New York, NY:ACM, SIGIR \u201907. https:\/\/doi.org\/10.1145\/1277741.1277756","DOI":"10.1145\/1277741.1277756"},{"key":"9357_CR31","doi-asserted-by":"crossref","unstructured":"Sakai, T. (2014). Designing test collections for comparing many systems. In Proceedings of the 23rd CIKM 2014, (pp. 61\u201370).","DOI":"10.1145\/2661829.2661893"},{"key":"9357_CR32","doi-asserted-by":"crossref","unstructured":"Sakai, T. (2016a). Statistical significance, power, and sample sizes: A systematic review of SIGIR and TOIS, 2006-2015. In Proceedings of the 39th SIGIR, (pp. 5\u201314). ACM.","DOI":"10.1145\/2911451.2911492"},{"issue":"3","key":"9357_CR33","doi-asserted-by":"publisher","first-page":"256","DOI":"10.1007\/s10791-015-9273-z","volume":"19","author":"T Sakai","year":"2016","unstructured":"Sakai, T. (2016b). Topic set size design. Information Retrieval Journal, 19(3), 256\u2013283.","journal-title":"Information Retrieval Journal"},{"key":"9357_CR34","doi-asserted-by":"crossref","unstructured":"Sanderson, M., & Soboroff, I. (2007). Problems with Kendall\u2019s Tau. In Proceedings of the 30th SIGIR, (pp. 839\u2013840).","DOI":"10.1145\/1277741.1277935"},{"key":"9357_CR35","doi-asserted-by":"crossref","unstructured":"Sanderson, M., & Zobel, J. (2005). Information retrieval system evaluation: Effort, sensitivity, and reliability. In Proceedings of the 28th SIGIR, (pp. 162\u2013169).","DOI":"10.1145\/1076034.1076064"},{"key":"9357_CR36","volume-title":"Handbook of parametric and nonparametric statistical procedures","author":"D Sheskin","year":"2007","unstructured":"Sheskin, D. (2007). Handbook of parametric and nonparametric statistical procedures (4th ed.). Boca Raton: CRC Press.","edition":"4"},{"issue":"3","key":"9357_CR37","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1007\/s10791-015-9274-y","volume":"19","author":"J Urbano","year":"2016","unstructured":"Urbano, J. (2016). Test collection reliability: A study of bias and robustness to statistical assumptions via stochastic simulation. Information Retrieval Journal, 19(3), 313\u2013350. https:\/\/doi.org\/10.1007\/s10791-015-9274-y.","journal-title":"Information Retrieval Journal"},{"key":"9357_CR38","doi-asserted-by":"crossref","unstructured":"Urbano, J., Marrero, M., & Mart\u00edn, D. (2013). On the measurement of test collection reliability. In Proceedings of the 36th SIGIR, (pp. 393\u2013402).","DOI":"10.1145\/2484028.2484038"},{"key":"9357_CR39","doi-asserted-by":"publisher","unstructured":"Urbano, J., & Nagler, T. (2018). Stochastic simulation of test collections: Evaluation scores. In The 41st international ACM SIGIR conference on research & development in information retrieval, (pp. 695\u2013704). New York, NY, USA: ACM, SIGIR \u201918. https:\/\/doi.org\/10.1145\/3209978.3210043.","DOI":"10.1145\/3209978.3210043"},{"key":"9357_CR40","doi-asserted-by":"crossref","unstructured":"Voorhees, E., & Buckley, C. (2002). The effect of topic set size on retrieval experiment error. InProceedings of the 25th SIGIR, (pp. 316\u2013323).","DOI":"10.1145\/564376.564432"},{"key":"9357_CR41","doi-asserted-by":"crossref","unstructured":"Webber, W., Moffat, A., & Zobel, J. (2008). Statistical power in retrieval experimentation. In Proceedings of the 17th CIKM, (pp. 571\u2013580).","DOI":"10.1145\/1458082.1458158"}],"container-title":["Information Retrieval Journal"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-019-09357-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10791-019-09357-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-019-09357-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,17]],"date-time":"2024-07-17T19:38:30Z","timestamp":1721245110000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10791-019-09357-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5,8]]},"references-count":41,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2020,2]]}},"alternative-id":["9357"],"URL":"https:\/\/doi.org\/10.1007\/s10791-019-09357-w","relation":{},"ISSN":["1386-4564","1573-7659"],"issn-type":[{"value":"1386-4564","type":"print"},{"value":"1573-7659","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,5,8]]},"assertion":[{"value":"23 November 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 April 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 May 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}