{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T05:16:22Z","timestamp":1778908582424,"version":"3.51.4"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2020,3,6]],"date-time":"2020-03-06T00:00:00Z","timestamp":1583452800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,3,6]],"date-time":"2020-03-06T00:00:00Z","timestamp":1583452800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100006595","name":"UEFISCDI","doi-asserted-by":"crossref","award":["No. PN-III-P1-1.2-PCCDI-2017-0734"],"award-info":[{"award-number":["No. PN-III-P1-1.2-PCCDI-2017-0734"]}],"id":[{"id":"10.13039\/501100006595","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Inf Syst Front"],"published-print":{"date-parts":[[2021,2]]},"DOI":"10.1007\/s10796-020-09999-y","type":"journal-article","created":{"date-parts":[[2020,3,9]],"date-time":"2020-03-09T13:30:42Z","timestamp":1583760642000},"page":"81-100","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["TextBenDS: a Generic Textual Data Benchmark for Distributed Systems"],"prefix":"10.1007","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7292-4462","authenticated-orcid":false,"given":"Ciprian-Octavian","family":"Truic\u0103","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Elena-Simona","family":"Apostol","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J\u00e9r\u00f4me","family":"Darmont","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ira","family":"Assent","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,3,6]]},"reference":[{"key":"9999_CR1","doi-asserted-by":"publisher","unstructured":"Agrawal, D., Butt, A., Doshi, K., Larriba-Pey, J. L., Li, M., Reiss, F. R., Raab, F., Schiefer, B., Suzumura, T., & Xia, Y. (2016). Sparkbench \u2013 a spark performance testing suite. In Performance evaluation and benchmarking: Traditional to big data to internet of things (pp. 26\u201344). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-31409-9-3.","DOI":"10.1007\/978-3-319-31409-9-3"},{"key":"9999_CR2","doi-asserted-by":"publisher","unstructured":"Armbrust, M., Xin, R. S., Lian, C., Huai, Y., Liu, D., Bradley, J. K., Meng, X., Kaftan, T., Franklin, M. J., Ghodsi, A., & Zaharia, M. (2015). Spark sql: Relational data processing in spark. In ACM SIGMOD International Conference on Management of Data (pp. 1383\u20131394). ACM Press. https:\/\/doi.org\/10.1145\/2723372.2742797.","DOI":"10.1145\/2723372.2742797"},{"key":"9999_CR3","doi-asserted-by":"publisher","unstructured":"Armstrong, T. G., Ponnekanti, V., Borthakur, D., & Callaghan, M. (2013). Linkbench: A database benchmark based on the facebook social graph. In ACM SIGMOD International Conference on Management of Data, SIGMOD \u201813 (pp. 1185\u20131196). ACM. https:\/\/doi.org\/10.1145\/2463676.2465296.","DOI":"10.1145\/2463676.2465296"},{"issue":"2","key":"9999_CR4","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1145\/2568388.2568393","volume":"47","author":"P Bellot","year":"2013","unstructured":"Bellot, P., Doucet, A., Geva, S., Gurajada, S., Kamps, J., Kazai, G., Koolen, M., Mishra, A., Moriceau, V., Mothe, J., Preminger, M., SanJuan, E., Schenkel, R., Tannier, X., Theobald, M., Trappett, M., Trotman, A., Sanderson, M., Scholer, F., & Wang, Q. (2013). Report on inex 2013. SIGIR Forum, 47(2), 21\u201332. https:\/\/doi.org\/10.1145\/2568388.2568393.","journal-title":"SIGIR Forum"},{"key":"9999_CR5","doi-asserted-by":"publisher","unstructured":"Bifet, A., & Frank, E. (2010). Sentiment knowledge discovery in twitter streaming data. In Discovery Science (pp. 1\u201315). Springer Berlin Heidelberg. https:\/\/doi.org\/10.1007\/978-3-642-16184-1-1.","DOI":"10.1007\/978-3-642-16184-1-1"},{"issue":"1","key":"9999_CR6","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1504\/ijbidm.2016.076425","volume":"11","author":"M Bouakkaz","year":"2016","unstructured":"Bouakkaz, M., Loudcher, S., & Ouinten, Y. (2016). OLAP textual aggregation approach using the google similarity distance. International Journal of Business Intelligence and Data Mining, 11(1), 31. https:\/\/doi.org\/10.1504\/ijbidm.2016.076425.","journal-title":"International Journal of Business Intelligence and Data Mining"},{"key":"9999_CR7","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1007\/978-3-642-23091-2_15","volume-title":"International Conference on Database and Expert Systems Applications","author":"S Bringay","year":"2011","unstructured":"Bringay, S., B\u00e9chet, N., Bouillot, F., Poncelet, P., Roche, M., & Teisseire, M. (2011). Towards an on-line analysis of tweets processing. In International Conference on Database and Expert Systems Applications (pp. 154\u2013161). https:\/\/doi.org\/10.1007\/978-3-642-23091-2_15."},{"key":"9999_CR8","doi-asserted-by":"publisher","unstructured":"Chowdhury, B., Rabl, T., Saadatpanah, P., Du, J., & Jacobsen, H. A. (2014). A bigbench implementation in the hadoop ecosystem. In Advancing big data benchmarks (pp. 3\u201318). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-10596-3-1.","DOI":"10.1007\/978-3-319-10596-3-1"},{"key":"9999_CR9","doi-asserted-by":"publisher","unstructured":"Crane, M., Culpepper, J. S., Lin, J., Mackenzie, J., & Trotman, A. (2017). A comparison of document-at-a-time and score-at-a-time query evaluation. In ACM International Conference on Web Search and Data Mining (pp. 201\u2013210). ACM. https:\/\/doi.org\/10.1145\/3018661.3018726.","DOI":"10.1145\/3018661.3018726"},{"issue":"1","key":"9999_CR10","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/1327452.1327492","volume":"51","author":"J Dean","year":"2008","unstructured":"Dean, J., & Ghemawat, S. (2008). Mapreduce: Simplified data processing on large clusters. Communications of the ACM, 51(1), 107\u2013113. https:\/\/doi.org\/10.1145\/1327452.1327492.","journal-title":"Communications of the ACM"},{"issue":"6","key":"9999_CR11","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1002\/(SICI)1097-4571(199009)41:6<391::AID-ASI1>3.0.CO;2-9","volume":"41","author":"S Deerwester","year":"1990","unstructured":"Deerwester, S., Dumais, S. T., Furnas, G. W., Landauer, T. K., & Harshman, R. (1990). Indexing by latent semantic analysis. Journal of the American Society for Information Science, 41(6), 391\u2013407. https:\/\/doi.org\/10.1002\/(SICI)1097-4571(199009)41:6<391::AID-ASI1>3.0.CO;2-9.","journal-title":"Journal of the American Society for Information Science"},{"key":"9999_CR12","doi-asserted-by":"publisher","unstructured":"Ferrarons, J., Adhana, M., Colmenares, C., Pietrowska, S., Bentayeb, F., Darmont, J. (2014). Primeball: a parallel processing framework benchmark for big data applications in the cloud. In: TPC Technology Conference on Performance Evaluation and Benchmarking, LNCS1, 839, pp. 109\u2013124. https:\/\/doi.org\/10.1007\/978-3-319-04936-6_8","DOI":"10.1007\/978-3-319-04936-6_8"},{"issue":"3\/4","key":"9999_CR13","doi-asserted-by":"publisher","first-page":"10:1","DOI":"10.1147\/JRD.2013.2240732","volume":"57","author":"AE Gattiker","year":"2013","unstructured":"Gattiker, A. E., Gebara, F. H., Hofstee, H. P., Hayes, J. D., & Hylick, A. (2013). Big data text-oriented benchmark creation for Hadoop. IBM Journal of Research and Development, 57(3\/4), 10:1\u201310:6. https:\/\/doi.org\/10.1147\/JRD.2013.2240732.","journal-title":"IBM Journal of Research and Development"},{"key":"9999_CR14","doi-asserted-by":"publisher","first-page":"1197","DOI":"10.1145\/2463676.2463712","volume-title":"ACM SIGMOD International Conference on Management of Data, SIGMOD \u201813","author":"A Ghazal","year":"2013","unstructured":"Ghazal, A., Rabl, T., Hu, M., Raab, F., Poess, M., Crolotte, A., & Jacobsen, H. A. (2013). Bigbench: Towards an industry standard benchmark for big data analytics. In ACM SIGMOD International Conference on Management of Data, SIGMOD \u201813 (pp. 1197\u20131208). https:\/\/doi.org\/10.1145\/2463676.2463712."},{"key":"9999_CR15","doi-asserted-by":"publisher","first-page":"1225","DOI":"10.1109\/ICDE.2017.167","volume-title":"2017 IEEE 33rd International Conference on Data Engineering","author":"A Ghazal","year":"2017","unstructured":"Ghazal, A., Ivanov, T., Kostamaa, P., Crolotte, A., Voong, R., Al-Kateb, M., Ghazal, W., & Zicari, R. V. (2017). Bigbench v2: The new and improved bigbench. In 2017 IEEE 33rd International Conference on Data Engineering (pp. 1225\u20131236). https:\/\/doi.org\/10.1109\/ICDE.2017.167."},{"key":"9999_CR16","unstructured":"Gray, J. (1993). The benchmark handbook for database and transaction systems (2nd ed.). Burlington: Morgan Kaufmann Publishers."},{"issue":"1","key":"9999_CR17","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1007\/s13278-015-0258-0","volume":"5","author":"A Guille","year":"2015","unstructured":"Guille, A., & Favre, C. (2015). Event detection, tracking, and visualization in twitter: a mention-anomaly-based approach. Social Network Analysis and Mining, 5(1), 18. https:\/\/doi.org\/10.1007\/s13278-015-0258-0.","journal-title":"Social Network Analysis and Mining"},{"issue":"2","key":"9999_CR18","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1145\/3130348.3130370","volume":"51","author":"T Hofmann","year":"2017","unstructured":"Hofmann, T. (2017). Probabilistic latent semantic indexing. SIGIR Forum, 51(2), 211\u2013218. https:\/\/doi.org\/10.1145\/3130348.3130370.","journal-title":"SIGIR Forum"},{"key":"9999_CR19","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1109\/ICDEW.2010.5452747","volume-title":"International Conference on Data Engineering","author":"S Huang","year":"2010","unstructured":"Huang, S., Huang, J., Dai, J., Xie, T., & Huang, B. (2010). The HiBench benchmark suite: Characterization of the MapReduce-based data analysis. In International Conference on Data Engineering (pp. 41\u201351). https:\/\/doi.org\/10.1109\/ICDEW.2010.5452747."},{"key":"9999_CR20","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1109\/IISWC.2014.6983058","volume-title":"2014 IEEE International Symposium on Workload Characterization","author":"Z Jia","year":"2014","unstructured":"Jia, Z., Zhan, J., Wang, L., Han, R., McKee, S. A., Yang, Q., Luo, C., & Li, J. (2014). Characterizing and subsetting big data workloads. In 2014 IEEE International Symposium on Workload Characterization (pp. 191\u2013201). https:\/\/doi.org\/10.1109\/IISWC.2014.6983058."},{"issue":"2","key":"9999_CR21","doi-asserted-by":"publisher","first-page":"174","DOI":"10.1177\/0165551515620551","volume":"43","author":"D K\u0131l\u0131\u00e7","year":"2017","unstructured":"K\u0131l\u0131\u00e7, D., \u00d6z\u00e7ift, A., Bozyigit, F., Yildirim, P., Y\u00fccalar, F., & Borandag, E. (2017). Ttc-3600: A new benchmark dataset for turkish text categorization. Journal of Information Science, 43(2), 174\u2013185. https:\/\/doi.org\/10.1177\/0165551515620551.","journal-title":"Journal of Information Science"},{"key":"9999_CR22","doi-asserted-by":"crossref","unstructured":"Krasnashchok, K., Jouili, S. (2018). Improving topic quality by promoting named entities in topic modeling. In: Annual Meeting of the Association for Computational Linguistics, pp. 247\u2013253.","DOI":"10.18653\/v1\/P18-2040"},{"issue":"2","key":"9999_CR23","doi-asserted-by":"publisher","first-page":"260","DOI":"10.1145\/3130348.3130376","volume":"51","author":"V Lavrenko","year":"2017","unstructured":"Lavrenko, V., & Croft, W. B. (2017). Relevance-based language models. SIGIR Forum, 51(2), 260\u2013267. https:\/\/doi.org\/10.1145\/3130348.3130376.","journal-title":"SIGIR Forum"},{"key":"9999_CR24","first-page":"361","volume":"5","author":"DD Lewis","year":"2004","unstructured":"Lewis, D. D., Yang, Y., Rose, T. G., & Li, F. (2004). Rcv1: A new benchmark collection for text categorization research. Journal of Machine Learning Research, 5, 361\u2013397 URL http:\/\/www.jmlr.org\/papers\/v5\/lewis04a.html.","journal-title":"Journal of Machine Learning Research"},{"key":"9999_CR25","doi-asserted-by":"publisher","unstructured":"Li, M., Tan, J., Wang, Y., Zhang, L., & Salapura, V. (2015). Sparkbench: A comprehensive benchmarking suite for in memory data analytic platform spark. In ACM International Conference on Computing Frontiers, CF \u201815 (pp. 53:1\u201353:8). ACM. https:\/\/doi.org\/10.1145\/2742854.2747283.","DOI":"10.1145\/2742854.2747283"},{"key":"9999_CR26","doi-asserted-by":"publisher","unstructured":"Lin, J., Crane, M., Trotman, A., Callan, J., Chattopadhyaya, I., Foley, J., Ingersoll, G., Macdonald, C., & Vigna, S. (2016). Toward reproducible baselines: The open-source ir reproducibility challenge. In Advances in information retrieval (pp. 408\u2013420). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-30671-1-30.","DOI":"10.1007\/978-3-319-30671-1-30"},{"key":"9999_CR27","doi-asserted-by":"crossref","unstructured":"Manning, C. D., Raghavan, P., & Sch\u00fctze, H. (2008). Introduction to information retrieval.\u00a0Cambridge: Cambridge University Press.","DOI":"10.1017\/CBO9780511809071"},{"key":"9999_CR28","doi-asserted-by":"publisher","unstructured":"Ming, Z., Luo, C., Gao, W., Han, R., Yang, Q., Wang, L., & Zhan, J. (2014). Bdgs: A scalable big data generator suite in big data benchmarking. In Advancing big data benchmarks (pp. 138\u2013154). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-10596-3-11.","DOI":"10.1007\/978-3-319-10596-3-11"},{"issue":"2","key":"9999_CR29","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1504\/IJIIDS.2010.032437","volume":"4","author":"J O\u2019Shea","year":"2010","unstructured":"O\u2019Shea, J., Bandar, Z., Crockett, K. A., & McLean, D. (2010). Benchmarking short text semantic similarity. International Journal of Intelligent Information and Database Systems, 4(2), 103\u2013120. https:\/\/doi.org\/10.1504\/IJIIDS.2010.032437.","journal-title":"International Journal of Intelligent Information and Database Systems"},{"key":"9999_CR30","unstructured":"Paltoglou, G., Thelwall, M. (2010). A study of information retrieval weighting schemes for sentiment analysis. In: Annual Meeting of the Association for Computational Linguistics, pp. 1386\u20131395. URL http:\/\/dl.acm.org\/citation.cfm?id=1858681.1858822."},{"key":"9999_CR31","unstructured":"Partalas, I., Kosmopoulos, A., Baskiotis, N., Arti\u00e8res, T., Paliouras, G., Gaussier, \u00c9., Androutsopoulos, I., Amini, M.R., Gallinari, P. (2015). Lshtc: A benchmark for large-scale text classification. CoRR. URL http:\/\/arxiv.org\/abs\/1503.08581."},{"key":"9999_CR32","doi-asserted-by":"publisher","first-page":"507","DOI":"10.1109\/BigData.2015.7363793","volume-title":"IEEE International Conference on Big Data","author":"P Pirzadeh","year":"2015","unstructured":"Pirzadeh, P., Carey, M. J., & Westmann, T. (2015). Bigfun: A performance study of big data management system functionality. In IEEE International Conference on Big Data (pp. 507\u2013514). https:\/\/doi.org\/10.1109\/BigData.2015.7363793."},{"key":"9999_CR33","doi-asserted-by":"publisher","unstructured":"Raiber, F., & Kurland, O. (2017). Kullback-leibler divergence revisited. In ACM SIGIR International Conference on Theory of Information Retrieval, ICTIR \u201817 (pp. 117\u2013124). ACM. https:\/\/doi.org\/10.1145\/3121050.3121062.","DOI":"10.1145\/3121050.3121062"},{"key":"9999_CR34","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1007\/978-3-540-85836-2-6","volume-title":"International Conference on Data Warehousing and Knowledge Discovery","author":"F Ravat","year":"2008","unstructured":"Ravat, F., Teste, O., Tournier, R., & Zurfluh, G. (2008). Top\u2212keyword: an aggregation function for textual document olap. In International Conference on Data Warehousing and Knowledge Discovery (pp. 55\u201364). https:\/\/doi.org\/10.1007\/978-3-540-85836-2-6."},{"key":"9999_CR35","doi-asserted-by":"publisher","first-page":"1357","DOI":"10.1145\/2723372.2742790","volume-title":"ACM SIGMOD International Conference on Management of Data","author":"B Saha","year":"2015","unstructured":"Saha, B., Shah, H., Seth, S., Vijayaraghavan, G., Murthy, A., & Curino, C. (2015). Apache tez: A unifying framework for modeling and building data processing applications. In ACM SIGMOD International Conference on Management of Data (pp. 1357\u20131369). New York: ACM. https:\/\/doi.org\/10.1145\/2723372.2742790."},{"key":"9999_CR36","doi-asserted-by":"publisher","unstructured":"Sangroya, A., Serrano, D., & Bouchenak, S. (2013). Mrbs: Towards dependability benchmarking for hadoop mapreduce. In Euro-Par 2012: Parallel Processing Workshops (pp. 3\u201312). Springer Berlin Heidelberg. https:\/\/doi.org\/10.1007\/978-3-642-36949-0-2.","DOI":"10.1007\/978-3-642-36949-0-2"},{"issue":"1","key":"9999_CR37","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1145\/3137597.3137600","volume":"19","author":"K Shu","year":"2017","unstructured":"Shu, K., Sliva, A., Wang, S., Tang, J., & Liu, H. (2017). Fake news detection on social media: A data mining perspective. ACM SIGKDD Explorations Newsletter, 19(1), 22\u201336. https:\/\/doi.org\/10.1145\/3137597.3137600.","journal-title":"ACM SIGKDD Explorations Newsletter"},{"key":"9999_CR38","unstructured":"Shu, K., Mahudeswaran, D., Wang, S., Lee, D., Liu, H. (2018). Fakenewsnet: A data repository with news content, social context and dynamic information for studying fake news on social media. arXiv preprint arXiv:1809.01286."},{"key":"9999_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/MSST.2010.5496972","volume-title":"Symposium on Mass Storage Systems and Technologies","author":"K Shvachko","year":"2010","unstructured":"Shvachko, K., Kuang, H., Radia, S., & Chansler, R. (2010). The hadoop distributed file system. In Symposium on Mass Storage Systems and Technologies (pp. 1\u201310). https:\/\/doi.org\/10.1109\/MSST.2010.5496972."},{"issue":"6","key":"9999_CR40","doi-asserted-by":"publisher","first-page":"779","DOI":"10.1016\/S0306-4573(00)00015-7","volume":"36","author":"K Sp\u00e4rck Jones","year":"2000","unstructured":"Sp\u00e4rck Jones, K., Walker, S., & Robertson, S. E. (2000a). A probabilistic model of information retrieval: development and comparative experiments: Part 1. Information Processing & Management, 36(6), 779\u2013808. https:\/\/doi.org\/10.1016\/S0306-4573(00)00015-7.","journal-title":"Information Processing & Management"},{"issue":"6","key":"9999_CR41","doi-asserted-by":"publisher","first-page":"809","DOI":"10.1016\/S0306-4573(00)00016-9","volume":"36","author":"K Sp\u00e4rck Jones","year":"2000","unstructured":"Sp\u00e4rck Jones, K., Walker, S., & Robertson, S. E. (2000b). A probabilistic model of information retrieval: development and comparative experiments: Part 2. Information Processing & Management, 36(6), 809\u2013840. https:\/\/doi.org\/10.1016\/S0306-4573(00)00016-9.","journal-title":"Information Processing & Management"},{"issue":"2","key":"9999_CR42","doi-asserted-by":"publisher","first-page":"1626","DOI":"10.14778\/1687553.1687609","volume":"2","author":"A Thusoo","year":"2009","unstructured":"Thusoo, A., Sarma, J. S., Jain, N., Shao, Z., Chakka, P., Anthony, S., Liu, H., Wyckoff, P., & Murthy, R. (2009). Hive: A warehousing solution over a map-reduce framework. VLDB Endowment, 2(2), 1626\u20131629. https:\/\/doi.org\/10.14778\/1687553.1687609.","journal-title":"VLDB Endowment"},{"key":"9999_CR43","unstructured":"Transaction Processing Performance Council (TPC) (2016). TPC express benchmark hs standard specification version 1.4.2.http:\/\/www.tpc.org Accessed March 2019."},{"key":"9999_CR44","unstructured":"Transaction Processing Performance Council (TPC) (2019). TPC-DS decision support benchmark 2.10.1.http:\/\/www.tpc.org Accessed March 2019."},{"key":"9999_CR45","doi-asserted-by":"publisher","unstructured":"Truic\u0103, C. O., & Darmont, J. (2017). T2K2: The twitter top-k keywords benchmark. In European Conference on Advances in Databases and Information Systems (pp. 21\u201328). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-67162-8_3.","DOI":"10.1007\/978-3-319-67162-8_3"},{"key":"9999_CR46","doi-asserted-by":"publisher","unstructured":"Truic\u0103, C. O., Darmont, J., & Velcine, J. (2016a). A scalable document-based architecture for text analysis. In International Conference on Advanced Data Mining and Applications (pp. 481\u2013494). Springer. https:\/\/doi.org\/10.1007\/978-3-319-49586-6-33.","DOI":"10.1007\/978-3-319-49586-6-33"},{"key":"9999_CR47","doi-asserted-by":"publisher","unstructured":"Truic\u0103, C.O., R\u0103dulescu, F., Boicea, A. (2016b). Comparing different term weighting schemas for topic modeling. In: International Symposium on Symbolic and Numeric Algorithms for Scientific Computing. IEEE. https:\/\/doi.org\/10.1109\/synasc.2016.055.","DOI":"10.1109\/synasc.2016.055"},{"key":"9999_CR48","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1016\/j.future.2018.02.037","volume":"85","author":"CO Truic\u0103","year":"2018","unstructured":"Truic\u0103, C. O., Darmont, J., Boicea, A., & R\u0103dulescu, F. (2018). Benchmarking top-k keyword and top-k document processing with T2K2 and T2K2D2. Future Generation Computer Systems, 85, 60\u201375. https:\/\/doi.org\/10.1016\/j.future.2018.02.037.","journal-title":"Future Generation Computer Systems"},{"key":"9999_CR49","doi-asserted-by":"publisher","first-page":"5:1","DOI":"10.1145\/2523616.2523633","volume-title":"Annual Symposium on Cloud Computing","author":"VK Vavilapalli","year":"2013","unstructured":"Vavilapalli, V. K., Murthy, A. C., Douglas, C., Agarwal, S., Konar, M., Evans, R., Graves, T., Lowe, J., Shah, H., Seth, S., Saha, B., Curino, C., O\u2019Malley, O., Radia, S., Reed, B., & Baldeschwieler, E. (2013). Apache hadoop yarn: Yet another resource negotiator. In Annual Symposium on Cloud Computing (pp. 5:1\u20135:16). https:\/\/doi.org\/10.1145\/2523616.2523633."},{"key":"9999_CR50","doi-asserted-by":"publisher","first-page":"488","DOI":"10.1109\/HPCA.2014.6835958","volume-title":"IEEE International Symposium on High Performance Computer Architecture","author":"L Wang","year":"2014","unstructured":"Wang, L., Zhan, J., Luo, C., Zhu, Y., Yang, Q., He, Y., Gao, W., Jia, Z., Shi, Y., Zhang, S., Zheng, C., Lu, G., Zhan, K., Li, X., & Qiu, B. (2014). BigDataBench: A big data benchmark suite from internet services. In IEEE International Symposium on High Performance Computer Architecture (pp. 488\u2013499). https:\/\/doi.org\/10.1109\/HPCA.2014.6835958."},{"issue":"10","key":"9999_CR51","doi-asserted-by":"publisher","first-page":"982","DOI":"10.1631\/FITEE.1500332","volume":"17","author":"L Wang","year":"2016","unstructured":"Wang, L., Dong, X., Zhang, X., Wang, Y., Ju, T., & Feng, G. (2016). Textgen: a realistic text data content generation method for modern storage system benchmarks. Frontiers of Information Technology & Electronic Engineering, 17(10), 982\u2013993. https:\/\/doi.org\/10.1631\/FITEE.1500332.","journal-title":"Frontiers of Information Technology & Electronic Engineering"},{"key":"9999_CR52","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/FUZZ-IEEE.2017.8015720","volume-title":"2017 IEEE International Conference on Fuzzy Systems","author":"X Wang","year":"2017","unstructured":"Wang, X., Ah-Pine, J., & Darmont, J. (2017). Shcoclust, a scalable similarity-based hierarchical co-clustering method and its application to textual collections. In 2017 IEEE International Conference on Fuzzy Systems (pp. 1\u20136). https:\/\/doi.org\/10.1109\/FUZZ-IEEE.2017.8015720."},{"key":"9999_CR53","doi-asserted-by":"publisher","unstructured":"Yin, J., Chao, D., Liu, Z., Zhang, W., Yu, X., & Wang, J. (2018). Model-based clustering of short text streams. In ACM SIGKDD International Conference on Knowledge Discovery & Data Mining (pp. 2634\u20132642). ACM Press. https:\/\/doi.org\/10.1145\/3219819.3220094.","DOI":"10.1145\/3219819.3220094"},{"issue":"11","key":"9999_CR54","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1145\/2934664","volume":"59","author":"M Zaharia","year":"2016","unstructured":"Zaharia, M., Xin, R. S., Wendell, P., Das, T., Armbrust, M., Dave, A., Meng, X., Rosen, J., Venkataraman, S., Franklin, M. J., Ghodsi, A., Gonzalez, J., Shenker, S., & Stoica, I. (2016). Apache spark: A unified engine for big data processing. Communications of the ACM, 59(11), 56\u201365. https:\/\/doi.org\/10.1145\/2934664.","journal-title":"Communications of the ACM"},{"key":"9999_CR55","doi-asserted-by":"publisher","unstructured":"Zhang, D., Zhai, C., Han, J. (2009). Topic cube: Topic modeling for OLAP on multidimensional text databases. In: SIAM International Conference on Data Mining, pp. 1124\u20131135. Society for Industrial and Applied Mathematics. https:\/\/doi.org\/10.1137\/1.9781611972795.96","DOI":"10.1137\/1.9781611972795.96"},{"issue":"3","key":"9999_CR56","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1002\/sam.11159","volume":"6","author":"D Zhang","year":"2012","unstructured":"Zhang, D., Zhai, C., & Han, J. (2012). MiTexCube: MicroTextCluster cube for online analysis of text cells and its applications. Statistical Analysis and Data Mining, 6(3), 243\u2013259. https:\/\/doi.org\/10.1002\/sam.11159.","journal-title":"Statistical Analysis and Data Mining"}],"container-title":["Information Systems Frontiers"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10796-020-09999-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10796-020-09999-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10796-020-09999-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,3,6]],"date-time":"2021-03-06T00:12:18Z","timestamp":1614989538000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10796-020-09999-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,3,6]]},"references-count":56,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2021,2]]}},"alternative-id":["9999"],"URL":"https:\/\/doi.org\/10.1007\/s10796-020-09999-y","relation":{},"ISSN":["1387-3326","1572-9419"],"issn-type":[{"value":"1387-3326","type":"print"},{"value":"1572-9419","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,3,6]]},"assertion":[{"value":"6 March 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}