{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T11:15:37Z","timestamp":1762341337589},"publisher-location":"Cham","reference-count":32,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030352875"},{"type":"electronic","value":"9783030352882"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-35288-2_21","type":"book-chapter","created":{"date-parts":[[2019,11,25]],"date-time":"2019-11-25T00:02:57Z","timestamp":1574640177000},"page":"253-264","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Evaluating the Boundaries of Big Data Environments for Machine Learning"],"prefix":"10.1007","author":[{"given":"Fathima Nuzla","family":"Ismail","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Brendon J.","family":"Woodford","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sherlock A.","family":"Licorish","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,11,25]]},"reference":[{"issue":"12","key":"21_CR1","doi-asserted-by":"publisher","first-page":"1840","DOI":"10.14778\/2824032.2824080","volume":"8","author":"M Armbrust","year":"2015","unstructured":"Armbrust, M., et al.: Scaling spark in the real world: performance and usability. Proc. VLDB Endow. 8(12), 1840\u20131843 (2015)","journal-title":"Proc. VLDB Endow."},{"key":"21_CR2","doi-asserted-by":"publisher","first-page":"1001","DOI":"10.1016\/j.procs.2016.03.127","volume":"79","author":"S Bende","year":"2016","unstructured":"Bende, S., Shedge, R.: Dealing with small files problem in hadoop distributed file system. Procedia Comput. Sci. 79, 1001\u20131012 (2016). Proceedings International Conference on Communication, Computing and Virtualization (ICCCV) 2016","journal-title":"Procedia Comput. Sci."},{"issue":"2","key":"21_CR3","doi-asserted-by":"publisher","first-page":"161","DOI":"10.14778\/2735471.2735477","volume":"8","author":"Y Bu","year":"2014","unstructured":"Bu, Y., Borkar, V., Jia, J., Carey, M.J., Condie, T.: Pregelix: Big(Ger) graph analytics on a dataflow engine. Proc. VLDB Endow. 8(2), 161\u2013172 (2014)","journal-title":"Proc. VLDB Endow."},{"key":"21_CR4","doi-asserted-by":"crossref","unstructured":"Cai, Z., Gao, Z.J., Luo, S., Perez, L.L., Vagena, Z., Jermaine, C.: A comparison of platforms for implementing and running very large scale machine learning algorithms. In: Proceedings 2014 ACM SIGMOD International Conference on Management of Data, SIGMOD\u201914, pp. 1371\u20131382. ACM, New York (2014)","DOI":"10.1145\/2588555.2593680"},{"key":"21_CR5","doi-asserted-by":"crossref","unstructured":"Chang, B.R., Tsai, H., Wang, Y., Huang, C.: Resilient distributed computing platforms for big data analysis using Spark and Hadoop. In: Proceedings 2016 International Conference on Applied System Innovation (ICASI), pp. 1\u20134, May 2016","DOI":"10.1109\/ICASI.2016.7539859"},{"issue":"2","key":"21_CR6","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1007\/s11036-013-0489-0","volume":"19","author":"M Chen","year":"2014","unstructured":"Chen, M., Mao, S., Liu, Y.: Big data: a survey. Mob. Netw. Appl. 19(2), 171\u2013209 (2014)","journal-title":"Mob. Netw. Appl."},{"key":"21_CR7","volume-title":"MongoDB - The Definitive Guide: Powerful and Scalable Data Storage","author":"K Chodorow","year":"2010","unstructured":"Chodorow, K., Dirolf, M.: MongoDB - The Definitive Guide: Powerful and Scalable Data Storage. O\u2019Reilly, Sebastopol (2010)"},{"key":"21_CR8","doi-asserted-by":"crossref","unstructured":"Dias, J., Ogasawara, E., de Oliveira, D., Porto, F., Valduriez, P., Mattoso, M.: Algebraic dataflows for big data analysis. In: Proceedings 2013 IEEE International Conference on Big Data, pp. 150\u2013155. IEEE Press, October 2013","DOI":"10.1109\/BigData.2013.6691567"},{"issue":"12","key":"21_CR9","doi-asserted-by":"publisher","first-page":"1295","DOI":"10.14778\/2732977.2733002","volume":"7","author":"A Floratou","year":"2014","unstructured":"Floratou, A., Minhas, U.F., \u00d6zcan, F.: SQL-on-Hadoop: full circle back to shared-nothing database architectures. Proc. VLDB Endow. 7(12), 1295\u20131306 (2014)","journal-title":"Proc. VLDB Endow."},{"key":"21_CR10","volume-title":"Mastering Apache Spark","author":"M Frampton","year":"2015","unstructured":"Frampton, M.: Mastering Apache Spark, 1st edn. Packt Publishing, Birmingham (2015)","edition":"1"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Fuad, A., Erwin, A., Ipung, H.P.: Processing performance on apache pig, apache hive and MySQL cluster. In: Proceedings International Conference on Information, Communication Technology and System (ICTS) 2014, pp. 297\u2013302. IEEE Press, September 2014","DOI":"10.1109\/ICTS.2014.7010600"},{"key":"21_CR12","volume-title":"Hbase: The Definitive Guide","author":"L George","year":"2011","unstructured":"George, L.: Hbase: The Definitive Guide, 1st edn. O\u2019Reilly Media Inc., Sebastopol (2011)","edition":"1"},{"issue":"1","key":"21_CR13","first-page":"8","volume":"113","author":"S Gopalani","year":"2015","unstructured":"Gopalani, S., Arora, R.: Article: comparing apache spark and map reduce with performance analysis using K-Means. Int. J. Comput. Appl. 113(1), 8\u201311 (2015)","journal-title":"Int. J. Comput. Appl."},{"key":"21_CR14","volume-title":"Mastering Apache Storm","author":"A Jain","year":"2017","unstructured":"Jain, A.: Mastering Apache Storm, 1st edn. Packt Publishing, Birmingham (2017)","edition":"1"},{"key":"21_CR15","unstructured":"Kalavri, V.: Performance optimization techniques and tools for data-intensive computation platforms: an overview of performance limitations in big data systems and proposed optimizations (2014). qC 20140605"},{"key":"21_CR16","first-page":"1","volume":"2014","author":"N Khan","year":"2014","unstructured":"Khan, N., et al.: Big data: survey, technologies, opportunities, and challenges. Sci. World J. 2014, 1\u201318 (2014)","journal-title":"Sci. World J."},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"Kupisz, B., Unold, O.: Collaborative filtering recommendation algorithm based on Hadoop and Spark. In: Proceedings 2015 IEEE International Conference on Industrial Technology (ICIT), pp. 1510\u20131514. IEEE Press, March 2015","DOI":"10.1109\/ICIT.2015.7125310"},{"issue":"1","key":"21_CR18","doi-asserted-by":"publisher","first-page":"24","DOI":"10.1186\/s40537-015-0032-1","volume":"2","author":"S Landset","year":"2015","unstructured":"Landset, S., Khoshgoftaar, T.M., Richter, A.N., Hasanin, T.: A survey of open source tools for machine learning with big data in the Hadoop ecosystem. J. Big Data 2(1), 24 (2015)","journal-title":"J. Big Data"},{"key":"21_CR19","doi-asserted-by":"publisher","first-page":"7776","DOI":"10.1109\/ACCESS.2017.2696365","volume":"5","author":"A L\u2019Heureux","year":"2017","unstructured":"L\u2019Heureux, A., Grolinger, K., Elyamany, H.F., Capretz, M.A.M.: Machine learning with big data: challenges and approaches. IEEE Access 5, 7776\u20137797 (2017)","journal-title":"IEEE Access"},{"key":"21_CR20","unstructured":"Ousterhout, K., Rasti, R., Ratnasamy, S., Shenker, S., Chun, B.G.: Making sense of performance in data analytics frameworks. In: Proceedings 12th USENIX Conference on Networked Systems Design and Implementation, NSDI\u201915, pp. 293\u2013307. USENIX Association, Berkeley, CA, USA (2015)"},{"key":"21_CR21","volume-title":"Getting Started with Impala","author":"J Russell","year":"2014","unstructured":"Russell, J.: Getting Started with Impala, 1st edn. O\u2019Reilly Media Inc., Sebastopol (2014)","edition":"1"},{"issue":"3","key":"21_CR22","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1007\/s41060-016-0027-9","volume":"1","author":"S Salloum","year":"2016","unstructured":"Salloum, S., Dautov, R., Chen, X., Peng, P.X., Huang, J.Z.: Big data analytics on Apache Spark. Int. J. Data Sci. Anal. 1(3), 145\u2013164 (2016)","journal-title":"Int. J. Data Sci. Anal."},{"key":"21_CR23","unstructured":"Shahrivari, S.: Beyond batch processing: towards real-time and streaming big data. CoRR abs\/1403.3375 (2014). \nhttp:\/\/arxiv.org\/abs\/1403.3375"},{"issue":"13","key":"21_CR24","doi-asserted-by":"publisher","first-page":"2110","DOI":"10.14778\/2831360.2831365","volume":"8","author":"J Shi","year":"2015","unstructured":"Shi, J., et al.: Clash of the titans: MapReduce vs. spark for large scale data analytics. Proc. VLDB Endow. 8(13), 2110\u20132121 (2015)","journal-title":"Proc. VLDB Endow."},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Shvachko, K., Kuang, H., Radia, S., Chansler, R.: The Hadoop distributed file system. In: Proceedings 2010 IEEE 26th Symposium on Mass Storage Systems and Technologies (MSST), MSST\u201910, pp. 1\u201310. IEEE Computer Society (2010)","DOI":"10.1109\/MSST.2010.5496972"},{"issue":"1","key":"21_CR26","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1186\/s40537-016-0051-6","volume":"3","author":"R Singh","year":"2016","unstructured":"Singh, R., Kaur, P.J.: Analyzing performance of Apache Tez and MapReduce with hadoop multinode cluster on Amazon cloud. J. Big Data 3(1), 19 (2016)","journal-title":"J. Big Data"},{"key":"21_CR27","unstructured":"Su, T., Dy, J.: A deterministic method for initializing K-means clustering. In: Proceedings 16th IEEE International Conference on Tools with Artificial Intelligence, pp. 784\u2013786. IEEE Press, November 2004"},{"key":"21_CR28","unstructured":"Taneja, R., Krishnamurthy, R., Liu, G.: Optimization of machine learning on Apache Spark. In: Proceedings 2016 International Conference on Parallel and Distributed Processing Techniques and Applications, pp. 163\u2013167. CSREA Press (2016)"},{"key":"21_CR29","doi-asserted-by":"publisher","first-page":"1626","DOI":"10.14778\/1687553.1687609","volume":"2","author":"A Thusoo","year":"2009","unstructured":"Thusoo, A., et al.: Hive - a warehousing solution over a map-reduce framework. Proc. VLDB Endow. 2, 1626\u20131629 (2009)","journal-title":"Proc. VLDB Endow."},{"key":"21_CR30","unstructured":"Zaharia, M., et al.: Resilient distributed datasets: a fault-tolerant abstraction for in-memory cluster computing. In: Proceedings 9th USENIX Conference on Networked Systems Design and Implementation, NSDI\u201912, p. 2. USENIX Association, Berkeley, CA, USA (2012)"},{"key":"21_CR31","unstructured":"Zaharia, M., Chowdhury, M., Franklin, M.J., Shenker, S., Stoica, I.: Spark: cluster computing with working sets. In: Proceedings 2nd USENIX Conference on Hot Topics in Cloud Computing, HotCloud\u201910, p. 10. USENIX Association, Berkeley, CA, USA (2010)"},{"issue":"1","key":"21_CR32","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/s10723-012-9204-9","volume":"10","author":"Y Zhang","year":"2012","unstructured":"Zhang, Y., Gao, Q., Gao, L., Wang, C.: iMapReduce: a distributed computing framework for iterative computation. J. Grid Comput. 10(1), 47\u201368 (2012)","journal-title":"J. Grid Comput."}],"container-title":["Lecture Notes in Computer Science","AI 2019: Advances in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-35288-2_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,11,25]],"date-time":"2019-11-25T00:30:20Z","timestamp":1574641820000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-35288-2_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030352875","9783030352882"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-35288-2_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"25 November 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"AI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australasian Joint Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Adelaide, SA","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"32","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ausai2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/nugget.unisa.edu.au\/AI2019\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"115","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"42% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.4","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}