{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T06:52:34Z","timestamp":1725864754705},"publisher-location":"Berlin, Heidelberg","reference-count":58,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783662534540"},{"type":"electronic","value":"9783662534557"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-662-53455-7_2","type":"book-chapter","created":{"date-parts":[[2016,9,9]],"date-time":"2016-09-09T18:21:18Z","timestamp":1473445278000},"page":"23-47","source":"Crossref","is-referenced-by-count":0,"title":["Divide-and-Conquer Parallelism for Learning Mixture Models"],"prefix":"10.1007","author":[{"given":"Takaya","family":"Kawakatsu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Akira","family":"Kinoshita","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Atsuhiro","family":"Takasu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Adachi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,9,10]]},"reference":[{"unstructured":"Zaharia, M., Chowdhury, M., Franklin, M.J., Shenkerand, S., Stoica, I.: Spark: cluster computing with working sets. In: Proceedings of the 2nd USENIX Conference on Hot Topics in Cloud Computing, June 2010","key":"2_CR1"},{"unstructured":"Zaharia, M., Chowdhury, M., Das, T., Dave, A., Ma, J., MacCauley, M., Franklin, M.J., Shenker, S., Stoica, I.: Resilient distributed datasets: a fault-tolerant abstraction for in-memory cluster computing. In: Proceedings of the 9th USENIX Conference on Networked Systems Design and Implementation, April 2012","key":"2_CR2"},{"unstructured":"Power, R., Li, J.: Piccolo: building fast, distributed programs with partitioned tables. In: Proceedings of the 9th USENIX Conference on Operating Systems Design and Implementation, October 2010","key":"2_CR3"},{"unstructured":"Huang, C., Chen, Q., Wang, Z., Power, R., Ortiz, J., Li, J., Xiao, Z.: Spartan: a distributed array framework with smart tiling. In: Proceedings of the USENIX Annual Technical Conference, July 2015","key":"2_CR4"},{"doi-asserted-by":"crossref","unstructured":"Dijkstra, E.W.: Cooperating sequential processes. EWD: EWD123 (1968)","key":"2_CR5","DOI":"10.1007\/978-1-4757-3472-0_2"},{"doi-asserted-by":"crossref","unstructured":"Mohr, E., Kranz Jr., D.A., Halstead, R.H.: Lazy task creation: a technique for increasing the granularity of parallel programs. In: Proceedings of the 1990 ACM Conference on LISP and Functional Programming, May 1990","key":"2_CR6","DOI":"10.1145\/91556.91631"},{"doi-asserted-by":"crossref","unstructured":"Blumofe, R.D., Joerg, C.F., Kuszmaul, B.C., Leiserson, C.E., Randall, K.H., Zhou, Y.: Cilk: an efficient multithreaded runtime system. In: Proceedings of the Fifth ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, August 1995","key":"2_CR7","DOI":"10.1145\/209936.209958"},{"issue":"1","key":"2_CR8","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","volume":"39","author":"AP Dempster","year":"1977","unstructured":"Dempster, A.P., Laird, N.M., Rubin, D.B.: Maximum likelihood from incomplete data via the EM algorithm. J. Roy. Stat. Soc. Ser. B (Methodol.) 39(1), 1\u201338 (1977)","journal-title":"J. Roy. Stat. Soc. Ser. B (Methodol.)"},{"key":"2_CR9","doi-asserted-by":"crossref","DOI":"10.1002\/9780470191613","volume-title":"The EM Algorithm and Extensions","author":"GJ McLachlan","year":"2008","unstructured":"McLachlan, G.J., Krishnan, T.: The EM Algorithm and Extensions. Wiley, Hoboken (2008)"},{"unstructured":"Kinoshita, A., Takasu, A., Adachi, J.: Traffic incident detection using probabilistic topic model. In: Proceedings of the Workshops of the EDBT\/ICDT 2014 Joint Conference, March 2014","key":"2_CR10"},{"issue":"C","key":"2_CR11","doi-asserted-by":"crossref","first-page":"169","DOI":"10.1016\/j.is.2015.07.002","volume":"54","author":"A Kinoshita","year":"2015","unstructured":"Kinoshita, A., Takasu, A., Adachi, J.: Real-time traffic incident detection using a probabilistic topic model. Inf. Syst. 54(C), 169\u2013188 (2015)","journal-title":"Inf. Syst."},{"issue":"6","key":"2_CR12","doi-asserted-by":"crossref","first-page":"595","DOI":"10.1109\/LSP.2013.2260329","volume":"20","author":"SS Pereira","year":"2013","unstructured":"Pereira, S.S., Lopez-Valcarce, R., Pages-Zamora, A.: A diffusion-based EM algorithm for distributed estimation in unreliable sensor networks. IEEE Signal Process. Lett. 20(6), 595\u2013598 (2013)","journal-title":"IEEE Signal Process. Lett."},{"issue":"8","key":"2_CR13","doi-asserted-by":"crossref","first-page":"7632","DOI":"10.3390\/s100807632","volume":"10","author":"J Chen","year":"2010","unstructured":"Chen, J., Salim, M.B., Matsumoto, M.: A gaussian mixture model-based continuous boundary detection for 3d sensor networks. Sensors 10(8), 7632\u20137650 (2010)","journal-title":"Sensors"},{"unstructured":"Miura, K., Noguchi, H., Kawaguchi, H., Yoshimoto, M.: A low memory bandwidth gaussian mixture model (GMM) processor for 20,000-word real-time speech recognition FPGA system. In: 2008 International Conference on ICECE Technology, December 2008","key":"2_CR14"},{"doi-asserted-by":"crossref","unstructured":"Gupta, K., Owens, J.D.: Three-layer optimizations for fast GMM computations on GPU-like parallel processors. In: IEEE Workshop on Automatic Speech Recognition & Understanding, December 2009","key":"2_CR15","DOI":"10.1109\/ASRU.2009.5373410"},{"doi-asserted-by":"crossref","unstructured":"Stauffer, C., Grimson, W.E.L.: Adaptive background mixture models for real-time tracking. In: IEEE Computer Society Conference on Computer Vision and Pattern Recognition, June 1999","key":"2_CR16","DOI":"10.1109\/CVPR.1999.784637"},{"unstructured":"Li, H., Achim, A., Bull, D.R.: GMM-based efficient foreground detection with adaptive region update. In: Proceedings of the 16th IEEE International Conference on Image Processing, November 2009","key":"2_CR17"},{"doi-asserted-by":"crossref","unstructured":"Patel, C.I., Patel, R.: Gaussian mixture model based moving object detection from video sequence. In: Proceedings of the International Conference and Workshop on Emerging Trends in Technology, February 2011","key":"2_CR18","DOI":"10.1145\/1980022.1980172"},{"doi-asserted-by":"crossref","unstructured":"Song, Y., Li, X., Liu, Q.: Fast moving object detection using improved gaussian mixture models. In: International Conference on Audio, Language and Image Processing, July 2014","key":"2_CR19","DOI":"10.1109\/ICALIP.2014.7009844"},{"unstructured":"Rumelhart, D.E., Hinton, G.E., Williams, R.J.: Learning representations by back-propagating errors. In: Neurocomputing: Foundations of Research, January 1988","key":"2_CR20"},{"doi-asserted-by":"crossref","unstructured":"Liu, Z., Li, H., Miao, G.: MapReduce-based backpropagation neural network over large scale mobile data. In: Sixth International Conference on Natural Computation, August 2010","key":"2_CR21","DOI":"10.1109\/ICNC.2010.5584323"},{"doi-asserted-by":"crossref","unstructured":"Gu, R., Shen, F., Huang, Y.: A parallel computing platform for training large scale neural networks. In: IEEE International Conference on Big Data, October 2013","key":"2_CR22","DOI":"10.1109\/BigData.2013.6691598"},{"issue":"12","key":"2_CR23","doi-asserted-by":"crossref","first-page":"1170","DOI":"10.1145\/7902.7903","volume":"29","author":"WD Hillis","year":"1986","unstructured":"Hillis, W.D., Steele Jr., G.L.: Data parallel algorithms. Commun. ACM Spec. Issue Parallelism 29(12), 1170\u20131183 (1986)","journal-title":"Commun. ACM Spec. Issue Parallelism"},{"issue":"9","key":"2_CR24","doi-asserted-by":"crossref","first-page":"948","DOI":"10.1109\/TC.1972.5009071","volume":"C\u201321","author":"MJ Flynn","year":"1972","unstructured":"Flynn, M.J.: Some computer organizations and their effectiveness. IEEE Trans. Comput. C\u201321(9), 948\u2013960 (1972)","journal-title":"IEEE Trans. Comput."},{"doi-asserted-by":"crossref","unstructured":"Kwedlo, W.: A parallel EM algorithm for Gaussian mixture models implemented on a NUMA system using OpenMP. In: 22nd Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP), February 2014","key":"2_CR25","DOI":"10.1109\/PDP.2014.77"},{"doi-asserted-by":"crossref","unstructured":"Yang, R., Xiong, T., Chen, T., Huang, Z., Feng, S.: DISTRIM: parallel GMM learning on multicore cluster. In: IEEE International Conference on Computer Science and Automation Engineering (CSAE), May 2012","key":"2_CR26","DOI":"10.1109\/CSAE.2012.6272849"},{"doi-asserted-by":"crossref","unstructured":"Wolfe, J., Haghighi, A., Klein, D.: Fully distributed EM for very large datasets. In: Proceedings of the 25th International Conference on Machine Learning, July 2008","key":"2_CR27","DOI":"10.1145\/1390156.1390305"},{"doi-asserted-by":"crossref","unstructured":"Kumar, N.S.L.P., Satoor, S., Buck, L.: Fast parallel expectation maximization for gaussian mixture models on GPUs using CUDA. In: 11th IEEE International Conference on High Performance Computing and Communications, June 2009","key":"2_CR28","DOI":"10.1109\/HPCC.2009.45"},{"doi-asserted-by":"crossref","unstructured":"Machlica, L., Vanek, J., Zajic, Z.: Fast estimation of gaussian mixture model parameters on GPU using CUDA. In: 12th International Conference on Parallel and Distributed Computing, Applications and Technologies (PDCAT), October 2011","key":"2_CR29","DOI":"10.1109\/PDCAT.2011.40"},{"doi-asserted-by":"crossref","unstructured":"Altinigneli, M.C., Plant, C., Bohm, C.: Massively parallel expectation maximization using graphics processing units. In: Proceedings of the 19th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, August 2013","key":"2_CR30","DOI":"10.1145\/2487575.2487628"},{"doi-asserted-by":"crossref","unstructured":"Bergstrom, L., Reppy, J.: Nested data-parallelism on the GPU. In: Proceedings of the 17th ACM SIGPLAN International Conference on Functional Programming, September 2012","key":"2_CR31","DOI":"10.1145\/2364527.2364563"},{"doi-asserted-by":"crossref","unstructured":"Lee, H., Brown, K.J., Sujeeth, A.K., Rompf, T., Olkotun, K.: Locality-aware mapping of nested parallel patterns on GPU. In: Proceedings of eht 47th Annual IEEE\/ACM International Symposium on Microarchitecture, December 2014","key":"2_CR32","DOI":"10.1109\/MICRO.2014.23"},{"key":"2_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1007\/BFb0018649","volume-title":"Parallel Symbolic Computing: Languages, Systems, and Applications","author":"M Feeley","year":"1993","unstructured":"Feeley, M.: A message passing implementation of lazy task creation. In: Halstead, R.H., Ito, T. (eds.) PSC 1992. LNCS, vol. 748, pp. 94\u2013107. Springer, Heidelberg (1993). doi: 10.1007\/BFb0018649"},{"key":"2_CR34","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"174","DOI":"10.1007\/978-3-540-39707-6_13","volume-title":"High Performance Computing","author":"S Umatani","year":"2003","unstructured":"Umatani, S., Yasugi, M., Komiya, T., Yuasa, T.: Pursuing laziness for efficient implementation of modern multithreaded languages. In: Veidenbaum, A., Joe, K., Amano, H., Aiso, H. (eds.) ISHPC 2003. LNCS, vol. 2858, pp. 174\u2013188. Springer, Heidelberg (2003). doi: 10.1007\/978-3-540-39707-6_13"},{"doi-asserted-by":"crossref","unstructured":"Acar, U.A., Chargueraud, A., Rainey, M.: Scheduling parallel programs by work stealing with private deques. In: Proceedings of the 18th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, February 2013","key":"2_CR35","DOI":"10.1145\/2442516.2442538"},{"doi-asserted-by":"crossref","unstructured":"Frigo, M., Leiserson, C.E., Randall, K.H.: The implementation of the Cilk-5 multithreaded language. In: Proceedings of the ACM SIGPLAN 1998 Conference on Programming Language Design and Implementation, May 1998","key":"2_CR36","DOI":"10.1145\/277650.277725"},{"unstructured":"Min, S.J., Iancu, C., Yelick, K.: Hierarchical work stealing on manycore clusters. In: Fifth Conference on Partitioned Global Address Space Programming Models, October 2011","key":"2_CR37"},{"doi-asserted-by":"crossref","unstructured":"Olivier, S.L., Porterfield, A.K., Wheeler, K.B., Prins, J.F.: Scheduling task parallelism on multi-socket multicore systems. In: Proceedings of the 1st International Workshop on Runtime and Operating Systems for Supercomputers, May 2011","key":"2_CR38","DOI":"10.1145\/1988796.1988804"},{"issue":"2","key":"2_CR39","doi-asserted-by":"crossref","first-page":"110","DOI":"10.1177\/1094342011434065","volume":"26","author":"SL Olivier","year":"2012","unstructured":"Olivier, S.L., Porterfield, A.K., Wheeler, K.B., Spiegel, M., Prins, J.F.: OpenMP task scheduling strategies for multicore numa systems. Int. J. High Perform. Comput. Appl. 26(2), 110\u2013124 (2012)","journal-title":"Int. J. High Perform. Comput. Appl."},{"doi-asserted-by":"crossref","unstructured":"Nakashima, J., Nakatani, S., Taura, K.: Design and implementation of a customizable work stealing scheduler. In: 3rd International Workshop on Runtime and Operating Systems for Supercomputers, June 2013","key":"2_CR40","DOI":"10.1145\/2491661.2481433"},{"doi-asserted-by":"crossref","unstructured":"Kranz, D.A., Halstead, R.H., Mohr Jr., E.: Mul-T: a high-performance parallel lisp. In: Proceedings of the ACM SIGPLAN 1989 Conference on Programming Language Design and Implementation, June 1989","key":"2_CR41","DOI":"10.1145\/73141.74825"},{"doi-asserted-by":"crossref","unstructured":"Wheeler, K.B., Murphy, R.C., Thain, D.: Qthreads: an API for programming with millions of lightweight threads. In: IEEE International Symposium on Parallel and Distributed Processing, April 2008","key":"2_CR42","DOI":"10.1109\/IPDPS.2008.4536359"},{"doi-asserted-by":"crossref","unstructured":"Molka, D., Hackenberg, D., Shone, R., Muller, M.S.: Memory performance and cache coherency effects on an intel nahalem multiprocessor system. In: 18th International Conference on Parallel Architectures and Compilation Techniques, September 2009","key":"2_CR43","DOI":"10.1109\/PACT.2009.22"},{"doi-asserted-by":"crossref","unstructured":"Molka, D., Hackenberg, D., Schone, R., Nagel, W.E.: Cache coherence protocol and memory performance of the intel haswell-EP architecture. In: 44th International Conference on Parallel Processing, September 2015","key":"2_CR44","DOI":"10.1109\/ICPP.2015.83"},{"doi-asserted-by":"crossref","unstructured":"Charles, P., Donawa, C., Ebcioglu, K., Grothoff, C., Kielstra, A., von Praun, C., Saraswat, V., Sarkar, V.: X10: an object-oriented approach to non-uniform cluster computing. In: Proceedings of the 20th Annual ACM SIGPLAN Conference on Object-Oriented Programming, Systems, Languages, and Applications, October 2005","key":"2_CR45","DOI":"10.1145\/1094811.1094852"},{"doi-asserted-by":"crossref","unstructured":"Callahan, D., Chamberlain, B.L., Zima, H.P.: The cascade high productivity language. In: 9th International Workshop on High-Level Parallel Programming Models and Supportive Environments, April 2004","key":"2_CR46","DOI":"10.1109\/HIPS.2004.1299190"},{"unstructured":"Dean, J., Ghemawat, S.: MapReduce: simplified data processing on large clusters. In: Proceedings of the 6th Conference on Symposium on Opearting Systems Design & Implementation, vol. 6, December 2004","key":"2_CR47"},{"unstructured":"Furmento, N., Goglin, B.: Enabling high-performance memory migration for multithreaded applications on Linux. In: IEEE International Symposium on Parallel & Distributed Processing, May 2009","key":"2_CR48"},{"issue":"7","key":"2_CR49","doi-asserted-by":"crossref","first-page":"40","DOI":"10.1145\/2508834.2513149","volume":"11","author":"C Lameter","year":"2013","unstructured":"Lameter, C.: NUMA (non-uniform memory access): an overview. Queue 11(7), 40 (2013)","journal-title":"Queue"},{"unstructured":"Low, Y., Gonzalez, J., Kyrola, A., Bickson, D., Guestrin, C., Hellerstein, J.: GraphLab: a new framework for parallel machine learning. In: Proceedings of the 26th Conference on Uncertainty in Artificial Intelligence, June 2010","key":"2_CR50"},{"doi-asserted-by":"crossref","unstructured":"Low, Y., Bickson, D., Gonzalez, J., Guestrin, C., Kyrola, A., Hellerstein, J.M.: Distributed GraphLab: a framework for machine learning and data mining in the cloud. In: Proceedings of the VLDB Endowment, April 2012","key":"2_CR51","DOI":"10.14778\/2212351.2212354"},{"unstructured":"Hamidouche, K., Falcou, J., Etiemble, D.: A framework for an automatic hybrid MPI+ openMP code generation. In: Proceedings of the 19th High Performance Computing Symposia, April 2011","key":"2_CR52"},{"doi-asserted-by":"crossref","unstructured":"Si, M., Pena, A.J., Balaji, P., Takagi, M., Ishikawa, Y.: MT-MPI: multithreaded MPI for many-core environments. In: Proceedings of the 28th ACM International Conference on Supercomputing, June 2014","key":"2_CR53","DOI":"10.1145\/2597652.2597658"},{"doi-asserted-by":"crossref","unstructured":"Luo, M., Lu, X., Hamidouche, K., Kandalla, K., Panda, D.K.: Initial study of multi-endpoint runtime for MPI+ openMP hybrid programming model on multi-core systems. In: Proceedings of the 19th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, February 2014","key":"2_CR54","DOI":"10.1145\/2555243.2555287"},{"key":"2_CR55","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1007\/978-3-319-22852-5_15","volume-title":"Database and Expert Systems Applications","author":"T Kawakatsu","year":"2015","unstructured":"Kawakatsu, T., Kinoshita, A., Takasu, A., Adachi, J.: Highly efficient parallel framework: a divide-and-conquer approach. In: Chen, Q., Hameurlain, A., Toumani, F., Wagner, R., Decker, H. (eds.) DEXA 2015. LNCS, vol. 9262, pp. 162\u2013176. Springer, Heidelberg (2015). doi: 10.1007\/978-3-319-22852-5_15"},{"doi-asserted-by":"crossref","unstructured":"Arora, N.S., Blumofe, R.D., Plaxton, C.G.: Thread scheduling for multiprogrammed multiprocessors. In: Proceedings of the Tenth Annual ACM Symposium on Parallel Algorithms and Architectures, June 1998","key":"2_CR56","DOI":"10.1145\/277651.277678"},{"key":"2_CR57","volume-title":"Processors, Programming Massively Parallel: A Hands-on Approach","author":"DB Kirk","year":"2010","unstructured":"Kirk, D.B., Hwu, W.W.: Processors, Programming Massively Parallel: A Hands-on Approach. Morgan Kaufmann, San Francisco (2010)"},{"unstructured":"Nvidia. CUDA C programming guide version 6.5, August 2014","key":"2_CR58"}],"container-title":["Lecture Notes in Computer Science","Transactions on Large-Scale Data- and Knowledge-Centered Systems XXVIII"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-662-53455-7_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,19]],"date-time":"2024-06-19T08:47:50Z","timestamp":1718786870000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-662-53455-7_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783662534540","9783662534557"],"references-count":58,"URL":"https:\/\/doi.org\/10.1007\/978-3-662-53455-7_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2016]]}}}