{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,18]],"date-time":"2026-01-18T22:16:29Z","timestamp":1768774589881,"version":"3.49.0"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2009,4,22]],"date-time":"2009-04-22T00:00:00Z","timestamp":1240358400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2012,4]]},"DOI":"10.1007\/s11227-009-0296-3","type":"journal-article","created":{"date-parts":[[2009,4,21]],"date-time":"2009-04-21T07:07:55Z","timestamp":1240297675000},"page":"141-162","source":"Crossref","is-referenced-by-count":33,"title":["Performance analysis and optimization of MPI collective operations on multi-core clusters"],"prefix":"10.1007","volume":"60","author":[{"given":"Bibo","family":"Tu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianping","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianfeng","family":"Zhan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaofang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2009,4,22]]},"reference":[{"key":"296_CR1","unstructured":"TOP500 Team, TOP500 Report for November 2007, http:\/\/www.top500.org"},{"key":"296_CR2","doi-asserted-by":"crossref","unstructured":"Mamidala AR, Kumar R, De D, Panda DK (2008) MPI collectives on modern multicore clusters: performance optimizations and communication characteristics. In: 8th IEEE international conference on cluster computing and the grid (CCGRID \u201908)","DOI":"10.1109\/CCGRID.2008.87"},{"key":"296_CR3","unstructured":"Rabenseifner R (1999) Automatic MPI counter profiling of all users: First results on a CRAY T3E 900-512. In: Proceedings of the message passing interface developer\u2019s and user\u2019s conference, pp\u00a077\u201385"},{"key":"296_CR4","unstructured":"Pjesivac-Grbovic J, Angskun T, Bosilca G et al (2005) Performance analysis of MPI collective operations. In: Proceedings of the 19th IEEE international parallel and distributed processing symposium (IPDPS\u201905)"},{"key":"296_CR5","unstructured":"Cameron KW, Sun X-H (2003) Quantifying locality effect in data access delay: memory logP. In: Proceedings of IEEE international parallel and distributed processing symposium (IPDPs 2003), Nice, France"},{"key":"296_CR6","unstructured":"Cameron KW, Ge R (2004) Predicting and evaluating distributed communication performance. In: Proceedings of the 2004 ACM\/IEEE supercomputing conference"},{"issue":"3","key":"296_CR7","doi-asserted-by":"crossref","first-page":"314","DOI":"10.1109\/TC.2007.38","volume":"56","author":"KW Cameron","year":"2007","unstructured":"Cameron KW, Ge R, Sun X-H (2007) log n P and log3P: accurate analytical models of point-to-point communication in distributed systems. IEEE Trans Comput 56(3):314\u2013327","journal-title":"IEEE Trans Comput"},{"key":"296_CR8","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1007\/978-3-540-39924-7_38","volume-title":"Recent advances in parallel virtual machine and message passing interface","author":"R Thakur","year":"2003","unstructured":"Thakur R, Gropp W (2003) Improving the performance of collective operations in MPICH. In: Dongarra J, Laforenza D, Orlando S (eds) Recent advances in parallel virtual machine and message passing interface. Lecture notes in computer science, vol\u00a02840. Springer, Berlin, pp\u00a0257\u2013267"},{"key":"296_CR9","volume-title":"Proceedings of EuroPVM\/MPI. Lecture notes in computer science","author":"R Rabenseifner","year":"2004","unstructured":"Rabenseifner R, Traff JL (2004) More efficient reduction algorithms for non-power-of-two number of processors in message-passing parallel systems. In: Proceedings of EuroPVM\/MPI. Lecture notes in computer science. Springer, Berlin"},{"key":"296_CR10","doi-asserted-by":"crossref","first-page":"131","DOI":"10.1145\/301104.301116","volume-title":"Proceedings of the seventh ACM SIGPLAN symposium on principles and practice of parallel programming","author":"T Kielmann","year":"1999","unstructured":"Kielmann T, Hofman RFH, Bal HE, Plaat A, Bhoedjang RAF (1999) MagPIe: MPI\u2019s collective communication operations for clustered wide area systems. In: Proceedings of the seventh ACM SIGPLAN symposium on principles and practice of parallel programming. ACM Press, New York, pp\u00a0131\u2013140"},{"key":"296_CR11","unstructured":"Park J-YL, Choi H-A, Nupairoj N, Ni LM (1996) Construction of optimal multicast trees based on the parameterized communication model. In: Proc int conference on parallel processing (ICPP), vol\u00a0I, pp\u00a0180\u2013187"},{"key":"296_CR12","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1145\/240455.240477","volume":"39","author":"DE Culler","year":"1996","unstructured":"Culler DE, Karp R, Patterson DA, Sahay A, Santos E, Schauser K, Subramonian R, von Eicken T (1996) LogP: a practical model of parallel computation. Commun ACM 39:78\u201385","journal-title":"Commun ACM"},{"key":"296_CR13","unstructured":"Alexandrov A, Ionescu MF, Schauser K, Scheiman C (1995) LogGP: incorporating long messages into the LogP model. In: Proceedings of seventh annual symposium on parallel algorithms and architecture, Santa Barbara, CA, pp\u00a095\u2013105"},{"key":"296_CR14","doi-asserted-by":"crossref","unstructured":"Kielmann T, Bal HE (2000) Fast measurement of LogP parameters for message passing platforms. In: Proceedings of the 15 IPDPS 2000 workshops on parallel and distributed processing, pp\u00a01176\u20131183","DOI":"10.1007\/3-540-45591-4_162"},{"key":"296_CR15","doi-asserted-by":"crossref","unstructured":"Frank MI, Agarwal A, Vernon MK (1997) LoPC: modeling contention in parallel algorithms. In: Proceedings of sixth symposium on principles and practice of parallel programming, Las Vegas, NV, pp\u00a0276\u2013287","DOI":"10.1145\/263764.263803"},{"key":"296_CR16","unstructured":"Moritz CA, Frank MI (1998) LoGPC: modeling network contention in message-passing programs. In: Proceedings of SIGMETRICS \u201998, Madison, WI, pp\u00a0254\u2013263"},{"key":"296_CR17","doi-asserted-by":"crossref","unstructured":"Ino F, Fujimoto N, Hagihara K (2001) LogGPS: a parallel computational model for synchronization analysis. In: Proceedings of PPoPP\u201901, Snowbird, Utah, pp\u00a0133\u2013142","DOI":"10.1145\/379539.379592"},{"key":"296_CR18","unstructured":"Barnett M, Littlefield R, Payne D, van de Geijn R (1993) Global combine on mesh architectures with wormhole routing. In: Proceedings of the 7th international parallel processing symposium, April"},{"key":"296_CR19","doi-asserted-by":"crossref","unstructured":"Scott D (1991) Efficient all-to-all communication patterns in hypercube and mesh topologies. In: Proceedings of the 6th distributed memory computing conference, pp\u00a0398\u2013403","DOI":"10.1109\/DMCC.1991.633174"},{"key":"296_CR20","doi-asserted-by":"crossref","unstructured":"Vadhiyar SS, Fagg GE, Dongarra J (1999) Automatically tuned collective communications. In: Proceedings of SC99: high performance networking and computing, November","DOI":"10.1109\/SC.2000.10024"},{"key":"296_CR21","doi-asserted-by":"crossref","unstructured":"Faraj A, Yuan X (2005) Automatic generation and tuning of MPI collective communication routines. In: Proceedings of the 19th annual international conference on supercomputing, pp\u00a0393\u2013402","DOI":"10.1145\/1088149.1088202"},{"key":"296_CR22","doi-asserted-by":"crossref","unstructured":"Karonis NT, de Supinski BR, Foster I, Gropp W et al (2000) Exploiting hierarchy in parallel computer networks to optimize collective operation performance. In: Proceedings of the 14th international parallel and distributed processing symposium (IPDPS\u20192000), pp\u00a0377\u2013384","DOI":"10.1109\/IPDPS.2000.846009"},{"key":"296_CR23","doi-asserted-by":"crossref","unstructured":"Husbands P, Hoe JC (1998) MPI-StarT: delivering network performance to numerical applications. In: Proceedings of the 1998 ACM\/IEEE SC98 conference (SC\u201998)","DOI":"10.1109\/SC.1998.10036"},{"key":"296_CR24","unstructured":"Tipparaju V, Nieplocha J, Panda DK (2003) Fast collective operations using shared and remote memory access protocols on clusters. In: International parallel and distributed processing symposium"},{"key":"296_CR25","unstructured":"Wu M-S, Kendall RA, Wright K (2005) Optimizing collective communications on SMP clusters. In: ICPP\u2019 2005"},{"key":"296_CR26","doi-asserted-by":"crossref","unstructured":"Chai L, Hartono A, Panda DK (2006) Designing high performance and scalable MPI intra-node communication support for clusters. In: The IEEE international conference on cluster computing","DOI":"10.1109\/CLUSTR.2006.311850"},{"key":"296_CR27","unstructured":"Asanovic K, Bodik R, Catanzaro BC et al (2006) The landscape of parallel computing research: a view from Berkeley. Electrical Engineering and Computer Sciences, University of California at Berkeley. Technical Report No: UCB\/EECS-2006-183, p\u00a012"},{"key":"296_CR28","doi-asserted-by":"crossref","unstructured":"Chai L, Gao Q, Panda DK (2007) Understanding the impact of multi-core architecture in cluster computing: a case study with intel dual-core system. In: Seventh IEEE international symposium on cluster computing and the grid (CCGrid\u201907), pp\u00a0471\u2013478","DOI":"10.1109\/CCGRID.2007.119"},{"key":"296_CR29","doi-asserted-by":"crossref","unstructured":"Alam SR, Barrett RF, Kuehn JA, Roth PC, Vetter JS (2006) Characterization of scientific workloads on systems with multi-core processors. In: International symposium on workload characterization","DOI":"10.1109\/IISWC.2006.302747"},{"key":"296_CR30","doi-asserted-by":"crossref","unstructured":"Liu J, Wu J, Panda DK (2004) High performance RDMA-based MPI implementation over InfiniBand. Int J Parallel Program","DOI":"10.1145\/782814.782855"},{"key":"296_CR31","doi-asserted-by":"crossref","unstructured":"Hoefler T, Lichei A, Rehm W (2007) Low-overhead LogGP parameter assessment for modern interconnection networks. In: Proceedings of IEEE international parallel and distributed processing symposium (IPDPS\u20192007)","DOI":"10.1109\/IPDPS.2007.370593"},{"key":"296_CR32","unstructured":"Curtis-Maury M, Ding X, Antonopoulos CD, Nikolopoulos DS (2005) An evaluation of OpenMP on current and emerging multithreaded\/multicore processors. In: IWOMP"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-009-0296-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-009-0296-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-009-0296-3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,6,1]],"date-time":"2019-06-01T06:23:58Z","timestamp":1559370238000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-009-0296-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2009,4,22]]},"references-count":32,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2012,4]]}},"alternative-id":["296"],"URL":"https:\/\/doi.org\/10.1007\/s11227-009-0296-3","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2009,4,22]]}}}