{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,4]],"date-time":"2025-05-04T05:40:08Z","timestamp":1746337208561,"version":"3.40.4"},"publisher-location":"Cham","reference-count":23,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319111964"},{"type":"electronic","value":"9783319111971"}],"license":[{"start":{"date-parts":[[2014,1,1]],"date-time":"2014-01-01T00:00:00Z","timestamp":1388534400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2014,1,1]],"date-time":"2014-01-01T00:00:00Z","timestamp":1388534400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014]]},"DOI":"10.1007\/978-3-319-11197-1_41","type":"book-chapter","created":{"date-parts":[[2014,8,13]],"date-time":"2014-08-13T14:54:29Z","timestamp":1407941669000},"page":"535-548","source":"Crossref","is-referenced-by-count":0,"title":["Customized Network-on-Chip for Message Reduction"],"prefix":"10.1007","author":[{"given":"Hongwei","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siyu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youhui","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangwen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weimin","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"41_CR1","unstructured":"Timothy, M.: The Future of Many Core Computing, http:\/\/i2pc.cs.illinois.edu\/presentations\/2010_05_06_Mattson_Slides.pdf"},{"key":"41_CR2","doi-asserted-by":"crossref","unstructured":"Rakesh, K., Timothy, G.M., Gilles, P., Rob, V.D.W.: The Case for Message Passing on Many-Core Chips. Multiprocessor System-on-Chip, pp. 115\u2013123 (2011)","DOI":"10.1007\/978-1-4419-6460-1_5"},{"key":"41_CR3","unstructured":"Jie, M., Daniel, R., Ayse, K.C.: 3D Systems with On-Chip DRAM for Enabling Low-Power High-Performance Computing. In: Proceedings of Fifteenth HPEC Workshop, Massachusetts, USA (September 2011)"},{"key":"41_CR4","unstructured":"Timothy, G.M., Rob, F.V.D.W., Michael, R., Thomas, L., Paul, B., Werner, H., Patrick, K., Jason, H., Sriram, V., Nitin, B., Greg, R., Saurabh, D.: The 48-core SCC processor: the programmer\u2019s view. In: Proceedings of 2010 International Conference for High Performance Computing, Networking, Storage and Analysis, New Orleans, LA (2010)"},{"key":"41_CR5","unstructured":"MULTICORE COMMUNICATIONS API WORKING GROUP, http:\/\/www.multicore-association.org\/workgroup\/mcapi.php"},{"key":"41_CR6","doi-asserted-by":"crossref","unstructured":"Dong, Y., Chen, J., Yang, X., Yang, C., Peng, L.: Low power optimization for MPI collective operations. In: The 9th International Conference for Young Computer Scientists, ICYCS 2008, IEEE (2008)","DOI":"10.1109\/ICYCS.2008.500"},{"key":"41_CR7","unstructured":"Rabenseifner, R.: Automatic MPI counter profiling of all users: First results on a CRAY t3e 900-512. In: Message Passing Interface Developer\u2019s and User\u2019s Conference (1999)"},{"key":"41_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-540-24685-5_1","volume-title":"Computational Science - ICCS 2004","author":"R. Rabenseifner","year":"2004","unstructured":"Rabenseifner, R.: Optimization of collective reduction operations. In: Bubak, M., van Albada, G.D., Sloot, P.M.A., Dongarra, J. (eds.) ICCS 2004. LNCS, vol.\u00a03036, pp. 1\u20139. Springer, Heidelberg (2004)"},{"key":"41_CR9","unstructured":"Open MPI Development Team, Open MPI: open source high-performance computing, http:\/\/www.open-mpi.org\/"},{"issue":"1","key":"41_CR10","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1177\/1094342005051521","volume":"19","author":"R. Thakur","year":"2005","unstructured":"Thakur, R., Rabenseifner, R., Gropp, W.: Optimization of collective communication operations in MPICH. High Performance Computing Applications\u00a019(1), 49\u201366 (2005)","journal-title":"High Performance Computing Applications"},{"key":"41_CR11","doi-asserted-by":"crossref","unstructured":"Rabenseifner, R.: Optimization of collective reduction operations. In: Proceedings of Int\u2019l Conference on Computational Science (ICCS), Krakow, Poland (2004)","DOI":"10.1007\/978-3-540-24685-5_1"},{"key":"41_CR12","doi-asserted-by":"crossref","unstructured":"Nicolas, F., Marc, H., Eric, L., Bernard, T.: MPI for the Clint Gb\/s Interconnect. In: Proceedings of the 10th European PVM\/MPI User\u2019s Group Meeting, pp. 395\u2013403 (2003)","DOI":"10.1007\/978-3-540-39924-7_54"},{"key":"41_CR13","unstructured":"Maximize Platform MPI Performance with Voltaire\u00ae Fabric Collective AcceleratorTM (FCATM) and HP, http:\/\/www.mellanox.com\/related-docs\/voltaire_acceleration_software\/FCA-Voltaire-Platform-HP-WEB111110.pdf"},{"issue":"7-8","key":"41_CR14","doi-asserted-by":"publisher","first-page":"751","DOI":"10.1002\/cpe.720","volume":"15","author":"K.D. Underwood","year":"2003","unstructured":"Underwood, K.D., Ligon, W.B., Sass, R.R.: Analysis of a prototype intelligent network interface. Concurrency and Computation: Practice and Experience\u00a015(7-8), 751\u2013777 (2003)","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"41_CR15","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"833","DOI":"10.1007\/978-3-540-27866-5_112","volume-title":"Euro-Par 2004 Parallel Processing","author":"G.S. Alm\u00e1si","year":"2004","unstructured":"Alm\u00e1si, G.S., et al.: Implementing MPI on the blueGene\/L supercomputer. In: Danelutto, M., Vanneschi, M., Laforenza, D. (eds.) Euro-Par 2004. LNCS, vol.\u00a03149, pp. 833\u2013845. Springer, Heidelberg (2004)"},{"key":"41_CR16","doi-asserted-by":"crossref","unstructured":"Gao, S., Schmidt, A.G., Sass, R.: Impact of reconfigurable hardware on accelerating mpi_reduce. In: 2010 International Conference on Field-Programmable Technology (FPT), pp. 29\u201336 (2010)","DOI":"10.1109\/FPT.2010.5681537"},{"key":"41_CR17","doi-asserted-by":"crossref","unstructured":"Libo, H., Zhiying, W., Nong, X.: Accelerating NoC-based MPI Primitives via Communication Architecture Customization. In: Proceedings of IEEE 23rd International Conference on Application-Specific Systems, Architectures and Processors, Delft, July 2012, pp. 141\u2013148. IEEE (2012)","DOI":"10.1109\/ASAP.2012.33"},{"key":"41_CR18","unstructured":"David, W., Patrick, G., Henry, H., Liewei, B., Bruce, E., Carl, R., Matthew, M., Chyi-Chang, M., John, F.B., John III, F.B., Anant, A.: On-chip Interconnection Architecture of the Tile Processor. IEEE Computer Society (September-October 2007)"},{"key":"41_CR19","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"488","DOI":"10.1007\/978-3-540-77220-0_45","volume-title":"High Performance Computing \u2013 HiPC 2007","author":"M.K. Velamati","year":"2007","unstructured":"Velamati, M.K., Kumar, A., Jayam, N., Senthilkumar, G., Baruah, P.K., Sharma, R., Kapoor, S., Srinivasan, A.: Optimization of collective communication in intra-cell MPI. In: Aluru, S., Parashar, M., Badrinath, R., Prasanna, V.K. (eds.) HiPC 2007. LNCS, vol.\u00a04873, pp. 488\u2013499. Springer, Heidelberg (2007)"},{"key":"41_CR20","doi-asserted-by":"crossref","unstructured":"Ali, Q., Midkiff, S.P., Pai, V.S.: Efficient high performance collective communication for the cell blade. In: Proceedings of the 23rd International Conference on Supercomputing, pp. 193\u2013203. ACM (2009)","DOI":"10.1145\/1542275.1542306"},{"key":"41_CR21","doi-asserted-by":"crossref","unstructured":"Kohler, A., Radetzki, M., Gschwandtner, P., Fahringer, T.: Low-latency collectives for the intel scc. In: 2012 IEEE International Conference on Cluster Computing (CLUSTER), pp. 346\u2013354. IEEE (2012)","DOI":"10.1109\/CLUSTER.2012.58"},{"key":"41_CR22","doi-asserted-by":"crossref","unstructured":"Peng, Y., Salda\u00f1a, M., Chow, P.: Hardware support for broadcast and reduce in mpsoc. In: 2011 International Conference on Field Programmable Logic and Applications (FPL), pp. 144\u2013150. IEEE (2011)","DOI":"10.1109\/FPL.2011.34"},{"issue":"2","key":"41_CR23","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1109\/40.848473","volume":"20","author":"R.E. Gonzalez","year":"2000","unstructured":"Gonzalez, R.E.: Xtensa: A configurable and extensible processor. IEEE Micro\u00a020(2), 60\u201370 (2000)","journal-title":"IEEE Micro"}],"container-title":["Lecture Notes in Computer Science","Algorithms and Architectures for Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-11197-1_41","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,4]],"date-time":"2025-05-04T05:01:47Z","timestamp":1746334907000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-11197-1_41"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014]]},"ISBN":["9783319111964","9783319111971"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-11197-1_41","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2014]]}}}