{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T18:13:12Z","timestamp":1725732792809},"publisher-location":"Berlin, Heidelberg","reference-count":22,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642387081"},{"type":"electronic","value":"9783642387098"}],"license":[{"start":{"date-parts":[[2014,1,1]],"date-time":"2014-01-01T00:00:00Z","timestamp":1388534400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014]]},"DOI":"10.1007\/978-3-662-44917-2_15","type":"book-chapter","created":{"date-parts":[[2014,8,23]],"date-time":"2014-08-23T01:18:23Z","timestamp":1408756703000},"page":"169-180","source":"Crossref","is-referenced-by-count":2,"title":["Designing Coalescing Network-on-Chip for Efficient Memory Accesses of GPGPUs"],"prefix":"10.1007","author":[{"given":"Chien-Ting","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yoshi Shih-Chieh","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan-Ying","family":"Chang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chiao-Yun","family":"Tu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chung-Ta","family":"King","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tai-Yuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Janche","family":"Sang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming-Hua","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"15_CR1","unstructured":"Nvidia gpu computing sdk suite, \n                    \n                      https:\/\/developer.nvidia.com\/gpu-computing-sdk"},{"key":"15_CR2","unstructured":"Parboil benchmark suite, \n                    \n                      http:\/\/impact.crhc.illinois.edu\/parboil.php"},{"key":"15_CR3","doi-asserted-by":"crossref","unstructured":"Ausavarungnirun, R., Chang, K., Subramanian, L., Loh, G., Mutlu, O.: Staged memory scheduling: Achieving high performance and scalability in heterogeneous systems. In: Proceedings of the 39th International Symposium on Computer Architecture, pp. 416\u2013427. IEEE Press (2012)","DOI":"10.1109\/ISCA.2012.6237036"},{"key":"15_CR4","doi-asserted-by":"crossref","unstructured":"Bakhoda, A., Kim, J., Aamodt, T.: Throughput-effective on-chip networks for manycore accelerators. In: Proceedings of the 2010 43rd Annual IEEE\/ACM International Symposium on Microarchitecture, pp. 421\u2013432. IEEE Computer Society (2010)","DOI":"10.1109\/MICRO.2010.50"},{"key":"15_CR5","doi-asserted-by":"crossref","unstructured":"Bakhoda, A., Yuan, G., Fung, W., Wong, H., Aamodt, T.: Analyzing cuda workloads using a detailed gpu simulator. In: Proceedings of IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS), pp. 163\u2013174. IEEE (2009)","DOI":"10.1109\/ISPASS.2009.4919648"},{"key":"15_CR6","doi-asserted-by":"crossref","unstructured":"Che, S., Boyer, M., Meng, J., Tarjan, D., Sheaffer, J., Lee, S., Skadron, K.: Rodinia: A benchmark suite for heterogeneous computing. In: Proceedings of IEEE International Symposium on Workload Characterization (IISWC), pp. 44\u201354. IEEE (2009)","DOI":"10.1109\/IISWC.2009.5306797"},{"issue":"2","key":"15_CR7","doi-asserted-by":"publisher","first-page":"194","DOI":"10.1109\/71.127260","volume":"3","author":"W. Dally","year":"1992","unstructured":"Dally, W.: Virtual-channel flow control. IEEE Transactions on Parallel and Distributed Systems\u00a03(2), 194\u2013205 (1992)","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"15_CR8","unstructured":"Dally, W., Towles, B.: Principles and practices of interconnection networks. Morgan Kaufmann (2004)"},{"issue":"1","key":"15_CR9","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1109\/MM.2010.98","volume":"31","author":"R. Das","year":"2011","unstructured":"Das, R., Mutlu, O., Moscibroda, T., Das, C.: A\u00e9rgia: A network-on-chip exploiting packet latency slack. IEEE Micro\u00a031(1), 29\u201341 (2011)","journal-title":"IEEE Micro"},{"key":"15_CR10","doi-asserted-by":"crossref","unstructured":"Kim, Y., Lee, H., Kim, J.: An alternative memory access scheduling in manycore accelerators. In: 2011 International Conference on Parallel Architectures and Compilation Techniques (PACT), pp. 195\u2013196. IEEE (2011)","DOI":"10.1109\/PACT.2011.37"},{"key":"15_CR11","doi-asserted-by":"crossref","unstructured":"Mutlu, O., Moscibroda, T.: Stall-time fair memory access scheduling for chip multiprocessors. In: Proceedings of the 40th IEEE\/ACM International Symposium on Microarchitecture, pp. 146\u2013160. IEEE Computer Society (2007)","DOI":"10.1109\/MICRO.2007.21"},{"key":"15_CR12","doi-asserted-by":"crossref","unstructured":"Mutlu, O., Moscibroda, T.: Parallelism-aware batch scheduling: Enhancing both performance and fairness of shared dram systems. In: ACM SIGARCH Computer Architecture News, vol.\u00a036, pp. 63\u201374. IEEE Computer Society (2008)","DOI":"10.1145\/1394608.1382128"},{"key":"15_CR13","doi-asserted-by":"crossref","unstructured":"Nesbit, K., Aggarwal, N., Laudon, J., Smith, J.: Fair queuing memory systems. In: Proceedings of the 39th IEEE\/ACM International Symposium on Microarchitecture, pp. 208\u2013222. IEEE (2006)","DOI":"10.1109\/MICRO.2006.24"},{"issue":"2","key":"15_CR14","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1109\/MM.2010.41","volume":"30","author":"J. Nickolls","year":"2010","unstructured":"Nickolls, J., Dally, W.: The gpu computing era. IEEE Micro\u00a030(2), 56\u201369 (2010)","journal-title":"IEEE Micro"},{"key":"15_CR15","unstructured":"NVIDIA: Nvidia\u2019s next generation cuda compute architecture: Fermi (2009)"},{"issue":"5","key":"15_CR16","doi-asserted-by":"publisher","first-page":"879","DOI":"10.1109\/JPROC.2008.917757","volume":"96","author":"J. Owens","year":"2008","unstructured":"Owens, J., Houston, M., Luebke, D., Green, S., Stone, J., Phillips, J.: Gpu computing. Proceedings of the IEEE\u00a096(5), 879\u2013899 (2008)","journal-title":"Proceedings of the IEEE"},{"key":"15_CR17","unstructured":"Pcchen: N-queens solver, \n                    \n                      http:\/\/forums.nvidia.com\/index.php?showtopic=76893"},{"key":"15_CR18","doi-asserted-by":"crossref","unstructured":"Rixner, S., Dally, W., Kapasi, U., Mattson, P., Owens, J.: Memory access scheduling. In: Proceedings of the 27th International Symposium on Computer Architecture, pp. 128\u2013138. IEEE (2000)","DOI":"10.1145\/342001.339668"},{"key":"15_CR19","unstructured":"Sanders, J., Kandrot, E.: CUDA by example: An introduction to general-purpose GPU programming. Addison-Wesley Professional (2010)"},{"issue":"3","key":"15_CR20","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1109\/MCSE.2010.69","volume":"12","author":"J. Stone","year":"2010","unstructured":"Stone, J., Gohara, D., Shi, G.: Opencl: A parallel programming standard for heterogeneous computing systems. Computing in Science and Engineering\u00a012(3), 66 (2010)","journal-title":"Computing in Science and Engineering"},{"key":"15_CR21","doi-asserted-by":"crossref","unstructured":"Yin, J., Zhou, P., Holey, A., Sapatnekar, S., Zhai, A.: Energy-efficient non-minimal path on-chip interconnection network for heterogeneous systems. In: Proceedings of the 2012 ACM\/IEEE International Symposium on Low Power Electronics and Design, pp. 57\u201362. ACM (2012)","DOI":"10.1145\/2333660.2333675"},{"key":"15_CR22","doi-asserted-by":"crossref","unstructured":"Yuan, G., Bakhoda, A., Aamodt, T.: Complexity effective memory access scheduling for many-core accelerator architectures. In: Proceedings of the 42nd IEEE\/ACM International Symposium on Microarchitecture, pp. 34\u201344. IEEE (2009)","DOI":"10.1145\/1669112.1669119"}],"container-title":["Lecture Notes in Computer Science","Advanced Information Systems Engineering"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-662-44917-2_15","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,27]],"date-time":"2019-05-27T17:11:52Z","timestamp":1558977112000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-662-44917-2_15"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014]]},"ISBN":["9783642387081","9783642387098"],"references-count":22,"URL":"https:\/\/doi.org\/10.1007\/978-3-662-44917-2_15","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2014]]}}}