{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:33:06Z","timestamp":1750221186573,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,8,13]],"date-time":"2018-08-13T00:00:00Z","timestamp":1534118400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key R&D Program of China","award":["2017YFB0202003"],"award-info":[{"award-number":["2017YFB0202003"]}]},{"name":"International (Regional) Cooperation and Exchange Program of National Natural Science Foundation of China","award":["61661146006"],"award-info":[{"award-number":["61661146006"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,8,13]]},"DOI":"10.1145\/3225058.3225107","type":"proceedings-article","created":{"date-parts":[[2018,8,8]],"date-time":"2018-08-08T19:13:06Z","timestamp":1533755586000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["UHCL-Darknet"],"prefix":"10.1145","author":[{"given":"Longlong","family":"Liao","sequence":"first","affiliation":[{"name":"National University of Defense Technology, State Key Laboratory of High Performance Computing, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kenli","family":"Li","sequence":"additional","affiliation":[{"name":"Hunan University, National Supercomputing Center in Changsha, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keqin","family":"Li","sequence":"additional","affiliation":[{"name":"State University of New York, National Supercomputing Center in Changsha, New Paltz, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Canqun","family":"Yang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, State Key Laboratory of High Performance Computing, Changsha, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Tian","sequence":"additional","affiliation":[{"name":"University of Texas at San Antonio, San Antonio, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,8,13]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation (OSDI'16)","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi , Paul Barham , Jianmin Chen , and et.al. 2016 . TensorFlow: A System for Large-scale Machine Learning . In Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation (OSDI'16) . USENIX Association, Berkeley, CA, USA, 265--283. http:\/\/dl.acm.org\/citation.cfm?id=3026877.3026899 Mart\u00edn Abadi, Paul Barham, Jianmin Chen, and et.al. 2016. TensorFlow: A System for Large-scale Machine Learning. In Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation (OSDI'16). USENIX Association, Berkeley, CA, USA, 265--283. http:\/\/dl.acm.org\/citation.cfm?id=3026877.3026899"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2016.05.006"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-36949-0_14"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPA.2011.28"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018743.3018769"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2017.71"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00607-016-0537-2"},{"key":"e_1_3_2_1_8_1","volume-title":"GFlink: An In-Memory Computing Architecture on Heterogeneous CPU-GPU Clusters for Big Data","author":"Chen Cen","year":"2016","unstructured":"Cen Chen , Kenli Li , Aijia Ouyang , ZhuoTang, and Keqin Li. 2016. GFlink: An In-Memory Computing Architecture on Heterogeneous CPU-GPU Clusters for Big Data . IEEE Transactions on Parallel & Distributed Systems ( 2016 ), 542--551. Cen Chen, Kenli Li, Aijia Ouyang, ZhuoTang, and Keqin Li. 2016. GFlink: An In-Memory Computing Architecture on Heterogeneous CPU-GPU Clusters for Big Data. IEEE Transactions on Parallel & Distributed Systems (2016), 542--551."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2017.2755657"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.3424"},{"key":"e_1_3_2_1_11_1","volume-title":"https:\/\/www.open-mpi.org. (May","author":"Gabriel Edgar","year":"2017","unstructured":"Edgar Gabriel , Graham E. Fagg , George Bosilca , and Thara Angskun . 2017. Open MPI. https:\/\/www.open-mpi.org. (May 2017 ). Edgar Gabriel, Graham E. Fagg, George Bosilca, and Thara Angskun. 2017. Open MPI. https:\/\/www.open-mpi.org. (May 2017)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3078155.3078160"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2909437.2909443"},{"key":"e_1_3_2_1_14_1","volume-title":"Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 770--778","author":"He Kaiming","year":"2016","unstructured":"Kaiming He , Xiangyu Zhang , Shaoqing Ren , and Jian Sun . 2016 . Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 770--778 . Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 770--778."},{"volume-title":"Euro-Par 2014 Parallel Processing","author":"Henry Sylvain","key":"e_1_3_2_1_15_1","unstructured":"Sylvain Henry , Alexandre Denis , Denis Barthou , Marie-Christine Counilh , and Raymond Namyst . 2014. Toward OpenCL Automatic Multi-Device Support . In Euro-Par 2014 Parallel Processing . Springer International Publishing , Porto, Portugal , 776--787. https:\/\/hal.inria.fr\/hal-01005765 Sylvain Henry, Alexandre Denis, Denis Barthou, Marie-Christine Counilh, and Raymond Namyst. 2014. Toward OpenCL Automatic Multi-Device Support. In Euro-Par 2014 Parallel Processing. Springer International Publishing, Porto, Portugal, 776--787. https:\/\/hal.inria.fr\/hal-01005765"},{"volume-title":"Densely Connected Convolutional Networks. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 2261--2269","author":"Huang Gao","key":"e_1_3_2_1_16_1","unstructured":"Gao Huang , Zhuang Liu , and Laurens van der Maaten. 2017 . Densely Connected Convolutional Networks. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 2261--2269 . Gao Huang, Zhuang Liu, and Laurens van der Maaten. 2017. Densely Connected Convolutional Networks. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 2261--2269."},{"key":"e_1_3_2_1_17_1","volume-title":"FireCaffe: Near-Linear Acceleration of Deep Neural Network Training on Compute Clusters. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","volume":"00","author":"Iandola Forrest N.","year":"2016","unstructured":"Forrest N. Iandola , Khalid Ashraf , Matthew W. Moskewicz , and Kurt Keutzer . 2016 . FireCaffe: Near-Linear Acceleration of Deep Neural Network Training on Compute Clusters. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) , Vol. 00 . 2592--2600. Forrest N. Iandola, Khalid Ashraf, Matthew W. Moskewicz, and Kurt Keutzer. 2016. FireCaffe: Near-Linear Acceleration of Deep Neural Network Training on Compute Clusters. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Vol. 00. 2592--2600."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654889"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2908080.2908094"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2304576.2304623"},{"key":"e_1_3_2_1_21_1","volume-title":"Fast Algorithms for Convolutional Neural Networks. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 4013--4021","author":"Lavin Andrew","year":"2016","unstructured":"Andrew Lavin and Scott Gray . 2016 . Fast Algorithms for Convolutional Neural Networks. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 4013--4021 . Andrew Lavin and Scott Gray. 2016. Fast Algorithms for Convolutional Neural Networks. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 4013--4021."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2016.11.046"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cpc.2012.01.019"},{"volume-title":"Microsoft Cognitive Toolkit (CNTK). https:\/\/github.com\/Microsoft\/CNTK. (March","year":"2018","key":"e_1_3_2_1_24_1","unstructured":"Microsoft. 2018. Microsoft Cognitive Toolkit (CNTK). https:\/\/github.com\/Microsoft\/CNTK. (March 2018 ). Microsoft. 2018. Microsoft Cognitive Toolkit (CNTK). https:\/\/github.com\/Microsoft\/CNTK. (March 2018)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSoC.2015.10"},{"key":"e_1_3_2_1_26_1","volume-title":"http:\/\/pytorch.org\/. (March","author":"Paszke Adam","year":"2018","unstructured":"Adam Paszke and Sam Gross . 2018. PyTorch. http:\/\/pytorch.org\/. (March 2018 ). Adam Paszke and Sam Gross. 2018. PyTorch. http:\/\/pytorch.org\/. (March 2018)."},{"key":"e_1_3_2_1_27_1","volume-title":"Based on OpenCL. CoRR abs\/1606.04884","author":"Perkins Hugh","year":"2016","unstructured":"Hugh Perkins . 2016. cltorch: a Hardware-Agnostic Backend for the Torch Deep Neural Network Library , Based on OpenCL. CoRR abs\/1606.04884 ( 2016 ). arXiv:1606.04884 Hugh Perkins. 2016. cltorch: a Hardware-Agnostic Backend for the Torch Deep Neural Network Library, Based on OpenCL. CoRR abs\/1606.04884 (2016). arXiv:1606.04884"},{"key":"e_1_3_2_1_28_1","unstructured":"Joseph Redmon. 2013--2017. Darknet: Open Source Neural Networks in C. http:\/\/pjreddie.com\/darknet\/. (2013-2017).  Joseph Redmon. 2013--2017. Darknet: Open Source Neural Networks in C. http:\/\/pjreddie.com\/darknet\/. (2013-2017)."},{"volume-title":"Computer Vision and Pattern Recognition (CVPR)","author":"Redmon Joseph","key":"e_1_3_2_1_29_1","unstructured":"Joseph Redmon and Ali Farhadi . 2017. YOLO9000 : Better, faster, stronger . In Computer Vision and Pattern Recognition (CVPR) . IEEE Computer Society , Honolulu, HI, USA , 7263--7271. Joseph Redmon and Ali Farhadi. 2017. YOLO9000: Better, faster, stronger. In Computer Vision and Pattern Recognition (CVPR). IEEE Computer Society, Honolulu, HI, USA, 7263--7271."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3078633.3081040"},{"key":"e_1_3_2_1_31_1","volume-title":"Portable Deep-Learning Library for GPGPUs. In 2016 45th International Conference on Parallel Processing (ICPP)","volume":"00","author":"Yang Yi","unstructured":"Yi Yang , Min Feng , and Srimat T. Chakradhar . 2016. HppCnn: A High-Performance , Portable Deep-Learning Library for GPGPUs. In 2016 45th International Conference on Parallel Processing (ICPP) , Vol. 00 . 582--587. Yi Yang, Min Feng, and Srimat T. Chakradhar. 2016. HppCnn: A High-Performance, Portable Deep-Learning Library for GPGPUs. In 2016 45th International Conference on Parallel Processing (ICPP), Vol. 00. 582--587."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2018.00061"}],"event":{"name":"ICPP 2018: 47th International Conference on Parallel Processing","sponsor":["University of Oregon University of Oregon"],"location":"Eugene OR USA","acronym":"ICPP 2018"},"container-title":["Proceedings of the 47th International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3225058.3225107","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3225058.3225107","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:39:07Z","timestamp":1750210747000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3225058.3225107"}},"subtitle":["An OpenCL-based Deep Neural Network Framework for Heterogeneous Multi-\/Many-core Clusters"],"short-title":[],"issued":{"date-parts":[[2018,8,13]]},"references-count":32,"alternative-id":["10.1145\/3225058.3225107","10.1145\/3225058"],"URL":"https:\/\/doi.org\/10.1145\/3225058.3225107","relation":{},"subject":[],"published":{"date-parts":[[2018,8,13]]},"assertion":[{"value":"2018-08-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}