{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,29]],"date-time":"2026-03-29T10:45:02Z","timestamp":1774781102317,"version":"3.50.1"},"reference-count":40,"publisher":"Informa UK Limited","issue":"1","license":[{"start":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T00:00:00Z","timestamp":1774656000000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"name":"Key Research and Development Programme of Shaanxi Province","award":["2023-ZDLGY-53"],"award-info":[{"award-number":["2023-ZDLGY-53"]}]},{"name":"National Key R&D Programme of China","award":["2022YFB2901103"],"award-info":[{"award-number":["2022YFB2901103"]}]}],"content-domain":{"domain":["www.tandfonline.com"],"crossmark-restriction":true},"short-container-title":["Connection Science"],"published-print":{"date-parts":[[2026,12,31]]},"DOI":"10.1080\/09540091.2026.2650981","type":"journal-article","created":{"date-parts":[[2026,3,29]],"date-time":"2026-03-29T05:08:09Z","timestamp":1774760889000},"update-policy":"https:\/\/doi.org\/10.1080\/tandf_crossmark_01","source":"Crossref","is-referenced-by-count":0,"title":["A near CXL memory processing architecture for distributed graph neural network inference and training"],"prefix":"10.1080","volume":"38","author":[{"given":"Haoyang","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University","place":["Xi'an, People's Republic of China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengbing","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University","place":["Xi'an, People's Republic of China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoya","family":"Fan","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University","place":["Xi'an, People's Republic of China"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University","place":["Xi'an, People's Republic of China"]}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"301","published-online":{"date-parts":[[2026,3,28]]},"reference":[{"key":"e_1_3_3_2_1","first-page":"908","article-title":"FAFNIR: accelerating sparse gathering by using efficient near-memory intelligent reduction","author":"Asgari B.","year":"2021","unstructured":"Asgari, B., Hadidi, R., Cao, J., Shim, D. E., Lim, S.-K., & Kim, H. (2021). FAFNIR: accelerating sparse gathering by using efficient near-memory intelligent reduction. in Proc. IEEE Int. Symp. High-Performance Comput. Archit. (HPCA), 908\u2013920.","journal-title":"in Proc. IEEE Int. Symp. High-Performance Comput. Archit. (HPCA)"},{"key":"e_1_3_3_3_1","volume-title":"in Proc. Int. Symp. Comput. Archit","author":"Chen Y.-H.","year":"2016","unstructured":"Chen, Y.-H., Emer, J. S., & Sze, V. (2016). Eyeriss: A spatial architecture for energy-efficient dataflow for convolutional neural networks, in Proc. Int. Symp. Comput. Archit. (ISCA)."},{"key":"e_1_3_3_4_1","doi-asserted-by":"publisher","DOI":"10.14778\/2824032.2824077"},{"key":"e_1_3_3_5_1","unstructured":"Compute Express Link Consortium. (2022a). Compute express link (CXL) specification revision 3.0."},{"key":"e_1_3_3_6_1","unstructured":"Compute Express Link Consortium. (2022b). \u201cCompute express link 3.0 white paper \u201d. [Online]. https:\/\/www.computeexpresslink.org\/"},{"key":"e_1_3_3_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308558.3313488"},{"key":"e_1_3_3_8_1","first-page":"6533","article-title":"Protein interface prediction using graph convolutional networks","author":"Fout A.","year":"2017","unstructured":"Fout, A., Byrd, J., Shariat, B., & Ben-Hur, A. (2017). Protein interface prediction using graph convolutional networks. in Adv. Neural Inf. Process. Syst. (NeurIPS). 6533\u20136542.","journal-title":"in Adv. Neural Inf. Process. Syst. (NeurIPS)"},{"key":"e_1_3_3_9_1","first-page":"551","volume-title":"in Proc. 15th USENIX Symp. Oper. Syst. Design Implement","author":"Gandhi S.","year":"2021","unstructured":"Gandhi, S., & Iyer, A. P. (2021). P3: distributed deep graph learning at scale, in Proc. 15th USENIX Symp. Oper. Syst. Design Implement (pp. 551\u2013568). (OSDI)."},{"key":"e_1_3_3_10_1","first-page":"1263","article-title":"Neural message passing for quantum chemistry","author":"Gilmer J.","year":"2017","unstructured":"Gilmer, J., Schoenholz, S. S., Riley, P. F., Vinyals, O., & Dahl, G. E. (2017). Neural message passing for quantum chemistry. in Proc. Int. Conf. Mach. Learn. (ICML). 1263\u20131272.","journal-title":"in Proc. Int. Conf. Mach. Learn. (ICML)"},{"key":"e_1_3_3_11_1","article-title":"Inductive representation learning on large graphs","volume":"30","author":"Hamilton W.","year":"2017","unstructured":"Hamilton, W., Ying, Z., & Leskovec, J. (2017). Inductive representation learning on large graphs. in Adv. Neural Inf. Process. Syst. (NeurIPS), 30","journal-title":"in Adv. Neural Inf. Process. Syst. (NeurIPS)"},{"key":"e_1_3_3_12_1","unstructured":"Hu W. Fey M. Zitnik M. Dong Y. Ren H. Liu B. Catasta M. & Leskovec J. (2020). Open graph benchmark: Datasets for machine learning on graphs. arXiv preprint arXiv:2005.00687."},{"key":"e_1_3_3_13_1","first-page":"727","article-title":"BEACON: scalable near-data-processing accelerators for genome analysis near memory pool with the CXL support","author":"Huangfu W.","year":"2022","unstructured":"Huangfu, W., Malladi, K. T., Chang, A., & Xie, Y. (2022). BEACON: scalable near-data-processing accelerators for genome analysis near memory pool with the CXL support. in Proc. IEEE\/ACM Int. Symp. Microarchitecture (MICRO), 727\u2013743.","journal-title":"in Proc. IEEE\/ACM Int. Symp. Microarchitecture (MICRO)"},{"key":"e_1_3_3_14_1","first-page":"1","article-title":"In-datacenter performance analysis of a tensor processing unit","author":"Jouppi N. P.","year":"2017","unstructured":"Jouppi, N. P., Young, C., Patil, N., Patterson, D., Agrawal, G., Bajwa, R., Bates, S., Bhatia, S., Boden, N., Borchers, A., Boyle, R., Cantin, P.-l., Chao, C., Clark, C., Coriell, J., Daley, M., Dau, M., Dean, J., Gelb, B., \u2026 Yoon, D. H. (2017). In-datacenter performance analysis of a tensor processing unit. in Proc. Int. Symp. Comput. Archit. (ISCA), 1\u201312.","journal-title":"in Proc. Int. Symp. Comput. Archit. (ISCA)"},{"key":"e_1_3_3_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2015.2414456"},{"key":"e_1_3_3_16_1","unstructured":"Kipf T. N. & Welling M. (2016). Semi-supervised classification with graph convolutional networks \u201d arXiv preprintarXiv:1609.02907"},{"key":"e_1_3_3_17_1","first-page":"271","article-title":"Algorithms for VLSI processor arrays","author":"Kung H. T.","year":"1980","unstructured":"Kung, H. T., & Leiserson, C. E. (1980). Algorithms for VLSI processor arrays. in Introduction to VLSI Systems. 271\u2013292.","journal-title":"in Introduction to VLSI Systems"},{"key":"e_1_3_3_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358284"},{"key":"e_1_3_3_19_1","article-title":"PyTorch-BigGraph: A large-scale graph embedding system","author":"Lerer A.","year":"2019","unstructured":"Lerer, A., Wu, L., Shen, J., Lacroix, T., Wehrstedt, L., Bose, A., & Peysakhovich, A. (2019). PyTorch-BigGraph: A large-scale graph embedding system. in Proc. 2nd SysML Conf.","journal-title":"in Proc. 2nd SysML Conf."},{"key":"e_1_3_3_20_1","unstructured":"Leskovec J. & Krevl A. (2014). SNAP datasets: Stanford large network dataset collection. Online]. Available. http:\/\/snap.stanford.edu\/data"},{"key":"e_1_3_3_21_1","first-page":"790","article-title":"RecNMP: accelerating personalized recommendation with near-memory processing","author":"Liu K.","year":"2020","unstructured":"Liu, K., Gupta, U., Cho, B. Y., Brooks, D., Chandra, V., Diril, U., Firoozshahian, A., Hazelwood, K., Jia, B., Lee, H.-S., Li, M., Maher, B., Mudigere, D., Naumov, M., Schatz, M., Smelyanskiy, M., Wang, X., Reagen, B., Wu, C.-J., \u2026 Zhang, X. (2020). RecNMP: accelerating personalized recommendation with near-memory processing. in Proc. Int. Symp. Comput. Archit. (ISCA), 790\u2013803.","journal-title":"in Proc. Int. Symp. Comput. Archit. (ISCA)"},{"key":"e_1_3_3_22_1","first-page":"574","article-title":"Pond: CXL-based memory pooling systems for cloud platforms","author":"Li H.","year":"2023","unstructured":"Li, H., Berger, D. S., Hsu, L., Ernst, D., Zardoshti, P., Novakovic, S., Shah, M., Rajadnya, S., Lee, S., Agarwal, I., Hill, M. D., Fontoura, M., & Bianchini, R. (2023). Pond: CXL-based memory pooling systems for cloud platforms. in Proc. Int. Conf. Archit. Support Program. Lang. Oper. Syst. (ASPLOS). 574\u2013587.","journal-title":"in Proc. Int. Conf. Archit. Support Program. Lang. Oper. Syst. (ASPLOS)"},{"key":"e_1_3_3_23_1","article-title":"Hyper-scale FPGA-as-a-service architecture for large-scale distributed graph neural network","author":"Li S.","year":"2022","unstructured":"Li, S., Niu, D., Wang, Y., Han, W., Zhang, Z., Guan, T., Guan, Y., Liu, H., Huang, L., Du, Z., Xue, F., Fang, Y., Zheng, H., & Xie, Y. (2022). Hyper-scale FPGA-as-a-service architecture for large-scale distributed graph neural network. in Proc. Int. Symp. Comput. Archit. (ISCA)","journal-title":"in Proc. Int. Symp. Comput. Archit. (ISCA)"},{"issue":"1","key":"e_1_3_3_24_1","first-page":"116","article-title":"Near-memory processing in action: accelerating personalized recommendation with AxDIMM","volume":"41","author":"Liu K.","year":"2021","unstructured":"Liu, K., Zhang, X., So, J., Lee, J.-G., Kang, S.-H., Lee, S., Han, S., Cho, Y., Kim, J. H., Kwon, Y., Kim, K., Jung, J., Yun, I., Park, S. J., Park, H., Song, J., Cho, J., Sohn, K., Kim, N. S., & Lee, H.-S. (2021). Near-memory processing in action: accelerating personalized recommendation with AxDIMM. IEEE Micro, 41(1), 116\u2013127.","journal-title":"IEEE Micro"},{"key":"e_1_3_3_25_1","unstructured":"NVIDIA. (2017). \u201cNVIDIA V100 Tensor Core GPU\u201d. [Online]. Available: https:\/\/www.nvidia.com\/en-us\/data-center\/v100\/"},{"key":"e_1_3_3_26_1","unstructured":"NVIDIA. (2020). NVIDIA A100 Tensor Core GPU NVIDIA. [Online]. Available: https:\/\/www.NVIDIA.com\/en-us\/data-center\/a100\/A100Tensor Core GPU"},{"key":"e_1_3_3_27_1","unstructured":"NVIDIA. (2022). NVIDIA H100 Tensor Core GPU. [Online]. Available: https:\/\/www.NVIDIA.com\/en-us\/data-center\/h100\/ H100 Tensor Core GPU"},{"key":"e_1_3_3_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480080"},{"key":"e_1_3_3_29_1","first-page":"970","article-title":"An LPDDR-based CXL-PNM platform for TCO-efficient inference of transformer-based large language models","author":"Park S.-S.","year":"2024","unstructured":"Park, S.-S., Kim, K., So, J., Jung, J., Lee, J., Woo, K., Kim, N., Lee, Y., Kim, H., Kwon, Y., Kim, J., Lee, J., Cho, Y., Tai, Y., Cho, J., Song, H., Ahn, J. H., & Kim, N. S. (2024). An LPDDR-based CXL-PNM platform for TCO-efficient inference of transformer-based large language models. in Proc. IEEE Int. Symp. High-Performance Comput. Archit. (HPCA), 970\u2013982.","journal-title":"in Proc. IEEE Int. Symp. High-Performance Comput. Archit. (HPCA)"},{"key":"e_1_3_3_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3652607"},{"key":"e_1_3_3_31_1","first-page":"338","article-title":"CLAY: CXL-based scalable NDP architecture accelerating embedding layers","author":"Sun Y.","year":"2024","unstructured":"Sun, Y., Nam, H., Kyung, K., Park, J., Kim, B., Kwon, Y., Lee, E., & Ahn, J. H. (2024). CLAY: CXL-based scalable NDP architecture accelerating embedding layers. in Proc. 38th ACM Int. Conf. Supercomputing (ICS), 338\u2013351.","journal-title":"in Proc. 38th ACM Int. Conf. Supercomputing (ICS)"},{"key":"e_1_3_3_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysarc.2022.102602"},{"key":"e_1_3_3_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219869"},{"key":"e_1_3_3_34_1","unstructured":"Wang M. Zheng D. Ye Z. Gan Q. Li M. Song X. Zhou J. Ma C. Yu L. Gai Y. Xiao T. He T. Karypis G. Li J. & Zhang Z. (2019). Deep graph library: A graph-centric highly-performant package for graph neural networks \u201d arXiv preprintarXiv:1909.01315."},{"key":"e_1_3_3_35_1","first-page":"346","article-title":"Session-based recommendation with heterogeneous graph neural networks","author":"Xu L.","year":"2021","unstructured":"Xu, L., Xi, W.-D., & Wang, C.-D. (2021). Session-based recommendation with heterogeneous graph neural networks. in Proc. Int. Joint Conf. Neural Netw. (IJCNN). 346\u2013353.","journal-title":"in Proc. Int. Joint Conf. Neural Netw. (IJCNN)"},{"key":"e_1_3_3_36_1","unstructured":"Xu K. Hu W. Leskovec J. & Jegelka S. (2019). How powerful are graph neural networks. in Int. Conf. Learn. Represent. (ICLR)"},{"key":"e_1_3_3_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219890"},{"key":"e_1_3_3_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2022.3182387"},{"key":"e_1_3_3_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330686"},{"key":"e_1_3_3_40_1","first-page":"36","article-title":"DistDGL: distributed graph neural network training for billion-scale graphs","author":"Zheng D.","year":"2020","unstructured":"Zheng, D., Ma, C., Wang, M., Zhou, J., Su, Q., Song, X., Gan, Q., Zhang, Z., & Karypis, G. (2020). DistDGL: distributed graph neural network training for billion-scale graphs. in Proc. IEEE\/ACM 10th Workshop Irregular Appl. Archit. Algo. (IA), 36\u201344.","journal-title":"in Proc. IEEE\/ACM 10th Workshop Irregular Appl. Archit. Algo. (IA)"},{"key":"e_1_3_3_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3559009.3569670"}],"container-title":["Connection Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.tandfonline.com\/doi\/pdf\/10.1080\/09540091.2026.2650981","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,29]],"date-time":"2026-03-29T05:08:12Z","timestamp":1774760892000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.tandfonline.com\/doi\/full\/10.1080\/09540091.2026.2650981"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,28]]},"references-count":40,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,12,31]]}},"alternative-id":["10.1080\/09540091.2026.2650981"],"URL":"https:\/\/doi.org\/10.1080\/09540091.2026.2650981","relation":{},"ISSN":["0954-0091","1360-0494"],"issn-type":[{"value":"0954-0091","type":"print"},{"value":"1360-0494","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,28]]},"assertion":[{"value":"The publishing and review policy for this title is described in its Aims & Scope.","order":1,"name":"peerreview_statement","label":"Peer Review Statement"},{"value":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=ccos20","URL":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=ccos20","order":2,"name":"aims_and_scope_url","label":"Aim & Scope"},{"value":"2025-09-15","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-02-15","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-03-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}],"article-number":"2650981"}}