{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:26:49Z","timestamp":1763202409265,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,8,17]],"date-time":"2020-08-17T00:00:00Z","timestamp":1597622400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the Beijing Municipal Natural Science Foundation","award":["4182041"],"award-info":[{"award-number":["4182041"]}]},{"name":"BUPT Excellent Ph.D. Students Foundation","award":["CX2019130"],"award-info":[{"award-number":["CX2019130"]}]},{"name":"the National Natural Science Foundation of China","award":["61671079 and 61771068"],"award-info":[{"award-number":["61671079 and 61771068"]}]},{"name":"the Ministry of Education and China Mobile Joint Fund","award":["MCM20180101"],"award-info":[{"award-number":["MCM20180101"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,8,17]]},"DOI":"10.1145\/3404397.3404425","type":"proceedings-article","created":{"date-parts":[[2020,8,9]],"date-time":"2020-08-09T03:54:26Z","timestamp":1596945266000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["DeepHop on Edge: Hop-by-hop Routing byDistributed Learning with Semantic Attention"],"prefix":"10.1145","author":[{"given":"Bo","family":"He","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingyu","family":"Wang","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Qi","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haifeng","family":"Sun","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zirui","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cong","family":"Liu","sequence":"additional","affiliation":[{"name":"China Mobile Research Institute, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianxin","family":"Liao","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,8,17]]},"reference":[{"volume-title":"Innovations in multi-agent systems and applications-1","author":"Bu\u015foniu L.","key":"e_1_3_2_1_1_1","unstructured":"L. Bu\u015foniu , R. Babu\u0161ka , and B. De\u00a0Schutter . 2010. Multi-agent reinforcement learning: An overview . In Innovations in multi-agent systems and applications-1 . Springer , 183\u2013221. L. Bu\u015foniu, R. Babu\u0161ka, and B. De\u00a0Schutter. 2010. Multi-agent reinforcement learning: An overview. In Innovations in multi-agent systems and applications-1. Springer, 183\u2013221."},{"key":"e_1_3_2_1_2_1","unstructured":"S. Chaudhari G. Polatkan R. Ramanath and V. Mithal. 2019. An attentive survey of attention models. arXiv preprint arXiv:1904.02874(2019).  S. Chaudhari G. Polatkan R. Ramanath and V. Mithal. 2019. An attentive survey of attention models. arXiv preprint arXiv:1904.02874(2019)."},{"key":"e_1_3_2_1_3_1","unstructured":"X. Chu. 2018. Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization. arXiv preprint arXiv:1807.00442(2018).  X. Chu. 2018. Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization. arXiv preprint arXiv:1807.00442(2018)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.2019.1800644"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jnca.2018.05.001"},{"key":"e_1_3_2_1_6_1","first-page":"1039","article-title":"Nash Q-learning for general-sum stochastic games","author":"Hu J.","year":"2003","unstructured":"J. Hu and M.\u00a0 P. Wellman . 2003 . Nash Q-learning for general-sum stochastic games . Journal of machine learning research 4 , Nov (2003), 1039 \u2013 1069 . J. Hu and M.\u00a0P. Wellman. 2003. Nash Q-learning for general-sum stochastic games. Journal of machine learning research 4, Nov (2003), 1039\u20131069.","journal-title":"Journal of machine learning research 4"},{"key":"e_1_3_2_1_7_1","unstructured":"B. Hubert J. Geul and S. S\u00e9hier. [n.d.]. The Wonder Shaper. https:\/\/github.com\/magnific0\/wondershaper.  B. Hubert J. Geul and S. S\u00e9hier. [n.d.]. The Wonder Shaper. https:\/\/github.com\/magnific0\/wondershaper."},{"key":"e_1_3_2_1_8_1","unstructured":"J. Jiang C. Dun and Z. Lu. 2018. Graph Convolutional Reinforcement Learning for Multi-Agent Cooperation. CoRR abs\/1810.09202(2018).  J. Jiang C. Dun and Z. Lu. 2018. Graph Convolutional Reinforcement Learning for Multi-Agent Cooperation. CoRR abs\/1810.09202(2018)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2011.111002"},{"key":"e_1_3_2_1_10_1","unstructured":"S.\u00a0M. Kakade. 2002. A natural policy gradient. In Advances in neural information processing systems. 1531\u20131538.  S.\u00a0M. Kakade. 2002. A natural policy gradient. In Advances in neural information processing systems. 1531\u20131538."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2018.2856587"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1137\/S0363012901385691"},{"key":"e_1_3_2_1_13_1","unstructured":"A.\u00a0H. Lashkari G. Draper-Gil M.\u00a0S.\u00a0I. Mamun and A.\u00a0A. Ghorbani. 2017. Characterization of Tor Traffic using Time based Features.. In ICISSP. 253\u2013262.  A.\u00a0H. Lashkari G. Draper-Gil M.\u00a0S.\u00a0I. Mamun and A.\u00a0A. Ghorbani. 2017. Characterization of Tor Traffic using Time based Features.. In ICISSP. 253\u2013262."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.comnet.2019.01.036"},{"volume-title":"4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings.","author":"Lillicrap P.","key":"e_1_3_2_1_15_1","unstructured":"T.\u00a0 P. Lillicrap , J.\u00a0 J. Hunt , A. Pritzel , N. Heess , T. Erez , Y. Tassa , D. Silver , and D. Wierstra . 2016. Continuous control with deep reinforcement learning . In 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings. T.\u00a0P. Lillicrap, J.\u00a0J. Hunt, A. Pritzel, N. Heess, T. Erez, Y. Tassa, D. Silver, and D. Wierstra. 2016. Continuous control with deep reinforcement learning. In 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings."},{"volume-title":"5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings.","author":"Lin Z.","key":"e_1_3_2_1_16_1","unstructured":"Z. Lin , M. Feng , C.\u00a0 N. Santos , M. Yu , B. Xiang , B. Zhou , and Y. Bengio . 2017. A structured self-attentive sentence embedding . In 5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings. Z. Lin, M. Feng, C.\u00a0N. Santos, M. Yu, B. Xiang, B. Zhou, and Y. Bengio. 2017. A structured self-attentive sentence embedding. In 5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings."},{"key":"e_1_3_2_1_17_1","unstructured":"R. Lowe Y. Wu A. Tamar J. Harb O.\u00a0P. Abbeel and I. Mordatch. 2017. Multi-agent actor-critic for mixed cooperative-competitive environments. In Advances in Neural Information Processing Systems. 6379\u20136390.  R. Lowe Y. Wu A. Tamar J. Harb O.\u00a0P. Abbeel and I. Mordatch. 2017. Multi-agent actor-critic for mixed cooperative-competitive environments. In Advances in Neural Information Processing Systems. 6379\u20136390."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2014.2349905"},{"volume-title":"International conference on machine learning. 1928\u20131937","author":"Mnih V.","key":"e_1_3_2_1_19_1","unstructured":"V. Mnih , A.\u00a0 P. Badia , M. Mirza , A. Graves , T. Lillicrap , T. Harley , D. Silver , and K. Kavukcuoglu . 2016. Asynchronous methods for deep reinforcement learning . In International conference on machine learning. 1928\u20131937 . V. Mnih, A.\u00a0P. Badia, M. Mirza, A. Graves, T. Lillicrap, T. Harley, D. Silver, and K. Kavukcuoglu. 2016. Asynchronous methods for deep reinforcement learning. In International conference on machine learning. 1928\u20131937."},{"key":"e_1_3_2_1_20_1","volume-title":"Human-level control through deep reinforcement learning. Nature 518, 7540","author":"Mnih V.","year":"2015","unstructured":"V. Mnih , K. Kavukcuoglu , D. Silver , A.\u00a0 A. Rusu , J. Veness , M.\u00a0 G. Bellemare , A. Graves , M. Riedmiller , A.\u00a0 K. Fidjeland , G. Ostrovski , 2015. Human-level control through deep reinforcement learning. Nature 518, 7540 ( 2015 ), 529. V. Mnih, K. Kavukcuoglu, D. Silver, A.\u00a0A. Rusu, J. Veness, M.\u00a0G. Bellemare, A. Graves, M. Riedmiller, A.\u00a0K. Fidjeland, G. Ostrovski, 2015. Human-level control through deep reinforcement learning. Nature 518, 7540 (2015), 529."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2017.08.058"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"J. Nash. 1951. Non-cooperative games. Annals of mathematics(1951) 286\u2013295.  J. Nash. 1951. Non-cooperative games. Annals of mathematics(1951) 286\u2013295.","DOI":"10.2307\/1969529"},{"volume-title":"Joint Service Placement and Request Routing in Multi-cell Mobile Edge Computing Networks. In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 10\u201318","author":"Poularakis K.","key":"e_1_3_2_1_23_1","unstructured":"K. Poularakis , J. Llorca , A.\u00a0 M. Tulino , I. Taylor , and L. Tassiulas . 2019 . Joint Service Placement and Request Routing in Multi-cell Mobile Edge Computing Networks. In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 10\u201318 . K. Poularakis, J. Llorca, A.\u00a0M. Tulino, I. Taylor, and L. Tassiulas. 2019. Joint Service Placement and Request Routing in Multi-cell Mobile Edge Computing Networks. In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 10\u201318."},{"volume-title":"International Conference on Machine Learning. 1889\u20131897","author":"Schulman J.","key":"e_1_3_2_1_24_1","unstructured":"J. Schulman , S. Levine , P. Abbeel , M. Jordan , and P. Moritz . 2015. Trust region policy optimization . In International Conference on Machine Learning. 1889\u20131897 . J. Schulman, S. Levine, P. Abbeel, M. Jordan, and P. Moritz. 2015. Trust region policy optimization. In International Conference on Machine Learning. 1889\u20131897."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2865661"},{"key":"e_1_3_2_1_26_1","unstructured":"A. Vaswani N. Shazeer N. Parmar J. Uszkoreit L. Jones A.\u00a0N. Gomez \u0141u. Kaiser and I. Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998\u20136008.  A. Vaswani N. Shazeer N. Parmar J. Uszkoreit L. Jones A.\u00a0N. Gomez \u0141u. Kaiser and I. Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998\u20136008."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2018.2880754"},{"volume-title":"IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 19\u201327","author":"Wang W.","key":"e_1_3_2_1_28_1","unstructured":"W. Wang , X. Liu , Y. Yao , Y. Pan , Z. Chi , and T. Zhu . 2019. CRF: Coexistent Routing and Flooding using WiFi Packets in Heterogeneous IoT Networks . In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 19\u201327 . W. Wang, X. Liu, Y. Yao, Y. Pan, Z. Chi, and T. Zhu. 2019. CRF: Coexistent Routing and Flooding using WiFi Packets in Heterogeneous IoT Networks. In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 19\u201327."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2019.02.050"},{"key":"e_1_3_2_1_30_1","unstructured":"Y. Wu E. Mansimov R.\u00a0B Grosse S. Liao and J. Ba. 2017. Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation. In Advances in neural information processing systems. 5279\u20135288.  Y. Wu E. Mansimov R.\u00a0B Grosse S. Liao and J. Ba. 2017. Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation. In Advances in neural information processing systems. 5279\u20135288."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8485853"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2019.2904358"},{"volume-title":"Proceedings of the 35th International Conference on Machine Learning, Stockholmsm\u00e4ssan","author":"Yang Y.","key":"e_1_3_2_1_33_1","unstructured":"Y. Yang , R. Luo , M. Li , M. Zhou , W. Zhang , and J. Wang . 2018. Mean field multi-agent reinforcement learning . In Proceedings of the 35th International Conference on Machine Learning, Stockholmsm\u00e4ssan , Stockholm, Sweden, July 10-15. 5567\u20135576. Y. Yang, R. Luo, M. Li, M. Zhou, W. Zhang, and J. Wang. 2018. Mean field multi-agent reinforcement learning. In Proceedings of the 35th International Conference on Machine Learning, Stockholmsm\u00e4ssan, Stockholm, Sweden, July 10-15. 5567\u20135576."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2018.2812868"},{"volume-title":"IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 1648\u20131656","author":"Zhang H.","key":"e_1_3_2_1_35_1","unstructured":"H. Zhang , W. Li , S. Gao , X. Wang , and B. Ye . 2019. ReLeS: A Neural Adaptive Multipath Scheduler based on Deep Reinforcement Learning . In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 1648\u20131656 . H. Zhang, W. Li, S. Gao, X. Wang, and B. Ye. 2019. ReLeS: A Neural Adaptive Multipath Scheduler based on Deep Reinforcement Learning. In IEEE INFOCOM 2019-IEEE Conference on Computer Communications. IEEE, 1648\u20131656."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2018.2889880"}],"event":{"name":"ICPP '20: 49th International Conference on Parallel Processing","acronym":"ICPP '20","location":"Edmonton AB Canada"},"container-title":["49th International Conference on Parallel Processing - ICPP"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3404397.3404425","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3404397.3404425","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:31:42Z","timestamp":1750195902000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3404397.3404425"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,8,17]]},"references-count":36,"alternative-id":["10.1145\/3404397.3404425","10.1145\/3404397"],"URL":"https:\/\/doi.org\/10.1145\/3404397.3404425","relation":{},"subject":[],"published":{"date-parts":[[2020,8,17]]},"assertion":[{"value":"2020-08-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}