{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T09:27:45Z","timestamp":1761989265816,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,9,10]],"date-time":"2018-09-10T00:00:00Z","timestamp":1536537600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key R&D Program of China","award":["2016QY02D0405"],"award-info":[{"award-number":["2016QY02D0405"]}]},{"name":"Youth Innovation Promotion Association CAS","award":["20144310, and 2016102"],"award-info":[{"award-number":["20144310, and 2016102"]}]},{"name":"973 Program of China","award":["2014CB340401"],"award-info":[{"award-number":["2014CB340401"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61773362, 61425016, 61472401, 61722211, 20180290"],"award-info":[{"award-number":["61773362, 61425016, 61472401, 61722211, 20180290"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,9,10]]},"DOI":"10.1145\/3234944.3234978","type":"proceedings-article","created":{"date-parts":[[2018,9,13]],"date-time":"2018-09-13T12:54:52Z","timestamp":1536843292000},"page":"83-90","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["MQGrad"],"prefix":"10.1145","author":[{"given":"Guoxin","family":"Cui","sequence":"first","affiliation":[{"name":"University of Chinese Academy of Sciences &amp; Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences &amp; Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Zeng","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences &amp; Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanyan","family":"Lan","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences &amp; Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiafeng","family":"Guo","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences &amp; Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueqi","family":"Cheng","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences &amp; Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,9,10]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"QSGD: Randomized Quantization for Communication-Optimal Stochastic Gradient Descent. arXiv preprint arXiv:1610.02132","author":"Alistarh Dan","year":"2016","unstructured":"Dan Alistarh , Jerry Li , Ryota Tomioka , and Milan Vojnovic . 2016 . QSGD: Randomized Quantization for Communication-Optimal Stochastic Gradient Descent. arXiv preprint arXiv:1610.02132 (2016). Dan Alistarh, Jerry Li, Ryota Tomioka, and Milan Vojnovic . 2016. QSGD: Randomized Quantization for Communication-Optimal Stochastic Gradient Descent. arXiv preprint arXiv:1610.02132 (2016)."},{"key":"e_1_3_2_1_2_1","unstructured":"Marcin Andrychowicz Misha Denil Sergio Gomez Matthew W Hoffman David Pfau Tom Schaul and Nando de Freitas . 2016. Learning to learn by gradient descent by gradient descent Advances in Neural Information Processing Systems. 3981--3989.   Marcin Andrychowicz Misha Denil Sergio Gomez Matthew W Hoffman David Pfau Tom Schaul and Nando de Freitas . 2016. Learning to learn by gradient descent by gradient descent Advances in Neural Information Processing Systems. 3981--3989."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Christian Daniel Jonathan Taylor and Sebastian Nowozin . 2016. Learning Step Size Controllers for Robust Neural Network Training. AAAI. 1519--1525.   Christian Daniel Jonathan Taylor and Sebastian Nowozin . 2016. Learning Step Size Controllers for Robust Neural Network Training. AAAI. 1519--1525.","DOI":"10.1609\/aaai.v30i1.10187"},{"key":"e_1_3_2_1_4_1","volume-title":"et almbox","author":"Dean Jeffrey","year":"2012","unstructured":"Jeffrey Dean , Greg Corrado , Rajat Monga , Kai Chen , Matthieu Devin , Mark Mao , Andrew Senior , Paul Tucker , Ke Yang , Quoc V Le , et almbox . . 2012 . Large scale distributed deep networks. In Advances in neural information processing systems. 1223--1231. Jeffrey Dean, Greg Corrado, Rajat Monga, Kai Chen, Matthieu Devin, Mark Mao, Andrew Senior, Paul Tucker, Ke Yang, Quoc V Le, et almbox. . 2012. Large scale distributed deep networks. In Advances in neural information processing systems. 1223--1231."},{"key":"e_1_3_2_1_5_1","volume-title":"Deep Q-Networks for Accelerating the Training of Deep Neural Networks. arXiv preprint arXiv:1606.01467","author":"Fu Jie","year":"2016","unstructured":"Jie Fu , Zichuan Lin , Miao Liu , Nicholas Leonard , Jiashi Feng , and Tat-Seng Chua . 2016. Deep Q-Networks for Accelerating the Training of Deep Neural Networks. arXiv preprint arXiv:1606.01467 ( 2016 ). Jie Fu, Zichuan Lin, Miao Liu, Nicholas Leonard, Jiashi Feng, and Tat-Seng Chua . 2016. Deep Q-Networks for Accelerating the Training of Deep Neural Networks. arXiv preprint arXiv:1606.01467 (2016)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2016.30"},{"key":"e_1_3_2_1_7_1","volume-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv:1510.00149","author":"Han Song","year":"2015","unstructured":"Song Han , Huizi Mao , and William J Dally . 2015. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv:1510.00149 ( 2015 ). Song Han, Huizi Mao, and William J Dally . 2015. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv:1510.00149 (2015)."},{"key":"e_1_3_2_1_8_1","unstructured":"Stephen Jos\u00e9 Hanson and Lorien Y Pratt . 1989. Comparing biases for minimal network construction with back-propagation Advances in neural information processing systems. 177--185.   Stephen Jos\u00e9 Hanson and Lorien Y Pratt . 1989. Comparing biases for minimal network construction with back-propagation Advances in neural information processing systems. 177--185."},{"key":"e_1_3_2_1_9_1","unstructured":"Babak Hassibi and David G Stork . 1993. Second order derivatives for network pruning: Optimal brain surgeon Advances in neural information processing systems. 164--171.   Babak Hassibi and David G Stork . 1993. Second order derivatives for network pruning: Optimal brain surgeon Advances in neural information processing systems. 164--171."},{"key":"e_1_3_2_1_10_1","volume-title":"Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing .","author":"Ho Qirong","year":"2013","unstructured":"Qirong Ho , James Cipar , Henggang Cui , Seunghak Lee , Jin Kyu Kim , Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing . 2013 . More effective distributed ml via a stale synchronous parallel parameter server Advances in neural information processing systems. 1223--1231. Qirong Ho, James Cipar, Henggang Cui, Seunghak Lee, Jin Kyu Kim, Phillip B Gibbons, Garth A Gibson, Greg Ganger, and Eric P Xing . 2013. More effective distributed ml via a stale synchronous parallel parameter server Advances in neural information processing systems. 1223--1231."},{"key":"e_1_3_2_1_11_1","unstructured":"Alex Krizhevsky and Geoffrey Hinton . 2009. Learning multiple layers of features from tiny images. (2009).  Alex Krizhevsky and Geoffrey Hinton . 2009. Learning multiple layers of features from tiny images. (2009)."},{"key":"e_1_3_2_1_12_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton . 2012. Imagenet classification with deep convolutional neural networks Advances in neural information processing systems. 1097--1105.   Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton . 2012. Imagenet classification with deep convolutional neural networks Advances in neural information processing systems. 1097--1105."},{"key":"e_1_3_2_1_13_1","unstructured":"Yann LeCun John S Denker and Sara A Solla . 1990. Optimal brain damage. In Advances in neural information processing systems. 598--605.   Yann LeCun John S Denker and Sara A Solla . 1990. Optimal brain damage. In Advances in neural information processing systems. 598--605."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.5555\/2685048.2685095"},{"key":"e_1_3_2_1_15_1","volume-title":"Device Placement Optimization with Reinforcement Learning. arXiv preprint arXiv:1706.04972","author":"Mirhoseini Azalia","year":"2017","unstructured":"Azalia Mirhoseini , Hieu Pham , Quoc V Le , Benoit Steiner , Rasmus Larsen , Yuefeng Zhou , Naveen Kumar , Mohammad Norouzi , Samy Bengio , and Jeff Dean . 2017. Device Placement Optimization with Reinforcement Learning. arXiv preprint arXiv:1706.04972 ( 2017 ). Azalia Mirhoseini, Hieu Pham, Quoc V Le, Benoit Steiner, Rasmus Larsen, Yuefeng Zhou, Naveen Kumar, Mohammad Norouzi, Samy Bengio, and Jeff Dean . 2017. Device Placement Optimization with Reinforcement Learning. arXiv preprint arXiv:1706.04972 (2017)."},{"key":"e_1_3_2_1_16_1","volume-title":"Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602","author":"Mnih Volodymyr","year":"2013","unstructured":"Volodymyr Mnih , Koray Kavukcuoglu , David Silver , Alex Graves , Ioannis Antonoglou , Daan Wierstra , and Martin Riedmiller . 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 ( 2013 ). Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller . 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178365"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-274"},{"key":"e_1_3_2_1_19_1","first-page":"1","article-title":"Phoneme probability estimation with dynamic sparsely connected artificial neural networks","volume":"5","author":"Nikko Str\u00f6m","year":"1997","unstructured":"Nikko Str\u00f6m . 1997 . Phoneme probability estimation with dynamic sparsely connected artificial neural networks . The Free Speech Journal Vol. 5 (1997), 1 -- 41 . Nikko Str\u00f6m . 1997. Phoneme probability estimation with dynamic sparsely connected artificial neural networks. The Free Speech Journal Vol. 5 (1997), 1--41.","journal-title":"The Free Speech Journal"},{"volume-title":"Reinforcement learning: An introduction","author":"Sutton Richard S","key":"e_1_3_2_1_20_1","unstructured":"Richard S Sutton and Andrew G Barto . 1998. Reinforcement learning: An introduction . Vol. Vol. 1 . MIT press Cambridge . Richard S Sutton and Andrew G Barto . 1998. Reinforcement learning: An introduction. Vol. Vol. 1. MIT press Cambridge."},{"key":"e_1_3_2_1_21_1","volume-title":"TernGrad: Ternary Gradients to Reduce Communication in Distributed Deep Learning. arXiv preprint arXiv:1705.07878","author":"Wen Wei","year":"2017","unstructured":"Wei Wen , Cong Xu , Feng Yan , Chunpeng Wu , Yandan Wang , Yiran Chen , and Hai Li . 2017. TernGrad: Ternary Gradients to Reduce Communication in Distributed Deep Learning. arXiv preprint arXiv:1705.07878 ( 2017 ). Wei Wen, Cong Xu, Feng Yan, Chunpeng Wu, Yandan Wang, Yiran Chen, and Hai Li . 2017. TernGrad: Ternary Gradients to Reduce Communication in Distributed Deep Learning. arXiv preprint arXiv:1705.07878 (2017)."},{"key":"e_1_3_2_1_22_1","volume-title":"Reinforcement Learning for Learning Rate Control. arXiv preprint arXiv:1705.11159","author":"Xu Chang","year":"2017","unstructured":"Chang Xu , Tao Qin , Gang Wang , and Tie-Yan Liu . 2017. Reinforcement Learning for Learning Rate Control. arXiv preprint arXiv:1705.11159 ( 2017 ). Chang Xu, Tao Qin, Gang Wang, and Tie-Yan Liu . 2017. Reinforcement Learning for Learning Rate Control. arXiv preprint arXiv:1705.11159 (2017)."},{"key":"e_1_3_2_1_23_1","volume-title":"Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578","author":"Zoph Barret","year":"2016","unstructured":"Barret Zoph and Quoc V Le . 2016. Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578 ( 2016 ). Barret Zoph and Quoc V Le . 2016. Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578 (2016)."}],"event":{"name":"ICTIR '18: The 2018 ACM SIGIR International Conference on the Theory of Information Retrieval","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Tianjin China","acronym":"ICTIR '18"},"container-title":["Proceedings of the 2018 ACM SIGIR International Conference on Theory of Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3234944.3234978","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3234944.3234978","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:08:16Z","timestamp":1750212496000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3234944.3234978"}},"subtitle":["Reinforcement Learning of Gradient Quantization in Parameter Server"],"short-title":[],"issued":{"date-parts":[[2018,9,10]]},"references-count":23,"alternative-id":["10.1145\/3234944.3234978","10.1145\/3234944"],"URL":"https:\/\/doi.org\/10.1145\/3234944.3234978","relation":{},"subject":[],"published":{"date-parts":[[2018,9,10]]},"assertion":[{"value":"2018-09-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}