{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T20:09:17Z","timestamp":1778789357763,"version":"3.51.4"},"reference-count":79,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2020,2,1]],"date-time":"2020-02-01T00:00:00Z","timestamp":1580515200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,2,1]],"date-time":"2020-02-01T00:00:00Z","timestamp":1580515200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,2,1]],"date-time":"2020-02-01T00:00:00Z","timestamp":1580515200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61673028"],"award-info":[{"award-number":["61673028"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Knowl. Data Eng."],"published-print":{"date-parts":[[2020,2,1]]},"DOI":"10.1109\/tkde.2018.2883613","type":"journal-article","created":{"date-parts":[[2018,11,27]],"date-time":"2018-11-27T20:01:56Z","timestamp":1543348916000},"page":"374-387","source":"Crossref","is-referenced-by-count":14,"title":["Training Simplification and Model Simplification for Deep Learning : A Minimal Effort Back Propagation Method"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8241-9320","authenticated-orcid":false,"given":"Xu","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6994-2114","authenticated-orcid":false,"given":"Xuancheng","family":"Ren","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuming","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8864-6153","authenticated-orcid":false,"given":"Bingzhen","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5763-8369","authenticated-orcid":false,"given":"Jingjing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Houfeng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-2053"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2009.5459469"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248110"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1162\/NECO_a_00052"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1008"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1013"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1027"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1088\/0954-898X\/7\/2\/014"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-2060"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(90)90006-7"},{"key":"ref78","first-page":"3260","article-title":"Deconvolution-based global decoding for neural machine translation","author":"lin","year":"2018","journal-title":"Proc 27th Int Conf Comput Linguistics"},{"key":"ref79","first-page":"1","article-title":"Towards neural phrase-based machine translation","author":"huang","year":"2018","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref33","article-title":"Google's neural machine translation system: Bridging the gap between human and machine translation","volume":"abs 1609 8144","author":"wu","year":"2016","journal-title":"CoRR"},{"key":"ref32","first-page":"1019","article-title":"A theoretically grounded application of dropout in recurrent neural networks","author":"gal","year":"2016","journal-title":"Proc Conf Neural Inf Process Syst"},{"key":"ref31","first-page":"2","article-title":"The IWSLT 2015 evaluation campaign","author":"cettolo","year":"2015","journal-title":"Proc Int Workshop Spoken Lang Translation"},{"key":"ref30","first-page":"3302","article-title":"Lattice-based recurrent neural network encoders for neural machine translation","author":"su","year":"2017","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICNN.1993.298623"},{"key":"ref36","first-page":"1","article-title":"DSD: Regularizing deep neural networks with dense-sparse-dense training flow","author":"han","year":"2017","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref35","first-page":"1","article-title":"Exploring sparsity in recurrent neural networks","author":"narang","year":"2017","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref34","first-page":"311","article-title":"BLEU: A method for automatic evaluation of machine translation","author":"papineni","year":"2002","journal-title":"Proc Annual Meeting of the Assoc Computational Linguistics"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1004"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.3115\/1599081.1599150"},{"key":"ref61","first-page":"46","article-title":"Task-oriented evaluation of syntactic parsers and their representations","author":"miyao","year":"2008","journal-title":"Proc Assoc Comput Linguistics"},{"key":"ref63","first-page":"149","article-title":"An efficient algorithm for projective dependency parsing","author":"nivre","year":"2003","journal-title":"Proc Int Workshop Parsing Technol"},{"key":"ref28","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","volume":"37","author":"ioffe","year":"2015","journal-title":"Proc Int Conf Int Conf Mach Learn"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1017\/S1351324906004505"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2003.1227801"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-2006"},{"key":"ref66","first-page":"1077","article-title":"Dynamic programming for linear-time incremental parsing","author":"huang","year":"2010","journal-title":"Proc Annual Meeting of the Assoc Computational Linguistics"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1166"},{"key":"ref67","first-page":"562","article-title":"A tale of two parsers: Investigating and combining graph-based and transition-based dependency parsing","author":"zhang","year":"2008","journal-title":"Proc Conf Empirical Methods Natural Language Process"},{"key":"ref68","first-page":"1391","article-title":"Analyzing the effect of global learning and beam-search on transition-based dependency parsing","author":"zhang","year":"2012","journal-title":"Proc 24th Int Conf Comput Linguistics"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00037"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/BFb0006203"},{"key":"ref1","first-page":"3299","article-title":"meProp: Sparsified back propagation for accelerated deep learning with reduced overfitting","author":"sun","year":"2017","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref20","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"J Mach Learn Res"},{"key":"ref22","article-title":"Distilling the knowledge in a neural network","author":"hinton","year":"2015"},{"key":"ref21","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"J Mach Learn Res"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/MLHPC.2016.004"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2005.251"},{"key":"ref26","first-page":"92","article-title":"Evaluation of pooling operations in convolutional architectures for object recognition","volume":"6354","author":"scherer","year":"2010","journal-title":"Proc Int Conf Artif Neural Netw"},{"key":"ref25","first-page":"1237","article-title":"Flexible, high performance convolutional neural networks for image classification","author":"ciresan","year":"2011","journal-title":"Proc 22nd Int Joint Conf Artif Intell"},{"key":"ref50","first-page":"1","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","author":"han","year":"2016","journal-title":"CoRR"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3115\/1117794.1117802"},{"key":"ref59","first-page":"238","article-title":"Learning with lookahead: Can history-based models rival globally optimized models?","author":"tsuruoka","year":"2011","journal-title":"Proc 15th Conf Computational Natural Language Learning"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.3115\/1687878.1687946"},{"key":"ref57","first-page":"382","article-title":"Developing a robust part-of-speech tagger for biomedical text","author":"tsuruoka","year":"2005","journal-title":"Proc 10th WWW Conf"},{"key":"ref56","first-page":"192","article-title":"Asynchronous parallel learning for neural networks and structured models with dense features","author":"sun","year":"2016","journal-title":"Proc 26th Int Conf Comput Linguistics"},{"key":"ref55","first-page":"2402","article-title":"Structure regularization for structured prediction","author":"sun","year":"2014","journal-title":"Proc 27th Int Conf Neural Inf Process Syst"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K15-1036"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.3115\/1687878.1687947"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.3115\/1073445.1073478"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"ref40","first-page":"1137","article-title":"Efficient learning of sparse representations with an energy-based model","author":"ranzato","year":"2006","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1082"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21236\/ADA273556"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/78.650093"},{"key":"ref15","first-page":"760","article-title":"Guided learning for bidirectional sequence classification","author":"shen","year":"2007","journal-title":"Proc 45th Annu Meeting Assoc Comput Linguistics"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.3115\/1118693.1118694"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"947","DOI":"10.1038\/35016072","article-title":"Digital selection and analogue amplification coexist in a cortex-inspired silicon circuit","volume":"405","author":"hahnloser","year":"2000","journal-title":"Nature"},{"key":"ref18","first-page":"807","article-title":"Rectified linear units improve restricted boltzmann machines","author":"nair","year":"2010","journal-title":"Proc 27th Int Conf Int Conf Mach Learn"},{"key":"ref19","first-page":"1","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"CoRR"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2014.09.003"},{"key":"ref3","doi-asserted-by":"crossref","DOI":"10.1038\/323533a0","article-title":"Learning representations by back-propagating errors","volume":"323","author":"rumelhart","year":"1986","journal-title":"Nature"},{"key":"ref6","first-page":"1","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2015","journal-title":"CoRR"},{"key":"ref5","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"2014","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref8","first-page":"6000","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc Conf Neural Inf Process Syst"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref49","first-page":"1135","article-title":"Learning both weights and connections for efficient neural network","author":"han","year":"2015","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref46","first-page":"1","article-title":"Deep gradient compression: Reducing the communication bandwidth for distributed training","author":"lin","year":"2018","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1045"},{"key":"ref48","first-page":"164","article-title":"Second order derivatives for network pruning: Optimal brain surgeon","author":"hassibi","year":"1992","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref47","first-page":"598","article-title":"Optimal brain damage","author":"lecun","year":"1989","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref42","first-page":"1","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","author":"shazeer","year":"2017","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P15-1001"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-354"},{"key":"ref43","first-page":"1058","article-title":"1-bit stochastic gradient descent and its application to data-parallel distributed training of speech dnns","author":"seide","year":"2014","journal-title":"Proc Annu Conf Int Speech Commun Assoc"}],"container-title":["IEEE Transactions on Knowledge and Data Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/69\/8956008\/08546786.pdf?arnumber=8546786","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T14:41:37Z","timestamp":1651070497000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8546786\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,2,1]]},"references-count":79,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tkde.2018.2883613","relation":{},"ISSN":["1041-4347","1558-2191","2326-3865"],"issn-type":[{"value":"1041-4347","type":"print"},{"value":"1558-2191","type":"electronic"},{"value":"2326-3865","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,2,1]]}}}