{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T16:44:33Z","timestamp":1773938673475,"version":"3.50.1"},"reference-count":116,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2018AAA0100701"],"award-info":[{"award-number":["2018AAA0100701"]}]},{"name":"Guoqiang Institute"},{"DOI":"10.13039\/501100004147","name":"Tsinghua University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004147","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2022,10,1]]},"DOI":"10.1109\/tpami.2021.3091717","type":"journal-article","created":{"date-parts":[[2021,6,23]],"date-time":"2021-06-23T19:51:45Z","timestamp":1624477905000},"page":"7167-7174","source":"Crossref","is-referenced-by-count":13,"title":["Recent Advances in Large Margin Learning"],"prefix":"10.1109","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0709-4877","authenticated-orcid":false,"given":"Yiwen","family":"Guo","sequence":"first","affiliation":[{"name":"ByteDance AI Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8088-367X","authenticated-orcid":false,"given":"Changshui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute for Artificial Intelligence, Tsinghua University (THUAI), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022627411411"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/72.788640"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4757-3264-1"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1162\/15324430260185628"},{"key":"ref5","first-page":"1453","article-title":"Large margin methods for structured and interdependent output variables","volume":"6","author":"Tsochantaridis","year":"2005","journal-title":"J. Mach. Learn. Res."},{"key":"ref6","first-page":"49","article-title":"1-norm support vector machines","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhu"},{"key":"ref7","first-page":"266","article-title":"Learning optimally sparse support vector machines","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Cotter"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/1113.001.0001"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511809682"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/1130.003.0007"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1162\/089976600300015565"},{"key":"ref14","first-page":"1485","article-title":"Robustness and regularization of support vector machines","volume":"10","author":"Xu","year":"2009","journal-title":"J. Mach. Learn. Res."},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-011-5268-1"},{"key":"ref16","article-title":"Intriguing properties of neural networks","volume-title":"Proc. 2nd Int. Conf. Learn. Representations","author":"Szegedy"},{"key":"ref17","article-title":"Explaining and harnessing adversarial examples","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Goodfellow"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.282"},{"key":"ref19","first-page":"1","article-title":"Adversarial machine learning at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kurakin"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.153"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1017\/9781009025096.003"},{"key":"ref22","first-page":"1567","article-title":"Learning imbalanced datasets with label-distribution-aware margin loss","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Cao"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2017.2708039"},{"key":"ref24","first-page":"1094","article-title":"Generalization error of invariant classifiers","volume-title":"Proc. 20th Int. Conf. Artif. Intell. Statist.","author":"Sokolic"},{"key":"ref25","first-page":"1333","article-title":"Discriminative robust transformation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Huang"},{"issue":"1","key":"ref26","first-page":"2822","article-title":"The implicit bias of gradient descent on separable data","volume":"19","author":"Soudry","year":"2018","journal-title":"J. Mach. Learn. Res."},{"key":"ref27","first-page":"9461","article-title":"Implicit bias of gradient descent on linear convolutional networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gunasekar"},{"key":"ref28","article-title":"Gradient descent aligns the layers of deep linear networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ji"},{"key":"ref29","first-page":"3420","article-title":"Convergence of gradient descent on separable data","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Nacson"},{"key":"ref30","first-page":"1772","article-title":"The implicit bias of gradient descent on nonseparable data","volume-title":"Proc. 32nd Conf. Learn. Theory","author":"Ji"},{"key":"ref31","article-title":"Regularization matters: Generalization and optimization of neural nets v.s. their induced kernel","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wei"},{"key":"ref32","first-page":"2422","article-title":"SGD learns the conjugate kernel class of the network","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Daniely"},{"key":"ref33","first-page":"8571","article-title":"Neural tangent kernel: Convergence and generalization in neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jacot"},{"key":"ref34","first-page":"8141","article-title":"On exact computation with an infinitely wide neural net","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Arora"},{"key":"ref35","first-page":"419","article-title":"Deep defense: Training DNNs with improved adversarial robustness","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yan"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2948348"},{"key":"ref37","first-page":"850","article-title":"Large margin deep networks for classification","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Elsayed"},{"key":"ref38","article-title":"Evaluating robustness of neural networks with mixed integer programming","author":"Tjeng","year":"2017"},{"key":"ref39","first-page":"5283","article-title":"Provable defenses against adversarial examples via the convex outer adversarial polytope","volume-title":"Proc. 35 th Int. Conf. Mach. Learn.","author":"Wong"},{"key":"ref40","first-page":"1310","article-title":"Certified adversarial robustness via randomized smoothing","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Cohen"},{"key":"ref41","first-page":"507","article-title":"Large-margin softmax loss for convolutional neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liu"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2018.2822810"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00482"},{"key":"ref44","first-page":"950","article-title":"A simple weight decay can improve generalization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Krogh"},{"key":"ref45","article-title":"Improving neural networks by preventing co-adaptation of feature detectors","author":"Hinton","year":"2012"},{"key":"ref46","first-page":"2266","article-title":"Formal guarantees on the robustness of a classifier against adversarial manipulation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hein"},{"key":"ref47","article-title":"Evaluating the robustness of neural networks: An extreme value theory approach","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Weng"},{"key":"ref48","first-page":"6240","article-title":"Spectrally-normalized margin bounds for neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Bartlett"},{"key":"ref49","first-page":"5947","article-title":"Exploring generalization in deep learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Neyshabur"},{"key":"ref50","first-page":"9722","article-title":"Data-dependent sample complexity of deep neural networks via Lipschitz augmentation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wei"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01258-8_32"},{"key":"ref52","first-page":"1","article-title":"Sensitivity and generalization in neural networks: An empirical study","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Novak"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11504"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00929"},{"key":"ref55","article-title":"Spectral norm regularization for improving the generalizability of deep learning","author":"Yoshida","year":"2017"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3006917"},{"key":"ref57","article-title":"Deep learning using linear support vector machines","volume-title":"Proc. Int. Conf. Mach. Learn. Workshop","author":"Tang"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-42051-1_16"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10243"},{"key":"ref60","first-page":"207","article-title":"Distance metric learning for large margin nearest neighbor classification","volume":"10","author":"Weinberger","year":"2009","journal-title":"J. Mach. Learn. Res."},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.713"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/138"},{"key":"ref64","first-page":"1942","article-title":"Virtual class enhanced discriminative embedding learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00552"},{"key":"ref66","first-page":"139","article-title":"Large margin in softmax cross-entropy loss","volume-title":"Proc. Brit. Mach. Vis. Conf.","author":"Kobayashi"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178777"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472806"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1680"},{"key":"ref70","article-title":"Neural machine translation by jointly learning to align and translate","author":"Bahdanau","year":"2014","journal-title":"arXiv:1409.0473"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1017\/9781108924238.008"},{"key":"ref72","article-title":"Large margin few-shot learning","author":"Wang","year":"2018"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1145\/1401890.1401920"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11698"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/419"},{"key":"ref76","article-title":"Support vector guided softmax loss for face recognition","author":"Wang","year":"2018"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.89"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.324"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11651"},{"key":"ref80","article-title":"SupportNet: Solving catastrophic forgetting in class incremental learning with support data","author":"Li","year":"2018"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00665"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.289"},{"key":"ref83","article-title":"A PAC-Bayesian approach to spectrally-normalized margin bounds for neural networks","author":"Neyshabur","year":"2017"},{"key":"ref84","article-title":"Improved sample complexities for deep networks and robust classification via an all-layer margin","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wei"},{"key":"ref85","article-title":"Minnorm training: An algorithm for training overcomplete deep neural networks","author":"Bansal","year":"2018"},{"key":"ref86","article-title":"Max-margin adversarial (MMA) training: Direct input space margin maximization through adversarial training","author":"Ding","year":"2019"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.06083"},{"key":"ref88","article-title":"Understanding adversarial robustness: The trade-off between minimum and average margin","author":"Wu","year":"2019"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2019.2897662"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.17"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.49"},{"key":"ref92","first-page":"242","article-title":"Sparse DNNs with improved adversarial robustness","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Guo"},{"key":"ref93","article-title":"Confidence-calibrated adversarial training: Towards robust models generalizing beyond the attack used during training","author":"Stutz","year":"2019"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-68167-2_18"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-63387-9_5"},{"key":"ref96","first-page":"550","article-title":"A dual approach to scalable verification of deep networks","volume-title":"Proc. Conf. Uncertainty Artif. Intell.","author":"Dvijotham"},{"key":"ref97","first-page":"10 802","article-title":"Fast and effective robustness certification","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Singh"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013240"},{"key":"ref99","first-page":"4939","article-title":"Efficient neural network robustness certification with general activation functions","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref100","first-page":"1","article-title":"Fast geometric projections for local robustness certification","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Fromherz"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref102","first-page":"1","article-title":"Towards robust, locally linear deep networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lee"},{"key":"ref103","first-page":"2057","article-title":"Provable robustness of ReLU networks via maximization of linear regions","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Croce"},{"key":"ref104","first-page":"6541","article-title":"Lipschitz-margin training: Scalable certification of perturbation invariance for deep neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Tsuzuku"},{"key":"ref105","first-page":"5321","article-title":"Does data augmentation lead to positive margin?","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Rajput"},{"key":"ref106","article-title":"Convergence and margin of adversarial training on separable data","author":"Charles","year":"2019"},{"key":"ref107","article-title":"Improved regularization of convolutional neural networks with cutout","author":"DeVries","year":"2017"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4899-7687-1_79"},{"key":"ref109","first-page":"1135","article-title":"Learning both weights and connections for efficient neural network","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Han"},{"key":"ref110","first-page":"1379","article-title":"Dynamic network surgery for efficient DNNs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Guo"},{"key":"ref111","first-page":"2074","article-title":"Learning structured sparsity in deep neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wen"},{"key":"ref112","article-title":"Binarized neural networks: Training deep neural networks with weights and activations constrained to +1 or \u22121","author":"Courbariaux","year":"2016"},{"key":"ref113","first-page":"1","article-title":"Incremental network quantization: Towards lossless CNNs with low-precision weights","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhou"},{"key":"ref114","first-page":"1","article-title":"DSD: Dense-sparse-dense training for deep neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Han"},{"key":"ref115","article-title":"Attacking binarized neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Galloway"},{"key":"ref116","article-title":"Predicting the generalization gap in deep networks with margin distributions","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jiang"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/9893033\/09463731.pdf?arnumber=9463731","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,11]],"date-time":"2024-01-11T23:43:07Z","timestamp":1705016587000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9463731\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,1]]},"references-count":116,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2021.3091717","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10,1]]}}}