{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T00:06:50Z","timestamp":1756512410101,"version":"3.44.0"},"reference-count":48,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,6,30]],"date-time":"2024-06-30T00:00:00Z","timestamp":1719705600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,6,30]],"date-time":"2024-06-30T00:00:00Z","timestamp":1719705600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,6,30]]},"DOI":"10.1109\/ijcnn60899.2024.10651013","type":"proceedings-article","created":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T13:35:05Z","timestamp":1725888905000},"page":"1-8","source":"Crossref","is-referenced-by-count":2,"title":["SEVEN: Pruning Transformer Model by Reserving Sentinels"],"prefix":"10.1109","author":[{"given":"Jinying","family":"Xiao","sequence":"first","affiliation":[{"name":"Changsha University of Science and Technology,School of Computer and Communication Engineering,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ping","family":"Li","sequence":"additional","affiliation":[{"name":"Changsha University of Science and Technology,School of Computer and Communication Engineering,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Nie","sequence":"additional","affiliation":[{"name":"Changsha University of Science and Technology,School of Computer and Communication Engineering,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhe","family":"Tang","sequence":"additional","affiliation":[{"name":"Changsha University of Science and Technology,School of Computer and Communication Engineering,Changsha,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proceedings of NAACL-HLT","author":"Kenton"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"issue":"8","key":"ref3","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"article-title":"Using deepspeed and megatron to train megatron-turing nlg 530b, a large-scale generative language model","year":"2022","author":"Smith","key":"ref4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.496"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.107"},{"article-title":"Prune once for all: Sparse pre-trained language models","year":"2021","author":"Zafrir","key":"ref7"},{"key":"ref8","first-page":"15 834","article-title":"The lottery ticket hypothesis for pre-trained bert networks","volume":"33","author":"Chen","year":"2020","journal-title":"Advances in neural information processing systems"},{"article-title":"Snip: Single-shot network pruning based on connection sensitivity","volume-title":"International Conference on Learning Representations","author":"Lee","key":"ref9"},{"article-title":"Picking winning tickets before training by preserving gradient flow","volume-title":"International Conference on Learning Representations","author":"Wang","key":"ref10"},{"key":"ref11","first-page":"26 809","article-title":"Platon: Pruning large transformer models with upper confidence bound of weight importance","volume-title":"International Conference on Machine Learning","author":"Zhang"},{"key":"ref12","first-page":"14 691","article-title":"Instant soup: Cheap pruning ensembles in a single pass can draw lottery tickets from large models","volume-title":"International Conference on Machine Learning","author":"Jaiswal"},{"key":"ref13","first-page":"20 378","article-title":"Movement pruning: Adaptive sparsity by fine-tuning","volume":"33","author":"Sanh","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref14","first-page":"10 367","article-title":"On the noisy gradient descent that generalizes as sgd","volume-title":"International Conference on Machine Learning","author":"Wu"},{"key":"ref15","first-page":"18 347","article-title":"Fishr: Invariant gradient variances for out-of-distribution generalization","volume-title":"International Conference on Machine Learning","author":"Rame"},{"key":"ref16","first-page":"4313","article-title":"Gradient descent with early stopping is provably robust to label noise for overparameterized neural networks","volume-title":"International conference on artificial intelligence and statistics","author":"Li"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00978"},{"key":"ref18","first-page":"15 383","article-title":"Why are adaptive methods good for attention models?","volume":"33","author":"Zhang","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Noise is not the main factor behind the gap between sgd and adam on transformers, but sign descent might be","volume-title":"The Eleventh International Conference on Learning Representations","author":"Kunstner","key":"ref19"},{"key":"ref20","first-page":"15 383","article-title":"Why are adaptive methods good for attention models?","volume":"33","author":"Zhang","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref21","first-page":"129","article-title":"What is the state of neural network pruning?","volume-title":"Proceedings of machine learning and systems","volume":"2","author":"Blalock"},{"article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","year":"2015","author":"Han","key":"ref22"},{"article-title":"The state of sparsity in deep neural networks","year":"2019","author":"Gale","key":"ref23"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.259"},{"key":"ref25","first-page":"5797","article-title":"Analyzing multi-head self-attention: Specialized heads do the heavy lifting, the rest can be pruned","volume-title":"ACL 2019-57th Annual Meeting of the Association for Computational Linguistics, Proceedings of the Conference","author":"Voita"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2021.05.003"},{"article-title":"The topological bert: Transforming attention into topology for natural language processing","year":"2022","author":"Perez","key":"ref27"},{"key":"ref28","article-title":"Are sixteen heads really better than one?","volume":"32","author":"Michel","year":"2019","journal-title":"Advances in neural information processing systems"},{"article-title":"Learning sparse neural networks through 1_0 regularization","volume-title":"International Conference on Learning Representations","author":"Louizos","key":"ref29"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.l803.03635"},{"article-title":"Progressive skeletonization: Trimming more fat from a network at initialization","volume-title":"International Conference on Learning Representations","author":"de Jorge","key":"ref31"},{"article-title":"Lnpt: Label-free network pruning and training","year":"2024","author":"Xiao","key":"ref32"},{"article-title":"Debertav3: Improving deberta using electra-style pre-training with gradient-disentangled embedding sharing","volume-title":"The Eleventh International Conference on Learning Representations","author":"He","key":"ref33"},{"key":"ref34","first-page":"21 285","article-title":"Towards theoretically understanding why sgd generalizes better than adam in deep learning","volume":"33","author":"Zhou","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref35","first-page":"7654","article-title":"The anisotropic noise in stochastic gradient descent: Its behavior of escaping from sharp minima and regularization effects","volume-title":"International Conference on Machine Learning","author":"Zhu"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ITA.2018.8503224"},{"key":"ref37","first-page":"5827","article-title":"A tail-index analysis of stochastic gradient noise in deep neural networks","volume-title":"International Conference on Machine Learning","author":"Simsekli"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1088\/1742-5468\/ac841d"},{"key":"ref39","first-page":"15 042","article-title":"Stochastic optimization with heavy-tailed noise via accelerated gradient clipping","volume":"33","author":"Gorbunov","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Adam: a method for stochastic optimization","volume-title":"Int Conf Learn Represent","author":"Kingma","key":"ref40"},{"key":"ref41","first-page":"6377","article-title":"Pruning neural networks without any data by iteratively conserving synaptic flow","volume":"33","author":"Tanaka","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.11929"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"ref44","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i9.26297"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.repl4nlp-1.18"},{"article-title":"Symbolic discovery of optimization algorithms","year":"2023","author":"Chen","key":"ref48"}],"event":{"name":"2024 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2024,6,30]]},"location":"Yokohama, Japan","end":{"date-parts":[[2024,7,5]]}},"container-title":["2024 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10649807\/10649898\/10651013.pdf?arnumber=10651013","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T17:38:39Z","timestamp":1756489119000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10651013\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,30]]},"references-count":48,"URL":"https:\/\/doi.org\/10.1109\/ijcnn60899.2024.10651013","relation":{},"subject":[],"published":{"date-parts":[[2024,6,30]]}}}