{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T03:41:01Z","timestamp":1743133261053,"version":"3.40.3"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030863395"},{"type":"electronic","value":"9783030863401"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-86340-1_18","type":"book-chapter","created":{"date-parts":[[2021,9,10]],"date-time":"2021-09-10T12:03:14Z","timestamp":1631275394000},"page":"222-234","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Growing Neural Networks Achieve Flatter Minima"],"prefix":"10.1007","author":[{"given":"Paul","family":"Caillon","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Christophe","family":"Cerisara","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,7]]},"reference":[{"issue":"1","key":"18_CR1","first-page":"115","volume":"14","author":"AR Barron","year":"1994","unstructured":"Barron, A.R.: Approximation and estimation bounds for artificial neural networks. Mach. Learn. 14(1), 115\u2013133 (1994)","journal-title":"Mach. Learn."},{"key":"18_CR2","unstructured":"Chaudhari, P., et al.: Entropy-SGD: biasing gradient descent into wide valleys. arXiv e-prints arXiv:1611.01838, November 2016"},{"key":"18_CR3","unstructured":"Choromanska, A., Henaff, M., Mathieu, M., Arous, G.B., LeCun, Y.: The loss surfaces of multilayer networks. arXiv e-prints arXiv:1412.0233, November 2014"},{"key":"18_CR4","unstructured":"Dai, X., Yin, H., Jha, N.K.: NeST: a neural network synthesis tool based on a grow-and-prune paradigm. arXiv e-prints arXiv:1711.02017, November 2017"},{"key":"18_CR5","unstructured":"Dai, X., Yin, H., Jha, N.K.: Grow and prune compact, fast, and accurate LSTMs. arXiv e-prints, page arXiv:1805.11797, May 2018"},{"key":"18_CR6","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. arXiv e-prints arXiv:1810.04805, October 2018"},{"key":"18_CR7","unstructured":"Dinh, L., Pascanu, R., Bengio, S., Bengio, Y.: Sharp minima can generalize for deep nets. arXiv e-prints arXiv:1703.04933, March 2017"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Elsken, T., Metzen, J.H., Hutter, F.: Neural architecture search: a survey. arXiv e-prints arXiv:1808.05377, August 2018","DOI":"10.1007\/978-3-030-05318-5_11"},{"key":"18_CR9","unstructured":"Foret, P., Kleiner, A., Mobahi, H., Neyshabur, B.: Sharpness-aware minimization for efficiently improving generalization. arXiv e-prints arXiv:2010.01412, October 2020"},{"key":"18_CR10","unstructured":"Goodfellow, I.J., et al.: Generative adversarial networks. arXiv e-prints arXiv:1406.2661, June 2014"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A.-R., Hinton, G.: Speech recognition with deep recurrent neural networks. arXiv e-prints arXiv:1303.5778, March 2013","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Healy, P., Nikolov, N.S.: How to layer a directed acyclic graph. In: Graph Drawing (2001)","DOI":"10.1007\/3-540-45848-4_2"},{"key":"18_CR13","unstructured":"Hung, S.C.Y., Tu, C.-H., Wu, C.-E., Chen, C.-H., Chan, Y.-M., Chen, C.-S.: Compacting, picking and growing for unforgetting continual learning. arXiv e-prints arXiv:1910.06562, October 2019"},{"key":"18_CR14","unstructured":"Karras, T., Aila, T., Laine, S., Lehtinen, J.: Progressive growing of GANs for improved quality, stability, and variation. arXiv e-prints arXiv:1710.10196, October 2017"},{"key":"18_CR15","unstructured":"Kawaguchi, K.: Deep learning without poor local minima. arXiv e-prints arXiv:1605.07110, May 2016"},{"key":"18_CR16","unstructured":"Kawaguchi, K., Pack Kaelbling, L.: Elimination of all bad local minima in deep learning. arXiv e-prints arXiv:1901.00279, January 2019"},{"key":"18_CR17","unstructured":"Kawaguchi, K., Kaelbling, L.P., Bengio, Y.: Generalization in deep learning. arXiv e-prints arXiv:1710.05468, October 2017"},{"issue":"11","key":"18_CR18","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y Lecun","year":"1998","unstructured":"Lecun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"issue":"7553","key":"18_CR19","first-page":"436","volume":"521","author":"Y Lecun","year":"2015","unstructured":"Lecun, Y., Bengio, Y., Hinton, G.: Deep learning. Nature Cell Biol. 521(7553), 436\u2013444 (2015)","journal-title":"Nature Cell Biol."},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Leshno, M., Lin, V.Y., Pinkus, A., Schocken, S.: Multilayer feedforward networks with a nonpolynomial activation function can approximate any function (1993)","DOI":"10.1016\/S0893-6080(05)80131-5"},{"key":"18_CR21","unstructured":"Li, L., Talwalkar, A.: Random search and reproducibility for neural architecture search. arXiv e-prints arXiv:1902.07638, February 2019"},{"key":"18_CR22","unstructured":"Li, X., Zhou, Y., Wu, T., Socher, R., Xiong, C.: Learn to grow: a continual structure learning framework for overcoming catastrophic forgetting. arXiv e-prints arXiv:1904.00310, March 2019"},{"key":"18_CR23","unstructured":"Liang, S., Sun, R., Lee, J.D., Srikant, R.: Adding one neuron can eliminate all bad local minima. arXiv e-prints arXiv:1805.08671, May 2018"},{"key":"18_CR24","unstructured":"Liu, Y., et al.: RoBERTa: a robustly optimized BERT pretraining approach. arXiv e-prints arXiv:1907.11692, July 2019"},{"key":"18_CR25","doi-asserted-by":"crossref","unstructured":"McCloskey, M., Cohen, N.J.: Catastrophic interference in connectionist networks: the sequential learning problem. Psychol. Learn. Motiv. Adv. Res. Theo. 24(C), 109\u2013165, January 1989. Funding Information: The research reported in this chapter was supported by NIH grant NS21047 to Michael McCloskey, and by a grant from the Sloan Foundation to Neal Cohen. We thank Sean Purcell and Andrew Olson for assistance in generating the figures, and Alfonso Caramazza, Walter Harley, Paul Macaruso, Jay McClelland, Andrew Olson, Brenda Rapp, Roger Rat-cliff, David Rumelhart, and Terry Sejnowski for helpful discussions","DOI":"10.1016\/S0079-7421(08)60536-8"},{"key":"18_CR26","unstructured":"Negrinho, R., Patil, D., Le, N., Ferreira, D., Gormley, M., Gordon, G.: Towards modular and programmable architecture search. arXiv e-prints arXiv:1909.13404, September 2019"},{"key":"18_CR27","unstructured":"Netzer, Y., Wang, T., Coates, A., Bissacco, A., Wu, B., Ng, A.Y.: Reading digits in natural images with unsupervised feature learning. In: NIPS Workshop on Deep Learning and Unsupervised Feature Learning 2011 (2011)"},{"key":"18_CR28","unstructured":"Neyshabur, B., Tomioka, R., Srebro, N.: Norm-based capacity control in neural networks. arXiv e-prints arXiv:1503.00036, February 2015"},{"key":"18_CR29","doi-asserted-by":"crossref","unstructured":"Parisi, G.I., Kemker, R., Part, J.L., Kanan, C., Wermter, S.: Continual lifelong learning with neural networks: a review. arXiv e-prints arXiv:1802.07569, February 2018","DOI":"10.1016\/j.neunet.2019.01.012"},{"key":"18_CR30","unstructured":"Petzka, H., Adilova, L., Kamp, M., Sminchisescu, C.: A reparameterization-invariant flatness measure for deep neural networks. arXiv e-prints arXiv:1912.00058, November 2019"},{"key":"18_CR31","unstructured":"Rangamani, A., Nguyen, N.H., Kumar, A., Phan, D., Chin, S.H., Tran, T.D.: A scale invariant flatness measure for deep network minima. arXiv e-prints arXiv:1902.02434, February 2019"},{"key":"18_CR32","unstructured":"Keskar, N.S., Mudigere, D., Nocedal, J., Smelyanskiy, M., Tang, P.T.P.: On large-batch training for deep learning: generalization gap and sharp minima. arXiv e-prints arXiv:1609.04836, September 2016"},{"key":"18_CR33","unstructured":"Sinha, S., Garg, A., Larochelle, H.: Curriculum by smoothing. arXiv e-prints arXiv:2003.01367, March 2020"},{"key":"18_CR34","unstructured":"Sutskever, I., Vinyals, O., Le, Q.V.: Sequence to sequence learning with neural networks. arXiv e-prints, page arXiv:1409.3215, September 2014"},{"key":"18_CR35","unstructured":"Wang, D., Li, M., Wu, L., Chandra, V., Liu, Q.: Energy-aware neural architecture optimization with fast splitting steepest descent. arXiv e-prints arXiv:1910.03103, October 2019"},{"key":"18_CR36","doi-asserted-by":"crossref","unstructured":"Warstadt, A., Singh, A., Bowman, S.R.: Neural network acceptability judgments. arXiv e-prints arXiv:1805.12471, May 2018","DOI":"10.1162\/tacl_a_00290"},{"key":"18_CR37","unstructured":"Wen, W, et al.: SmoothOut: smoothing out sharp minima to improve generalization in deep learning. arXiv e-prints arXiv:1805.07898, May 2018"},{"key":"18_CR38","unstructured":"Wu, L., Liu, B., Stone, P., Liu, Q.: Firefly neural architecture descent: a general approach for growing neural networks. arXiv e-prints, page arXiv:2102.08574, February 2021"}],"container-title":["Lecture Notes in Computer Science","Artificial Neural Networks and Machine Learning \u2013 ICANN 2021"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-86340-1_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,9,10]],"date-time":"2021-09-10T12:08:06Z","timestamp":1631275686000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-86340-1_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030863395","9783030863401"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-86340-1_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"7 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICANN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bratislava","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Slovakia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 September 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icann2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/e-nns.org\/icann2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OCS","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"496","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"265","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"53% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Conference was held online due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}