{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,12]],"date-time":"2025-08-12T21:43:33Z","timestamp":1755035013485,"version":"3.37.3"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2023,5,8]],"date-time":"2023-05-08T00:00:00Z","timestamp":1683504000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,5,8]],"date-time":"2023-05-08T00:00:00Z","timestamp":1683504000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100010665","name":"H2020 Marie Sk\u0142odowska-Curie Actions","doi-asserted-by":"publisher","award":["H2020-MSCAIF-2017-EF-797805-STRUDEL"],"award-info":[{"award-number":["H2020-MSCAIF-2017-EF-797805-STRUDEL"]}],"id":[{"id":"10.13039\/100010665","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005363","name":"Universidad de Buenos Aires","doi-asserted-by":"publisher","award":["UBACyT 20020170100470BA"],"award-info":[{"award-number":["UBACyT 20020170100470BA"]}],"id":[{"id":"10.13039\/501100005363","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002923","name":"Consejo Nacional de Investigaciones Cient\u00edficas y T\u00e9cnicas","doi-asserted-by":"publisher","award":["Beca Postdoctoral"],"award-info":[{"award-number":["Beca Postdoctoral"]}],"id":[{"id":"10.13039\/501100002923","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2023,9]]},"DOI":"10.1007\/s10994-023-06337-6","type":"journal-article","created":{"date-parts":[[2023,5,8]],"date-time":"2023-05-08T22:01:37Z","timestamp":1683583297000},"page":"3105-3150","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["The role of mutual information in variational classifiers"],"prefix":"10.1007","volume":"112","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9180-7595","authenticated-orcid":false,"given":"Matias","family":"Vera","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Leonardo","family":"Rey Vega","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pablo","family":"Piantanida","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,5,8]]},"reference":[{"key":"6337_CR1","first-page":"1","volume":"19","author":"A Achille","year":"2018","unstructured":"Achille, A., & Soatto, S. (2018a). Emergence of invariance and disentangling in deep representations. Journal of Machine Learning Research (JMLR), 19, 1\u201334.","journal-title":"Journal of Machine Learning Research (JMLR)"},{"issue":"12","key":"6337_CR2","doi-asserted-by":"publisher","first-page":"2897","DOI":"10.1109\/TPAMI.2017.2784440","volume":"40","author":"A Achille","year":"2018","unstructured":"Achille, A., & Soatto, S. (2018b). Information dropout: Learning optimal representations through noisy computation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 40(12), 2897\u20132905.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"6337_CR4","unstructured":"Alemi, A. A., Fischer, I., Dillon, J. V., & Murphy, K. (2016). Deep variational information bottleneck. CoRR arxiv:1612.00410."},{"key":"6337_CR5","unstructured":"Amjad, R. A., & Geiger, B. C. (2018). Learning representations for neural network-based classification using the information bottleneck principle. CoRR arxiv:1802.09766."},{"key":"6337_CR6","unstructured":"Bassily, R., Moran, S., Nachum, I., Shafer, J., & Yehudayoff, A. (2018). Learners that use little information. In Proceedings of machine learning research, PMLR, (Vol. 83, pp. 25\u201355)."},{"key":"6337_CR7","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1016\/S0168-1699(99)00046-0","volume":"24","author":"J Blackard","year":"1999","unstructured":"Blackard, J., & Dean, D. (1999). Comparative accuracies of artificial neural networks and discriminant analysis in predicting forest cover types from cartographic variables. Elsevier Computers and Electronics in Agriculture, 24, 131\u2013151.","journal-title":"Elsevier Computers and Electronics in Agriculture"},{"key":"6337_CR8","doi-asserted-by":"publisher","first-page":"499","DOI":"10.1162\/153244302760200704","volume":"2","author":"O Bousquet","year":"2002","unstructured":"Bousquet, O., & Elisseeff, A. (2002). Stability and generalization. Journal of Machine Learning Research, 2, 499\u2013526. https:\/\/doi.org\/10.1162\/153244302760200704","journal-title":"Journal of Machine Learning Research"},{"issue":"1","key":"6337_CR9","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1613\/jair.953","volume":"16","author":"NV Chawla","year":"2002","unstructured":"Chawla, N. V., Bowyer, K. W., Hall, L. O., & Kegelmeyer, W. P. (2002). SMOTE: Synthetic minority over-sampling technique. Journal of Artificial Intelligence Research, 16(1), 321\u2013357.","journal-title":"Journal of Artificial Intelligence Research"},{"issue":"1","key":"6337_CR10","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s40747-017-0054-8","volume":"4","author":"P Chopra","year":"2018","unstructured":"Chopra, P., & Yadav, S. K. (2018). Restricted Boltzmann machine and softmax regression for fault detection and classification. Complex & Intelligent Systems, 4(1), 67\u201377.","journal-title":"Complex & Intelligent Systems"},{"key":"6337_CR11","volume-title":"Elements of information theory (Wiley series in telecommunications and signal processing)","author":"TM Cover","year":"2006","unstructured":"Cover, T. M., & Thomas, J. A. (2006). Elements of information theory (Wiley series in telecommunications and signal processing). Wiley-Interscience."},{"key":"6337_CR12","volume-title":"A probabilistic theory of pattern recognition, applications of mathematics","author":"L Devroye","year":"1997","unstructured":"Devroye, L., Gy\u00f6rfi, L., & Lugosi, G. (1997). A probabilistic theory of pattern recognition, applications of mathematics (2nd ed., Vol. 31). Springer.","edition":"2"},{"key":"6337_CR13","unstructured":"Devroye, L., Mehrabian, A., & Reddad, T. (2020). The total variation distance between high-dimensional Gaussians. arXiv:1810.08693 [math, stat]."},{"issue":"2","key":"6337_CR14","doi-asserted-by":"publisher","first-page":"183","DOI":"10.1002\/cpa.3160360204","volume":"36","author":"MD Donsker","year":"1983","unstructured":"Donsker, M. D., & Varadhan, S. R. S. (1983). Asymptotic evaluation of certain Markov process expectations for large time. IV. Communications on Pure and Applied Mathematics, 36(2), 183\u2013212. https:\/\/doi.org\/10.1002\/cpa.3160360204","journal-title":"Communications on Pure and Applied Mathematics"},{"key":"6337_CR15","unstructured":"Goodfellow, I., Bengio, Y., & Courville, A. (2016). Deep learning. MIT Press, http:\/\/www.deeplearningbook.org"},{"key":"6337_CR16","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1007\/s10994-005-0462-7","volume":"59","author":"T Graepel","year":"2005","unstructured":"Graepel, T., Herbrich, R., & Shawe-Taylor, J. (2005). PAC-Bayesian compression bounds on the prediction error of learning algorithms for classification. Springer Machine Learning, 59, 55\u201376.","journal-title":"Springer Machine Learning"},{"key":"6337_CR17","unstructured":"Guo, C., Pleiss, G., Sun, Y., Weinberger, & K. Q. (2017). On calibration of modern neural networks. In Proceedings of the international conference on machine learning ICML, Sydney."},{"key":"6337_CR18","doi-asserted-by":"publisher","first-page":"1039","DOI":"10.1007\/s10994-020-05869-5","volume":"109","author":"D Halbersberg","year":"2020","unstructured":"Halbersberg, D., Wienreb, M., & Lerner, B. (2020). Joint maximization of accuracy and information for learning the structure of a Bayesian network classifier. Springer Machine Learning, 109, 1039\u20131099.","journal-title":"Springer Machine Learning"},{"key":"6337_CR19","unstructured":"Higgins, I., Matthey, L., Pal, A., Burgess, C., Glorot, X., Botvinick, M., Mohamed, S., & Lerchner, A. (2017). $$\\beta $$-VAE: Learning basic visual concepts with a constrained variational framework. In Proceedings of the international conference on learning representations ICLR, Toulon."},{"issue":"8","key":"6337_CR20","doi-asserted-by":"publisher","first-page":"1771","DOI":"10.1162\/089976602760128018","volume":"14","author":"GE Hinton","year":"2002","unstructured":"Hinton, G. E. (2002). Training products of experts by minimizing contrastive divergence. Neural Computation, 14(8), 1771\u20131800. https:\/\/doi.org\/10.1162\/089976602760128018","journal-title":"Neural Computation"},{"key":"6337_CR21","doi-asserted-by":"crossref","unstructured":"Hinton, G. E. (2012). A practical guide to training restricted Boltzmann machines. In Proceedings of neural networks: Tricks of the trade (2nd ed., pp. 599\u2013619), Springer.","DOI":"10.1007\/978-3-642-35289-8_32"},{"issue":"7","key":"6337_CR22","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton, G. E., Osindero, S., & Teh, Y. W. (2006). A fast learning algorithm for deep belief nets. Neural Computation, 18(7), 1527\u20131554.","journal-title":"Neural Computation"},{"key":"6337_CR23","unstructured":"Kingma, D. P., & Welling, M. (2013). Auto-encoding variational Bayes. In Proceedings of the 2nd international conference on learning representations (ICLR)."},{"key":"6337_CR24","unstructured":"Krizhevsky, A. (2009). Learning multiple layers of features from tiny images. Technical report, University of Toronto."},{"key":"6337_CR25","unstructured":"Li, Y., Bradshaw, J., & Sharma, Y. (2019) Are generative classifiers more robust to adversarial attacks? In Proceedings of machine learning research, PMLR, Long Beach, California, USA (Vol.\u00a097, pp. 3804\u20133814)."},{"key":"6337_CR26","doi-asserted-by":"crossref","unstructured":"Maggipinto, M., Terzi, M., & Susto, G. A. (2020). $$\\beta $$-variational classifiers under attack. CoRR arxiv:2008.09010.","DOI":"10.1016\/j.ifacol.2020.12.1979"},{"key":"6337_CR27","series-title":"Adaptive computation and machine learning","volume-title":"Foundations of machine learning","author":"M Mohri","year":"2018","unstructured":"Mohri, M., Rostamizadeh, A., & Talwalkar, A. (2018). Foundations of machine learning. Adaptive computation and machine learning (2nd ed.). MIT Press.","edition":"2"},{"key":"6337_CR28","unstructured":"Neyshabur, B., Tomioka, R., Salakhutdinov, R., & Srebro, N. (2017). Geometry of optimization and implicit regularization in deep learning. CoRR arxiv:abs\/1705.03071."},{"key":"6337_CR29","unstructured":"Pichler, G., Piantanida, P., & Koliander, G. (2020). On the estimation of information measures of continuous distributions. arxiv:2002.02851."},{"key":"6337_CR30","unstructured":"Pinsker, M. (1964). Information and information stability of random variables and processes. Holden-Day series in time series analysis, Holden-Day."},{"key":"6337_CR31","volume-title":"Principles of mathematical analysis","author":"W Rudin","year":"1986","unstructured":"Rudin, W. (1986). Principles of mathematical analysis. McGraw-Hill Book Company."},{"key":"6337_CR32","unstructured":"Russo, D., & Zou, J. (2015). How much does your data exploration overfit? Controlling bias via information usage. arXiv:1511.05219 [cs, stat]."},{"key":"6337_CR33","doi-asserted-by":"crossref","unstructured":"Saxe, A., Bansal, Y., Dapello, J., Advani, M., Kolchinsky, A., Tracey, B., & Cox, D. (2018). On the information bottleneck theory of deep learning. In Proceedings of the 6th international conference on learning representations (ICLR).","DOI":"10.1088\/1742-5468\/ab3985"},{"key":"6337_CR34","unstructured":"Schwartz-Ziv, R., & Tishby, N. (2017). Opening the black box of deep neural networks via information. CoRR arxiv:1703.00810."},{"issue":"29\u201330","key":"6337_CR35","doi-asserted-by":"publisher","first-page":"2696","DOI":"10.1016\/j.tcs.2010.04.006","volume":"411","author":"O Shamir","year":"2010","unstructured":"Shamir, O., Sabato, S., & Tishby, N. (2010). Learning and generalization with the information bottleneck. Theoretical Computer Science, 411(29\u201330), 2696\u20132711. https:\/\/doi.org\/10.1016\/j.tcs.2010.04.006","journal-title":"Theoretical Computer Science"},{"key":"6337_CR36","unstructured":"Sokolic, J., Giryes, R., Sapiro, G., & Rodrigues, M. (2017). Generalization error of invariant classifiers. In Proceedings of machine learning research, PMLR, Fort Lauderdale, FL, USA (Vol. 54, pp. 1094\u20131103)."},{"key":"6337_CR37","doi-asserted-by":"publisher","first-page":"4265","DOI":"10.1109\/TSP.2017.2708039","volume":"65","author":"J Sokolic","year":"2017","unstructured":"Sokolic, J., Giryes, R., Sapiro, G., & Rodrigues, M. R. D. (2017). Robust large margin deep neural networks. IEEE Transactions on Signal Processing, 65, 4265\u20134280.","journal-title":"IEEE Transactions on Signal Processing"},{"issue":"1","key":"6337_CR39","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G. E., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: A simple way to prevent neural networks from overfitting. Journal of Machine Learning Research, 15(1), 1929\u20131958.","journal-title":"Journal of Machine Learning Research"},{"key":"6337_CR38","unstructured":"Srivastava, N., Salakhutdinov, R., & Hinton, G. E. (2013). Modeling documents with deep Boltzmann machines. In Proceedings of the twenty-ninth conference on uncertainty in artificial intelligence, UAI 2013, Bellevue, WA, USA, August 11\u201315, 2013."},{"key":"6337_CR41","unstructured":"Tishby, N., Pereira, F. C., & Bialek, W. (1999). The information bottleneck method. In Proceedings of the 37th annual Allerton conference on communication, control and computing (pp. 368\u2013377)."},{"key":"6337_CR40","doi-asserted-by":"crossref","unstructured":"Tishby, N., & Zaslavsky, N. (2015). Deep learning and the information bottleneck principle. CoRR arxiv:1503.02406.","DOI":"10.1109\/ITW.2015.7133169"},{"key":"6337_CR42","doi-asserted-by":"crossref","unstructured":"Vera, M., Piantanida, P., & Rey\u00a0Vega, L. (2018a). The role of the information bottleneck in representation learning. In IEEE International symposium on information theory (ISIT).","DOI":"10.1109\/ISIT.2018.8437679"},{"issue":"5","key":"6337_CR43","doi-asserted-by":"publisher","first-page":"1063","DOI":"10.1109\/JSTSP.2018.2846218","volume":"12","author":"M Vera","year":"2018","unstructured":"Vera, M., Rey Vega, L., & Piantanida, P. (2018b). Compression-based regularization with an application to multitask learning. IEEE Journal of Selected Topics in Signal Processing, 12(5), 1063\u20131076.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"6337_CR44","first-page":"3371","volume":"11","author":"P Vincent","year":"2010","unstructured":"Vincent, P., Larochelle, H., Lajoie, I., Bengio, Y., & Manzagol, P. A. (2010). Stacked denoising autoencoders: Learning useful representations in a deep network with a local denoising criterion. Journal of Machine Learning Research, 11, 3371\u20133408.","journal-title":"Journal of Machine Learning Research"},{"key":"6337_CR45","unstructured":"Xu, A., & Raginsky, M. (2017). Information-theoretic analysis of generalization capability of learning algorithms. arXiv:1705.07809 [cs, math, stat]."},{"issue":"3","key":"6337_CR46","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1007\/s10994-011-5268-1","volume":"86","author":"H Xu","year":"2012","unstructured":"Xu, H., & Mannor, S. (2012). Robustness and generalization. Machine Learning, 86(3), 391\u2013423. https:\/\/doi.org\/10.1007\/s10994-011-5268-1","journal-title":"Machine Learning"},{"key":"6337_CR47","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1007\/BF00992676","volume":"9","author":"K Yamanishi","year":"1992","unstructured":"Yamanishi, K. (1992). A learning criterion for stochastic rules. Springer Machine Learning, 9, 165\u2013203.","journal-title":"Springer Machine Learning"},{"key":"6337_CR48","unstructured":"Zhang, C., Bengio, S., Hardt, M., Recht, B., & Vinyals, O. (2017). Understanding deep learning requires rethinking generalization. In 5th international conference on learning representations, ICLR 2017, Toulon, France, April 24\u201326, 2017, conference track proceedings, https:\/\/openreview.net\/forum?id=Sy8gdB9xx"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-023-06337-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-023-06337-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-023-06337-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,8]],"date-time":"2024-05-08T00:02:15Z","timestamp":1715126535000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-023-06337-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,8]]},"references-count":47,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2023,9]]}},"alternative-id":["6337"],"URL":"https:\/\/doi.org\/10.1007\/s10994-023-06337-6","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"type":"print","value":"0885-6125"},{"type":"electronic","value":"1573-0565"}],"subject":[],"published":{"date-parts":[[2023,5,8]]},"assertion":[{"value":"16 October 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 April 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 May 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"The authors voluntarily agree to take part in this study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}