{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,7]],"date-time":"2025-08-07T20:22:16Z","timestamp":1754598136372,"version":"3.40.3"},"reference-count":69,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tpami.2024.3392724","type":"journal-article","created":{"date-parts":[[2024,4,23]],"date-time":"2024-04-23T18:51:01Z","timestamp":1713898261000},"page":"7529-7541","source":"Crossref","is-referenced-by-count":1,"title":["Elodi: Ensemble Logit Difference Inhibition for Positive-Congruent Training"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2753-5921","authenticated-orcid":false,"given":"Yue","family":"Zhao","sequence":"first","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5413-2445","authenticated-orcid":false,"given":"Yantao","family":"Shen","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6391-4921","authenticated-orcid":false,"given":"Yuanjun","family":"Xiong","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5754-3183","authenticated-orcid":false,"given":"Shuo","family":"Yang","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1073-1533","authenticated-orcid":false,"given":"Wei","family":"Xia","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1900-2124","authenticated-orcid":false,"given":"Zhuowen","family":"Tu","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9683-5237","authenticated-orcid":false,"given":"Bernt","family":"Schiele","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2902-6362","authenticated-orcid":false,"given":"Stefano","family":"Soatto","sequence":"additional","affiliation":[{"name":"AWS AI Labs, Seattle, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01407"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33012429"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.515"},{"article-title":"Distilling the knowledge in a neural network","year":"2015","author":"Hinton","key":"ref4"},{"key":"ref5","first-page":"6405","article-title":"Simple and scalable predictive uncertainty estimation using deep ensembles","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Lakshminarayanan"},{"article-title":"An empirical study of example forgetting during deep neural network learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Toneva","key":"ref6"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00640"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403379"},{"article-title":"Churn reduction via distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jiang","key":"ref9"},{"key":"ref10","first-page":"116","article-title":"Backward-compatible prediction updates: A probabilistic approach","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"34","author":"Tr\u00e4uble"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/BF00058655"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.1997.1504"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.29172\/7c2a6982-6d72-4cd8-bba6-2fccb06a7011"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1214\/aos\/1024691352"},{"article-title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Allen-Zhu","key":"ref15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1214\/ss\/1009212519"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1111\/insr.12243"},{"article-title":"Snapshot ensembles: Train 1, get m for free","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Huang","key":"ref18"},{"key":"ref19","first-page":"8803","article-title":"Loss surfaces, mode connectivity, and fast ensembling of DNNs","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"31","author":"Garipov"},{"key":"ref20","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"Srivastava","year":"2014","journal-title":"J. Mach. Learn. Res."},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.5555\/3045390.3045502"},{"article-title":"FractalNet: Ultra-deep neural networks without residuals","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Larsson","key":"ref22"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_39"},{"article-title":"Batchensemble: An alternative approach to efficient ensemble and lifelong learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wen","key":"ref24"},{"article-title":"Training independent subnetworks for robust prediction","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Havasi","key":"ref25"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00381"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00396"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.450"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-614"},{"key":"ref30","first-page":"953","article-title":"Ensemble knowledge distillation for learning improved and efficient networks","volume-title":"Proc. Eur. Conf. Artif. Intell.","author":"Asif"},{"article-title":"Ensemble distribution distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Malinin","key":"ref31"},{"key":"ref32","first-page":"2351","article-title":"Ensemble distillation for robust model fusion in federated learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"33","author":"Lin"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022859003006"},{"article-title":"No one representation to rule them all: Overlapping features of training methods","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gontijo-Lopes","key":"ref34"},{"key":"ref35","first-page":"23965","article-title":"Model soups: Averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wortsman"},{"key":"ref36","first-page":"183","article-title":"On domains of attraction of multi-dimensional distributions","volume":"2","author":"Rva\u010deva","year":"1962","journal-title":"Sel. Translations Math. Statist. Probability"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1201\/9781315136288"},{"volume-title":"Quadratic Forms in Random Variables: Theory and Applications","year":"1992","author":"Mathai","key":"ref38"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1167\/jov.21.10.1"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"article-title":"Understanding the logit distributions of adversarially-trained deep neural networks","year":"2021","author":"Seguin","key":"ref41"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177704472"},{"key":"ref43","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Glorot"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref45","article-title":"On lazy training in differentiable programming","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"32","author":"Chizat"},{"article-title":"FitNets: Hints for thin deep nets","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Romero","key":"ref46"},{"article-title":"Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zagoruyko","key":"ref47"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.754"},{"article-title":"Partial FC: Training 10 million identities on a single machine","year":"2020","author":"An","key":"ref49"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00914"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref52","first-page":"649","article-title":"Character-level convolutional networks for text classification","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Zhang"},{"article-title":"Training deep nets with sublinear memory cost","year":"2016","author":"Chen","key":"ref53"},{"article-title":"Accurate, large minibatch SGD: Training imageNet in 1 hour","year":"2017","author":"Goyal","key":"ref54"},{"article-title":"Hot-refresh model upgrades with regression-free compatible training in image retrieval","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang","key":"ref55"},{"key":"ref56","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics","author":"Devlin"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-32381-3_16"},{"key":"ref58","first-page":"1321","article-title":"On calibration of modern neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Guo"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.4324\/9781410605337-29"},{"article-title":"Very deep convolutional networks for large-scale image recognition","year":"2014","author":"Simonyan","key":"ref61"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2983686"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00255"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01044"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01352"},{"article-title":"PyTorch image models","year":"2019","author":"Wightman","key":"ref68"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01065"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10746266\/10507026.pdf?arnumber=10507026","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,5]],"date-time":"2025-04-05T04:12:44Z","timestamp":1743826364000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10507026\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":69,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3392724","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"type":"print","value":"0162-8828"},{"type":"electronic","value":"2160-9292"},{"type":"electronic","value":"1939-3539"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}