{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T21:14:44Z","timestamp":1773090884573,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T00:00:00Z","timestamp":1691107200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100006785","name":"Google","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006785","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000006","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["ONR-N000142112841"],"award-info":[{"award-number":["ONR-N000142112841"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,6]]},"DOI":"10.1145\/3580305.3599278","type":"proceedings-article","created":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T18:10:58Z","timestamp":1691172658000},"page":"832-844","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["COMET: Learning Cardinality Constrained Mixture of Experts with Trees and Local Search"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3300-0213","authenticated-orcid":false,"given":"Shibal","family":"Ibrahim","sequence":"first","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7952-7735","authenticated-orcid":false,"given":"Wenyu","family":"Chen","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4501-0678","authenticated-orcid":false,"given":"Hussein","family":"Hazimeh","sequence":"additional","affiliation":[{"name":"Google Research, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6761-1468","authenticated-orcid":false,"given":"Natalia","family":"Ponomareva","sequence":"additional","affiliation":[{"name":"Google Research, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6847-0186","authenticated-orcid":false,"given":"Zhe","family":"Zhao","sequence":"additional","affiliation":[{"name":"Google DeepMind, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1384-9743","authenticated-orcid":false,"given":"Rahul","family":"Mazumder","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,8,4]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Zemel","author":"Adams Ryan Prescott","year":"2011","unstructured":"Ryan Prescott Adams and Richard S . Zemel . 2011 . Ranking via Sinkhorn Propagation . https:\/\/doi.org\/10.48550\/ARXIV.1106.1925 10.48550\/ARXIV.1106.1925 Ryan Prescott Adams and Richard S. Zemel. 2011. Ranking via Sinkhorn Propagation. https:\/\/doi.org\/10.48550\/ARXIV.1106.1925"},{"key":"e_1_3_2_2_2_1","volume-title":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics.","author":"Artetxe Mikel","year":"2022","unstructured":"Mikel Artetxe , Shruti Bhosale , Naman Goyal , 2022 . Efficient Large Scale Language Modeling with Mixtures of Experts . In Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics. Mikel Artetxe, Shruti Bhosale, Naman Goyal, et al. 2022. Efficient Large Scale Language Modeling with Mixtures of Experts. In Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1137\/120869778"},{"key":"e_1_3_2_2_4_1","volume-title":"International Conference on Learning Representations Workshop Tract. https:\/\/openreview.net\/forum?id=B1ckMDqlg","author":"Bengio Emmanuel","year":"2016","unstructured":"Emmanuel Bengio , Pierre-Luc Bacon , Joelle Pineau , and Doina Precup . 2016 . Conditional Computation in Neural Networks for faster models . In International Conference on Learning Representations Workshop Tract. https:\/\/openreview.net\/forum?id=B1ckMDqlg Emmanuel Bengio, Pierre-Luc Bacon, Joelle Pineau, and Doina Precup. 2016. Conditional Computation in Neural Networks for faster models. In International Conference on Learning Representations Workshop Tract. https:\/\/openreview.net\/forum?id=B1ckMDqlg"},{"key":"e_1_3_2_2_5_1","volume-title":"Courville","author":"Bengio Yoshua","year":"2013","unstructured":"Yoshua Bengio , Nicholas L\u00e9onard , and Aaron C . Courville . 2013 . Estimating or Propagating Gradients Through Stochastic Neurons for Conditional Computation. CoRR abs\/1308.3432 (2013). arXiv:1308.3432 http:\/\/arxiv.org\/abs\/1308.3432 Yoshua Bengio, Nicholas L\u00e9onard, and Aaron C. Courville. 2013. Estimating or Propagating Gradients Through Stochastic Neurons for Conditional Computation. CoRR abs\/1308.3432 (2013). arXiv:1308.3432 http:\/\/arxiv.org\/abs\/1308.3432"},{"key":"e_1_3_2_2_6_1","volume-title":"Introduction to linear optimization","author":"Bertsimas Dimitris","unstructured":"Dimitris Bertsimas and John N Tsitsiklis . 1997. Introduction to linear optimization . Vol. 6 . Athena Scientific Belmont , MA. Dimitris Bertsimas and John N Tsitsiklis. 1997. Introduction to linear optimization. Vol. 6. Athena Scientific Belmont, MA."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/S17-2001"},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"4086","author":"Clark Aidan","year":"2022","unstructured":"Aidan Clark , Diego De Las Casas , Aurelia Guy , Arthur Mensch , Michela Paganini , Jordan Hoffmann , Bogdan Damoc , Blake Hechtman , Trevor Cai , Sebastian Borgeaud , George Bm Van Den Driessche , Eliza Rutherford , Tom Hennigan , Matthew J Johnson , Albin Cassirer , Chris Jones , Elena Buchatskaya , David Budden , Laurent Sifre , Simon Osindero , Oriol Vinyals , Marc'Aurelio Ranzato , Jack Rae , Erich Elsen , Koray Kavukcuoglu , and Karen Simonyan . 2022 . Unified Scaling Laws for Routed Language Models . In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research , Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 4057-- 4086 . https: \/\/proceedings.mlr.press\/v162\/clark22a.html Aidan Clark, Diego De Las Casas, Aurelia Guy, Arthur Mensch, Michela Paganini, Jordan Hoffmann, Bogdan Damoc, Blake Hechtman, Trevor Cai, Sebastian Borgeaud, George Bm Van Den Driessche, Eliza Rutherford, Tom Hennigan, Matthew J Johnson, Albin Cassirer, Chris Jones, Elena Buchatskaya, David Budden, Laurent Sifre, Simon Osindero, Oriol Vinyals, Marc'Aurelio Ranzato, Jack Rae, Erich Elsen, Koray Kavukcuoglu, and Karen Simonyan. 2022. Unified Scaling Laws for Routed Language Models. In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 4057--4086. https: \/\/proceedings.mlr.press\/v162\/clark22a.html"},{"key":"e_1_3_2_2_9_1","volume-title":"Machine Learning Challenges. Evaluating Predictive Uncertainty, Visual Object Classification, and Recognising Tectual Entailment, Joaquin Qui\u00f1onero-Candela, Ido Dagan","author":"Dagan Ido","unstructured":"Ido Dagan , Oren Glickman , and Bernardo Magnini . 2006. The PASCAL Recognising Textual Entailment Challenge . In Machine Learning Challenges. Evaluating Predictive Uncertainty, Visual Object Classification, and Recognising Tectual Entailment, Joaquin Qui\u00f1onero-Candela, Ido Dagan , Bernardo Magnini , and Florence d'Alch\u00e9 Buc (Eds.). Springer Berlin Heidelberg , Berlin, Heidelberg, 177--190. Ido Dagan, Oren Glickman, and Bernardo Magnini. 2006. The PASCAL Recognising Textual Entailment Challenge. In Machine Learning Challenges. Evaluating Predictive Uncertainty, Visual Object Classification, and Recognising Tectual Entailment, Joaquin Qui\u00f1onero-Candela, Ido Dagan, Bernardo Magnini, and Florence d'Alch\u00e9 Buc (Eds.). Springer Berlin Heidelberg, Berlin, Heidelberg, 177--190."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2211477"},{"key":"e_1_3_2_2_11_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2018 . BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding . http:\/\/arxiv.org\/abs\/1810.04805 cite arxiv:1810.04805Comment: 13 pages. Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. http:\/\/arxiv.org\/abs\/1810.04805 cite arxiv:1810.04805Comment: 13 pages."},{"key":"e_1_3_2_2_12_1","first-page":"05","volume-title":"Proceedings of the Third International Workshop on Paraphrasing (IWP2005)","author":"William","unstructured":"William B. Dolan and Chris Brockett. 2005. Automatically Constructing a Corpus of Sentential Paraphrases . In Proceedings of the Third International Workshop on Paraphrasing (IWP2005) . https:\/\/aclanthology.org\/I 05 - 5002 William B. Dolan and Chris Brockett. 2005. Automatically Constructing a Corpus of Sentential Paraphrases. In Proceedings of the Third International Workshop on Paraphrasing (IWP2005). https:\/\/aclanthology.org\/I05-5002"},{"key":"e_1_3_2_2_13_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"5569","author":"Du Nan","year":"2022","unstructured":"Nan Du , Yanping Huang , Andrew M Dai , Simon Tong , Dmitry Lepikhin , Yuanzhong Xu , Maxim Krikun , Yanqi Zhou , Adams Wei Yu , Orhan Firat , Barret Zoph , Liam Fedus , Maarten P Bosma , Zongwei Zhou , Tao Wang , Emma Wang , Kellie Webster , Marie Pellat , Kevin Robinson , Kathleen Meier-Hellstern , Toju Duke , Lucas Dixon , Kun Zhang , Quoc Le , Yonghui Wu , Zhifeng Chen , and Claire Cui . 2022 . GLaM: Efficient Scaling of Language Models with Mixture-of-Experts . In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research , Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 5547-- 5569 . https:\/\/proceedings.mlr.press\/v162\/du22c.html Nan Du, Yanping Huang, Andrew M Dai, Simon Tong, Dmitry Lepikhin, Yuanzhong Xu, Maxim Krikun, Yanqi Zhou, Adams Wei Yu, Orhan Firat, Barret Zoph, Liam Fedus, Maarten P Bosma, Zongwei Zhou, Tao Wang, Emma Wang, Kellie Webster, Marie Pellat, Kevin Robinson, Kathleen Meier-Hellstern, Toju Duke, Lucas Dixon, Kun Zhang, Quoc Le, Yonghui Wu, Zhifeng Chen, and Claire Cui. 2022. GLaM: Efficient Scaling of Language Models with Mixture-of-Experts. In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 5547--5569. https:\/\/proceedings.mlr.press\/v162\/du22c.html"},{"key":"e_1_3_2_2_14_1","first-page":"1","article-title":"Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity","volume":"23","author":"Fedus William","year":"2022","unstructured":"William Fedus , Barret Zoph , and Noam Shazeer . 2022 . Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity . Journal of Machine Learning Research 23 , 120 (2022), 1 -- 39 . http:\/\/jmlr.org\/papers\/v23\/21-0998.html William Fedus, Barret Zoph, and Noam Shazeer. 2022. Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity. Journal of Machine Learning Research 23, 120 (2022), 1--39. http:\/\/jmlr.org\/papers\/v23\/21-0998.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_15_1","unstructured":"Nicholas Frosst and Geoffrey Hinton. 2017. Distilling a Neural Network Into a Soft Decision Tree. arXiv:1711.09784  Nicholas Frosst and Geoffrey Hinton. 2017. Distilling a Neural Network Into a Soft Decision Tree. arXiv:1711.09784"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1011419012209"},{"key":"e_1_3_2_2_17_1","article-title":"The MovieLens Datasets","volume":"5","author":"Maxwell Harper F.","year":"2015","unstructured":"F. Maxwell Harper and Joseph A. Konstan . 2015 . The MovieLens Datasets : History and Context. ACM Trans. Interact. Intell. Syst. 5 , 4, Article 19 (dec 2015), 19 pages. https:\/\/doi.org\/10.1145\/2827872 10.1145\/2827872 F. Maxwell Harper and Joseph A. Konstan. 2015. The MovieLens Datasets: History and Context. ACM Trans. Interact. Intell. Syst. 5, 4, Article 19 (dec 2015), 19 pages. https:\/\/doi.org\/10.1145\/2827872","journal-title":"History and Context. ACM Trans. Interact. Intell. Syst."},{"key":"e_1_3_2_2_18_1","first-page":"5","article-title":"Fast Best Subset Selection","volume":"68","author":"Hazimeh Hussein","year":"2020","unstructured":"Hussein Hazimeh and Rahul Mazumder . 2020 . Fast Best Subset Selection : Coordinate Descent and Local Combinatorial Optimization Algorithms. Oper. Res. 68 , 5 (Sept. 2020), 1517--1537. https:\/\/doi.org\/10.1287\/opre.2019.1919 10.1287\/opre.2019.1919 Hussein Hazimeh and Rahul Mazumder. 2020. Fast Best Subset Selection: Coordinate Descent and Local Combinatorial Optimization Algorithms. Oper. Res. 68, 5 (Sept. 2020), 1517--1537. https:\/\/doi.org\/10.1287\/opre.2019.1919","journal-title":"Coordinate Descent and Local Combinatorial Optimization Algorithms. Oper. Res."},{"key":"e_1_3_2_2_19_1","volume-title":"International Conference on Machine Learning. PMLR, 4138--4148","author":"Hazimeh Hussein","year":"2020","unstructured":"Hussein Hazimeh , Natalia Ponomareva , Petros Mol , Zhenyu Tan , and Rahul Mazumder . 2020 . The tree ensemble layer: Differentiability meets conditional computation . In International Conference on Machine Learning. PMLR, 4138--4148 . Hussein Hazimeh, Natalia Ponomareva, Petros Mol, Zhenyu Tan, and Rahul Mazumder. 2020. The tree ensemble layer: Differentiability meets conditional computation. In International Conference on Machine Learning. PMLR, 4138--4148."},{"key":"e_1_3_2_2_20_1","volume-title":"Chi","author":"Hazimeh Hussein","year":"2021","unstructured":"Hussein Hazimeh , Zhe Zhao , Aakanksha Chowdhery , Maheswaran Sathiamoorthy , Yihua Chen , Rahul Mazumder , Lichan Hong , and Ed Chi . 2021 . DSelect-k: Differentiable Selection in the Mixture of Experts with Applications to Multi-Task Learning. In Advances in Neural Information Processing Systems, A. Beygelzimer, Y. Dauphin, P. Liang, and J. Wortman Vaughan (Eds .). https: \/\/openreview.net\/forum?id=tKlYQJLYN8v Hussein Hazimeh, Zhe Zhao, Aakanksha Chowdhery, Maheswaran Sathiamoorthy, Yihua Chen, Rahul Mazumder, Lichan Hong, and Ed Chi. 2021. DSelect-k: Differentiable Selection in the Mixture of Experts with Applications to Multi-Task Learning. In Advances in Neural Information Processing Systems, A. Beygelzimer, Y. Dauphin, P. Liang, and J. Wortman Vaughan (Eds.). https: \/\/openreview.net\/forum?id=tKlYQJLYN8v"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01237-6"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539412"},{"key":"e_1_3_2_2_23_1","volume-title":"Convolutional Networks and the Models in-Between. CoRR abs\/1603.01250","author":"Ioannou Yani","year":"2016","unstructured":"Yani Ioannou , Duncan P. Robertson , Darko Zikic , Peter Kontschieder , Jamie Shotton , Matthew Brown , and Antonio Criminisi . 2016. Decision Forests , Convolutional Networks and the Models in-Between. CoRR abs\/1603.01250 ( 2016 ). arXiv:1603.01250 http:\/\/arxiv.org\/abs\/1603.01250 Yani Ioannou, Duncan P. Robertson, Darko Zikic, Peter Kontschieder, Jamie Shotton, Matthew Brown, and Antonio Criminisi. 2016. Decision Forests, Convolutional Networks and the Models in-Between. CoRR abs\/1603.01250 (2016). arXiv:1603.01250 http:\/\/arxiv.org\/abs\/1603.01250"},{"key":"e_1_3_2_2_24_1","volume-title":"Proceedings of the 21st International Conference on Pattern Recognition (ICPR2012)","author":"Irsoy Ozan","year":"2012","unstructured":"Ozan Irsoy , O. T. Yildiz , and Ethem Alpaydin . 2012 . Soft decision trees . Proceedings of the 21st International Conference on Pattern Recognition (ICPR2012) (2012), 1819--1822. Ozan Irsoy, O. T. Yildiz, and Ethem Alpaydin. 2012. Soft decision trees. Proceedings of the 21st International Conference on Pattern Recognition (ICPR2012) (2012), 1819--1822."},{"key":"e_1_3_2_2_25_1","volume-title":"Interpretable Mixture of Experts. Transactions on Machine Learning Research","author":"Ismail Aya Abdelsalam","year":"2023","unstructured":"Aya Abdelsalam Ismail , Sercan O Arik , Jinsung Yoon , Ankur Taly , Soheil Feizi , and Tomas Pfister . 2023. Interpretable Mixture of Experts. Transactions on Machine Learning Research ( 2023 ). https:\/\/openreview.net\/forum?id=DdZoPUPm0a Aya Abdelsalam Ismail, Sercan O Arik, Jinsung Yoon, Ankur Taly, Soheil Feizi, and Tomas Pfister. 2023. Interpretable Mixture of Experts. Transactions on Machine Learning Research (2023). https:\/\/openreview.net\/forum?id=DdZoPUPm0a"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.2.369"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.79"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0893-6080(99)00066-0"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.1993.716791"},{"key":"e_1_3_2_2_30_1","volume-title":"Deep Neural Decision Forests. In 2015 IEEE International Conference on Computer Vision (ICCV). 1467--1475","author":"Kontschieder Peter","year":"2015","unstructured":"Peter Kontschieder , Madalina Fiterau , Antonio Criminisi , 2015 . Deep Neural Decision Forests. In 2015 IEEE International Conference on Computer Vision (ICCV). 1467--1475 . Peter Kontschieder, Madalina Fiterau, Antonio Criminisi, et al. 2015. Deep Neural Decision Forests. In 2015 IEEE International Conference on Computer Vision (ICCV). 1467--1475."},{"key":"e_1_3_2_2_31_1","first-page":"304","volume-title":"Beyond Distillation: Task-level Mixture-of-Experts for Efficient Inference. In Findings of the Association for Computational Linguistics: EMNLP 2021","author":"Kudugunta Sneha","year":"2021","unstructured":"Sneha Kudugunta , Yanping Huang , Ankur Bapna , Maxim Krikun , Dmitry Lepikhin , Minh-Thang Luong , and Orhan Firat . 2021 . Beyond Distillation: Task-level Mixture-of-Experts for Efficient Inference. In Findings of the Association for Computational Linguistics: EMNLP 2021 . Association for Computational Linguistics, Punta Cana, Dominican Republic, 3577--3599. https:\/\/doi.org\/10. 18653\/v1\/2021. findings-emnlp. 304 10.18653\/v1 Sneha Kudugunta, Yanping Huang, Ankur Bapna, Maxim Krikun, Dmitry Lepikhin, Minh-Thang Luong, and Orhan Firat. 2021. Beyond Distillation: Task-level Mixture-of-Experts for Efficient Inference. In Findings of the Association for Computational Linguistics: EMNLP 2021. Association for Computational Linguistics, Punta Cana, Dominican Republic, 3577--3599. https:\/\/doi.org\/10.18653\/v1\/2021. findings-emnlp.304"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1002\/nav.3800020109"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.5555\/3031843.3031909"},{"key":"e_1_3_2_2_34_1","volume-title":"Sparse Models. In International Conference on Machine Learning.","author":"Lewis Mike","year":"2021","unstructured":"Mike Lewis , Shruti Bhosale , Tim Dettmers , Naman Goyal , and Luke Zettlemoyer . 2021 . BASE Layers: Simplifying Training of Large , Sparse Models. In International Conference on Machine Learning. Mike Lewis, Shruti Bhosale, Tim Dettmers, Naman Goyal, and Luke Zettlemoyer. 2021. BASE Layers: Simplifying Training of Large, Sparse Models. In International Conference on Machine Learning."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.425"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220007"},{"key":"e_1_3_2_2_37_1","volume-title":"Learning Latent Permutations with Gumbel-Sinkhorn Networks. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Byt3oJ-0W","author":"Mena Gonzalo","year":"2018","unstructured":"Gonzalo Mena , David Belanger , Scott Linderman , and Jasper Snoek . 2018 . Learning Latent Permutations with Gumbel-Sinkhorn Networks. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Byt3oJ-0W Gonzalo Mena, David Belanger, Scott Linderman, and Jasper Snoek. 2018. Learning Latent Permutations with Gumbel-Sinkhorn Networks. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Byt3oJ-0W"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622826.1622827"},{"key":"e_1_3_2_2_39_1","volume-title":"Joan Puigcerver, Rodolphe Jenatton, and Neil Houlsby.","author":"Mustafa Basil","year":"2022","unstructured":"Basil Mustafa , Carlos Riquelme Ruiz , Joan Puigcerver, Rodolphe Jenatton, and Neil Houlsby. 2022 . Multimodal Contrastive Learning with LIMoE: the Language-Image Mixture of Experts. In Advances in Neural Information Processing Systems, Alice H. Oh, Alekh Agarwal, Danielle Belgrave, and Kyunghyun Cho (Eds .). https:\/\/openreview.net\/forum?id=Qy1D9JyMBg0 Basil Mustafa, Carlos Riquelme Ruiz, Joan Puigcerver, Rodolphe Jenatton, and Neil Houlsby. 2022. Multimodal Contrastive Learning with LIMoE: the Language-Image Mixture of Experts. In Advances in Neural Information Processing Systems, Alice H. Oh, Alekh Agarwal, Danielle Belgrave, and Kyunghyun Cho (Eds.). https:\/\/openreview.net\/forum?id=Qy1D9JyMBg0"},{"key":"e_1_3_2_2_40_1","unstructured":"Yuval Netzer Tao Wang Adam Coates A. Bissacco Bo Wu and A. Ng. 2011. Reading Digits in Natural Images with Unsupervised Feature Learning.  Yuval Netzer Tao Wang Adam Coates A. Bissacco Bo Wu and A. Ng. 2011. Reading Digits in Natural Images with Unsupervised Feature Learning."},{"key":"#cr-split#-e_1_3_2_2_41_1.1","unstructured":"Xiaonan Nie Xupeng Miao Shijie Cao Lingxiao Ma Qibin Liu Jilong Xue Youshan Miao Yi Liu Zhi Yang and Bin Cui. 2021. EvoMoE: An Evolutional Mixture-of-Experts Training Framework via Dense-To-Sparse Gate. https: \/\/doi.org\/10.48550\/ARXIV.2112.14397 10.48550\/ARXIV.2112.14397"},{"key":"#cr-split#-e_1_3_2_2_41_1.2","unstructured":"Xiaonan Nie Xupeng Miao Shijie Cao Lingxiao Ma Qibin Liu Jilong Xue Youshan Miao Yi Liu Zhi Yang and Bin Cui. 2021. EvoMoE: An Evolutional Mixture-of-Experts Training Framework via Dense-To-Sparse Gate. https: \/\/doi.org\/10.48550\/ARXIV.2112.14397"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-2124"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1264"},{"key":"e_1_3_2_2_44_1","unstructured":"Stephen Roller Sainbayar Sukhbaatar Arthur Szlam and Jason E Weston. 2021. Hash Layers For Large Sparse Models. In Advances in Neural Information Processing Systems A. Beygelzimer Y. Dauphin P. Liang and J. Wortman Vaughan (Eds.). https:\/\/openreview.net\/forum?id=lMgDDWb1ULW  Stephen Roller Sainbayar Sukhbaatar Arthur Szlam and Jason E Weston. 2021. Hash Layers For Large Sparse Models. In Advances in Neural Information Processing Systems A. Beygelzimer Y. Dauphin P. Liang and J. Wortman Vaughan (Eds.). https:\/\/openreview.net\/forum?id=lMgDDWb1ULW"},{"key":"e_1_3_2_2_45_1","volume-title":"Daniel Keysers, and Neil Houlsby.","author":"Ruiz Carlos Riquelme","year":"2021","unstructured":"Carlos Riquelme Ruiz , Joan Puigcerver , Basil Mustafa , Maxim Neumann , Rodolphe Jenatton , Andr\u00e9 Susano Pinto , Daniel Keysers, and Neil Houlsby. 2021 . Scaling Vision with Sparse Mixture of Experts. In Advances in Neural Information Processing Systems, A. Beygelzimer, Y. Dauphin, P. Liang, and J. Wortman Vaughan (Eds .). https:\/\/openreview.net\/forum?id=NGPmH3vbAA_ Carlos Riquelme Ruiz, Joan Puigcerver, Basil Mustafa, Maxim Neumann, Rodolphe Jenatton, Andr\u00e9 Susano Pinto, Daniel Keysers, and Neil Houlsby. 2021. Scaling Vision with Sparse Mixture of Experts. In Advances in Neural Information Processing Systems, A. Beygelzimer, Y. Dauphin, P. Liang, and J. Wortman Vaughan (Eds.). https:\/\/openreview.net\/forum?id=NGPmH3vbAA_"},{"key":"e_1_3_2_2_46_1","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","author":"Sabour Sara","unstructured":"Sara Sabour , Nicholas Frosst , and Geoffrey E. Hinton . 2017. Dynamic Routing between Capsules . In Proceedings of the 31st International Conference on Neural Information Processing Systems ( Long Beach, California, USA) (NIPS'17). Curran Associates Inc., Red Hook, NY, USA, 3859--3869. Sara Sabour, Nicholas Frosst, and Geoffrey E. Hinton. 2017. Dynamic Routing between Capsules. In Proceedings of the 31st International Conference on Neural Information Processing Systems (Long Beach, California, USA) (NIPS'17). Curran Associates Inc., Red Hook, NY, USA, 3859--3869."},{"key":"e_1_3_2_2_47_1","volume-title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1ckMDqlg","author":"Shazeer Noam","year":"2017","unstructured":"Noam Shazeer , * Azalia Mirhoseini , * Krzysztof Maziarz , Andy Davis , Quoc Le , Geoffrey Hinton , and Jeff Dean . 2017 . Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1ckMDqlg Noam Shazeer, *Azalia Mirhoseini, *Krzysztof Maziarz, Andy Davis, Quoc Le, Geoffrey Hinton, and Jeff Dean. 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1ckMDqlg"},{"key":"e_1_3_2_2_48_1","volume-title":"Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics","author":"Socher Richard","year":"2013","unstructured":"Richard Socher , Alex Perelygin , Jean Wu , Jason Chuang , Christopher D. Manning , Andrew Ng , and Christopher Potts . 2013 . Recursive Deep Models for Semantic Compositionality Over a Sentiment Treebank . In Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics , Seattle, Washington, USA, 1631--1642. https:\/\/aclanthology.org\/D13-1170 Richard Socher, Alex Perelygin, Jean Wu, Jason Chuang, Christopher D. Manning, Andrew Ng, and Christopher Potts. 2013. Recursive Deep Models for Semantic Compositionality Over a Sentiment Treebank. In Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics, Seattle, Washington, USA, 1631--1642. https:\/\/aclanthology.org\/D13-1170"},{"key":"e_1_3_2_2_49_1","volume-title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rJ4km2R5t7","author":"Wang Alex","unstructured":"Alex Wang , Amanpreet Singh , Julian Michael , Felix Hill , Omer Levy , and Samuel R. Bowman . 2019 . GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rJ4km2R5t7 Alex Wang, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel R. Bowman. 2019. GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rJ4km2R5t7"},{"key":"e_1_3_2_2_50_1","volume-title":"SkipNet: Learning Dynamic Routing in Convolutional Networks. In The European Conference on Computer Vision (ECCV).","author":"Wang Xin","unstructured":"Xin Wang , Fisher Yu , Zi-Yi Dou , Trevor Darrell , and Joseph E. Gonzalez . 2018 . SkipNet: Learning Dynamic Routing in Convolutional Networks. In The European Conference on Computer Vision (ECCV). Xin Wang, Fisher Yu, Zi-Yi Dou, Trevor Darrell, and Joseph E. Gonzalez. 2018. SkipNet: Learning Dynamic Routing in Convolutional Networks. In The European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_2_51_1","volume-title":"Chi","author":"Wang Yuyan","year":"2020","unstructured":"Yuyan Wang , Zhe Zhao , Bo Dai , Christopher Fifty , Dong Lin , Lichan Hong , and Ed H . Chi . 2020 . Small Towers Make Big Differences. ArXiv abs\/2008.05808 (2020). Yuyan Wang, Zhe Zhao, Bo Dai, Christopher Fifty, Dong Lin, Lichan Hong, and Ed H. Chi. 2020. Small Towers Make Big Differences. ArXiv abs\/2008.05808 (2020)."},{"key":"e_1_3_2_2_52_1","volume-title":"Bowman","author":"Warstadt Alex","year":"2019","unstructured":"Alex Warstadt , Amanpreet Singh , and Samuel R . Bowman . 2019 . Neural Network Acceptability Judgments. Transactions of the Association for Computational Linguistics 7 (09 2019), 625--641. https: \/\/doi.org\/10.1162\/tacl_a_00290 arXiv:https:\/\/direct.mit.edu\/tacl\/article-pdf\/doi\/10.1162\/tacl_a_00290\/1923083\/tacl_a_00290.pdf 10.1162\/tacl_a_00290 Alex Warstadt, Amanpreet Singh, and Samuel R. Bowman. 2019. Neural Network Acceptability Judgments. Transactions of the Association for Computational Linguistics 7 (09 2019), 625--641. https: \/\/doi.org\/10.1162\/tacl_a_00290 arXiv:https:\/\/direct.mit.edu\/tacl\/article-pdf\/doi\/10.1162\/tacl_a_00290\/1923083\/tacl_a_00290.pdf"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1101"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_2_55_1","volume-title":"Residual Mixture of Experts. ArXiv abs\/2204.09636","author":"Wu Lemeng","year":"2022","unstructured":"Lemeng Wu , Mengchen Liu , Yinpeng Chen , Dongdong Chen , Xiyang Dai , and Lu Yuan . 2022. Residual Mixture of Experts. ArXiv abs\/2204.09636 ( 2022 ). Lemeng Wu, Mengchen Liu, Yinpeng Chen, Dongdong Chen, Xiyang Dai, and Lu Yuan. 2022. Residual Mixture of Experts. ArXiv abs\/2204.09636 (2022)."},{"key":"e_1_3_2_2_56_1","unstructured":"Yanqi Zhou Tao Lei Hanxiao Liu Nan Du Yanping Huang Vincent Y Zhao Andrew M. Dai Zhifeng Chen Quoc V Le and James Laudon. 2022. Mixture-of-Experts with Expert Choice Routing. In Advances in Neural Information Processing Systems Alice H. Oh Alekh Agarwal Danielle Belgrave and Kyunghyun Cho (Eds.). https:\/\/openreview.net\/forum?id=jdJo1HIVinI  Yanqi Zhou Tao Lei Hanxiao Liu Nan Du Yanping Huang Vincent Y Zhao Andrew M. Dai Zhifeng Chen Quoc V Le and James Laudon. 2022. Mixture-of-Experts with Expert Choice Routing. In Advances in Neural Information Processing Systems Alice H. Oh Alekh Agarwal Danielle Belgrave and Kyunghyun Cho (Eds.). https:\/\/openreview.net\/forum?id=jdJo1HIVinI"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/1060745.1060754"},{"key":"e_1_3_2_2_58_1","unstructured":"Barret Zoph Irwan Bello Sameer Kumar Nan Du Yanping Huang Jeff Dean Noam M. Shazeer and William Fedus. 2022. ST-MoE: Designing Stable and Transferable Sparse Expert Models.  Barret Zoph Irwan Bello Sameer Kumar Nan Du Yanping Huang Jeff Dean Noam M. Shazeer and William Fedus. 2022. ST-MoE: Designing Stable and Transferable Sparse Expert Models."},{"key":"e_1_3_2_2_59_1","volume-title":"Taming Sparsely Activated Transformer with Stochastic Experts. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B72HXs80q4","author":"Zuo Simiao","year":"2022","unstructured":"Simiao Zuo , Xiaodong Liu , Jian Jiao , Young Jin Kim , Hany Hassan , Ruofei Zhang , Jianfeng Gao , and Tuo Zhao . 2022 . Taming Sparsely Activated Transformer with Stochastic Experts. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B72HXs80q4 Simiao Zuo, Xiaodong Liu, Jian Jiao, Young Jin Kim, Hany Hassan, Ruofei Zhang, Jianfeng Gao, and Tuo Zhao. 2022. Taming Sparsely Activated Transformer with Stochastic Experts. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B72HXs80q4"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.116"}],"event":{"name":"KDD '23: The 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Long Beach CA USA","acronym":"KDD '23","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3580305.3599278","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3580305.3599278","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3580305.3599278","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:51:16Z","timestamp":1750182676000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3580305.3599278"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,4]]},"references-count":61,"alternative-id":["10.1145\/3580305.3599278","10.1145\/3580305"],"URL":"https:\/\/doi.org\/10.1145\/3580305.3599278","relation":{},"subject":[],"published":{"date-parts":[[2023,8,4]]},"assertion":[{"value":"2023-08-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}