{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,16]],"date-time":"2024-09-16T19:59:19Z","timestamp":1726516759402},"publisher-location":"Cham","reference-count":212,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030061630"},{"type":"electronic","value":"9783030061647"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-06164-7_11","type":"book-chapter","created":{"date-parts":[[2020,5,8]],"date-time":"2020-05-08T02:03:20Z","timestamp":1588903400000},"page":"341-388","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Statistical Computational Learning"],"prefix":"10.1007","author":[{"given":"Antoine","family":"Cornuejols","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fr\u00e9d\u00e9ric","family":"Koriche","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Richard","family":"Nock","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,5,8]]},"reference":[{"key":"11_CR1","doi-asserted-by":"crossref","unstructured":"Aggarwal C, Reddy C (2013) Data clustering: algorithms and applications. Taylor and Francis","DOI":"10.1201\/b15410"},{"issue":"1","key":"11_CR2","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1016\/j.jcss.2007.04.011","volume":"74","author":"M Alekhnovich","year":"2008","unstructured":"Alekhnovich M, Braverman M, Feldman V, Klivans A, Pitassi T (2008) The complexity of properly learning simple concept classes. J Comput Syst Sci 74(1):16\u201334","journal-title":"J Comput Syst Sci"},{"issue":"1","key":"11_CR3","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1137\/050623905","volume":"20","author":"N Alon","year":"2006","unstructured":"Alon N (2006) Ranking tournaments. SIAM J Discret Math 20(1):137\u2013142","journal-title":"SIAM J Discret Math"},{"issue":"4","key":"11_CR4","doi-asserted-by":"publisher","first-page":"615","DOI":"10.1145\/263867.263927","volume":"44","author":"N Alon","year":"1997","unstructured":"Alon N, Ben-David S, Cesa-Bianchi N, Haussler D (1997) Scale-sensitive dimensions, uniform convergence, and learnability. J ACM (JACM) 44(4):615\u2013631","journal-title":"J ACM (JACM)"},{"key":"11_CR5","unstructured":"Alpaydin E (2009) Introduction to machine learning. MIT, USA"},{"issue":"4","key":"11_CR6","first-page":"343","volume":"2","author":"D Angluin","year":"1987","unstructured":"Angluin D, Laird PD (1987) Learning from noisy examples. Mach Learn 2(4):343\u2013370","journal-title":"Mach Learn"},{"key":"11_CR7","doi-asserted-by":"crossref","unstructured":"Anthony M (2001) Discrete mathematics of neural networks: selected topics. SIAM monographs on. discrete mathematics and applications","DOI":"10.1137\/1.9780898718539"},{"key":"11_CR8","doi-asserted-by":"crossref","unstructured":"Anthony M (2010) Probabilistic learning and boolean functions. In: Crama Y, Hammer P (eds) Boolean models and methods in mathematics, computer science, and engineering, encyclopedia of mathematics and its applications. Cambridge University, Cambridge, pp 197\u2013220","DOI":"10.1017\/CBO9780511780448.009"},{"key":"11_CR9","doi-asserted-by":"crossref","unstructured":"Anthony M, Barlett P (1999) Neural network learning: theoretical foundations. Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511624216"},{"key":"11_CR10","unstructured":"Anthony M, Biggs N (1997) Computational learning theory. Cambridge University, Cambridge"},{"key":"11_CR11","doi-asserted-by":"publisher","first-page":"385","DOI":"10.1017\/S0963548300000778","volume":"2","author":"M Anthony","year":"1993","unstructured":"Anthony M, Shawe-Taylor J (1993) Using the perceptron algorithm to find consistent hypotheses. Comb, Probab Comput 2:385\u2013387","journal-title":"Comb, Probab Comput"},{"issue":"2","key":"11_CR12","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1006\/jcss.1997.1472","volume":"54","author":"S Arora","year":"1997","unstructured":"Arora S, Babai L, Stern J, Sweedyk Z (1997) The hardness of approximate optima in lattices, codes, and systems of linear equations. J Comput Syst Sci 54(2):317\u2013331","journal-title":"J Comput Syst Sci"},{"key":"11_CR13","unstructured":"Bach FR (2008) Exploring large feature spaces with hierarchical multiple kernel learning. In: Advances in neural information processing systems 21 (NIPS 2008), pp 105\u2013112"},{"key":"11_CR14","first-page":"53","volume":"18:19:1\u201319","author":"FR Bach","year":"2017","unstructured":"Bach FR (2017) Breaking the curse of dimensionality with convex neural networks. J Mach Learn Res 18:19:1\u201319:53","journal-title":"J Mach Learn Res"},{"issue":"1","key":"11_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1561\/2200000015","volume":"4","author":"FR Bach","year":"2012","unstructured":"Bach FR, Jenatton R, Mairal J, Obozinski G (2012) Optimization with sparsity-inducing penalties. Found Trends Mach Learn 4(1):1\u2013106","journal-title":"Found Trends Mach Learn"},{"key":"11_CR16","first-page":"807","volume":"14","author":"S Bahmani","year":"2013","unstructured":"Bahmani S, Raj B, Boufounos P (2013) Greedy sparsity-constrained optimization. J Mach Learn Res 14:807\u2013841","journal-title":"J Mach Learn Res"},{"key":"11_CR17","doi-asserted-by":"publisher","first-page":"258","DOI":"10.1016\/j.artint.2015.03.003","volume":"244","author":"M Bartlett","year":"2017","unstructured":"Bartlett M, Cussens J (2017) Integer linear programming for the bayesian network structure learning problem. Artif Intell 244:258\u2013271","journal-title":"Artif Intell"},{"issue":"3","key":"11_CR18","doi-asserted-by":"publisher","first-page":"167","DOI":"10.1016\/S0167-6377(02)00231-6","volume":"31","author":"A Beck","year":"2003","unstructured":"Beck A, Teboulle M (2003) Mirror descent and nonlinear projected subgradient methods for convex optimization. Oper Res Lett 31(3):167\u2013175","journal-title":"Oper Res Lett"},{"issue":"3","key":"11_CR19","doi-asserted-by":"publisher","first-page":"496","DOI":"10.1016\/S0022-0000(03)00038-2","volume":"66","author":"S Ben-David","year":"2003","unstructured":"Ben-David S, Eiron N, Long PM (2003) On the difficulty of approximately maximizing agreements. J Comput Syst Sci 66(3):496\u2013514","journal-title":"J Comput Syst Sci"},{"key":"11_CR20","unstructured":"Berg J, J\u00e4rvisalo M, Malone B (2014) Learning optimal bounded treewidth bayesian networks via maximum satisfiability. In: Proceedings of the 17th international conference on artificial intelligence and statistics (AISTATS 2014), pp 86\u201395"},{"key":"11_CR21","unstructured":"Bertsekas D (2015) Convex optimization algorithms. MIT, USA"},{"key":"11_CR22","doi-asserted-by":"crossref","unstructured":"Birge J, Louveaux F (2011) Introduction to stochastic programming. Springer, Berlin","DOI":"10.1007\/978-1-4614-0237-4"},{"key":"11_CR23","unstructured":"Bishop C (2006) Pattern recognition and machine learning. Springer, Berlin"},{"key":"11_CR24","doi-asserted-by":"crossref","unstructured":"Blum A (1992) Rank-$$r$$ decision trees are a subclass of $$r$$-decision lists. Inf Process Lett 42(4):183\u2013185","DOI":"10.1016\/0020-0190(92)90237-P"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Blum A, Rivest RL (1992) Training a 3-node neural network is NP-complete. Neural Netw 5(1):117\u2013127","DOI":"10.1016\/S0893-6080(05)80010-3"},{"issue":"4","key":"11_CR26","doi-asserted-by":"publisher","first-page":"929","DOI":"10.1145\/76359.76371","volume":"36","author":"A Blumer","year":"1989","unstructured":"Blumer A, Ehrenfeucht A, Haussler D, Warmuth M (1989) Learnability and the Vapnik-Chervonenkis dimension. J ACM (JACM) 36(4):929\u2013965","journal-title":"J ACM (JACM)"},{"key":"11_CR27","doi-asserted-by":"crossref","unstructured":"Boser BE, Guyon L, Vapnik V (1992) A training algorithm for optimal margin classifiers. In: Proceedings of the 5th annual ACM conference on computational learning theory (COLT 1992), pp 144\u2013152","DOI":"10.1145\/130385.130401"},{"key":"11_CR28","doi-asserted-by":"crossref","unstructured":"Bottou L (2007) Large-scale kernel machines, Neural information processing series. MIT, USA","DOI":"10.7551\/mitpress\/7496.001.0001"},{"issue":"Mar","key":"11_CR29","first-page":"499","volume":"2","author":"O Bousquet","year":"2002","unstructured":"Bousquet O, Elisseeff A (2002) Stability and generalization. J Mach Learn Res 2(Mar):499\u2013526","journal-title":"J Mach Learn Res"},{"key":"11_CR30","doi-asserted-by":"crossref","unstructured":"Boyd S, Vandenberghe L (2004) Convex optimization. Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511804441"},{"issue":"2","key":"11_CR31","first-page":"123","volume":"24","author":"L Breiman","year":"1996","unstructured":"Breiman L (1996) Bagging predictors. Mach Learn 24(2):123\u2013140","journal-title":"Mach Learn"},{"issue":"1","key":"11_CR32","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1023\/A:1010933404324","volume":"45","author":"L Breiman","year":"2001","unstructured":"Breiman L (2001) Random forests. Mach Learn 45(1):5\u201332","journal-title":"Mach Learn"},{"issue":"3\u20134","key":"11_CR33","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1561\/2200000050","volume":"8","author":"S Bubeck","year":"2015","unstructured":"Bubeck S (2015) Convex optimization: algorithms and complexity. Found Trends Mach Learn 8(3\u20134):231\u2013358","journal-title":"Found Trends Mach Learn"},{"key":"11_CR34","doi-asserted-by":"crossref","unstructured":"Cao Y, Xu J, Liu T, Li H, Huang Y, Hon H (2006) Adapting ranking SVM to document retrieval. In: Proceedings of the 29th annual international ACM conference on research and development in information retrieval (SIGIR 2006), pp 186\u2013193","DOI":"10.1145\/1148170.1148205"},{"key":"11_CR35","doi-asserted-by":"crossref","unstructured":"Caruana R, Karampatziakis N, Yessenalina A (2008) An empirical evaluation of supervised learning in high dimensions. In: Proceedings of the 25th international conference on machine learning (ICML 2008), pp 96\u2013103","DOI":"10.1145\/1390156.1390169"},{"key":"11_CR36","unstructured":"Cauchy A (1847) M\u00e9thode g\u00e9n\u00e9rale pour la r\u00e9solution des syst\u00e8mes d\u2019\u00e9quations simultan\u00e9es. C. R. Acad. Sci. Paris 25:536\u2013538"},{"key":"11_CR37","unstructured":"Censor Y, Zenios SA (1997) Parallel optimization. Oxford University, Oxford"},{"key":"11_CR38","doi-asserted-by":"crossref","unstructured":"Chickering DM (1996) Learning Bayesian networks is NP-complete. In: Learning from data: artificial intelligence and statistics V, Springer, Berlin, pp 121\u2013130","DOI":"10.1007\/978-1-4612-2404-4_12"},{"key":"11_CR39","doi-asserted-by":"crossref","unstructured":"Cho K, van Merrienboer B, G\u00fcl\u00e7ehre \u00c7, Bahdanau D, Bougares F, Schwenk H, Bengio Y (2014). Learning phrase representations using RNN encoder-decoder for statistical machine translation. In: Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP 2014), pp 1724\u20131734","DOI":"10.3115\/v1\/D14-1179"},{"issue":"3","key":"11_CR40","doi-asserted-by":"publisher","first-page":"462","DOI":"10.1109\/TIT.1968.1054142","volume":"14","author":"CK Chow","year":"1968","unstructured":"Chow CK, Liu CN (1968) Approximating discrete probability distributions with dependence trees. IEEE Trans Inf Theory 14(3):462\u2013467","journal-title":"IEEE Trans Inf Theory"},{"key":"11_CR41","first-page":"30","volume":"6(4):63:1\u201363","author":"KL Clarkson","year":"2010","unstructured":"Clarkson KL (2010) Coresets, sparse greedy approximation, and the Frank-Wolfe algorithm. ACM Trans Algorithms 6(4):63:1\u201363:30","journal-title":"ACM Trans Algorithms"},{"key":"11_CR42","first-page":"2671","volume":"8","author":"S Cl\u00e9men\u00e7on","year":"2007","unstructured":"Cl\u00e9men\u00e7on S, Vayatis N (2007) Ranking the best instances. J Mach Learn Res 8:2671\u20132699","journal-title":"J Mach Learn Res"},{"key":"11_CR43","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1613\/jair.587","volume":"10","author":"W Cohen","year":"1999","unstructured":"Cohen W, Schapire R, Singer Y (1999) Learning to order things. J Artif Intell Res (JAIR) 10:243\u2013270","journal-title":"J Artif Intell Res (JAIR)"},{"key":"11_CR44","unstructured":"Cohen WW, Singer Y (1999). A simple, fast, and effictive rule learner. In: Proceedings of the 16th national conference on artificial intelligence (AAAI 1999), pp 335\u2013342"},{"key":"11_CR45","doi-asserted-by":"crossref","unstructured":"Collobert R, Weston J (2008) A unified architecture for natural language processing: deep neural networks with multitask learning. In: Proceedings of the twenty-fifth international conference in machine learning (ICML 2008), pp 160\u2013167","DOI":"10.1145\/1390156.1390177"},{"key":"11_CR46","unstructured":"Cortes C, Mohri M, Rostamizadeh A (2009) Learning non-linear combinations of kernels. In: Advances in neural information processing systems 22 (NIPS 2009), pp 396\u2013404"},{"key":"11_CR47","unstructured":"Cortes C, Mohri M, Rostamizadeh A (2010) Generalization bounds for learning kernels. In: Proceedings of the 27th international conference on machine learning (ICML 2010), pp 247\u2013254"},{"key":"11_CR48","first-page":"1025","volume":"3","author":"K Crammer","year":"2003","unstructured":"Crammer K, Singer Y (2003) A family of additive online algorithms for category ranking. J Mach Learn Res 3:1025\u20131058","journal-title":"J Mach Learn Res"},{"key":"11_CR49","unstructured":"Criminisi A, Shotton J, Konukoglu E (2012) Decision forests: a unified framework for classification, regression, density estimation, manifold learning and semi-supervised learning. Found Trends Comput Graph Vis 7(2\u20133):81\u2013227"},{"key":"11_CR50","doi-asserted-by":"crossref","unstructured":"Cristianini N, Shawe-Taylor J (2000) An introduction to support vector machines and other kernel-based learning methods. Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511801389"},{"key":"11_CR51","unstructured":"Cussens J (2008) Bayesian network learning by compiling to weighted MAX-SAT. In: Proceedings of the 24th conference in uncertainty in artificial intelligence (UAI 2008), pp 105\u2013112"},{"key":"11_CR52","unstructured":"Cussens J (2011) Bayesian network learning with cutting planes. In: Proceedings of the 27th conference on uncertainty in artificial intelligence (UAI 2011), pp 153\u2013160"},{"key":"11_CR53","unstructured":"Daniely A, Shalev-Shwartz S (2016) Complexity theoretic limitations on learning dnf\u2019s. In: Proceedings of the 29th conference on learning theory (COLT 2016), pp 815\u2013830"},{"key":"11_CR54","doi-asserted-by":"crossref","unstructured":"Darwiche A (2009) Modeling and reasoning with bayesian networks. Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511811357"},{"key":"11_CR55","doi-asserted-by":"crossref","unstructured":"DasGupta A (2011) Probability for statistics and machine learning: fundamentals and. advanced topics. Springer, Berlin","DOI":"10.1007\/978-1-4419-9634-3"},{"key":"11_CR56","unstructured":"Dasgupta S (1999) Learning polytrees. In: Proceedings of the fifteenth conference on uncertainty in artificial intelligence (UAI 1999), pp 134\u2013141"},{"key":"11_CR57","doi-asserted-by":"crossref","unstructured":"De Raedt L (2008) Logical and relational learning. Springer, Berlin","DOI":"10.1007\/978-3-540-68856-3"},{"key":"11_CR58","unstructured":"Dekel O, Manning CD, Singer Y (2003) Log-linear models for label ranking. In: Advances in neural information processing systems 16 (NIPS 2003), pp 497\u2013504"},{"issue":"1","key":"11_CR59","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","volume":"39","author":"AP Dempster","year":"1977","unstructured":"Dempster AP, Laird NM, Rubin DB (1977) Maximum likelihood from incomplete data via the EM algorithm. J R Stat Society Ser B (Methodological) 39(1):1\u201338","journal-title":"J R Stat Society Ser B (Methodological)"},{"key":"11_CR60","unstructured":"Devroye L, Gy\u00f6rfi L, Lugosi G (2013) A probabilistic theory of pattern recognition. Springer, Berlin"},{"key":"11_CR61","doi-asserted-by":"crossref","unstructured":"Du K-L, Swamy MNS (2013) Neural networks and statistical learning. Springer, Berlin","DOI":"10.1007\/978-1-4471-5571-3"},{"key":"11_CR62","doi-asserted-by":"crossref","unstructured":"Duchi JC, Shalev-Shwartz S, Singer Y, Chandra T (2008) Efficient projections onto the l$${}_{\\text{1}}$$-ball for learning in high dimensions. In: Machine learning, proceedings of the twenty-fifth international conference (ICML 2008), pp 272\u2013279","DOI":"10.1145\/1390156.1390191"},{"key":"11_CR63","doi-asserted-by":"crossref","unstructured":"Engel A, Broeck C (2001) Statistical mechanics of learning. Cambridge University, Cambridge","DOI":"10.1017\/CBO9781139164542"},{"issue":"2","key":"11_CR64","doi-asserted-by":"publisher","first-page":"606","DOI":"10.1137\/070684914","volume":"39","author":"V Feldman","year":"2009","unstructured":"Feldman V, Gopalan P, Khot S, Ponnuswami AK (2009) On agnostic learning of parities, monomials, and halfspaces. SIAM J Comput 39(2):606\u2013645","journal-title":"SIAM J Comput"},{"issue":"6","key":"11_CR65","doi-asserted-by":"publisher","first-page":"1558","DOI":"10.1137\/120865094","volume":"41","author":"V Feldman","year":"2012","unstructured":"Feldman V, Guruswami V, Raghavendra P, Wu Y (2012) Agnostic learning of monomials by halfspaces is hard. SIAM J Comput 41(6):1558\u20131590","journal-title":"SIAM J Comput"},{"key":"11_CR66","doi-asserted-by":"crossref","unstructured":"Flach P (2012) Machine learning: the art and science of algorithms that make sense of data. Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511973000"},{"issue":"3","key":"11_CR67","doi-asserted-by":"crossref","first-page":"359","DOI":"10.1111\/j.2517-6161.1986.tb01420.x","volume":"48","author":"MA Fligner","year":"1986","unstructured":"Fligner MA, Verducci JS (1986) Distance based ranking models. J R Stat Soc 48(3):359\u2013369","journal-title":"J R Stat Soc"},{"key":"11_CR68","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1002\/nav.3800030109","volume":"3","author":"M Franck","year":"1956","unstructured":"Franck M, Wolfe P (1956) An algorithm for quadratic programming. Naval Res Logis Q 3:95\u2013110","journal-title":"Naval Res Logis Q"},{"issue":"1\u20132","key":"11_CR69","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1007\/s10107-014-0841-6","volume":"155","author":"RM Freund","year":"2016","unstructured":"Freund RM, Grigas P (2016) New analysis and results for the Frank-Wolfe method. Math Program 155(1\u20132):199\u2013230","journal-title":"Math Program"},{"key":"11_CR70","first-page":"933","volume":"4","author":"Y Freund","year":"2003","unstructured":"Freund Y, Iyer RD, Schapire RE, Singer Y (2003) An efficient boosting algorithm for combining preferences. J Mach Learn Res 4:933\u2013969","journal-title":"J Mach Learn Res"},{"key":"11_CR71","unstructured":"Freund Y, Mason L (1999) The alternating decision tree learning algorithm. In: Proceedings of the 16th international conference on machine learning (ICML 1999), pp 124\u2013133"},{"issue":"1","key":"11_CR72","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1006\/jcss.1997.1504","volume":"55","author":"Y Freund","year":"1997","unstructured":"Freund Y, Schapire RE (1997) A decision-theoretic generalization of on-line learning and an application to boosting. J Comput Syst Sci 55(1):119\u2013139","journal-title":"J Comput Syst Sci"},{"key":"11_CR73","doi-asserted-by":"crossref","unstructured":"F\u00fcrnkranz J, H\u00fcllermeier E (2010) Preference learning. Springer, Berlin","DOI":"10.1007\/978-3-642-14125-6"},{"issue":"3","key":"11_CR74","doi-asserted-by":"publisher","first-page":"1493","DOI":"10.1137\/140985366","volume":"26","author":"D Garber","year":"2016","unstructured":"Garber D, Hazan E (2016) A linearly convergent variant of the conditional gradient algorithm under strong convexity, with applications to online and stochastic optimization. SIAM J Optim 26(3):1493\u20131528","journal-title":"SIAM J Optim"},{"key":"11_CR75","unstructured":"Garber D, Meshi O (2016) Linear-memory and decomposition-invariant linearly convergent conditional gradient algorithm for structured polytopes. In: Advances in neural information processing systems 29 (NIPS 2016), pp 1001\u20131009"},{"issue":"2\u20133","key":"11_CR76","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1007\/s10994-009-5129-3","volume":"76","author":"T G\u00e4rtner","year":"2009","unstructured":"G\u00e4rtner T, Vembu S (2009) On structured output training: hard cases and an efficient alternative. Mach Learn 76(2\u20133):227\u2013242","journal-title":"Mach Learn"},{"key":"11_CR77","doi-asserted-by":"crossref","unstructured":"Gentile C (2003) The robustness of the $$p$$-norm algorithms. Mach Learn 53(3):265\u2013299","DOI":"10.1023\/A:1026319107706"},{"key":"11_CR78","doi-asserted-by":"crossref","unstructured":"Getoor L, Taskar B (2007) Introduction to statistical relational learning. MIT, Cambridge","DOI":"10.7551\/mitpress\/7432.001.0001"},{"key":"11_CR79","unstructured":"Goodfellow I, Bengio Y, Courville A (2016) Deep learning. MIT, USA"},{"key":"11_CR80","doi-asserted-by":"crossref","unstructured":"Graves A, Mohamed A, Hinton GE (2013) Speech recognition with deep recurrent neural networks. In: IEEE international conference on acoustics, speech and signal processing (ICASSP 2013), pp 6645\u20136649","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"11_CR81","doi-asserted-by":"crossref","unstructured":"Gr\u00fcnwald P (2007) The minimum description length principle. MIT, USA","DOI":"10.7551\/mitpress\/4643.001.0001"},{"key":"11_CR82","doi-asserted-by":"crossref","unstructured":"Hastie T, Tibshirani R, Friedman J (2009) The elements of statistical learning: data mining, inference, and prediction. Springer, Berlin","DOI":"10.1007\/978-0-387-84858-7"},{"issue":"2","key":"11_CR83","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1016\/0004-3702(88)90002-1","volume":"36","author":"D Haussler","year":"1988","unstructured":"Haussler D (1988) Quantifying inductive bias: AI learning algorithms and Valiant\u2019s learning framework. Artif Intell 36(2):177\u2013221","journal-title":"Artif Intell"},{"issue":"1","key":"11_CR84","doi-asserted-by":"publisher","first-page":"78","DOI":"10.1016\/0890-5401(92)90010-D","volume":"100","author":"D Haussler","year":"1992","unstructured":"Haussler D (1992) Decision theoretic generalizations of the PAC model for neural net and other learning applications. Inf Comput 100(1):78\u2013150","journal-title":"Inf Comput"},{"key":"11_CR85","unstructured":"Hazan E, Kale S (2012) Projection-free online learning. In: Proceedings of the 29th international conference on machine learning (ICML 2012)"},{"key":"11_CR86","unstructured":"Hegde C, Indyk P, Schmidt L (2015) A nearly-linear time framework for graph-structured sparsity. In: Proceedings of the 32nd international conference on machine learning (ICML 2015), pp 928\u2013937"},{"issue":"2","key":"11_CR87","doi-asserted-by":"publisher","first-page":"240","DOI":"10.1137\/0221019","volume":"21","author":"DP Helmbold","year":"1992","unstructured":"Helmbold DP, Sloan RH, Warmuth MK (1992) Learning integer lattices. SIAM J Comput 21(2):240\u2013266","journal-title":"SIAM J Comput"},{"key":"11_CR88","doi-asserted-by":"crossref","unstructured":"Herbrich R (2002) Learning kernel classifiers: theory and algorithms. MIT, USA","DOI":"10.7551\/mitpress\/4170.001.0001"},{"key":"11_CR89","doi-asserted-by":"crossref","unstructured":"Herbrich R, Graepel T, Obermayer K (2000) Large margin rank boundaries for ordinal regression. In: Advances in large margin classifiers. MIT Press, USA, pp 115\u2013132","DOI":"10.7551\/mitpress\/1113.003.0010"},{"issue":"6","key":"11_CR90","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton G, Deng L, Yu D, Dahl GE, r. Mohamed A, Jaitly N, Senior A, Vanhoucke V, Nguyen P, Sainath TN, Kingsbury B (2012) Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process Mag 29(6):82\u201397","journal-title":"IEEE Signal Process Mag"},{"key":"11_CR91","unstructured":"Hiriart-Urrut JB, Lemar\u00e9chal C (2004) Fundamentals of convex analysis. Springer, Berlin"},{"key":"11_CR92","doi-asserted-by":"crossref","unstructured":"H\u00f6ffgen K, Simon HU (1992) Robust trainability of single neurons. In: Proceedings of the fifth annual acm conference on computational learning theory (COLT 1992), pp 428\u2013439","DOI":"10.1145\/130385.130431"},{"key":"11_CR93","doi-asserted-by":"crossref","unstructured":"Hsieh C, Chang K, Lin C, Keerthi SS, Sundararajan S (2008) A dual coordinate descent method for large-scale linear SVM. In: Proceedings of the 25th international conference on machine learning, pp 408\u2013415","DOI":"10.1145\/1390156.1390208"},{"issue":"16\u201317","key":"11_CR94","doi-asserted-by":"publisher","first-page":"1897","DOI":"10.1016\/j.artint.2008.08.002","volume":"172","author":"E H\u00fcllermeier","year":"2008","unstructured":"H\u00fcllermeier E, F\u00fcrnkranz J, Cheng W, Brinker K (2008) Label ranking by learning pairwise preferences. Artif Intell 172(16\u201317):1897\u20131916","journal-title":"Artif Intell"},{"key":"11_CR95","unstructured":"Jaggi M (2013) Revisiting frank-wolfe: projection-free sparse convex optimization. In: Proceedings of the 30th international conference on machine learning (ICML 2013), pp 427\u2013435"},{"key":"11_CR96","unstructured":"Jain P, Rao N, Dhillon I (2016) Structured sparse regression via greedy hard thresholding. In: Advances in neural information processing systems 29 (NIPS 2016), pp 1516\u20131524"},{"key":"11_CR97","unstructured":"Jain P, Tewari A, Kar P (2014) On iterative hard thresholding methods for high-dimensional M-estimation. In: Advances in neural information processing systems 27 (NIPS 2014), pp 685\u2013693"},{"key":"11_CR98","doi-asserted-by":"crossref","unstructured":"James G, Witten D, Hastie T, Tibshirani R (2013) An introduction to statistical learning: with applications in R. Springer texts in statistics, Springer, New York","DOI":"10.1007\/978-1-4614-7138-7"},{"key":"11_CR99","doi-asserted-by":"crossref","unstructured":"Joachims T (2002) Optimizing search engines using clickthrough data. In: Proceedings of the 8th ACM international conference on knowledge discovery and data mining (SIGKDD 2002), pp 133\u2013142","DOI":"10.1145\/775047.775067"},{"key":"11_CR100","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1016\/0304-3975(78)90006-3","volume":"6","author":"DS Johnson","year":"1978","unstructured":"Johnson DS, Preparata FP (1978) The densest hemisphere problem. Theorertical Comput Sci 6:93\u2013107","journal-title":"Theorertical Comput Sci"},{"key":"11_CR101","first-page":"1865","volume":"13","author":"SM Kakade","year":"2012","unstructured":"Kakade SM, Shalev-Shwartz S, Tewari A (2012) Regularization techniques for learning with matrices. J Mach Learn Res 13:1865\u20131890","journal-title":"J Mach Learn Res"},{"key":"11_CR102","doi-asserted-by":"crossref","unstructured":"Kalchbrenner N, Grefenstette E, Blunsom P (2014) A convolutional neural network for modelling sentences. In: Proceedings of the 52nd annual meeting of the association for computational linguistics (ACL 2014), pp 655\u2013665","DOI":"10.3115\/v1\/P14-1062"},{"key":"11_CR103","doi-asserted-by":"crossref","unstructured":"Kamishima T, Kazawa H, Akaho S (2010) A survey and empirical comparison of object ranking methods. Preference learning. Springer, Berlin, pp 181\u2013201","DOI":"10.1007\/978-3-642-14125-6_9"},{"key":"11_CR104","unstructured":"Kawaguchi K (2016) Deep learning without poor local minima. In: Advances in neural information processing systems 29 (NIPS 2016), pp 586\u2013594"},{"key":"11_CR105","unstructured":"Kawaguchi K, Kaelbling LP, Bengio Y (2017) Generalization in deep learning. CoRR. arXiv:1710.05468"},{"issue":"1","key":"11_CR106","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1002\/scj.10630","volume":"36","author":"H Kazawa","year":"2005","unstructured":"Kazawa H, Hirao T, Maeda E (2005) Order SVM: a kernel method for order learning based on generalized order statistics. Syst Comput Jpn 36(1):35\u201343","journal-title":"Syst Comput Jpn"},{"issue":"4","key":"11_CR107","doi-asserted-by":"publisher","first-page":"807","DOI":"10.1137\/0222052","volume":"22","author":"M Kearns","year":"1993","unstructured":"Kearns M, Li M (1993) Learning in the presence of malicious errors. SIAM J Comput 22(4):807\u2013837","journal-title":"SIAM J Comput"},{"key":"11_CR108","doi-asserted-by":"crossref","unstructured":"Kearns M, Li M, Pitt L, Valiant L (1987) Recent results on boolean concept learning. In: Proceedings of the fourth international workshop on machine learning (ICML 1987), pp 337\u2013352","DOI":"10.1016\/B978-0-934613-41-5.50037-4"},{"issue":"6","key":"11_CR109","doi-asserted-by":"publisher","first-page":"1298","DOI":"10.1145\/195613.195656","volume":"41","author":"M Kearns","year":"1994","unstructured":"Kearns M, Li M, Valiant LG (1994a) Learning boolean formulas. J. ACM 41(6):1298\u20131328","journal-title":"J. ACM"},{"issue":"2","key":"11_CR110","first-page":"115","volume":"17","author":"M Kearns","year":"1994","unstructured":"Kearns M, Schapire R, Sellie L (1994b) Toward efficient agnostic learning. Mach Learn 17(2):115\u2013141","journal-title":"Mach Learn"},{"key":"11_CR111","doi-asserted-by":"crossref","unstructured":"Kearns M, Vazirani U (1994) An introduction to computational learning theory. MIT, USA","DOI":"10.7551\/mitpress\/3897.001.0001"},{"issue":"1","key":"11_CR112","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1006\/inco.1996.2612","volume":"132","author":"J Kivinen","year":"1997","unstructured":"Kivinen J, Warmuth MK (1997) Exponentiated gradient versus gradient descent for linear predictors. Inf Comput 132(1):1\u201363","journal-title":"Inf Comput"},{"issue":"4","key":"11_CR113","doi-asserted-by":"publisher","first-page":"808","DOI":"10.1016\/j.jcss.2003.11.002","volume":"68","author":"AR Klivans","year":"2004","unstructured":"Klivans AR, O\u2019Donnell R, Servedio RA (2004) Learning intersections and thresholds of halfspaces. J Comput Syst Sci 68(4):808\u2013840","journal-title":"J Comput Syst Sci"},{"key":"11_CR114","unstructured":"Klivans AR, Servedio RA (2004) Learning DNF in time 2$$^{\\tilde{o}\\text{(n }^{\\text{1\/3 }}\\text{) }}$$. J Comput Syst Sci 68(2):303\u2013318"},{"issue":"1","key":"11_CR115","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1016\/j.jcss.2008.07.008","volume":"75","author":"AR Klivans","year":"2009","unstructured":"Klivans AR, Sherstov AA (2009) Cryptographic hardness for learning intersections of halfspaces. J Comput Syst Sci 75(1):2\u201312","journal-title":"J Comput Syst Sci"},{"key":"11_CR116","unstructured":"Koller D, Friedman N (2009) Probabilistic graphical models. MIT, USA"},{"key":"11_CR117","doi-asserted-by":"crossref","unstructured":"Krichene W, Krichene S, Bayen AM (2015) Efficient bregman projections onto the simplex. In: Proceedings of the 54th IEEE conference on decision and control, (CDC 2015), pp 3291\u20133298","DOI":"10.1109\/CDC.2015.7402714"},{"key":"11_CR118","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. In: Advances in neural information processing systems 25 (NIPS 2012), pp 1106\u20131114"},{"key":"11_CR119","doi-asserted-by":"crossref","unstructured":"Kulkarni S, Harman G (2011) An elementary introduction to statistical learning theory. Wiley series in probability and statistics, Wiley, New York","DOI":"10.1002\/9781118023471"},{"key":"11_CR120","unstructured":"Kumar KSS, Bach FR (2013) Convex relaxations for learning bounded-treewidth decomposable graphs. In: Proceedings of the 30th international conference on machine learning (ICML 2013), pp 525\u2013533"},{"key":"11_CR121","doi-asserted-by":"crossref","unstructured":"Kung S (2014) Kernel methods and machine learning. Cambridge University, Cambridge","DOI":"10.1017\/CBO9781139176224"},{"key":"11_CR122","unstructured":"Kushner HJ, Yin GG (2010) Stochastic approximation and recursive algorithms and applications. Springer, Berlin"},{"key":"11_CR123","unstructured":"Lacoste-Julien S, Jaggi M (2015) On the global linear convergence of frank-wolfe optimization variants. In: Advances in neural information processing systems 28 (NIPS 2015), pp 496\u2013504"},{"key":"11_CR124","unstructured":"Lacoste-Julien S, Jaggi M, Schmidt MW, Pletscher P (2013) Block-coordinate frank-wolfe optimization for structural SVMs. In: Proceedings of the 30th international conference on machine learning (ICML 2013), pp 53\u201361"},{"key":"11_CR125","first-page":"27","volume":"5","author":"GRG Lanckriet","year":"2004","unstructured":"Lanckriet GRG, Cristianini N, Bartlett PL, Ghaoui LE, Jordan MI (2004) Learning the kernel matrix with semidefinite programming. J Mach Learn Res 5:27\u201372","journal-title":"J Mach Learn Res"},{"issue":"2","key":"11_CR126","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1016\/0167-9473(93)E0056-A","volume":"19","author":"SL Lauritzen","year":"1995","unstructured":"Lauritzen SL (1995) The em algorithm for graphical association models with missing data. Comput Stat Data Anal 19(2):191\u2013201","journal-title":"Comput Stat Data Anal"},{"key":"11_CR127","unstructured":"Lebanon G, Lafferty J (2002) Conditional models on the ranking poset. In: Advances in neural information processing systems 15 (NIPS 2002), pp 415\u2013422"},{"key":"11_CR128","doi-asserted-by":"crossref","unstructured":"LeCun Y, Kavukcuoglu K, Farabet C (2010) Convolutional networks and applications in vision. In: Proceedings of the international symposium on circuits and systems (ISCAS 2010), pp 253\u2013256","DOI":"10.1109\/ISCAS.2010.5537907"},{"key":"11_CR129","unstructured":"Lim CH, Wright SJ (2016) Efficient bregman projections onto the permutahedron and related polytopes. In: Proceedings of the 19th international conference on artificial intelligence and statistics (AISTATS 2016), pp 1205\u20131213"},{"key":"11_CR130","unstructured":"Little R, Rubin D (2014) Statistical analysis with missing data. Wiley, New York"},{"key":"11_CR131","unstructured":"Liu T, Lugosi G, Neu G, Tao D (2017) Algorithmic stability and hypothesis complexity. In: Proceedings of the 34th international conference on machine learning (ICML 2017), pp 2159\u20132167"},{"issue":"1","key":"11_CR132","first-page":"3783","volume":"15","author":"T Lu","year":"2014","unstructured":"Lu T, Boutilier C (2014) Effective sampling and learning for Mallows models with pairwise-preference data. J Mach Learn Res 15(1):3783\u20133829","journal-title":"J Mach Learn Res"},{"issue":"1\u20132","key":"11_CR133","doi-asserted-by":"publisher","first-page":"615","DOI":"10.1007\/s10107-014-0800-2","volume":"152","author":"Z Lu","year":"2015","unstructured":"Lu Z, Xiao L (2015) On the complexity analysis of randomized block-coordinate descent methods. Math Program 152(1\u20132):615\u2013642","journal-title":"Math Program"},{"key":"11_CR134","doi-asserted-by":"crossref","unstructured":"Ma Y, Fu Y (2011) Manifold learning theory and applications. CRC","DOI":"10.1201\/b11431"},{"key":"11_CR135","unstructured":"Mahdavi M, Yang T, Jin R, Zhu S, Yi J (2012) Stochastic gradient descent with only one projection. In: Advances in neural information processing systems 25 (NIPS 2012), pp 503\u2013511"},{"issue":"1\u20132","key":"11_CR136","doi-asserted-by":"publisher","first-page":"114","DOI":"10.1093\/biomet\/44.1-2.114","volume":"44","author":"CL Mallows","year":"1957","unstructured":"Mallows CL (1957) Non-null ranking models. Biometrika 44(1\u20132):114\u2013130","journal-title":"Biometrika"},{"issue":"4","key":"11_CR137","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1007\/BF02187916","volume":"3","author":"N Megiddo","year":"1988","unstructured":"Megiddo N (1988) On the complexity of polyhedral separability. Discret Comput Geom 3(4):325\u2013337","journal-title":"Discret Comput Geom"},{"key":"11_CR138","unstructured":"Meila M, Chen H (2010) Dirichlet process mixtures of generalized mallows models. In: Proceedings of the twenty-sixth conference on uncertainty in artificial intelligence (UAI 2010), pp 358\u2013367"},{"issue":"1","key":"11_CR139","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1007\/s11222-006-5535-3","volume":"16","author":"M Meila","year":"2006","unstructured":"Meila M, Jaakkola TS (2006) Tractable bayesian learning of tree belief networks. Stat Comput 16(1):77\u201392","journal-title":"Stat Comput"},{"issue":"2","key":"11_CR140","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1016\/0004-3702(82)90040-6","volume":"18","author":"T Mitchell","year":"1982","unstructured":"Mitchell T (1982) Generalization as search. Artif Intell 18(2):203\u2013226","journal-title":"Artif Intell"},{"key":"11_CR141","unstructured":"Mitchell T (1997) Machine learning. McGraw-Hill Education"},{"key":"11_CR142","first-page":"2027","volume":"6","author":"L Mohammadi","year":"2005","unstructured":"Mohammadi L, van de Geer S (2005) Asymptotics in empirical risk minimization. J Mach Learn Res 6:2027\u20132047","journal-title":"J Mach Learn Res"},{"key":"11_CR143","unstructured":"Mohri M, Rostamizadeh A, Talwalkar A (2012) Foundations of machine learning. MIT, USA"},{"issue":"1","key":"11_CR144","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1007\/s10444-004-7634-z","volume":"25","author":"S Mukherjee","year":"2006","unstructured":"Mukherjee S, Niyogi P, Poggio T, Rifkin R (2006) Learning theory: stability is sufficient for generalization and necessary and sufficient for consistency of empirical risk minimization. Adv Comput Math 25(1):161\u2013193","journal-title":"Adv Comput Math"},{"key":"11_CR145","unstructured":"Murphy K (2012) Machine learning: a probabilistic perspective. MIT, USA"},{"key":"11_CR146","unstructured":"Natarajan B (1991) Machine learning: a theoretical approach. M. Kaufmann Publishers"},{"issue":"2","key":"11_CR147","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1137\/S0097539792240406","volume":"24","author":"B Natarajan","year":"1995","unstructured":"Natarajan B (1995) Sparse approximate solutions to linear systems. SIAM J Comput 24(2):227\u2013234","journal-title":"SIAM J Comput"},{"key":"11_CR148","unstructured":"Nemirovski A (1995) Efficient methods in convex programming. http:\/\/www2.isye.gatech.edu\/~nemirovs\/Lec_EMCO.pdf"},{"key":"11_CR149","unstructured":"Nemirovski AS, Yudin DB (1983) Problem complexity and method efficiency in optimization. J. Wiley, New York"},{"key":"11_CR150","doi-asserted-by":"crossref","unstructured":"Nesterov Y (2004) Introductory lectures on convex optimization: a basic course. Kluwer Academic Publishers","DOI":"10.1007\/978-1-4419-8853-9"},{"key":"11_CR151","doi-asserted-by":"crossref","unstructured":"Nesterov Y (2012) Efficiency of coordinate descent methods on huge-scale optimization problems. SIAM J Optim 22(2):341\u2013362","DOI":"10.1137\/100802001"},{"key":"11_CR152","unstructured":"Nie S, Mau\u00e1 DD, de Campos CP, Ji Q (2014) Advances in learning bayesian networks of bounded treewidth. In: Advances in neural information processing systems 27 (NIPS 2014), pp 2285\u20132293"},{"key":"11_CR153","doi-asserted-by":"crossref","unstructured":"Parberry I (1994) Circuit complexity and neural networks. MIT, USA","DOI":"10.7551\/mitpress\/1836.001.0001"},{"key":"11_CR154","volume-title":"Probabilistic reasoning in intelligent systems: networks of plausible inference","author":"J Pearl","year":"1988","unstructured":"Pearl J (1988) Probabilistic reasoning in intelligent systems: networks of plausible inference. Morgan Kaufmann, San Mateo"},{"key":"11_CR155","unstructured":"Pinheiro PHO, Collobert R (2014) Recurrent convolutional neural networks for scene labeling. In: Proceedings of the 31th international conference on machine learning (ICML 2014), pp 82\u201390"},{"issue":"4","key":"11_CR156","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1145\/48014.63140","volume":"35","author":"L Pitt","year":"1988","unstructured":"Pitt L, Valiant L (1988) Computational limitations on learning from examples. J ACM 35(4):965\u2013984","journal-title":"J ACM"},{"issue":"10","key":"11_CR157","first-page":"193","volume":"24","author":"RL Plackett","year":"1975","unstructured":"Plackett RL (1975) The analysis of permutations. J R Stat Soc 24(10):193\u2013202","journal-title":"J R Stat Soc"},{"issue":"6981","key":"11_CR158","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1038\/nature02341","volume":"428","author":"T Poffio","year":"2004","unstructured":"Poffio T, Rifkin R, Kukherjee S, Niyogi P (2004) General conditions for predictivity in learning theory. Nature 428(6981):419","journal-title":"Nature"},{"issue":"1","key":"11_CR159","first-page":"81","volume":"1","author":"JR Quinlan","year":"1986","unstructured":"Quinlan JR (1986) Induction of decision trees. Mach Learn 1(1):81\u2013106","journal-title":"Mach Learn"},{"key":"11_CR160","unstructured":"Quinlan JR (1993) C4. 5: Programs for machine learning. Morgan Kaufmann"},{"key":"11_CR161","unstructured":"Quinlan JR (1996) Bagging, boosting, and C4.5. In: Proceedings of the 30th national conference on artificial intelligence (AAAI 1996), pp 725\u2013730"},{"key":"11_CR162","unstructured":"Rakhlin A, Shamir O, Sridharan K (2012) Making gradient descent optimal for strongly convex stochastic optimization. In: Proceedings of the 29th international conference on machine learning (ICML 2012)"},{"key":"11_CR163","doi-asserted-by":"crossref","unstructured":"Rish I, Grabarnik G (2014) Sparse modeling: theory, algorithms, and applications. CRC","DOI":"10.1201\/b17758"},{"key":"11_CR164","doi-asserted-by":"crossref","unstructured":"Rissanen J (1983) A universal prior for integers and estimation by minimum description length. Ann stat 416\u2013431","DOI":"10.1214\/aos\/1176346150"},{"key":"11_CR165","unstructured":"Rissanen J (1985) Minimum description length principle. Wiley, New York"},{"issue":"3","key":"11_CR166","first-page":"229","volume":"2","author":"RL Rivest","year":"1987","unstructured":"Rivest RL (1987) Learning decision lists. Mach Learn 2(3):229\u2013246","journal-title":"Mach Learn"},{"issue":"3","key":"11_CR167","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method. Ann Math Stat 22(3):400\u2013407","journal-title":"Ann Math Stat"},{"key":"11_CR168","doi-asserted-by":"crossref","unstructured":"Rockafellar T (1970) Convex analysis. Princeton University, Princeton","DOI":"10.1515\/9781400873173"},{"key":"11_CR169","doi-asserted-by":"publisher","first-page":"386","DOI":"10.1037\/h0042519","volume":"65","author":"F Rosenblatt","year":"1958","unstructured":"Rosenblatt F (1958) The perceptron: a probabilistic model for information storage and organization in the brain. Psychol Rev 65:386\u2013408","journal-title":"Psychol Rev"},{"key":"11_CR170","unstructured":"Rumelhart DE, Hinton GE, Williams RJ (1986) Learning internal representations by error propagation. In: Parallel distributed processing: explorations in the microstructure of cognition, vol 1. MIT, USA, pp 318\u2013362"},{"key":"11_CR171","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1016\/0097-3165(72)90019-2","volume":"13","author":"N Sauer","year":"1972","unstructured":"Sauer N (1972) On the density of families of sets. J Comb Theory 13:145\u2013147","journal-title":"J Comb Theory"},{"key":"11_CR172","first-page":"197","volume":"5","author":"RE Schapire","year":"1990","unstructured":"Schapire RE (1990) The strength of weak learnability. Mach Learn 5:197\u2013227","journal-title":"Mach Learn"},{"key":"11_CR173","doi-asserted-by":"crossref","unstructured":"Schapire RE, Freund Y (2012) Boosting. MIT, USA","DOI":"10.7551\/mitpress\/8291.001.0001"},{"issue":"3","key":"11_CR174","doi-asserted-by":"publisher","first-page":"297","DOI":"10.1023\/A:1007614523901","volume":"37","author":"RE Schapire","year":"1999","unstructured":"Schapire RE, Singer Y (1999) Improved boosting algorithms using confidence-rated predictions. Mach Learn 37(3):297\u2013336","journal-title":"Mach Learn"},{"key":"11_CR175","doi-asserted-by":"crossref","unstructured":"Sch\u00f6lkopf B, Herbrich R, Smola AJ (2001) A generalized representer theorem. In: Proceedings of the 14th annual conference on computational on computational learning theory (COLT 2001), pp 416\u2013426","DOI":"10.1007\/3-540-44581-1_27"},{"key":"11_CR176","doi-asserted-by":"crossref","unstructured":"Sch\u00f6lkopf B, Smola A (2002) Learning with Kernels: support vector machines, regularization, optimization, and beyond. Adaptive computation and machine learning, MIT, USA","DOI":"10.7551\/mitpress\/4175.001.0001"},{"key":"11_CR177","doi-asserted-by":"crossref","unstructured":"Shalev-Shwartz S, Ben-David S (2014) Understanding machine learning: from theory to algorithms. Cambridge University, Cambridge","DOI":"10.1017\/CBO9781107298019"},{"key":"11_CR178","unstructured":"Shalev-Shwartz S, Shamir O, Shammah S (2017) Failures of gradient-based deep learning. In: Proceedings of the 34th international conference on machine learning (ICML 2017), pp 3067\u20133075"},{"key":"11_CR179","unstructured":"Shalev-Shwartz S, Shamir O, Srebro N, Sridharan K (2009) Stochastic convex optimization. In: Proceedings of the 22nd conference on learning theory (COLT 2009), pp 177\u2013186"},{"key":"11_CR180","first-page":"2635","volume":"11","author":"S Shalev-Shwartz","year":"2010","unstructured":"Shalev-Shwartz S, Shamir O, Srebro N, Sridharan K (2010) Learnability, stability and uniform convergence. J Mach Learn Res 11:2635\u20132670","journal-title":"J Mach Learn Res"},{"key":"11_CR181","doi-asserted-by":"crossref","unstructured":"Shalev-Shwartz S, Singer Y, Srebro N (2007) Pegasos: primal estimated sub-gradient solver for SVM. In: Proceedings of the 24th international conference on machine learning (ICML 2007), pp 807\u2013814","DOI":"10.1145\/1273496.1273598"},{"issue":"6","key":"11_CR182","doi-asserted-by":"publisher","first-page":"2807","DOI":"10.1137\/090759574","volume":"20","author":"S Shalev-Shwartz","year":"2010","unstructured":"Shalev-Shwartz S, Srebro N, Zhang T (2010) Trading accuracy for sparsity in optimization problems with sparsity constraints. SIAM J Optim 20(6):2807\u20132832","journal-title":"SIAM J Optim"},{"key":"11_CR183","unstructured":"Shalev-Shwartz S, Tewari A (2011) Stochastic methods for l$${}_{\\text{1 }}$$-regularized loss minimization. J Mach Learn Res 12:1865\u20131892"},{"key":"11_CR184","doi-asserted-by":"crossref","unstructured":"Shawe-Taylor J, Cristianini N (2004) Kernel methods for pattern analysis. Kernel methods for pattern analysis, Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511809682"},{"key":"11_CR185","unstructured":"Song L, Vempala S, Wilmes J, Xie B (2017) On the complexity of learning neural networks. CoRR. arXiv:1707.04615"},{"key":"11_CR186","doi-asserted-by":"crossref","unstructured":"Sra S, Nowozin S, Wright S (2012) Optimization for machine learning. Neural information processing series, MIT, USA","DOI":"10.7551\/mitpress\/8996.001.0001"},{"issue":"1","key":"11_CR187","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/S0004-3702(02)00360-0","volume":"143","author":"N Srebro","year":"2003","unstructured":"Srebro N (2003) Maximum likelihood bounded tree-width Markov networks. Artif Intell 143(1):123\u2013138","journal-title":"Artif Intell"},{"key":"11_CR188","unstructured":"Sridharan K (2012) Learning from an optimization viewpoint. Ph.D. thesis, Technicological Institute of Chicago, Toyota"},{"key":"11_CR189","unstructured":"Steinwart I, Christmann A (2008) Support vector machines. Information science and statistics, Springer, Berlin"},{"key":"11_CR190","doi-asserted-by":"crossref","unstructured":"Sugiyama M (2015) Introduction to statistical machine learning. Elsevier Science","DOI":"10.1016\/B978-0-12-802121-7.00012-1"},{"key":"11_CR191","doi-asserted-by":"crossref","unstructured":"Theodoridis S (2015) Machine learning: a bayesian and optimization perspective. Elsevier Science","DOI":"10.1016\/B978-0-12-801522-3.00012-4"},{"issue":"5","key":"11_CR192","first-page":"195","volume":"39","author":"A Tikhonov","year":"1943","unstructured":"Tikhonov A (1943) On the stability of inverse problems. Doklady Akademii Nauk SSSR 39(5):195\u2013198","journal-title":"Doklady Akademii Nauk SSSR"},{"issue":"1\u20132","key":"11_CR193","doi-asserted-by":"publisher","first-page":"387","DOI":"10.1007\/s10107-007-0170-0","volume":"117","author":"P Tseng","year":"2009","unstructured":"Tseng P, Yun S (2009) A coordinate gradient descent method for nonsmooth separable minimization. Math Program 117(1\u20132):387\u2013423","journal-title":"Math Program"},{"key":"11_CR194","doi-asserted-by":"publisher","first-page":"433","DOI":"10.1093\/mind\/LIX.236.433","volume":"59","author":"A Turing","year":"1950","unstructured":"Turing A (1950) Computing machinery and intelligence. Mind 59:433\u2013460","journal-title":"Mind"},{"issue":"11","key":"11_CR195","doi-asserted-by":"publisher","first-page":"1134","DOI":"10.1145\/1968.1972","volume":"27","author":"LG Valiant","year":"1984","unstructured":"Valiant LG (1984) A theory of the learnable. Commun ACM 27(11):1134\u20131142","journal-title":"Commun ACM"},{"key":"11_CR196","doi-asserted-by":"crossref","unstructured":"van Beek P, Hoffmann H (2015) Machine learning of bayesian networks using constraint programming. In: Proceedings of the 21st confernce on principles and practice of constraint programming (CP 2015), pp 429\u2013445","DOI":"10.1007\/978-3-319-23219-5_31"},{"key":"11_CR197","unstructured":"Vapnik V (1998) Statistical learning theory. Wiley, New York"},{"key":"11_CR198","volume-title":"The nature of statistical learning theory","author":"V Vapnik","year":"2013","unstructured":"Vapnik V (2013) The nature of statistical learning theory, 3rd edn. Springer, Berlin","edition":"3"},{"key":"11_CR199","unstructured":"Vapnik V, Chervonenkis A (1974) Theory of pattern recognition. Nauka, Moskow (in Russian)"},{"key":"11_CR200","doi-asserted-by":"crossref","unstructured":"Vembu S, G\u00e4rtner T (2010) Label ranking algorithms: a survey. In: Preference learning. Springer, Berlin, pp 45\u201364","DOI":"10.1007\/978-3-642-14125-6_3"},{"key":"11_CR201","unstructured":"Vembu S, G\u00e4rtner T, Boley M (2009) Probabilistic structured predictors. In: Proceedings of the twenty-fifth conference on uncertainty in artificial intelligence (UAI 2009), pp 557\u2013564"},{"key":"11_CR202","unstructured":"Viola PA, Jones MJ (2001) Robust real-time face detection. In: Proceedings of the 8th international conference on computer vision ICCV 2001, p 747"},{"issue":"1\u20132","key":"11_CR203","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1561\/2200000001","volume":"1","author":"M Wainwright","year":"2008","unstructured":"Wainwright M, Jordan M (2008) Graphical models, exponential families, and variational inference. Found Trends Mach Learn 1(1\u20132):1\u2013305","journal-title":"Found Trends Mach Learn"},{"key":"11_CR204","doi-asserted-by":"crossref","unstructured":"Watanabe S (2009) Algebraic Geometry and Statistical Learning Theory. Cambridge University, Cambridge","DOI":"10.1017\/CBO9780511800474"},{"key":"11_CR205","doi-asserted-by":"crossref","unstructured":"Webb A, Copsey K (2011) Statistical pattern recognition. Wiley, New York","DOI":"10.1002\/9781119952954"},{"key":"11_CR206","unstructured":"Wibisono A, Rosasco L, Poggio T (2009) Sufficient conditions for uniform stability of regularization algorithms. Technical Report MIT-CSAIL-TR-2009-060. MIT, Computer Science and artificial intelligence laboratory"},{"issue":"1","key":"11_CR207","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s10107-015-0892-3","volume":"151","author":"SJ Wright","year":"2015","unstructured":"Wright SJ (2015) Coordinate descent algorithms. Math Program 151(1):3\u201334","journal-title":"Math Program"},{"key":"11_CR208","doi-asserted-by":"crossref","unstructured":"Xu J, Li H (2007) AdaRank: a boosting algorithm for information retrieval. In: Proceedings of the 30th annual international ACM conference on research and development in information retrieval (SIGIR 2007), pp 391\u2013398","DOI":"10.1145\/1277741.1277809"},{"key":"11_CR209","unstructured":"Zhang L, Yang T, Jin R, He X (2013) $$O(\\log T)$$ projections for stochastic optimization of smooth and strongly convex functions. In: Proceedings of the 30th international conference on machine learning (ICML 2013), pp 1121\u20131129"},{"key":"11_CR210","unstructured":"Zhang X (2010) Empirical risk minimization. In: Sammut C, Webb G, (eds) Encyclopedia of machine learning, Springer, Berlin, p 312"},{"key":"11_CR211","unstructured":"Zhang Y, Liang P, Wainwright M (2017) Convexified convolutional neural networks. In: Proceedings of the 34th international conference on machine learning (ICML 2017), pp 4044\u20134053"},{"key":"11_CR212","unstructured":"Zhao Z, Piech P, Xia L (2016) Learning mixtures of plackett-luce models. In: Proceedings of the 33nd international conference on machine learning (ICML 2016), pp 2906\u20132914"}],"container-title":["A Guided Tour of Artificial Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-06164-7_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,5]],"date-time":"2024-08-05T14:02:20Z","timestamp":1722866540000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-06164-7_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030061630","9783030061647"],"references-count":212,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-06164-7_11","relation":{},"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"8 May 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}