{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T15:40:28Z","timestamp":1735746028945,"version":"3.32.0"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"1-3","license":[{"start":{"date-parts":[[2005,6,2]],"date-time":"2005-06-02T00:00:00Z","timestamp":1117670400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2005,9]]},"DOI":"10.1007\/s10994-005-0928-7","type":"journal-article","created":{"date-parts":[[2005,6,10]],"date-time":"2005-06-10T14:37:33Z","timestamp":1118414253000},"page":"229-250","source":"Crossref","is-referenced-by-count":11,"title":["Combining Statistical Language Models via the Latent Maximum Entropy Principle"],"prefix":"10.1007","volume":"60","author":[{"given":"Shaojun","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dale","family":"Schuurmans","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fuchun","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunxin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2005,6,2]]},"reference":[{"issue":"4","key":"928_CR1","first-page":"597","volume":"23","author":"S. Abney","year":"1997","unstructured":"Abney, S. (1997). Stochastic attribute-value grammars. Computational Linguistics, 23:4, 597\u2013618.","journal-title":"Computational Linguistics"},{"issue":"8","key":"928_CR2","doi-asserted-by":"crossref","first-page":"1279","DOI":"10.1109\/5.880084","volume":"88","author":"J. Bellegarda","year":"2000","unstructured":"Bellegarda, J. (2000). Exploiting latent semantic information in statistical language modeling. Proceedings of IEEE, 88:8, 1279\u20131296.","journal-title":"Proceedings of IEEE"},{"key":"928_CR3","first-page":"1137","volume":"3","author":"Y. Bengio","year":"2003","unstructured":"Bengio, Y., Ducharme, R., Vincent, P., & Jauvin, C. (2003). A neural probabilistic language model. Journal of Machine Learning Research, 3, 1137\u20131155.","journal-title":"Journal of Machine Learning Research"},{"issue":"1","key":"928_CR4","first-page":"39","volume":"22","author":"A. Berger","year":"1996","unstructured":"Berger, A., Della Pietra, S.,& Della Pietra, V. (1996).Amaximum entropy approach to natural language processing. Computational Linguistics, 22:1, 39\u201371.","journal-title":"Computational Linguistics"},{"key":"928_CR5","doi-asserted-by":"crossref","unstructured":"Borwein, J., & Lewis, A. (2000). Convex analysis and nonlinear optimization: Theory and examples. Springer.","DOI":"10.1007\/978-1-4757-9859-3"},{"key":"928_CR6","unstructured":"Brown, P., Della Pietra, S., Della Pietra, V., Mercer, R., Nadas, A., & Roukos, S. (1992). A maximum entropy construction of conditional log-linear language and translation models using learned features and a generalized Csiszar algorithm. IBM Report."},{"issue":"4","key":"928_CR7","doi-asserted-by":"crossref","first-page":"283","DOI":"10.1006\/csla.2000.0147","volume":"14","author":"C., Chelba","year":"2000","unstructured":"Chelba, C., & Jelinek, F. (2000). Structured language modeling. Computer Speech and Language, 14:4, 283\u2013332.","journal-title":"Computer Speech and Language"},{"issue":"4","key":"928_CR8","doi-asserted-by":"crossref","first-page":"319","DOI":"10.1006\/csla.1999.0128","volume":"13","author":"S., Chen","year":"1999","unstructured":"Chen, S., & Goodman, J. (1999). An empirical study of smoothing techniques for language modeling. Computer Speech and Language, 13:4, 319\u2013358.","journal-title":"Computer Speech and Language"},{"issue":"1","key":"928_CR9","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1109\/89.817452","volume":"8","author":"S., Chen","year":"2000","unstructured":"Chen, S., & Rosenfeld, R. (2000). A survey of smoothing techniques for ME models. IEEE Trans. on Speech and Audio Processing, 8:1, 37\u2013244.","journal-title":"IEEE Trans. on Speech and Audio Processing"},{"key":"928_CR10","doi-asserted-by":"crossref","unstructured":"Clarkson, P., & Rosenfeld, R. (1997). Statistical language modeling using the CMU-Cambridge toolkit. Proceedings of Eurospeech, 2707\u20132710.","DOI":"10.21437\/Eurospeech.1997-683"},{"key":"928_CR11","doi-asserted-by":"crossref","unstructured":"Csiszar, I. (1996). Maxent, mathematics, and information theory. In K. Hanson and R. Silver (Eds.), Maximum Entropy and Bayesian Methods (pp. 35\u201350). Kluwer Academic Publishers","DOI":"10.1007\/978-94-011-5430-7_5"},{"issue":"5","key":"928_CR12","doi-asserted-by":"crossref","first-page":"1470","DOI":"10.1214\/aoms\/1177692379","volume":"43","author":"J., Darroch","year":"1972","unstructured":"Darroch, J., & Ratchliff, D. (1972).Generalized iterative scaling for log-linear models. The Annals ofMathematical Statistics, 43:5, 1470\u20131480.","journal-title":"The Annals of Mathematical Statistics"},{"issue":"4","key":"928_CR13","doi-asserted-by":"crossref","first-page":"380","DOI":"10.1109\/34.588021","volume":"19","author":"S. Della Pietra","year":"1997","unstructured":"Della Pietra, S., Della Pietra, V., & Lafferty, J. (1997). Inducing features of random fields. IEEE Transactions on Pattern Analysis and Machine Intelligence, 19:4, 380\u2013393.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"928_CR14","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","volume":"39","author":"A. Dempster","year":"1977","unstructured":"Dempster, A., Laird, N., & Rubin, D. (1977). Maximum likelihood estimation from incomplete data via the EM algorithm. Journal of Royal Statistical Society, Series B, 39, 1\u201338.","journal-title":"Journal of Royal Statistical Society, Series B"},{"key":"928_CR15","doi-asserted-by":"crossref","unstructured":"Durbin, R., Eddy, S., Krogh, A., & Mitchison, G. (1998). Biological sequence analysis: Probabilistic models of proteins and nucleic acids. Cambridge University Press.","DOI":"10.1017\/CBO9780511790492"},{"issue":"1","key":"928_CR16","doi-asserted-by":"crossref","first-page":"177","DOI":"10.1023\/A:1007617005950","volume":"42","author":"T. Hofmann","year":"2001","unstructured":"Hofmann, T. (2001). Unsupervised learning by probabilistic latent semantic analysis. Machine Learning, 42:1, 177\u2013196.","journal-title":"Machine Learning"},{"key":"928_CR17","unstructured":"Jaynes, E. (1983). Papers on probability, statistics, and statistical physics. In R. Rosenkrantz & D. Reidel, Publishing Company."},{"key":"928_CR18","unstructured":"Jelinek, F., & Mercer, R. (1980). Interpolated estimation of Markov source parameters from sparse data. In E. Gelsema, & L. Kanal, (Eds.), Pattern Recognition in Practice. (pp. 381\u2013397) North Holland."},{"key":"928_CR19","unstructured":"Jelinek, F. (1998). Statistical methods for speech recognition. MIT Press."},{"issue":"4","key":"928_CR20","doi-asserted-by":"crossref","first-page":"355","DOI":"10.1006\/csla.2000.0149","volume":"14","author":"S., Khudanpur","year":"2000","unstructured":"Khudanpur, S., & Wu, J. (2000).Maximum entropy techniques for exploiting syntactic, semantic and collocational dependencies in language modeling. Computer Speech and Language, 14:4, 355\u2013372.","journal-title":"Computer Speech and Language"},{"key":"928_CR21","doi-asserted-by":"crossref","unstructured":"Lafferty, J., & Zhai, C. (2001). Document language models, query models, and risk minimization for information retrieval. In ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR).","DOI":"10.1145\/383952.383970"},{"key":"928_CR22","doi-asserted-by":"crossref","first-page":"191","DOI":"10.1016\/0167-9473(93)E0056-A","volume":"1","author":"S. Lauritzen","year":"1995","unstructured":"Lauritzen, S. (1995). The EM-algorithm for graphical association models with missing data. Computational Statistics and Data Analysis, 1, 191\u2013201.","journal-title":"Computational Statistics and Data Analysis"},{"issue":"3\u20134","key":"928_CR23","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1023\/B:INRT.0000011209.19643.e2","volume":"7","author":"F. Peng","year":"2004","unstructured":"Peng, F., Schuurmans, D., & Wang, S. (2004). Augumenting naive Bayes text classifier using statistical language models. Information Retrieval, 7:3\u20134, 317\u2013345.","journal-title":"Information Retrieval"},{"key":"928_CR24","unstructured":"Riezler, S. (1999). Probabilistic constraint logic programming. Ph.D. Dissertation, University of Stuttgart, Germany."},{"issue":"2","key":"928_CR25","doi-asserted-by":"crossref","first-page":"249","DOI":"10.1162\/089120101750300526","volume":"27","author":"B. Roark","year":"2001","unstructured":"Roark, B. (2001). Probabilistic top-down parsing and language modeling. Computational Linguistics, 27:2, 249\u2013285.","journal-title":"Computational Linguistics"},{"key":"928_CR26","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1006\/csla.1996.0011","volume":"10","author":"R. Rosenfeld","year":"1996","unstructured":"Rosenfeld, R. (1996). A maximum entropy approach to adaptive statistical language modeling. Computer Speech and Language, 10, 187\u2013228.","journal-title":"Computer Speech and Language"},{"issue":"8","key":"928_CR27","doi-asserted-by":"crossref","first-page":"1270","DOI":"10.1109\/5.880083","volume":"88","author":"R. Rosenfeld","year":"2000","unstructured":"Rosenfeld, R. (2000). Two decades of statistical language modeling: Where do we go from here?. Proceedings of the IEEE, 88:8, 1270\u20131278.","journal-title":"Proceedings of the IEEE"},{"key":"928_CR28","doi-asserted-by":"crossref","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","volume":"27","author":"C. Shannon","year":"1948","unstructured":"Shannon, C. (1948). A mathematical theory of communication. Bell System Technical Journal, 27, 379\u2013423.","journal-title":"Bell System Technical Journal"},{"key":"928_CR29","unstructured":"Wainwright, M., & Jordan, M. (2003). Graphical models, exponential families, and variational inference. Technical Report 649, Department of Statistics, University of California, Berkeley."},{"key":"928_CR30","unstructured":"Wang, S., Schuurmans, D., & Zhao, Y. (2003). The latent maximum entropy principle. Manuscript submitted."},{"key":"928_CR31","unstructured":"Wang, S., Schuurmans, D., Peng, F., & Zhao, Y. (2004). Learning mixture models with the regularized latent maximum entropy principle. IEEE Transactions on Neural Networks: Special Issue on Information Theoretic Learning, 154."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-005-0928-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10994-005-0928-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-005-0928-7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T15:14:29Z","timestamp":1735744469000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10994-005-0928-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2005,6,2]]},"references-count":31,"journal-issue":{"issue":"1-3","published-print":{"date-parts":[[2005,9]]}},"alternative-id":["928"],"URL":"https:\/\/doi.org\/10.1007\/s10994-005-0928-7","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"type":"print","value":"0885-6125"},{"type":"electronic","value":"1573-0565"}],"subject":[],"published":{"date-parts":[[2005,6,2]]}}}