{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T23:41:52Z","timestamp":1782862912566,"version":"3.54.5"},"reference-count":76,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2019,11,12]],"date-time":"2019-11-12T00:00:00Z","timestamp":1573516800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,11,12]],"date-time":"2019-11-12T00:00:00Z","timestamp":1573516800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Computing"],"published-print":{"date-parts":[[2020,3]]},"DOI":"10.1007\/s00607-019-00768-7","type":"journal-article","created":{"date-parts":[[2019,11,12]],"date-time":"2019-11-12T15:56:35Z","timestamp":1573574195000},"page":"717-740","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":161,"title":["A survey of word embeddings based on deep learning"],"prefix":"10.1007","volume":"102","author":[{"given":"Shirui","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,11,12]]},"reference":[{"issue":"1","key":"768_CR1","doi-asserted-by":"publisher","first-page":"100","DOI":"10.1017\/S1351324909005129","volume":"16","author":"C Manning","year":"2010","unstructured":"Manning C, Raghavan P, Sch\u00fctze H (2010) Introduction to information retrieval. Nat Lang Eng 16(1):100\u2013103","journal-title":"Nat Lang Eng"},{"issue":"1\u20134","key":"768_CR2","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1007\/s13042-010-0001-0","volume":"1","author":"Y Zhang","year":"2010","unstructured":"Zhang Y, Jin R, Zhou Z-H (2010) Understanding bag-of-words model: a statistical framework. Int J Mach Learn Cybern 1(1\u20134):43\u201352","journal-title":"Int J Mach Learn Cybern"},{"key":"768_CR3","unstructured":"Firth JR (1957) A synopsis of linguistic theory, 1930\u20131955. In: Studies in linguistic analysis, Philological Society, Oxford"},{"issue":"2\u20133","key":"768_CR4","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1080\/00437956.1954.11659520","volume":"10","author":"ZS Harris","year":"1954","unstructured":"Harris ZS (1954) Distributional structure. Word 10(2\u20133):146\u2013162","journal-title":"Word"},{"issue":"1\u20133","key":"768_CR5","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1023\/A:1007537716579","volume":"34","author":"I Dagan","year":"1999","unstructured":"Dagan I, Lee L, Pereira FCN (1999) Similarity-based models of word co-occurrence probabilities. Mach Learn 34(1\u20133):43\u201369","journal-title":"Mach Learn"},{"key":"768_CR6","doi-asserted-by":"crossref","unstructured":"Dagan I, Marcus S, Markovitch S (1993) Contextual word similarity and estimation from sparse data. In: Proceedings of the 31st annual meeting on association for computational linguistics, pp 164\u2013171","DOI":"10.3115\/981574.981596"},{"key":"768_CR7","unstructured":"Sch\u00fctze H (1992) Context space. In: AAAI fall symposium on probabilistic approaches to natural language, pp 113\u2013120"},{"key":"768_CR8","doi-asserted-by":"crossref","unstructured":"Sch\u00fctze H (1992) Dimensions of meaning. In: Supercomputing\u201992: proceedings of the 1992 ACM\/IEEE conference on supercomputing, IEEE, pp 787\u2013796","DOI":"10.1109\/SUPERC.1992.236684"},{"key":"768_CR9","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning CD (2014) Glove: global vectors for word representation. In: Empirical methods in natural language processing (EMNLP), pp 1532\u20131543","DOI":"10.3115\/v1\/D14-1162"},{"key":"768_CR10","doi-asserted-by":"crossref","unstructured":"Baroni M, Dinu G, Kruszewski G (2014) Don\u2019t count, predict! A systematic comparison of context-counting vs. context-predicting semantic vectors. In: Proceedings of the 52nd annual meeting of the association for computational linguistics, vol 1, pp 238\u2013247","DOI":"10.3115\/v1\/P14-1023"},{"key":"768_CR11","unstructured":"Turian J, Ratinov L, Bengio Y (2010) Word representations: a simple and general method for semi-supervised learning. In: Proceedings of the 48th annual meeting of the association for computational linguistics, Association for Computational Linguistics, pp 384\u2013394"},{"key":"768_CR12","doi-asserted-by":"crossref","unstructured":"Lebret R, Collobert R (2014) Word embeddings through hellinger pca. In: EACL, p 482","DOI":"10.3115\/v1\/E14-1051"},{"issue":"2\u20133","key":"768_CR13","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1080\/01638539809545028","volume":"25","author":"TK Landauer","year":"1998","unstructured":"Landauer TK, Foltz PW, Laham D (1998) An introduction to latent semantic analysis. Discourse Process 25(2\u20133):259\u2013284","journal-title":"Discourse Process"},{"key":"768_CR14","unstructured":"Dhillon PS, Foster DP, Ungar LH (2011) Multi-view learning of word embeddings via CCA. In: Advances in neural information processing systems, pp 199\u2013207"},{"key":"768_CR15","first-page":"3035","volume":"16","author":"PS Dhillon","year":"2015","unstructured":"Dhillon PS, Foster DP, Ungar LH (2015) Eigenwords: spectral word embeddings. J Mach Learn Res 16:3035\u20133078","journal-title":"J Mach Learn Res"},{"key":"768_CR16","doi-asserted-by":"crossref","unstructured":"Pereira F, Tishby N, Lee L (1993) Distributional clustering of english words. In: Proceedings of the 31st annual meeting on association for computational linguistics, pp 183\u2013190","DOI":"10.3115\/981574.981598"},{"issue":"4","key":"768_CR17","first-page":"467","volume":"18","author":"PF Brown","year":"1992","unstructured":"Brown PF, Desouza PV, Mercer RL, Pietra VJD, Lai JC (1992) Class-based n-gram models of natural language. Comput Linguist 18(4):467\u2013479","journal-title":"Comput Linguist"},{"key":"768_CR18","doi-asserted-by":"crossref","unstructured":"Lin D, Wu X (2009) Phrase clustering for discriminative learning. In: Proceedings of the joint conference of the 47th annual meeting of the ACL and the 4th international joint conference on natural language processing, pp 1030\u20131038","DOI":"10.3115\/1690219.1690290"},{"key":"768_CR19","unstructured":"Bengio Y, Ducharme R, Vincent P (2001) A neural probabilistic language model. In: Advances in neural information processing systems, pp 932\u2013938"},{"key":"768_CR20","first-page":"1137","volume":"3","author":"Y Bengio","year":"2003","unstructured":"Bengio Y, Ducharme R, Vincent P, Jauvin C (2003) A neural probabilistic language model. J Mach Learn Res 3:1137\u20131155","journal-title":"J Mach Learn Res"},{"key":"768_CR21","unstructured":"Turian J, Ratinov L, Bengio Y (2010) Word representations: a simple and general method for semi-supervised learning. In: Proceedings of the 48th annual meeting of the association for computational linguistics (ACL), pp 384\u2013394"},{"key":"768_CR22","doi-asserted-by":"crossref","unstructured":"Xu W, Rudnicky A (2000) Can artificial neural networks learn language models? In: Sixth international conference on spoken language processing","DOI":"10.21437\/ICSLP.2000-50"},{"key":"768_CR23","doi-asserted-by":"crossref","unstructured":"Mnih A, Hinton G (2007) Three new graphical models for statistical language modelling. In: Proceedings of the 24th international conference on machine learning, pp 641\u2013648","DOI":"10.1145\/1273496.1273577"},{"key":"768_CR24","unstructured":"Mnih A, Hinton GE (2008) A scalable hierarchical distributed language model. In: Advances in neural information processing systems, pp 1081\u20131088"},{"key":"768_CR25","unstructured":"Mnih A, Kavukcuoglu K (2013) Learning word embeddings efficiently with noise-contrastive estimation. In: Advances in neural information processing systems, pp 2265\u20132273"},{"key":"768_CR26","unstructured":"Morin F, Bengio Y (2005) Hierarchical probabilistic neural network language model. In: Proceedings of the international workshop on artificial intelligence and statistics, pp 246\u2013252"},{"key":"768_CR27","doi-asserted-by":"crossref","unstructured":"Collobert R, Weston J (2008) A unified architecture for natural language processing: deep neural networks with multitask learning. In: International conference on machine learning","DOI":"10.1145\/1390156.1390177"},{"key":"768_CR28","doi-asserted-by":"crossref","unstructured":"Mikolov T, Karafi\u00e1t M, Burget L, Cernocky\u0300 J, Khudanpur S (2010) Recurrent neural network based language model. In: 11th Annual conference of the international speech communication association, INTERSPEECH 2010, pp 1045\u20131048","DOI":"10.21437\/Interspeech.2010-343"},{"key":"768_CR29","unstructured":"Mikolov T, Chen K, Corrado G, Dean J (2013) Efficient estimation of word representations in vector space. In: International conference on learning representations workshop Track"},{"key":"768_CR30","unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS, Dean J (2013) Distributed representations of words and phrases and their compositionality. In: Advances in neural information processing systems, pp 3111\u20133119"},{"key":"768_CR31","unstructured":"Mikolov T, Yih WT, Zweig G (2013) Linguistic regularities in continuous space word representations. In: NAACL-HLT, pp 746\u2013751"},{"key":"768_CR32","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1162\/tacl_a_00051","volume":"5","author":"P Bojanowski","year":"2017","unstructured":"Bojanowski P, Grave E, Joulin A, Mikolov T (2017) Enriching word vectors with subword information. Trans Assoc Comput Linguist 5:135\u2013146","journal-title":"Trans Assoc Comput Linguist"},{"key":"768_CR33","doi-asserted-by":"crossref","unstructured":"Tissier J, Gravier C, Habrard A (2017) Dict2vec: learning word embeddings using lexical dictionaries. In: Conference on empirical methods in natural language processing (EMNLP), pp 254\u2013263","DOI":"10.18653\/v1\/D17-1024"},{"key":"768_CR34","doi-asserted-by":"crossref","unstructured":"Cao S, Lu W, Zhou J, Li X (2018) cw2vec: learning chinese word embeddings with stroke n-gram information. In: Thirty-second AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.12029"},{"key":"768_CR35","doi-asserted-by":"crossref","unstructured":"Xu J, Liu J, Zhang L, Li Z, Chen H (2016) Improve chinese word embeddings by exploiting internal structure. In: NAACL-HLT","DOI":"10.18653\/v1\/N16-1119"},{"key":"768_CR36","first-page":"2493","volume":"12","author":"R Collobert","year":"2011","unstructured":"Collobert R, Weston J, Bottou L, Karlen M, Kavukcuoglu K, Kuksa P (2011) Natural language processing (almost) from scratch. J Mach Learn Res 12:2493\u20132537","journal-title":"J Mach Learn Res"},{"key":"768_CR37","first-page":"1899","volume":"2014","author":"J Botha","year":"2014","unstructured":"Botha J, Blunsom P (2014) Compositional morphology for word representations and language modelling. Comput Sci 2014:1899\u20131907","journal-title":"Comput Sci"},{"key":"768_CR38","unstructured":"Chen X, Lei X, Liu Z, Sun M, Luan H (2015) Joint learning of character and word embeddings. In: International conference on artificial intelligence"},{"key":"768_CR39","unstructured":"Kalchbrenner N, Blunsom P (2013) Recurrent convolutional neural networks for discourse compositionality. In: Workshop on CVSC, pp 119\u2013126"},{"key":"768_CR40","doi-asserted-by":"crossref","unstructured":"Kalchbrenner N, Grefenstette E, Blunsom P (2014) A convolutional neural network for modelling sentences. In: Proceedings of the 52nd annual meeting of the association for computational linguistics, pp 655\u2013665","DOI":"10.3115\/v1\/P14-1062"},{"key":"768_CR41","doi-asserted-by":"crossref","unstructured":"Kim Y (2014) Convolutional neural networks for sentence classification. In: Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP), pp 1746\u20131751","DOI":"10.3115\/v1\/D14-1181"},{"key":"768_CR42","unstructured":"Xu Y, Liu J (2017) Implicitly incorporating morphological information into word embedding. arXiv preprint arXiv:170102481"},{"key":"768_CR43","doi-asserted-by":"crossref","unstructured":"Conneau A, Kiela D, Schwenk H, Barrault L, Bordes A (2018) Supervised learning of universal sentence representations from natural language inference data. In: Conference on empirical methods in natural language processing","DOI":"10.18653\/v1\/D17-1070"},{"key":"768_CR44","unstructured":"Talman A, Yli-Jyra A, Tiedemann J (2018) Natural language inference with hierarchical Bilstm max pooling architecture. arXiv preprint arXiv:180808762"},{"key":"768_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.engappai.2019.01.009","volume":"80","author":"T Chung","year":"2019","unstructured":"Chung T, Xu B, Liu Y, Ouyang C, Li S, Luo L (2019) Empirical study on character level neural network classifier for chinese text. Eng Appl Artif Intell 80:1\u20137","journal-title":"Eng Appl Artif Intell"},{"key":"768_CR46","doi-asserted-by":"publisher","first-page":"248","DOI":"10.1016\/j.engappai.2018.11.012","volume":"78","author":"JR Martinez-Rico","year":"2019","unstructured":"Martinez-Rico JR, Martinez-Romo J, Araujo L (2019) Can deep learning techniques improve classification performance of vandalism detection in wikipedia? Eng Appl Artif Intell 78:248\u2013259","journal-title":"Eng Appl Artif Intell"},{"key":"768_CR47","doi-asserted-by":"publisher","first-page":"432","DOI":"10.1016\/j.engappai.2017.06.024","volume":"64","author":"L Yao","year":"2017","unstructured":"Yao L, Zhang Y, Chen Q, Qian H, Wei B, Hu Z (2017) Mining coherent topics in documents using word embeddings and large-scale text data. Eng Appl Artif Intell 64:432\u2013439","journal-title":"Eng Appl Artif Intell"},{"key":"768_CR48","doi-asserted-by":"crossref","unstructured":"Ma X, Hovy E (2016) End-to-end sequence labeling via bi-directional lstm-cnns-crf. arXiv preprint arXiv:160301354","DOI":"10.18653\/v1\/P16-1101"},{"key":"768_CR49","unstructured":"Shijia E, Xiang Y (2017) Chinese named entity recognition with character word mixed embedding. In: ACM on conference on information knowledge management"},{"key":"768_CR50","unstructured":"Sun Y, Lei L, Tang D, Nan Y, Ji Z, Wang X (2015) Modeling mention, context and entity with neural networks for entity disambiguation. In: Twenty-fourth international joint conference on artificial intelligence"},{"key":"768_CR51","first-page":"1","volume":"2018","author":"J Li","year":"2018","unstructured":"Li J, Zhao S, Yang J et al (2018) WCP-RNN: a novel RNN-based approach for bio-NER in chinese EMRs. J Supercomput 2018:1\u201318","journal-title":"J Supercomput"},{"key":"768_CR52","doi-asserted-by":"crossref","unstructured":"Peters ME, Neumann M, Iyyer M, Gardner M, Clark C, Lee K, Zettlemoyer L (2018) Deep contextualized word representations. arXiv preprint arXiv:1802.05365","DOI":"10.18653\/v1\/N18-1202"},{"key":"768_CR53","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"768_CR54","unstructured":"Radford A, Narasimhan K, Salimans T, Sutskever I (2018) Improving language understanding by generative pre-training. URL https:\/\/s3-us-west-2.amazonaws.com\/openai-assets\/research-covers\/languageunsupervised\/languageunderstandingpaper.pdf"},{"key":"768_CR55","unstructured":"Joulin A, Grave E, Bojanowski P, Mikolov T (2016) Bag of tricks for efficient text classification. arXiv preprint arXiv:1607.01759"},{"key":"768_CR56","unstructured":"Luong T, Socher R, Manning CD (2013) Better word representations with recursive neural networks for morphology. In: Proceedings of the conference"},{"key":"768_CR57","unstructured":"Socher R, Lin C, Manning C, Ng AY (2011) Parsing natural scenes and naturallanguage with recursive neural networks. In: Proceedings of the 28th international conference on machine learning (ICML-11), pp 129\u2013136"},{"key":"768_CR58","unstructured":"Sennrich R, Haddow B, Birch A (2015) Neural machine translation of rare words with subword units. arXiv preprint arXiv:150807909"},{"key":"768_CR59","doi-asserted-by":"crossref","unstructured":"Cotterell R, Sch\u00fctze H (2015) Morphological word-embeddings. In: Proceedings of the 2015 conference of the North American chapter of the association for computational linguistics: human language technologies, pp 1287\u20131292","DOI":"10.3115\/v1\/N15-1140"},{"key":"768_CR60","doi-asserted-by":"crossref","unstructured":"Bian J, Gao B, Liu TY (2014) Knowledge-powered deep learning for word embedding. In: Joint European conference on machine learning and knowledge discovery in databases, Springer, pp 132\u2013148","DOI":"10.1007\/978-3-662-44848-9_9"},{"key":"768_CR61","doi-asserted-by":"crossref","unstructured":"Cao K, Rei M (2016) A joint model for word embedding and word morphology. arXiv preprint arXiv:1606.02601","DOI":"10.18653\/v1\/W16-1603"},{"key":"768_CR62","doi-asserted-by":"crossref","unstructured":"Kim Y, Jernite Y, Sontag D, Rush AM (2016) Character-aware neural language models. In: Thirtieth AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v30i1.10362"},{"key":"768_CR63","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Polosukhin I (2017) Attention is all you need. In: Advances in neural information processing systems, pp 5998\u20136008"},{"key":"768_CR64","unstructured":"Mitchell J, Lapata M (2008) Vector-based models of semantic composition. In: Proceedings of ACL-08: HLT, pp 236\u2013244"},{"key":"768_CR65","unstructured":"Blacoe W, Lapata M (2012) A comparison of vector-based representations for semantic composition. In: Proceedings of the 2012 joint conference on empirical methods in natural language processing and computational natural language learning, Association for Computational Linguistics, pp 546\u2013556"},{"key":"768_CR66","unstructured":"Le QV, Mikolov T (2014) Distributed representations of sentences and documents. In: International conference on machine learning, pp 1188\u20131196"},{"key":"768_CR67","unstructured":"Kiros R, Zhu Y, Salakhutdinov RR, Zemel R, Urtasun R, Torralba A, Fidler S (2015) Skip-thought vectors. In: Advances in neural information processing systems, pp 3294\u20133302"},{"key":"768_CR68","unstructured":"Logeswaran L, Lee H (2018) An efficient framework for learning sentence representations. arXiv preprint arXiv:1803.02893"},{"key":"768_CR69","doi-asserted-by":"crossref","unstructured":"Conneau A, Kiela D, Schwenk H et al (2017) Supervised learning of universal sentence representations from natural language inference data. arXiv preprint arXiv:1705.02364","DOI":"10.18653\/v1\/D17-1070"},{"key":"768_CR70","doi-asserted-by":"crossref","unstructured":"Levy O, Goldberg Y (2014) Linguistic regularities in sparse and explicit word representations. In: Proceedings of the eighteenth conference on computational natural language learning, pp 171\u2013180","DOI":"10.3115\/v1\/W14-1618"},{"key":"768_CR71","unstructured":"Heinzerling B, Strube M (2017) Bpemb: tokenization-free pre-trained subword embeddings in 275 languages. arXiv preprint arXiv:1710.02187"},{"key":"768_CR72","unstructured":"Xin JYXJHH, Song Y (2017) Joint embeddings of chinese words, characters, and fine-grained subcharacter components. In: EMNLP"},{"key":"768_CR73","doi-asserted-by":"crossref","unstructured":"Li Y, Li W, Sun F, Li S (2015) Component-enhanced chinese character embeddings. arXiv preprint arXiv:150806669","DOI":"10.18653\/v1\/D15-1098"},{"key":"768_CR74","unstructured":"Su TR, Lee HY (2017) Learning chinese word representations from glyphs of characters. arXiv preprint arXiv:170804755"},{"key":"768_CR75","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/978-3-319-12640-1_34","volume":"8835","author":"Y Sun","year":"2014","unstructured":"Sun Y, Lei L, Nan Y, Ji Z, Wang X (2014) Radical-enhanced chinese character embedding. Lect Not Comput Sci 8835:279\u2013286","journal-title":"Lect Not Comput Sci"},{"key":"768_CR76","doi-asserted-by":"crossref","unstructured":"Yang L, Sun M (2015) Improved learning of Chinese word embeddings with semantic knowledge. In: Chinese computational linguistic and natural language processing based on naturally annotated big data. Springer, pp 15\u201325","DOI":"10.1007\/978-3-319-25816-4_2"}],"container-title":["Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00607-019-00768-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00607-019-00768-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00607-019-00768-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,4]],"date-time":"2022-10-04T12:49:55Z","timestamp":1664887795000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00607-019-00768-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,12]]},"references-count":76,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2020,3]]}},"alternative-id":["768"],"URL":"https:\/\/doi.org\/10.1007\/s00607-019-00768-7","relation":{},"ISSN":["0010-485X","1436-5057"],"issn-type":[{"value":"0010-485X","type":"print"},{"value":"1436-5057","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,11,12]]},"assertion":[{"value":"4 June 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 November 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 November 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethical standards"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}