{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:20:04Z","timestamp":1757618404258,"version":"3.44.0"},"publisher-location":"Singapore","reference-count":37,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819681792"},{"type":"electronic","value":"9789819681808"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-8180-8_38","type":"book-chapter","created":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T09:16:07Z","timestamp":1750324567000},"page":"480-491","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Handling Korean Out-of-Vocabulary Words with\u00a0Phoneme Representation Learning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7971-1627","authenticated-orcid":false,"given":"Nayeon","family":"Kim","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7503-987X","authenticated-orcid":false,"given":"Eojin","family":"Jeon","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7900-3743","authenticated-orcid":false,"given":"Jun-Hyung","family":"Park","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6249-8217","authenticated-orcid":false,"given":"SangKeun","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"38_CR1","doi-asserted-by":"crossref","unstructured":"International Phonetic Association: Handbook of the International Phonetic Association: A guide to the use of the International Phonetic Alphabet. Cambridge University Press (1999)","DOI":"10.1017\/9780511807954"},{"key":"38_CR2","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1162\/tacl_a_00051","volume":"5","author":"P Bojanowski","year":"2017","unstructured":"Bojanowski, P., Grave, E., Joulin, A., Mikolov, T.: Enriching word vectors with subword information. Trans. Assoc. Comput. Linguist. 5, 135\u2013146 (2017)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"38_CR3","unstructured":"Carroll, D.W.: Psychology of Language. Thomson Brooks\/Cole Publishing Co (1986)"},{"key":"38_CR4","doi-asserted-by":"crossref","unstructured":"Chen, L., Varoquaux, G., Suchanek, F.M.: Imputing out-of-vocabulary embeddings with LOVE makes language models robust with little cost. In: ACL 2022, pp. 3488\u20133504 (2022)","DOI":"10.18653\/v1\/2022.acl-long.245"},{"key":"38_CR5","doi-asserted-by":"crossref","unstructured":"Hu, Z., Chen, T., Chang, K., Sun, Y.: Few-shot representation learning for out-of-vocabulary words. In: ACL 2019, pp. 4102\u20134112 (2019)","DOI":"10.18653\/v1\/P19-1402"},{"key":"38_CR6","unstructured":"Huang, Z., Xu, W., Yu, K.: Bidirectional LSTM-CRF models for sequence tagging. CoRR abs\/1508.01991 (2015)"},{"key":"38_CR7","doi-asserted-by":"crossref","unstructured":"Jeong, Y., et al.: KOLD: Korean offensive language dataset. In: EMNLP 2022, pp. 10818\u201310833 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.744"},{"key":"38_CR8","doi-asserted-by":"crossref","unstructured":"Jin, D., Jin, Z., Zhou, J.T., Szolovits, P.: Is BERT really robust? A strong baseline for natural language attack on text classification and entailment. In: AAAI 2020, pp. 8018\u20138025 (2020)","DOI":"10.1609\/aaai.v34i05.6311"},{"key":"38_CR9","doi-asserted-by":"crossref","unstructured":"Kim, N., Park, J., Choi, J., Jeon, E., Kang, Y., Lee, S.: Break it down into BTS: basic, tiniest subword units for Korean. In: EMNLP 2022, pp. 7007\u20137024 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.472"},{"key":"38_CR10","doi-asserted-by":"crossref","unstructured":"Kim, S., Park, J., Kim, Y., Lee, S.: KOMBO: Korean character representations based on the combination rules of subcharacters. In: Findings of the Association for Computational Linguistics ACL 2024, pp. 5102\u20135119 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.302"},{"key":"38_CR11","doi-asserted-by":"crossref","unstructured":"Kwon, O., Kim, D., Lee, S., Choi, J., Lee, S.: Handling out-of-vocabulary problem in Hangeul word embeddings. In: EACL 2021, pp. 3213\u20133221 (2021)","DOI":"10.18653\/v1\/2021.eacl-main.280"},{"key":"38_CR12","doi-asserted-by":"crossref","unstructured":"Lee, J., et al.: Length-aware byte pair encoding for mitigating over-segmentation in Korean machine translation. In: Findings of the Association for Computational Linguistics ACL 2024, pp. 2287\u20132303 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.135"},{"key":"38_CR13","doi-asserted-by":"crossref","unstructured":"Li, J., Wang, Q., Mao, Z., Guo, J., Yang, Y., Zhang, Y.: Improving Chinese spelling check by character pronunciation prediction: the effects of adaptivity and granularity. In: EMNLP 2022, pp. 4275\u20134286 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.287"},{"key":"38_CR14","doi-asserted-by":"crossref","unstructured":"Liang, B., Li, H., Su, M., Bian, P., Li, X., Shi, W.: Deep text classification can be fooled. In: IJCAI 2018, pp. 4208\u20134215 (2018)","DOI":"10.24963\/ijcai.2018\/585"},{"key":"38_CR15","doi-asserted-by":"crossref","unstructured":"Liang, Z., Quan, X., Wang, Q.: Disentangled phonetic representation for Chinese spelling correction. In: ACL 2023, pp. 13509\u201313521 (2023)","DOI":"10.18653\/v1\/2023.acl-long.755"},{"key":"38_CR16","doi-asserted-by":"crossref","unstructured":"Liang, Z., Lu, Y., Chen, H., Rao, Y.: Graph-based relation mining for context-free out-of-vocabulary word embedding learning. In: ACL 2023, pp. 14133\u201314149 (2023)","DOI":"10.18653\/v1\/2023.acl-long.790"},{"key":"38_CR17","doi-asserted-by":"crossref","unstructured":"Park, K., Lee, J., Jang, S., Jung, D.: An empirical study of tokenization strategies for various Korean NLP tasks. In: AACL\/IJCNLP 2020, pp. 133\u2013142 (2020)","DOI":"10.18653\/v1\/2020.aacl-main.17"},{"key":"38_CR18","doi-asserted-by":"crossref","unstructured":"Park, S., Byun, J., Baek, S., Cho, Y., Oh, A.: Subword-level word vector representations for Korean. In: ACL 2018, pp. 2429\u20132438 (2018)","DOI":"10.18653\/v1\/P18-1226"},{"key":"38_CR19","unstructured":"Park, S., et al.: KLUE: Korean language understanding evaluation. In: NeurIPS 2021 (2021)"},{"key":"38_CR20","unstructured":"Pil\u00e1n, I., Volodina, E.: Exploring word embeddings and phonological similarity for the unsupervised correction of language learner errors. In: Proceedings of the Second Joint SIGHUM Workshop on Computational Linguistics for Cultural Heritage, Social Sciences, Humanities and Literature, pp. 119\u2013128 (2018)"},{"key":"38_CR21","doi-asserted-by":"crossref","unstructured":"Pinter, Y., Guthrie, R., Eisenstein, J.: Mimicking word embeddings using subword RNNs. In: EMNLP 2017, pp. 102\u2013112 (2017)","DOI":"10.18653\/v1\/D17-1010"},{"key":"38_CR22","volume-title":"Writing Systems","author":"G Sampson","year":"1985","unstructured":"Sampson, G.: Writing Systems. Hutchinson, London, UK (1985)"},{"key":"38_CR23","doi-asserted-by":"crossref","unstructured":"Sasaki, S., Suzuki, J., Inui, K.: Subword-based compact reconstruction of word embeddings. In: NAACL-HLT 2019, pp. 3498\u20133508 (2019)","DOI":"10.18653\/v1\/N19-1353"},{"key":"38_CR24","doi-asserted-by":"crossref","unstructured":"Schick, T., Sch\u00fctze, H.: Attentive mimicking: better word embeddings by attending to informative contexts. In: NAACL-HLT 2019, pp. 489\u2013494 (2019)","DOI":"10.18653\/v1\/N19-1048"},{"key":"38_CR25","doi-asserted-by":"crossref","unstructured":"Schick, T., Sch\u00fctze, H.: Learning semantic representations for novel words: leveraging both form and context. In: AAAI 2019, pp. 6965\u20136973 (2019)","DOI":"10.1609\/aaai.v33i01.33016965"},{"key":"38_CR26","doi-asserted-by":"crossref","unstructured":"Seo, J., Moon, H., Lee, J., Eo, S., Park, C., Lim, H.: CHEF in the language kitchen: a generative data augmentation leveraging Korean morpheme ingredients. In: EMNLP 2023, pp. 6014\u20136029 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.367"},{"key":"38_CR27","doi-asserted-by":"crossref","unstructured":"Sofroniev, P., \u00c7\u00f6ltekin, \u00c7.: Phonetic vector representations for sound sequence alignment. In: Proceedings of the Fifteenth Workshop on Computational Research in Phonetics, Phonology, and Morphology. Brussels, Belgium (2018)","DOI":"10.18653\/v1\/W18-5812"},{"key":"38_CR28","doi-asserted-by":"crossref","unstructured":"Su, H., et al.: RocBERT: robust Chinese BERT with multimodal contrastive pretraining. In: ACL 2022, pp. 921\u2013931 (2022)","DOI":"10.18653\/v1\/2022.acl-long.65"},{"key":"38_CR29","unstructured":"Sun, L., et al.: Adv-BERT: BERT is not robust on misspellings! generating nature adversarial samples on BERT. arXiv preprint arXiv:2003.04985 (2020)"},{"key":"38_CR30","doi-asserted-by":"crossref","unstructured":"Sundararaman, M.N., Kumar, A., Vepa, J.: PhonemeBERT: joint language modelling of phoneme sequence and ASR transcript. In: Interspeech 2021, pp. 3236\u20133240 (2021)","DOI":"10.21437\/Interspeech.2021-1582"},{"key":"38_CR31","doi-asserted-by":"crossref","unstructured":"\u00dcst\u00fcn, A., Kurfali, M., Can, B.: Characters or morphemes: how to represent words? In: Proceedings of The Third Workshop on Representation Learning for NLP, Rep4NLP@ACL 2018, pp. 144\u2013153 (2018)","DOI":"10.18653\/v1\/W18-3019"},{"key":"38_CR32","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NIPS 2017, pp. 5998\u20136008 (2017)"},{"key":"38_CR33","doi-asserted-by":"crossref","unstructured":"Wieting, J., Bansal, M., Gimpel, K., Livescu, K.: Charagram: embedding words and sentences via character n-grams. In: EMNLP 2016, pp. 1504\u20131515 (2016)","DOI":"10.18653\/v1\/D16-1157"},{"key":"38_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, R., et al.: Correcting Chinese spelling errors with phonetic pre-training. In: Findings of the Association for Computational Linguistics: ACL\/IJCNLP 2021, pp. 2250\u20132261 (2021)","DOI":"10.18653\/v1\/2021.findings-acl.198"},{"key":"38_CR35","doi-asserted-by":"crossref","unstructured":"Zhao, J., Mudgal, S., Liang, Y.: Generalizing word embeddings using bag of subwords. In: EMNLP 2018, pp. 601\u2013606 (2018)","DOI":"10.18653\/v1\/D18-1059"},{"key":"38_CR36","doi-asserted-by":"crossref","unstructured":"Zhu, J., Yang, C., Samir, F., Islam, J.: The taste of IPA: towards open-vocabulary keyword spotting and forced alignment in any language. In: NAACL 2024, pp. 750\u2013772 (2024)","DOI":"10.18653\/v1\/2024.naacl-long.43"},{"key":"38_CR37","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Vulic, I., Korhonen, A.: A systematic study of leveraging subword information for learning word representations. In: NAACL-HLT 2019, pp. 912\u2013932 (2019)","DOI":"10.18653\/v1\/N19-1097"}],"container-title":["Lecture Notes in Computer Science","Advances in Knowledge Discovery and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-8180-8_38","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T20:44:45Z","timestamp":1757191485000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-8180-8_38"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819681792","9789819681808"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-8180-8_38","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"20 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Sydney, NSW","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 June 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/pakdd2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}