{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:06:39Z","timestamp":1750309599020,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T00:00:00Z","timestamp":1732233600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,22]]},"DOI":"10.1145\/3722237.3722267","type":"proceedings-article","created":{"date-parts":[[2025,4,30]],"date-time":"2025-04-30T06:56:56Z","timestamp":1745996216000},"page":"175-179","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Improving Fully Non-Autoregressive Translation with Pre-Trained Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9872-4346","authenticated-orcid":false,"given":"Zijian","family":"Fu","sequence":"first","affiliation":[{"name":"School of Computer and Software, Nanyang Institute of Technology, Nanyang, Henan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1816-0902","authenticated-orcid":false,"given":"Shuheng","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer and Software, Nanyang Institute of Technology, Nanyang, Henan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,4,29]]},"reference":[{"key":"e_1_3_3_1_1_2","unstructured":"Bahdanau D. Cho K. Bengio Y.: 2014. Neural machine translation by jointly learningto align and translate. arXiv preprint arXiv:1409.0473."},{"key":"e_1_3_3_1_2_2","unstructured":"Gehring J. Auli M. Grangier D. Yarats D. Dauphin Y.N.: 2017. Convolutional sequence to sequence learning. In: ICML."},{"key":"e_1_3_3_1_3_2","first-page":"6008","volume-title":"In: Advances in Neural Information Processing Systems","author":"Vaswani A.","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: 2017. Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008."},{"key":"e_1_3_3_1_4_2","unstructured":"Gu J. Bradbury J. Xiong C. Li V.O. Socher R.: 2017. Non-autoregressive neural machine translation. arXiv preprint arXiv:1711.02281."},{"key":"e_1_3_3_1_5_2","volume-title":"Flowseq: Non-autoregressive conditional sequence generation with generative flow. arXiv preprint arXiv:1909.02480.","author":"Ma X.","year":"2019","unstructured":"Ma, X., Zhou, C., Li, X., Neubig, G., Hovy, E.: 2019. Flowseq: Non-autoregressive conditional sequence generation with generative flow. arXiv preprint arXiv:1909.02480."},{"key":"e_1_3_3_1_6_2","first-page":"8853","volume-title":"Proceedings of the Aaai Conference on Artificial Intelligence","volume":"34","author":"Shu R.","unstructured":"Shu, R., Lee, J., Nakayama, H., Cho, K.: 2020. Latent-variable non-autoregressive neural machine translation with deterministic inference using a delta posterior. In: Proceedings of the Aaai Conference on Artificial Intelligence, vol. 34, pp. 8846\u20138853."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Saharia C. Chan W. Saxena S. Norouzi M.: 2020. Non-autoregressive machine translation with latent alignments. arXiv preprint arXiv:2004.07437.","DOI":"10.18653\/v1\/2020.emnlp-main.83"},{"key":"e_1_3_3_1_8_2","first-page":"1182","volume-title":"Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing","author":"Lee J.","unstructured":"Lee, J., Mansimov, E., Cho, K.: 2018. Deterministic non-autoregressive neural sequence modeling by iterative refinement. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 1173\u20131182."},{"key":"e_1_3_3_1_9_2","volume-title":"Mask-predict: Parallel decoding of conditional masked language models. arXiv preprint arXiv:1904.09324.","author":"Ghazvininejad M.","year":"2019","unstructured":"Ghazvininejad, M., Levy, O., Liu, Y., Zettlemoyer, L.: 2019. Mask-predict: Parallel decoding of conditional masked language models. arXiv preprint arXiv:1904.09324."},{"key":"e_1_3_3_1_10_2","volume-title":"Helcl, J.","author":"Libovick","year":"2018","unstructured":"Libovick`y, J., Helcl, J.: 2018. End-to-end non-autoregressive neural machine translation with connectionist temporal classification. arXiv preprint arXiv:1811.04719."},{"key":"e_1_3_3_1_11_2","first-page":"5384","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"33","author":"Wang Y.","unstructured":"Wang, Y., Tian, F., He, D., Qin, T., Zhai, C., Liu, T.-Y.: 2019. Non-autoregressive machine translation with auxiliary regularization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 5377\u20135384."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Wei B. Wang M. Zhou H. Lin J. Xie J. Sun X.: 2019. Imitation learning for non-autoregressive neural machine translation. arXiv preprint arXiv:1906.02041.","DOI":"10.18653\/v1\/P19-1125"},{"key":"e_1_3_3_1_13_2","first-page":"3523","volume-title":"International Conference on Machine Learning","author":"Ghazvininejad M.","unstructured":"Ghazvininejad, M., Karpukhin, V., Zettlemoyer, L., Levy, O.: 2020. Aligned cross entropy for non-autoregressive machine translation. In: International Conference on Machine Learning, pp. 3515\u20133523."},{"key":"e_1_3_3_1_14_2","unstructured":"Sun Z. Li Z. Wang H. Lin Z. He D. Deng Z.-H.: 2019. Fast structured decoding for sequence models. arXiv preprint arXiv:1910.11555."},{"key":"e_1_3_3_1_15_2","first-page":"205","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"34","author":"Shao C.","unstructured":"Shao, C., Zhang, J., Feng, Y., Meng, F., Zhou, J.: 2020. Minimizing the bag-of-ngrams difference for non-autoregressive neural machine translation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 198\u2013205."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Shao C. Feng Y. Zhang J. Meng F. Chen X. Zhou J.: 2019. Retrieving sequential information for non-autoregressive neural machine translation. arXiv preprint arXiv:1906.09444.","DOI":"10.18653\/v1\/P19-1288"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Qian L. Zhou H. Bao Y. Wang M. Qiu L. Zhang W. Yu Y. Li L.: 2020. Glancing transformer for non-autoregressive neural machine translation. arXiv preprint arXiv:2008.07905.","DOI":"10.18653\/v1\/2021.acl-long.155"},{"key":"e_1_3_3_1_18_2","unstructured":"Li Z. He D. Tian F. Qin T. Wang L. Liu T.-Y.: 2018. Hint-based training for non-autoregressive translation."},{"key":"e_1_3_3_1_19_2","first-page":"7846","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"34","author":"Guo J.","unstructured":"Guo, J., Tan, X., Xu, L., Qin, T., Chen, E., Liu, T.-Y.: 2020. Fine-tuning by curriculum learning for non-autoregressive neural machine translation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 7839\u20137846."},{"key":"e_1_3_3_1_20_2","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805.","author":"Devlin J.","year":"2018","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_3_1_21_2","first-page":"376","volume-title":"Proceedings of the 23rd International Conference on Machine Learning","author":"Graves A.","unstructured":"Graves, A., Fern\u00b4andez, S., Gomez, F., Schmidhuber, J.: 2006. Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376."},{"key":"e_1_3_3_1_22_2","first-page":"10843","article-title":"Incorporating bert into parallel sequence decoding with adapters","volume":"33","author":"Guo J.","year":"2020","unstructured":"Guo, J., Zhang, Z., Xu, L., Wei, H.-R., Chen, B., Chen, E.: 2020. Incorporating bert into parallel sequence decoding with adapters. Advances in Neural Information Processing Systems 33, 10843\u201310854.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Su Y. Cai D. Wang Y. Vandyke D. Baker S. Li P. Collier N.: 2021. Non-autoregressive text generation with pre-trained language models. arXiv preprint arXiv:2102.08220.","DOI":"10.18653\/v1\/2021.eacl-main.18"},{"key":"e_1_3_3_1_24_2","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980.","author":"Kingma D.P.","year":"2014","unstructured":"Kingma, D.P., Ba, J.: 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980."},{"key":"e_1_3_3_1_25_2","first-page":"318","volume-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics","author":"Papineni K.","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: 2002. Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318."},{"key":"e_1_3_3_1_26_2","unstructured":"Zhou C. Neubig G. Gu J.: 2019. Understanding knowledge distillation in non-autoregressive machine translation. arXiv preprint arXiv:1911.02727."}],"event":{"name":"ICAIE 2024: 2024 3rd International Conference on Artificial Intelligence and Education","acronym":"ICAIE 2024","location":"Xiamen China"},"container-title":["Proceedings of the 2024 3rd International Conference on Artificial Intelligence and Education"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3722237.3722267","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3722237.3722267","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:57:05Z","timestamp":1750298225000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3722237.3722267"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,22]]},"references-count":26,"alternative-id":["10.1145\/3722237.3722267","10.1145\/3722237"],"URL":"https:\/\/doi.org\/10.1145\/3722237.3722267","relation":{},"subject":[],"published":{"date-parts":[[2024,11,22]]},"assertion":[{"value":"2025-04-29","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}