{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:21:55Z","timestamp":1750220515964,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,12,11]],"date-time":"2020-12-11T00:00:00Z","timestamp":1607644800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,12,11]]},"DOI":"10.1145\/3445815.3445856","type":"proceedings-article","created":{"date-parts":[[2021,3,17]],"date-time":"2021-03-17T17:05:28Z","timestamp":1616000728000},"page":"254-260","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Using Triangle Exchange Mechanism to Accelerate the Pre-training Convergence Speed of BERT"],"prefix":"10.1145","author":[{"given":"Jingjing","family":"Liao","sequence":"first","affiliation":[{"name":"College of Computer &amp; Information Science Southwest University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nanzhi","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computer &amp; Information Science Southwest University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guocai","family":"Yang","sequence":"additional","affiliation":[{"name":"College of Computer &amp; Information Science Southwest University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,3,17]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"3079","volume-title":"Advances in neural information processing systems.","author":"Dai A.M","unstructured":"Dai A.M , Le Q.V. 2015. Semi-supervised sequence learning . In: Advances in neural information processing systems. pp. 3079 - 3087 . Dai A.M, Le Q.V. 2015. Semi-supervised sequence learning. In: Advances in neural information processing systems. pp. 3079-3087."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Howard J Ruder S. 2018. Universal language model fine-tuning for text classification. arXiv preprint arXiv:1801.06146.  Howard J Ruder S. 2018. Universal language model fine-tuning for text classification. arXiv preprint arXiv:1801.06146.","DOI":"10.18653\/v1\/P18-1031"},{"key":"e_1_3_2_1_3_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805.","author":"Devlin J","year":"2018","unstructured":"Devlin J , Chang M.W , Lee K , 2018 . Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805. Devlin J, Chang M.W, Lee K, 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Peters M.E Neumann M Iyyer M. 2018. Deep contextualized word representations. arXiv preprint arXiv:1802.05365.  Peters M.E Neumann M Iyyer M. 2018. Deep contextualized word representations. arXiv preprint arXiv:1802.05365.","DOI":"10.18653\/v1\/N18-1202"},{"key":"e_1_3_2_1_5_1","unstructured":"Radford A Narasimhan K Salimans T 2018. Improving language understanding by generative pre-training. URL https:\/\/s3-us-west-2. amazonaws. com\/openai-assets\/researchcovers\/languageunsupervised\/language understanding paper. pdf.  Radford A Narasimhan K Salimans T 2018. Improving language understanding by generative pre-training. URL https:\/\/s3-us-west-2. amazonaws. com\/openai-assets\/researchcovers\/languageunsupervised\/language understanding paper. pdf."},{"issue":"2","key":"e_1_3_2_1_6_1","first-page":"1137","article-title":"A neural probabilistic language model. In","volume":"3","author":"Bengio Y","year":"2003","unstructured":"Bengio Y , Ducharme R , Vincent P , 2003 . A neural probabilistic language model. In : Journal of machine learning research. 3 ( 2 ), 1137 - 1155 . Bengio Y, Ducharme R, Vincent P, 2003. A neural probabilistic language model. In: Journal of machine learning research. 3(2), 1137-1155.","journal-title":"Journal of machine learning research."},{"key":"e_1_3_2_1_7_1","unstructured":"Mikolov T Chen K Corrado G 2013. Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781.  Mikolov T Chen K Corrado G 2013. Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Mikolov T Karafi\u00e1t M Burget L 2010. Recurrent neural network based language model. In: Eleventh annual conference of the international speech communication association.  Mikolov T Karafi\u00e1t M Burget L 2010. Recurrent neural network based language model. In: Eleventh annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2010-343"},{"key":"e_1_3_2_1_9_1","first-page":"3104","volume-title":"Advances in neural information processing systems.","author":"Sutskever I","unstructured":"Sutskever I , Vinyals O , Le Q.V. 2014. Sequence to sequence learning with neural networks . In: Advances in neural information processing systems. pp. 3104 - 3112 . Sutskever I, Vinyals O, Le Q.V. 2014. Sequence to sequence learning with neural networks. In: Advances in neural information processing systems. pp. 3104-3112."},{"issue":"2","key":"e_1_3_2_1_10_1","first-page":"1137","article-title":"A neural probabilistic language model. In","volume":"3","author":"Bengio Y","year":"2003","unstructured":"Bengio Y , Ducharme R , Vincent P , 2003 . A neural probabilistic language model. In : Journal of machine learning research. 3 ( 2 ): 1137 - 1155 . Bengio Y, Ducharme R, Vincent P, 2003. A neural probabilistic language model. In: Journal of machine learning research.3(2): 1137-1155.","journal-title":"Journal of machine learning research."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Kalchbrenner N Grefenstette E Blunsom P.2014. A convolutional neural network for modelling sentences. arXiv preprint arXiv:1404.2188.  Kalchbrenner N Grefenstette E Blunsom P.2014. A convolutional neural network for modelling sentences. arXiv preprint arXiv:1404.2188.","DOI":"10.3115\/v1\/P14-1062"},{"key":"e_1_3_2_1_12_1","unstructured":"Bahdanau D Cho K Bengio Y.2014. Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473.  Bahdanau D Cho K Bengio Y.2014. Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473."},{"key":"e_1_3_2_1_13_1","first-page":"5998","volume-title":"Advances in neural information processing systems.","author":"Vaswani A","unstructured":"Vaswani A , Shazeer N , Parmar N , 2017. Attention is all you need . In: Advances in neural information processing systems. pp. 5998 - 6008 . Vaswani A, Shazeer N, Parmar N, 2017. Attention is all you need. In: Advances in neural information processing systems. pp. 5998-6008 ."},{"key":"e_1_3_2_1_14_1","unstructured":"Yang Z Dai Z Yang Y 2019. XLNet: Generalized Autoregressive Pretraining for Language Understanding. arXiv preprint arXiv:1906.08237.  Yang Z Dai Z Yang Y 2019. XLNet: Generalized Autoregressive Pretraining for Language Understanding. arXiv preprint arXiv:1906.08237."},{"key":"e_1_3_2_1_15_1","volume-title":"Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860.","author":"Dai Z","year":"2019","unstructured":"Dai Z , Yang Z , Yang Y , 2019 . Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860. Dai Z, Yang Z, Yang Y, 2019. Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860."},{"key":"e_1_3_2_1_16_1","volume-title":"Tinybert: Distilling bert for natural language understanding. arXiv preprint arXiv:1909.10351.","author":"Jiao X","year":"2019","unstructured":"Jiao X , Yin Y , Shang L , 2019 . Tinybert: Distilling bert for natural language understanding. arXiv preprint arXiv:1909.10351. Jiao X, Yin Y, Shang L, 2019. Tinybert: Distilling bert for natural language understanding. arXiv preprint arXiv:1909.10351."},{"key":"e_1_3_2_1_17_1","unstructured":"Sanh V Debut L Chaumond J 2019. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. arXiv preprint arXiv:1910.01108.  Sanh V Debut L Chaumond J 2019. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. arXiv preprint arXiv:1910.01108."},{"key":"e_1_3_2_1_18_1","volume-title":"ALBERT: A lite BERT for self-supervised learning of language representations. arXiv preprint arXiv:1909.11942.","author":"Lan Z","year":"2019","unstructured":"Lan Z , Chen M , Goodman S , 2019 . ALBERT: A lite BERT for self-supervised learning of language representations. arXiv preprint arXiv:1909.11942. Lan Z, Chen M, Goodman S, 2019. ALBERT: A lite BERT for self-supervised learning of language representations. arXiv preprint arXiv:1909.11942."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Baevski A Edunov S Liu Y 2019. Cloze-driven pretraining of self-attention networks. arXiv preprint arXiv:1903.07785.  Baevski A Edunov S Liu Y 2019. Cloze-driven pretraining of self-attention networks. arXiv preprint arXiv:1903.07785.","DOI":"10.18653\/v1\/D19-1539"},{"issue":"8","key":"e_1_3_2_1_20_1","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford A","year":"2019","unstructured":"Radford A , Wu J , Child R , 2019 . Language models are unsupervised multitask learners . OpenAI Blog , 1 ( 8 ): 9 . Radford A, Wu J, Child R, 2019. Language models are unsupervised multitask learners. OpenAI Blog, 1(8): 9.","journal-title":"OpenAI Blog"},{"key":"e_1_3_2_1_21_1","first-page":"9051","volume-title":"Advances in Neural Information Processing Systems.","author":"Zellers R","unstructured":"Zellers R , Holtzman A , Rashkin H , 2019. Defending against neural fake news . In: Advances in Neural Information Processing Systems. pp. 9051 - 9062 . Zellers R, Holtzman A, Rashkin H, 2019. Defending against neural fake news. In: Advances in Neural Information Processing Systems. pp. 9051-9062."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Ott M Edunov S Grangier D 2018. Scaling neural machine translation. arXiv preprint arXiv:1806.00187.  Ott M Edunov S Grangier D 2018. Scaling neural machine translation. arXiv preprint arXiv:1806.00187.","DOI":"10.18653\/v1\/W18-6301"},{"key":"e_1_3_2_1_23_1","unstructured":"You Y Li J Hseu J 2019. Reducing BERT pre-training time from 3 days to 76 minutes. arXiv preprint arXiv:1904.00962.  You Y Li J Hseu J 2019. Reducing BERT pre-training time from 3 days to 76 minutes. arXiv preprint arXiv:1904.00962."},{"key":"e_1_3_2_1_24_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692.","author":"Liu Y","year":"2019","unstructured":"Liu Y , Ott M , Goyal N , 2019 . Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692. Liu Y, Ott M, Goyal N, 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692."},{"key":"e_1_3_2_1_25_1","unstructured":"Lample G Conneau A. 2019. Cross-lingual language model pretraining. arXiv preprint arXiv:1901.07291.  Lample G Conneau A. 2019. Cross-lingual language model pretraining. arXiv preprint arXiv:1901.07291."},{"key":"e_1_3_2_1_26_1","volume-title":"Spanbert: Improving pre-training by representing and predicting spans. arXiv preprint arXiv:1907.10529.","author":"Joshi M","year":"2019","unstructured":"Joshi M , Chen D , Liu Y , 2019 . Spanbert: Improving pre-training by representing and predicting spans. arXiv preprint arXiv:1907.10529. Joshi M, Chen D, Liu Y, 2019. Spanbert: Improving pre-training by representing and predicting spans. arXiv preprint arXiv:1907.10529."},{"key":"e_1_3_2_1_27_1","volume-title":"ERNIE: Enhanced Language Representation with Informative Entities. arXiv preprint arXiv:1905.07129.","author":"Zhang Z","year":"2019","unstructured":"Zhang Z , Han X , Liu Z , 2019 . ERNIE: Enhanced Language Representation with Informative Entities. arXiv preprint arXiv:1905.07129. Zhang Z, Han X, Liu Z, 2019. ERNIE: Enhanced Language Representation with Informative Entities. arXiv preprint arXiv:1905.07129."},{"key":"e_1_3_2_1_28_1","volume-title":"ERNIE: Enhanced Representation through Knowledge Integration. arXiv preprint arXiv:1904.09223","author":"Sun Y.","year":"2019","unstructured":"Sun , Y. , Wang , S. , Li , Y. , : ERNIE: Enhanced Representation through Knowledge Integration. arXiv preprint arXiv:1904.09223 ( 2019 ). Sun, Y., Wang, S., Li, Y., : ERNIE: Enhanced Representation through Knowledge Integration. arXiv preprint arXiv:1904.09223 (2019)."},{"key":"e_1_3_2_1_29_1","volume-title":"Mass: Masked sequence to sequence pre-training for language generation. arXiv preprint arXiv:1905.02450.","author":"Song K","year":"2019","unstructured":"Song K , Tan X , Qin T , 2019 . Mass: Masked sequence to sequence pre-training for language generation. arXiv preprint arXiv:1905.02450. Song K, Tan X, Qin T, 2019. Mass: Masked sequence to sequence pre-training for language generation. arXiv preprint arXiv:1905.02450."},{"key":"e_1_3_2_1_30_1","unstructured":"Dong L Yang N Wang W 2019. Unified Language Model Pre-training for Natural Language Understanding and Generation. arXiv preprint arXiv:1905.03197.  Dong L Yang N Wang W 2019. Unified Language Model Pre-training for Natural Language Understanding and Generation. arXiv preprint arXiv:1905.03197."}],"event":{"name":"CSAI 2020: 2020 4th International Conference on Computer Science and Artificial Intelligence","acronym":"CSAI 2020","location":"Zhuhai China"},"container-title":["2020 4th International Conference on Computer Science and Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3445815.3445856","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3445815.3445856","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:24:33Z","timestamp":1750195473000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3445815.3445856"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,12,11]]},"references-count":30,"alternative-id":["10.1145\/3445815.3445856","10.1145\/3445815"],"URL":"https:\/\/doi.org\/10.1145\/3445815.3445856","relation":{},"subject":[],"published":{"date-parts":[[2020,12,11]]},"assertion":[{"value":"2021-03-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}