{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,29]],"date-time":"2025-11-29T08:02:26Z","timestamp":1764403346739,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,1,26]],"date-time":"2024-01-26T00:00:00Z","timestamp":1706227200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Zhejiang Electric Power Co., Ltd.","award":["Science and Technology Project No.B311YF230001"],"award-info":[{"award-number":["Science and Technology Project No.B311YF230001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,1,26]]},"DOI":"10.1145\/3640824.3640825","type":"proceedings-article","created":{"date-parts":[[2024,3,8]],"date-time":"2024-03-08T12:05:28Z","timestamp":1709899528000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Self-Supervised Learning Representations for Dialect Identification with Sparse Transformers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-6851-6931","authenticated-orcid":false,"given":"Ran","family":"Shen","sequence":"first","affiliation":[{"name":"Marketing Service Center, State Grid Zhejiang Electric Power Co., Ltd., China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7449-2496","authenticated-orcid":false,"given":"Yiling","family":"Li","sequence":"additional","affiliation":[{"name":"Marketing Service Center, State Grid Zhejiang Electric Power Co., Ltd., China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8179-1829","authenticated-orcid":false,"given":"Hongjie","family":"Gu","sequence":"additional","affiliation":[{"name":"Marketing Service Center, State Grid Zhejiang Electric Power Co., Ltd., China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2531-2213","authenticated-orcid":false,"given":"Yifan","family":"Wang","sequence":"additional","affiliation":[{"name":"Marketing Service Center, State Grid Zhejiang Electric Power Co., Ltd., China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2663-0848","authenticated-orcid":false,"given":"Junjie","family":"Huang","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2931-704X","authenticated-orcid":false,"given":"Qingshun","family":"She","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,3,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv preprint arXiv:2111.09296","author":"Babu Arun","year":"2021","unstructured":"Arun Babu, Changhan Wang, Andros Tjandra, Kushal Lakhotia, Qiantong Xu, Naman Goyal, Kritika Singh, Patrick von Platen, Yatharth Saraf, Juan Pino, 2021. XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv preprint arXiv:2111.09296 (2021)."},{"key":"e_1_3_2_1_2_1","volume-title":"International Conference on Machine Learning. PMLR, 1298\u20131312","author":"Baevski Alexei","year":"2022","unstructured":"Alexei Baevski, Wei-Ning Hsu, Qiantong Xu, Arun Babu, Jiatao Gu, and Michael Auli. 2022. Data2vec: A general framework for self-supervised learning in speech, vision and language. In International Conference on Machine Learning. PMLR, 1298\u20131312."},{"key":"e_1_3_2_1_3_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33 (2020), 12449\u201312460."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682386"},{"key":"e_1_3_2_1_5_1","volume-title":"Unsupervised cross-lingual representation learning for speech recognition. arXiv preprint arXiv:2006.13979","author":"Conneau Alexis","year":"2020","unstructured":"Alexis Conneau, Alexei Baevski, Ronan Collobert, Abdelrahman Mohamed, and Michael Auli. 2020. Unsupervised cross-lingual representation learning for speech recognition. arXiv preprint arXiv:2006.13979 (2020)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413952"},{"key":"e_1_3_2_1_7_1","volume-title":"Exploring wav2vec 2.0 on speaker verification and language identification. arXiv preprint arXiv:2012.06185","author":"Fan Zhiyun","year":"2020","unstructured":"Zhiyun Fan, Meng Li, Shiyu Zhou, and Bo Xu. 2020. Exploring wav2vec 2.0 on speaker verification and language identification. arXiv preprint arXiv:2012.06185 (2020)."},{"key":"e_1_3_2_1_8_1","volume-title":"Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, 2020. Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746050"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Ma Jin Yan Song Ian\u00a0Vince McLoughlin Wu Guo and Li-Rong Dai. 2017. End-to-end language identification using high-order utterance representation with bilinear pooling. (2017).","DOI":"10.21437\/Interspeech.2017-44"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747515"},{"key":"e_1_3_2_1_13_1","volume-title":"Segment anything. arXiv preprint arXiv:2304.02643","author":"Kirillov Alexander","year":"2023","unstructured":"Alexander Kirillov, Eric Mintun, Nikhila Ravi, Hanzi Mao, Chloe Rolland, Laura Gustafson, Tete Xiao, Spencer Whitehead, Alexander\u00a0C Berg, Wan-Yen Lo, 2023. Segment anything. arXiv preprint arXiv:2304.02643 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"Dynamic multi-scale convolution for dialect identification. arXiv preprint arXiv:2108.07787","author":"Kong Tianlong","year":"2021","unstructured":"Tianlong Kong, Shouyi Yin, Dawei Zhang, Wang Geng, Xin Wang, Dandan Song, Jinwen Huang, Huiyu Shi, and Xiaorui Wang. 2021. Dynamic multi-scale convolution for dialect identification. arXiv preprint arXiv:2108.07787 (2021)."},{"key":"e_1_3_2_1_15_1","volume-title":"A survey of transformers. AI Open","author":"Lin Tianyang","year":"2022","unstructured":"Tianyang Lin, Yuxin Wang, Xiangyang Liu, and Xipeng Qiu. 2022. A survey of transformers. AI Open (2022)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3201445"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3207050"},{"key":"e_1_3_2_1_18_1","volume-title":"Attentive statistics pooling for deep speaker embedding. arXiv preprint arXiv:1803.10963","author":"Okabe Koji","year":"2018","unstructured":"Koji Okabe, Takafumi Koshinaka, and Koichi Shinoda. 2018. Attentive statistics pooling for deep speaker embedding. arXiv preprint arXiv:1803.10963 (2018)."},{"key":"e_1_3_2_1_19_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arxiv:2303.08774\u00a0[cs.CL]"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICACCI.2016.7732177"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"G Ramesh C\u00a0Shiva Kumar and Sri Rama\u00a0Murty Kodukula. 2021. Self-supervised phonotactic representations for language identification. (2021).","DOI":"10.21437\/Interspeech.2021-1310"},{"key":"e_1_3_2_1_22_1","volume-title":"wav2vec: Unsupervised pre-training for speech recognition. arXiv preprint arXiv:1904.05862","author":"Schneider Steffen","year":"2019","unstructured":"Steffen Schneider, Alexei Baevski, Ronan Collobert, and Michael Auli. 2019. wav2vec: Unsupervised pre-training for speech recognition. arXiv preprint arXiv:1904.05862 (2019)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"David Snyder Daniel Garcia-Romero Alan McCree Gregory Sell Daniel Povey and Sanjeev Khudanpur. 2018. Spoken language recognition using x-vectors.. In Odyssey Vol.\u00a02018. 105\u2013111.","DOI":"10.21437\/Odyssey.2018-15"},{"key":"e_1_3_2_1_24_1","volume-title":"Kespeech: An open source speech dataset of mandarin and its eight subdialects.","author":"Tang Zhiyuan","year":"2021","unstructured":"Zhiyuan Tang, Dong Wang, Yanguang Xu, Jianwei Sun, Xiaoning Lei, Shuaijiang Zhao, Cheng Wen, Xingjun Tan, Chuandong Xie, Shuran Zhou, 2021. Kespeech: An open source speech dataset of mandarin and its eight subdialects. (2021)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747667"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461972"},{"key":"e_1_3_2_1_27_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_28_1","volume-title":"Attentive temporal pooling for conformer-based streaming language identification in long-form speech. arXiv preprint arXiv:2202.12163","author":"Wang Quan","year":"2022","unstructured":"Quan Wang, Yang Yu, Jason Pelecanos, Yiling Huang, and Ignacio\u00a0Lopez Moreno. 2022. Attentive temporal pooling for conformer-based streaming language identification in long-form speech. arXiv preprint arXiv:2202.12163 (2022)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746639"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003870"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.3390\/app121910154"},{"key":"e_1_3_2_1_32_1","volume-title":"Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages. arXiv preprint arXiv:2303.01037","author":"Zhang Yu","year":"2023","unstructured":"Yu Zhang, Wei Han, James Qin, Yongqiang Wang, Ankur Bapna, Zhehuai Chen, Nanxin Chen, Bo Li, Vera Axelrod, Gary Wang, 2023. Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages. arXiv preprint arXiv:2303.01037 (2023)."}],"event":{"name":"CCEAI 2024: 2024 8th International Conference on Control Engineering and Artificial Intelligence","acronym":"CCEAI 2024","location":"Shanghai China"},"container-title":["2024 8th International Conference on Control Engineering and Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640824.3640825","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3640824.3640825","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T16:46:11Z","timestamp":1756485971000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640824.3640825"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,26]]},"references-count":32,"alternative-id":["10.1145\/3640824.3640825","10.1145\/3640824"],"URL":"https:\/\/doi.org\/10.1145\/3640824.3640825","relation":{},"subject":[],"published":{"date-parts":[[2024,1,26]]},"assertion":[{"value":"2024-03-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}