{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T19:12:04Z","timestamp":1770059524589,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","funder":[{"name":"China National Social Science Foundation (Special Project for Rare and Precious Studies)","award":["No. 22VJXG012"],"award-info":[{"award-number":["No. 22VJXG012"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,12]]},"DOI":"10.1145\/3784833.3784908","type":"proceedings-article","created":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T05:22:31Z","timestamp":1770009751000},"page":"171-175","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Scalable Synthetic Data Pipeline and End-to-End Sound Synthesis: A Case Study on the Chinese Dizi"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-0743-4418","authenticated-orcid":false,"given":"Junchen","family":"Liu","sequence":"first","affiliation":[{"name":"Hainan Advanced Digital Technology and Systems Laboratory, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5595-3701","authenticated-orcid":false,"given":"Rongfeng","family":"Li","sequence":"additional","affiliation":[{"name":"Hainan Advanced Digital Technology and Systems Laboratory, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2052-3472","authenticated-orcid":false,"given":"Zijin","family":"Li","sequence":"additional","affiliation":[{"name":"Central Conservatory of Music, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9763-965X","authenticated-orcid":false,"given":"Ya","family":"Li","sequence":"additional","affiliation":[{"name":"Music College, Shanghai Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3952-1742","authenticated-orcid":false,"given":"Linfeng","family":"Fan","sequence":"additional","affiliation":[{"name":"Central Conservatory of Music, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1352-3441","authenticated-orcid":false,"given":"Pei","family":"Huang","sequence":"additional","affiliation":[{"name":"School of Digital Media and Design Arts, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,2]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Igor\u00a0Barros Barbosa Marco Cristani Barbara Caputo Aleksander Rognhaugen and Theoharis Theoharis. 2018. Looking beyond appearances: Synthetic training data for deep cnns in re-identification. Computer Vision and Image Understanding 167 (2018) 50\u201362.","DOI":"10.1016\/j.cviu.2017.12.002"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1002\/9781118680605.ch15"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447265"},{"key":"e_1_3_3_1_5_2","volume-title":"International Conference on Learning Representations","author":"Engel Jesse","year":"2020","unstructured":"Jesse Engel, Chenjie Gu, Adam Roberts, et\u00a0al. 2020. DDSP: Differentiable Digital Signal Processing. In International Conference on Learning Representations."},{"key":"e_1_3_3_1_6_2","first-page":"553","volume-title":"Proc. of the 18th International Congress on Acoustics, 2004","author":"Goto Masataka","year":"2004","unstructured":"Masataka Goto et\u00a0al. 2004. Development of the RWC Music Database. In Proc. of the 18th International Congress on Acoustics, 2004. 553\u2013556."},{"key":"e_1_3_3_1_7_2","unstructured":"Curtis Hawthorne Andriy Stasyuk Adam Roberts Ian Simon Cheng-Zhi\u00a0Anna Huang Sander Dieleman Erich Elsen Jesse Engel and Douglas Eck. 2018. Enabling Factorized Piano Music Modeling and Generation with the MAESTRO Dataset. (2018)."},{"key":"e_1_3_3_1_8_2","volume-title":"International Conference on Learning Representations","author":"Huang Cheng-Zhi\u00a0Anna","unstructured":"Cheng-Zhi\u00a0Anna Huang, Ashish Vaswani, Jakob Uszkoreit, Ian Simon, Curtis Hawthorne, Noam Shazeer, Andrew\u00a0M Dai, Matthew\u00a0D Hoffman, Monica Dinculescu, and Douglas Eck. [n. d.]. Music Transformer: Generating Music with Long-Term Structure. In International Conference on Learning Representations."},{"key":"e_1_3_3_1_9_2","first-page":"3","volume-title":"International Conference on Technologies and Applications of Artificial Intelligence","author":"Hung Tzu-Yun","year":"2024","unstructured":"Tzu-Yun Hung, Jui-Te Wu, Yu-Chia Kuo, Yo-Wei Hsiao, Ting-Wei Lin, and Li Su. 2024. A Study on Synthesizing Expressive Violin Performances: Approaches and Comparisons. In International Conference on Technologies and Applications of Artificial Intelligence. Springer, 3\u201316."},{"key":"e_1_3_3_1_10_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Kreuk Felix","year":"2022","unstructured":"Felix Kreuk, Gabriel Synnaeve, Adam Polyak, Uriel Singer, Alexandre D\u00e9fossez, Jade Copet, Devi Parikh, Yaniv Taigman, and Yossi Adi. 2022. AudioGen: Textually Guided Audio Generation. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_1_11_2","first-page":"18319","volume-title":"International Conference on Machine Learning","author":"Lai Yuhang","year":"2023","unstructured":"Yuhang Lai, Chengxi Li, Yiming Wang, Tianyi Zhang, Ruiqi Zhong, Luke Zettlemoyer, Wen-tau Yih, Daniel Fried, Sida Wang, and Tao Yu. 2023. DS-1000: A natural and reliable benchmark for data science code generation. In International Conference on Machine Learning. PMLR, 18319\u201318345."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Bochen Li Xinzhao Liu Karthik Dinesh Zhiyao Duan and Gaurav Sharma. 2018. Creating A Multi-track Classical Music Performance Dataset for Multi-modal Music Analysis: Challenges Insights and Applications. IEEE Transactions on Multimedia 21 2 (2018) 522\u2013535.","DOI":"10.1109\/TMM.2018.2856090"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.5555\/1128011.1128202"},{"key":"e_1_3_3_1_14_2","volume-title":"Forty-second International Conference on Machine Learning","author":"Liu Zihan","year":"2025","unstructured":"Zihan Liu, Shuangrui Ding, Zhixiong Zhang, Xiaoyi Dong, Pan Zhang, Yuhang Zang, Yuhang Cao, Dahua Lin, and Jiaqi Wang. 2025. SongGen: A Single Stage Auto-regressive Transformer for Text-to-Song Generation. In Forty-second International Conference on Machine Learning."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Zuxin Liu Thai Hoang Jianguo Zhang Ming Zhu Tian Lan Juntao Tan Weiran Yao Zhiwei Liu Yihao Feng Rithesh RN et\u00a0al. 2024. Apigen: Automated pipeline for generating verifiable and diverse function-calling datasets. Advances in Neural Information Processing Systems 37 (2024) 54463\u201354482.","DOI":"10.52202\/079017-1725"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-36159-6_23"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Fabian Ostermann Igor Vatolkin and Martin Ebeling. 2023. AAM: a dataset of Artificial Audio Multitracks for diverse music information retrieval tasks. EURASIP Journal on Audio Speech and Music Processing 2023 1 (2023) 13.","DOI":"10.1186\/s13636-023-00278-7"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Yurii Pushkarenko and Volodymyr Zaslavskyi. 2024. Synthetic Data Generation for Fraud Detection Using Diffusion Models. Information & Security 55 2 (2024) 185\u2013198.","DOI":"10.11610\/isij.5534"},{"key":"e_1_3_3_1_19_2","first-page":"126","volume-title":"Proceedings of the Detection and Classification of Acoustic Scenes and Events 2024 Workshop (DCASE2024)","author":"Ronchini Francesca","year":"2024","unstructured":"Francesca Ronchini, Luca Comanducci, Fabio Antonacci, et\u00a0al. 2024. Synthetic Training Set Generation using Text-To-Audio Models for Environmental Sound Classification. In Proceedings of the Detection and Classification of Acoustic Scenes and Events 2024 Workshop (DCASE2024). 126\u2013130."},{"key":"e_1_3_3_1_20_2","unstructured":"Rameel Sethi. 2018. A Framework for Synthesis of Musical Training Examples for Polyphonic Instrument Recognition. Ph.\u00a0D. Dissertation. University of Alberta."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Matthias Templ Bernhard Meindl Alexander Kowarik and Olivier Dupriez. 2017. Simulation of synthetic complex data: The R package simPop. Journal of Statistical Software 79 (2017) 1\u201338.","DOI":"10.18637\/jss.v079.i10"},{"key":"e_1_3_3_1_22_2","unstructured":"Cheng-Han Wu Pimpa Cheewaprakobkit Timothy\u00a0K Shih Yu-Cheng Lin and Bing-Ze Liu. 2025. Automatic Timbre Transformation using Enhanced Diffusion Model. IEEE Access (2025)."},{"key":"e_1_3_3_1_23_2","volume-title":"Proc. of the International Conference on Learning Representations (ICLR)","author":"Wu Yusong","year":"2022","unstructured":"Yusong Wu, Ethan Manilow, Yi Deng, Rigel\u00a0Jacob Swavely, Kyle Kastner, Tim Cooijmans, Aaron Courville, Anna Huang, and Jesse Engel. 2022. MIDI-DDSP: Hierarchical modeling of music for detailed control. In Proc. of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Jaime Yan. 2025. A Novel Pipeline for Generating Realistic Synthetic CDISC ADaM Datasets Using Large Language Models and Knowledge Graphs. Authorea Preprints (2025).","DOI":"10.36227\/techrxiv.174234967.75285658\/v1"},{"key":"e_1_3_3_1_26_2","unstructured":"Jianguo Zhang Tian Lan Rithesh Murthy Zhiwei Liu Weiran Yao Juntao Tan Thai Hoang Liangwei Yang Yihao Feng Zuxin Liu et\u00a0al. 2024. AgentOhana: Design Unified Data and Training Pipeline for Effective Agent Learning. CoRR (2024)."}],"event":{"name":"ICCIP 2025: 2025 the 11th International Conference on Communication and Information Processing","location":"Lingshui Hainan China","acronym":"ICCIP 2025"},"container-title":["Proceedings of the 2025 11th International Conference on Communication and Information Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3784833.3784908","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T07:44:36Z","timestamp":1770018276000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3784833.3784908"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,12]]},"references-count":25,"alternative-id":["10.1145\/3784833.3784908","10.1145\/3784833"],"URL":"https:\/\/doi.org\/10.1145\/3784833.3784908","relation":{},"subject":[],"published":{"date-parts":[[2025,11,12]]},"assertion":[{"value":"2026-02-01","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}