{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,21]],"date-time":"2026-04-21T15:05:39Z","timestamp":1776783939601,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","funder":[{"name":"STI 2030?Major Projects","award":["2021ZD0201500"],"award-info":[{"award-number":["2021ZD0201500"]}]},{"name":"National Natural Science Foundation of China (NSFC)","award":["62201002,6247077204"],"award-info":[{"award-number":["62201002,6247077204"]}]},{"name":"Excellent Youth Foundation of Anhui Scientific Committee","award":["2408085Y034"],"award-info":[{"award-number":["2408085Y034"]}]},{"name":"Cloud Ginger XR-1"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755501","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:47:42Z","timestamp":1761371262000},"page":"6977-6985","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["DMF2Mel: A Dynamic Multiscale Fusion Network for EEG-Driven Mel Spectrogram Reconstruction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6318-8803","authenticated-orcid":false,"given":"Cunhang","family":"Fan","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology,State Key Laboratory of Opto-Electronic Information Acquisition and Protection Technology, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5291-5807","authenticated-orcid":false,"given":"Sheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1173-2970","authenticated-orcid":false,"given":"Jingjing","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2248-9920","authenticated-orcid":false,"given":"Enrui","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5522-0037","authenticated-orcid":false,"given":"Xinhui","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8441-5462","authenticated-orcid":false,"given":"Gangming","family":"Zhao","sequence":"additional","affiliation":[{"name":"School of Electronics and Information Engineering, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4530-4422","authenticated-orcid":false,"given":"Zhao","family":"Lv","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology,State Key Laboratory of Opto-Electronic Information Acquisition and Protection Technology, Anhui University, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Hugo Van Hamme, and Tom Francart","author":"Accou Bernd","year":"2023","unstructured":"Bernd Accou, Lies Bollens, Marlies Gillis, Wendy Verheijen, Hugo Van Hamme, and Tom Francart. 2023a. SparrKULee: A speech-evoked auditory response repository of the KU Leuven, containing EEG of 85 participants. BioRxiv (2023), 2023-07."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-022-27332-2"},{"key":"e_1_3_2_1_3_1","volume-title":"Towards reconstructing intelligible speech from the human auditory cortex. Scientific reports","author":"Akbari Hassan","year":"2019","unstructured":"Hassan Akbari, Bahar Khalighinejad, Jose L Herrero, Ashesh D Mehta, and Nima Mesgarani. 2019. Towards reconstructing intelligible speech from the human auditory cortex. Scientific reports, Vol. 9, 1 (2019), 874."},{"key":"e_1_3_2_1_4_1","first-page":"493","volume-title":"Nature","volume":"568","author":"Anumanchipalli Gopala K","year":"2019","unstructured":"Gopala K Anumanchipalli, Josh Chartier, and Edward F Chang. 2019. Speech synthesis from neural decoding of spoken sentences. Nature, Vol. 568, 7753 (2019), 493-498."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-024-00824-8"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.08.035"},{"key":"e_1_3_2_1_7_1","volume-title":"M3ANet: Multi-scale and Multi-Modal Alignment Network for Brain-Assisted Target Speaker Extraction. arXiv preprint arXiv:2506.00466","author":"Fan Cunhang","year":"2025","unstructured":"Cunhang Fan, Ying Chen, Jian Zhou, Zexu Pan, Jingjing Zhang, Youdian Gao, Xiaoke Yang, Zhengqi Wen, and Zhao Lv. 2025a. M3ANet: Multi-scale and Multi-Modal Alignment Network for Brain-Assisted Target Speaker Extraction. arXiv preprint arXiv:2506.00466 (2025)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3533359"},{"key":"e_1_3_2_1_9_1","volume-title":"ListenNet: A Lightweight Spatio-Temporal Enhancement Nested Network for Auditory Attention Detection. arXiv preprint arXiv:2505.10348","author":"Fan Cunhang","year":"2025","unstructured":"Cunhang Fan, Xiaoke Yang, Hongyu Zhang, Ying Chen, Lu Li, Jian Zhou, and Zhao Lv. 2025c. ListenNet: A Lightweight Spatio-Temporal Enhancement Nested Network for Auditory Attention Detection. arXiv preprint arXiv:2505.10348 (2025)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106580"},{"key":"e_1_3_2_1_11_1","volume-title":"Seeing helps hearing: A multi-modal dataset and a mamba-based dual branch parallel network for auditory attention decoding. Information Fusion","author":"Fan Cunhang","year":"2025","unstructured":"Cunhang Fan, Hongyu Zhang, Qinke Ni, Jingjing Zhang, Jianhua Tao, Jian Zhou, Jiangyan Yi, Zhao Lv, and Xiaopei Wu. 2025d. Seeing helps hearing: A multi-modal dataset and a mamba-based dual branch parallel network for auditory attention decoding. Information Fusion (2025), 102946."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681550"},{"key":"e_1_3_2_1_13_1","volume-title":"2025 e. SSM2Mel: State Space Model to Reconstruct Mel Spectrogram from the EEG. arXiv preprint arXiv:2501.10402","author":"Fan Cunhang","year":"2025","unstructured":"Cunhang Fan, Sheng Zhang, Jingjing Zhang, Zexu Pan, and Zhao Lv. 2025 e. SSM2Mel: State Space Model to Reconstruct Mel Spectrogram from the EEG. arXiv preprint arXiv:2501.10402 (2025)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1162\/0898929053279522"},{"key":"e_1_3_2_1_15_1","volume-title":"Brain-computer interfaces: Revolutionizing human-computer interaction","author":"Graimann Bernhard","unstructured":"Bernhard Graimann, Brendan Allison, and Gert Pfurtscheller. 2010. Brain-computer interfaces: A gentle introduction. In Brain-computer interfaces: Revolutionizing human-computer interaction. Springer, 1-27."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2024.3370299"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053340"},{"key":"e_1_3_2_1_18_1","volume-title":"Hippo-kan: Efficient kan model for time series analysis. arXiv preprint arXiv:2410.14939","author":"Lee SangJong","year":"2024","unstructured":"SangJong Lee, Jin-Kwang Kim, JunHo Kim, TaeHan Kim, and James Lee. 2024. Hippo-kan: Efficient kan model for time series analysis. arXiv preprint arXiv:2410.14939 (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i5.25745"},{"key":"e_1_3_2_1_20_1","volume-title":"ConDSeg: A General Medical Image Segmentation Framework via Contrast-Driven Feature Enhancement. arXiv preprint arXiv:2412.08345","author":"Lei Mengqi","year":"2024","unstructured":"Mengqi Lei, Haochen Wu, Xinhua Lv, and Xin Wang. 2024. ConDSeg: A General Medical Image Segmentation Framework via Contrast-Driven Feature Enhancement. arXiv preprint arXiv:2412.08345 (2024)."},{"key":"e_1_3_2_1_21_1","first-page":"2620","article-title":"Cross-Attention-Guided WaveNet for EEG-to-MEL Spectrogram Reconstruction","volume":"2024","author":"Li Hao","year":"2024","unstructured":"Hao Li, Yuan Fang, Xueliang Zhang, Fei Chen, and Guanglai Gao. 2024. Cross-Attention-Guided WaveNet for EEG-to-MEL Spectrogram Reconstruction. In Proc. Interspeech 2024. 2620-2624.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_22_1","volume-title":"Human cortical encoding of pitch in tonal and non-tonal languages. Nature communications","author":"Li Yuanning","year":"2021","unstructured":"Yuanning Li, Claire Tang, Junfeng Lu, Jinsong Wu, and Edward F Chang. 2021. Human cortical encoding of pitch in tonal and non-tonal languages. Nature communications, Vol. 12, 1 (2021), 1161."},{"key":"e_1_3_2_1_23_1","volume-title":"Vmamba: Visual state space model. Advances in neural information processing systems","author":"Liu Yue","year":"2024","unstructured":"Yue Liu, Yunjie Tian, Yuzhong Zhao, Hongtian Yu, Lingxi Xie, Yaowei Wang, Qixiang Ye, Jianbin Jiao, and Yunfan Liu. 2024a. Vmamba: Visual state space model. Advances in neural information processing systems, Vol. 37 (2024), 103031-103063."},{"key":"e_1_3_2_1_24_1","volume-title":"Kan: Kolmogorov-arnold networks. arXiv preprint arXiv:2404.19756","author":"Liu Ziming","year":"2024","unstructured":"Ziming Liu, Yixuan Wang, Sachin Vaidya, Fabian Ruehle, James Halverson, Marin Solja\u010di\u0107, Thomas Y Hou, and Max Tegmark. 2024b. Kan: Kolmogorov-arnold networks. arXiv preprint arXiv:2404.19756 (2024)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1088\/1741-2552\/acc976"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Sean L Metzger Jessie R Liu David A Moses Maximilian E Dougherty Margaret P Seaton Kaylo T Littlejohn Josh Chartier Gopala K Anumanchipalli Adelyn Tu-Chan Karunesh Ganguly et al. 2022. Generalizable spelling using a speech neuroprosthesis in an individual with severe limb and vocal paralysis. Nature communications Vol. 13 1 (2022) 6510.","DOI":"10.1038\/s41467-022-33611-3"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1088\/1741-2560\/13\/5\/056004"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1056\/NEJMoa2027540"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095477"},{"key":"e_1_3_2_1_30_1","first-page":"234","volume-title":"Munich","author":"Ronneberger Olaf","year":"2015","unstructured":"Olaf Ronneberger, Philipp Fischer, and Thomas Brox. 2015. U-net: Convolutional networks for biomedical image segmentation. In Medical image computing and computer-assisted intervention-MICCAI 2015: 18th international conference, Munich, Germany, October 5-9, 2015, proceedings, part III 18. Springer, 234-241."},{"key":"e_1_3_2_1_31_1","volume-title":"Structured neuronal encoding and decoding of human speech features. Nature communications","author":"Tankus Ariel","year":"2012","unstructured":"Ariel Tankus, Itzhak Fried, and Shy Shoham. 2012. Structured neuronal encoding and decoding of human speech features. Nature communications, Vol. 3, 1 (2012), 1015."},{"key":"e_1_3_2_1_32_1","volume-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-2.","author":"Dyck Bob Van","year":"2023","unstructured":"Bob Van Dyck, Liuyin Yang, and Marc M Van Hulle. 2023. Decoding auditory eeg responses using an adapted wavenet. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-2."},{"key":"e_1_3_2_1_33_1","volume-title":"Representation of internal speech by single neurons in human supramarginal gyrus. Nature human behaviour","author":"Wandelt Sarah K","year":"2024","unstructured":"Sarah K Wandelt, David A Bj\u00e5nes, Kelsie Pejsa, Brian Lee, Charles Liu, and Richard A Andersen. 2024. Representation of internal speech by single neurons in human supramarginal gyrus. Nature human behaviour, Vol. 8, 6 (2024), 1136-1149."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i5.20472"},{"key":"e_1_3_2_1_35_1","first-page":"249","volume-title":"Nature","volume":"593","author":"Willett Francis R","year":"2021","unstructured":"Francis R Willett, Donald T Avansino, Leigh R Hochberg, Jaimie M Henderson, and Krishna V Shenoy. 2021. High-performance brain-to-text communication via handwriting. Nature, Vol. 593, 7858 (2021), 249-254."},{"key":"e_1_3_2_1_36_1","first-page":"1031","volume-title":"Nature","volume":"620","author":"Willett Francis R","year":"2023","unstructured":"Francis R Willett, Erin M Kunz, Chaofei Fan, Donald T Avansino, Guy H Wilson, Eun Young Choi, Foram Kamdar, Matthew F Glasser, Leigh R Hochberg, Shaul Druckmann, et al., 2023. A high-performance speech neuroprosthesis. Nature, Vol. 620, 7976 (2023), 1031-1036."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"e_1_3_2_1_38_1","volume-title":"NeurIPS Workshop on Time Series in the Age of Large Models.","author":"Xu Kunpeng","year":"2024","unstructured":"Kunpeng Xu, Lifei Chen, and Shengrui Wang. 2024a. KAN4Drift: Are KAN Effective for Identifying and Tracking Concept Drift in Time Series?. In NeurIPS Workshop on Time Series in the Age of Large Models."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSPW62465.2024.10626859"},{"key":"e_1_3_2_1_40_1","first-page":"31688","article-title":"Darnet: Dual attention refinement network with spatiotemporal construction for auditory attention detection","volume":"37","author":"Yan Sheng","year":"2024","unstructured":"Sheng Yan, Cunhang Fan, Hongyu Zhang, Xiaoke Yang, Jianhua Tao, and Zhao Lv. 2024. Darnet: Dual attention refinement network with spatiotemporal construction for auditory attention detection. Advances in Neural Information Processing Systems, Vol. 37 (2024), 31688-31707.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-025-91674-w"},{"key":"e_1_3_2_1_42_1","volume-title":"KAN-AD: Time series anomaly detection with Kolmogorov-Arnold networks. arXiv preprint arXiv:2411.00278","author":"Zhou Quan","year":"2024","unstructured":"Quan Zhou, Changhua Pei, Fei Sun, Jing Han, Zhengwei Gao, Dan Pei, Haiming Zhang, Gaogang Xie, and Jianhui Li. 2024. KAN-AD: Time series anomaly detection with Kolmogorov-Arnold networks. arXiv preprint arXiv:2411.00278 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755501","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:22:52Z","timestamp":1765308172000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755501"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":42,"alternative-id":["10.1145\/3746027.3755501","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755501","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}