{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,28]],"date-time":"2026-02-28T17:58:21Z","timestamp":1772301501449,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819981373","type":"print"},{"value":"9789819981380","type":"electronic"}],"license":[{"start":{"date-parts":[[2023,11,26]],"date-time":"2023-11-26T00:00:00Z","timestamp":1700956800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,26]],"date-time":"2023-11-26T00:00:00Z","timestamp":1700956800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-99-8138-0_5","type":"book-chapter","created":{"date-parts":[[2023,11,25]],"date-time":"2023-11-25T10:02:23Z","timestamp":1700906543000},"page":"48-61","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["TPTGAN: Two-Path Transformer-Based Generative Adversarial Network Using Joint Magnitude Masking and\u00a0Complex Spectral Mapping for\u00a0Speech Enhancement"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0678-5940","authenticated-orcid":false,"given":"Zhaoyi","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0746-5015","authenticated-orcid":false,"given":"Zhuohang","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1112-0336","authenticated-orcid":false,"given":"Wendian","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1812-7360","authenticated-orcid":false,"given":"Zhuoyao","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8620-0537","authenticated-orcid":false,"given":"Haoda","family":"Di","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8140-4417","authenticated-orcid":false,"given":"Yufan","family":"Long","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1197-5906","authenticated-orcid":false,"given":"Haizhou","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,26]]},"reference":[{"key":"5_CR1","doi-asserted-by":"crossref","unstructured":"Zheng, C., Peng, X., Zhang, Y., Srinivasan, S., Lu, Y.: Interactive speech and noise modeling for speech enhancement. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, no. 16, pp. 14549\u201314557 (2021)","DOI":"10.1609\/aaai.v35i16.17710"},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Valentini-Botinhao, C., Wang, X., Takaki, S., Yamagishi, J.: Investigating RNN-based speech enhancement methods for noise-robust text-to-speech. In: Proceedings of ISCA Workshop on Speech Synthesis Workshop, pp. 146\u2013152 (2016)","DOI":"10.21437\/SSW.2016-24"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Yin, D., Luo, C., Xiong, Z., Zeng, W.: PHASEN: a phase-and-harmonics-aware speech enhancement network. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, no. 05, pp. 9458\u20139465 (2020)","DOI":"10.1609\/aaai.v34i05.6489"},{"key":"5_CR4","doi-asserted-by":"crossref","unstructured":"Rethage, D., Pons, J., Serra, X.: A Wavenet for speech denoising. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 5069\u20135073 (2018)","DOI":"10.1109\/ICASSP.2018.8462417"},{"key":"5_CR5","doi-asserted-by":"crossref","unstructured":"Pandey, A., Wang, D.: TCNN: temporal convolutional neural network for real-time speech enhancement in the time domain. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 6875\u20136879 (2019)","DOI":"10.1109\/ICASSP.2019.8683634"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Wang, K., He, B., Zhu, W.-P.: TSTNN: two-stage transformer based neural network for speech enhancement in the time domain. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 7098\u20137102 (2021)","DOI":"10.1109\/ICASSP39728.2021.9413740"},{"key":"5_CR7","doi-asserted-by":"crossref","unstructured":"Pascual, S., Bonafonte, A., Serra, J.: SEGAN: speech enhancement generative adversarial network. In: Proceedings of Interspeech, pp. 3642\u20133646 (2017)","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"5_CR8","doi-asserted-by":"crossref","unstructured":"Defossez, A., Synnaeve, G., Adi, Y.: Real time speech enhancement in the waveform domain. In: Proceedings of Interspeech, pp. 3291\u20133295 (2020)","DOI":"10.21437\/Interspeech.2020-2409"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Fu, S.-W., Hu, T., Tsao, Y., Lu, X.: Complex spectrogram enhancement by convolutional neural network with multi-metrics learning. In: Proceedings of International Workshop on Machine Learning for Signal Processing, pp. 1\u20136 (2017)","DOI":"10.1109\/MLSP.2017.8168119"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Soni, M.H., Shah, N., Patil, H.A.: Time-frequency masking-based speech enhancement using generative adversarial network. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 5039\u20135043 (2018)","DOI":"10.1109\/ICASSP.2018.8462068"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Takahashi, N., Agrawal, P., Goswami, N., Mitsufuji, Y.: PhaseNet: discretized phase modeling with deep neural networks for audio source separation. In: Proceedings of INTERSPEECH, pp. 2713\u20132717 (2018)","DOI":"10.21437\/Interspeech.2018-1773"},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Tan, K., Wang, D.: Learning complex spectral mapping with gated convolutional recurrent networks for monaural speech enhancement. In: IEEE\/ACM Transactions on Audio, Speech, and Language Processing, vol. 28, pp. 380\u2013390 (2020)","DOI":"10.1109\/TASLP.2019.2955276"},{"key":"5_CR13","doi-asserted-by":"publisher","first-page":"2018","DOI":"10.1109\/LSP.2021.3116502","volume":"28","author":"Z-Q Wang","year":"2021","unstructured":"Wang, Z.-Q., Wichern, G., Le Roux, J.: On the compensation between magnitude and phase in speech separation. IEEE Sig. Process. Lett. 28, 2018\u20132022 (2021)","journal-title":"IEEE Sig. Process. Lett."},{"key":"5_CR14","unstructured":"Fu, S.-W., Liao, C.-F., Tsao, Y., Lin, S.D.: MetricGAN: generative adversarial networks based black-box metric scores optimization for speech enhancement. In: Proceedings of International Conference on Machine Learning, pp. 2031\u20132041 (2019)"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Kim, E., Seo, H.: SE-conformer: time-domain speech enhancement using conformer. In: Proceedings of Interspeech, pp. 2736\u20132740 (2021)","DOI":"10.21437\/Interspeech.2021-2207"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Cao, R., Abdulatif, S., Yang, B.: CMGAN: conformer-based metric GAN for speech enhancement. In: Proceedings of INTERSPEECH, pp. 936\u2013940 (2022)","DOI":"10.36227\/techrxiv.21187846.v2"},{"key":"5_CR17","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of International Conference on Neural Information Processing Systems, pp. 6000\u20136010 (2017)"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Dang, F., Chen, H., Zhang, P.: DPT-FSNet: dual-path transformer based full-band and sub-band fusion network for speech enhancement. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 6857\u20136861 (2022)","DOI":"10.1109\/ICASSP43922.2022.9746171"},{"key":"5_CR19","unstructured":"Hendrycks, D., Gimpel, K.: Bridging nonlinearities and stochastic regularizers with gaussian error linear units. arXiv preprint arXiv:1606.08415 (2016)"},{"key":"5_CR20","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"5_CR21","doi-asserted-by":"crossref","unstructured":"Braun, S., Tashev, I.: A consolidated view of loss functions for supervised deep learning-based speech enhancement. In: Proceedings of International Conference on Telecommunications and Signal Processing, pp. 72\u201376 (2021)","DOI":"10.1109\/TSP52935.2021.9522648"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Pandey, A., Wang, D.: Densely connected neural network with dilated convolutions for real-time speech enhancement in the time domain. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 6629\u20136633 (2020)","DOI":"10.1109\/ICASSP40776.2020.9054536"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Delving deep into rectifiers: surpassing human-level performance on ImageNet classification. In: Proceedings of International Conference on Computer Vision, pp. 1026\u20131034 (2015)","DOI":"10.1109\/ICCV.2015.123"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Fu, S.-W., et al.: MetricGAN+: an improved version of MetricGAN for speech enhancement. In: Proceedings of INTERSPEECH, pp. 201\u2013205 (2021)","DOI":"10.21437\/Interspeech.2021-599"},{"key":"5_CR25","doi-asserted-by":"crossref","unstructured":"Mao, X., et al.: Least squares generative adversarial networks. In: Proceedings of International Conference on Computer Vision, pp. 2813\u20132821 (2017)","DOI":"10.1109\/ICCV.2017.304"},{"key":"5_CR26","doi-asserted-by":"crossref","unstructured":"Veaux, C., Yamagishi, J., King, S.: The voice bank corpus: design, collection and data analysis of a large regional accent speech database. In: Proceedings of International Conference Oriental COCOSDA held jointly with Conference on Asian Spoken Language Research and Evaluation, pp. 1\u20134 (2013)","DOI":"10.1109\/ICSDA.2013.6709856"},{"issue":"5","key":"5_CR27","first-page":"3591","volume":"133","author":"T Joachim","year":"2013","unstructured":"Joachim, T., Ito, N., Vincent, E.: The diverse environments multi-channel acoustic noise database: a database of multichannel environmental noise recordings. J. Acoust. Soc. Am. 133(5), 3591\u20133591 (2013)","journal-title":"J. Acoust. Soc. Am."},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Chen, J., Mao, Q., Liu, D.: Dual-path transformer network: direct context-aware modeling for end-to-end monaural speech separation. In: Proceedings of INTERSPEECH, pp. 2642\u20132646 (2020)","DOI":"10.21437\/Interspeech.2020-2205"},{"key":"5_CR29","doi-asserted-by":"publisher","DOI":"10.1201\/b14529","volume-title":"Speech Enhancement: Theory and Practice","author":"PC Loizou","year":"2013","unstructured":"Loizou, P.C.: Speech Enhancement: Theory and Practice, 2nd edn. CRC Press, Boca Raton (2013)","edition":"2"},{"issue":"1","key":"5_CR30","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y Hu","year":"2008","unstructured":"Hu, Y., Loizou, P.C.: Evaluation of objective quality measures for speech enhancement. IEEE Trans. Audio Speech Lang. Process. 16(1), 229\u2013238 (2008)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Hansen, J.H.L., Pellom, B.L.: An effective quality evaluation protocol for speech enhancement algorithms. In: Proceedings of International Conference on Spoken Language Processing (1998)","DOI":"10.21437\/ICSLP.1998-350"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Yu, G., Li, A., Zheng, C., Guo, Y., Wang, Y., Wang, H.: Dual-branch attention-in-attention transformer for single-channel speech enhancement. In Proceedings of International Conference on Acoustics, Speech and Signal Processing, pp. 7847\u20137851 (2022)","DOI":"10.1109\/ICASSP43922.2022.9746273"},{"key":"5_CR33","unstructured":"Macartney, C., Weyde, T.: Improved speech enhancement with the wave-U-Net. arXiv preprint arXiv:1811.11307 (2018)"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Schr\u00f6ter, H., Escalante-B., A.N., Rosenkranz, T., Maier, A.: DeepFilterNet2: towards real-time speech enhancement on embedded devices for full-band audio. arXiv preprint arXiv:2205.05474 (2022)","DOI":"10.1109\/IWAENC53105.2022.9914782"},{"key":"5_CR35","unstructured":"Dang, F., Hu, Q., Zhang, P.: THLNet: two-stage heterogeneous lightweight network for monaural speech enhancement. arXiv preprint arXiv:2301.07939 (2023)"}],"container-title":["Communications in Computer and Information Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-99-8138-0_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,13]],"date-time":"2024-03-13T17:30:06Z","timestamp":1710351006000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-99-8138-0_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,26]]},"ISBN":["9789819981373","9789819981380"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-99-8138-0_5","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,26]]},"assertion":[{"value":"26 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Changsha","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 November 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2023.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1274","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"650","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"51% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.14","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.46","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}