{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T01:58:00Z","timestamp":1743127080714,"version":"3.40.3"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783031301070"},{"type":"electronic","value":"9783031301087"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-30108-7_9","type":"book-chapter","created":{"date-parts":[[2023,4,12]],"date-time":"2023-04-12T04:03:04Z","timestamp":1681272184000},"page":"101-112","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MVNet: Memory Assistance and\u00a0Vocal Reinforcement Network for\u00a0Speech Enhancement"],"prefix":"10.1007","author":[{"given":"Jianrong","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaomin","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuewei","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mei","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,4,13]]},"reference":[{"key":"9_CR1","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. In: Advances in Neural Information Processing Systems, vol. 33, pp. 12449\u201312460 (2020)"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Chen, G., et al.: Who is real bob? Adversarial attacks on speaker recognition systems. In: 2021 IEEE Symposium on Security and Privacy (SP), pp. 694\u2013711. IEEE (2021)","DOI":"10.1109\/SP40001.2021.00004"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Chen, J., Mao, Q., Liu, D.: Dual-path transformer network: direct context-aware modeling for end-to-end monaural speech separation. arXiv preprint arXiv:2007.13975 (2020)","DOI":"10.21437\/Interspeech.2020-2205"},{"key":"9_CR4","unstructured":"Cosentino, J., Pariente, M., Cornell, S., Deleforge, A., Vincent, E.: Librimix: an open-source dataset for generalizable speech separation. arXiv preprint arXiv:2005.11262 (2020)"},{"key":"9_CR5","doi-asserted-by":"publisher","unstructured":"Faraji, F., Attabi, Y., Champagne, B., Zhu, W.P.: On the use of audio fingerprinting features for speech enhancement with generative adversarial network. In: 2020 IEEE Workshop on Signal Processing Systems (SiPS), pp. 1\u20136 (2020). https:\/\/doi.org\/10.1109\/SiPS50750.2020.9195238","DOI":"10.1109\/SiPS50750.2020.9195238"},{"key":"9_CR6","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1109\/LSP.2019.2953810","volume":"27","author":"SW Fu","year":"2019","unstructured":"Fu, S.W., Liao, C.F., Tsao, Y.: Learning with learned loss function: speech enhancement with quality-net to improve perceptual evaluation of speech quality. IEEE Signal Process. Lett. 27, 26\u201330 (2019)","journal-title":"IEEE Signal Process. Lett."},{"key":"9_CR7","doi-asserted-by":"crossref","unstructured":"Fu, S.W., Yu, C., Hung, K.H., Ravanelli, M., Tsao, Y.: Metricgan-u: unsupervised speech enhancement\/dereverberation based only on noisy\/reverberated speech. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7412\u20137416. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747180"},{"key":"9_CR8","doi-asserted-by":"crossref","unstructured":"Garcia-Romero, D., Espy-Wilson, C.Y.: Analysis of i-vector length normalization in speaker recognition systems. In: Twelfth Annual Conference of the International Speech Communication Association (2011)","DOI":"10.21437\/Interspeech.2011-53"},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Hao, X., Su, X., Horaud, R., Li, X.: Fullsubnet: a full-band and sub-band fusion model for real-time single-channel speech enhancement. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6633\u20136637. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9414177"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Hsieh, T.A., Yu, C., Fu, S.W., Lu, X., Tsao, Y.: Improving perceptual quality by phone-fortified perceptual loss using wasserstein distance for speech enhancement. arXiv preprint arXiv:2010.15174 (2020)","DOI":"10.21437\/Interspeech.2021-582"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Hu, Y., et al.: DCCRN: deep complex convolution recurrent network for phase-aware speech enhancement. arXiv preprint arXiv:2008.00264 (2020)","DOI":"10.21437\/Interspeech.2020-2537"},{"issue":"1","key":"9_CR12","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y Hu","year":"2007","unstructured":"Hu, Y., Loizou, P.C.: Evaluation of objective quality measures for speech enhancement. IEEE Trans. Audio Speech Lang. Process. 16(1), 229\u2013238 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"5","key":"9_CR13","doi-asserted-by":"publisher","first-page":"808","DOI":"10.1016\/j.solener.2011.01.013","volume":"85","author":"W Ji","year":"2011","unstructured":"Ji, W., Chee, K.C.: Prediction of hourly solar radiation using a novel hybrid model of ARMA and TDNN. Sol. Energy 85(5), 808\u2013817 (2011)","journal-title":"Sol. Energy"},{"key":"9_CR14","first-page":"17022","volume":"33","author":"J Kong","year":"2020","unstructured":"Kong, J., Kim, J., Bae, J.: HiFi-GAN: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural. Inf. Process. Syst. 33, 17022\u201317033 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"9_CR15","doi-asserted-by":"publisher","first-page":"1829","DOI":"10.1109\/TASLP.2021.3079813","volume":"29","author":"A Li","year":"2021","unstructured":"Li, A., Liu, W., Zheng, C., Fan, C., Li, X.: Two heads are better than one: a two-stage complex spectral mapping approach for monaural speech enhancement. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 1829\u20131843 (2021). https:\/\/doi.org\/10.1109\/TASLP.2021.3079813","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR16","unstructured":"Li, C., et al.: Deep speaker: an end-to-end neural speaker embedding system. arXiv preprint arXiv:1705.02304 (2017)"},{"key":"9_CR17","doi-asserted-by":"publisher","first-page":"292","DOI":"10.1109\/TMM.2020.2976493","volume":"23","author":"L Liu","year":"2020","unstructured":"Liu, L., Feng, G., Beautemps, D., Zhang, X.P.: Re-synchronization using the hand preceding model for multi-modal fusion in automatic continuous cued speech recognition. IEEE Trans. Multimedia 23, 292\u2013305 (2020)","journal-title":"IEEE Trans. Multimedia"},{"key":"9_CR18","doi-asserted-by":"crossref","unstructured":"Liu, L., Hueber, T., Feng, G., Beautemps, D.: Visual recognition of continuous cued speech using a tandem CNN-HMM approach. In: Interspeech, pp. 2643\u20132647 (2018)","DOI":"10.21437\/Interspeech.2018-2434"},{"issue":"8","key":"9_CR19","doi-asserted-by":"publisher","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","volume":"27","author":"Y Luo","year":"2019","unstructured":"Luo, Y., Mesgarani, N.: Conv-TasNet: surpassing ideal time-frequency magnitude masking for speech separation. IEEE\/ACM Trans. Audio Speech Lang. Process. 27(8), 1256\u20131266 (2019)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR20","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: an ASR corpus based on public domain audio books. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5206\u20135210. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Pandey, A., Wang, D.: Densely connected neural network with dilated convolutions for real-time speech enhancement in the time domain. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6629\u20136633. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054536"},{"key":"9_CR22","unstructured":"Quackenbush, S.R.: Objective measures of speech quality (subjective) (1986)"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Rix, A.W., Beerends, J.G., Hollier, M.P., Hekstra, A.P.: Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In: 2001 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No. 01CH37221), vol. 2, pp. 749\u2013752. IEEE (2001)","DOI":"10.1109\/ICASSP.2001.941023"},{"issue":"7","key":"9_CR24","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","volume":"19","author":"CH Taal","year":"2011","unstructured":"Taal, C.H., Hendriks, R.C., Heusdens, R., Jensen, J.: An algorithm for intelligibility prediction of time-frequency weighted noisy speech. IEEE Trans. Audio Speech Lang. Process. 19(7), 2125\u20132136 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Tan, K., Wang, D.: A convolutional recurrent neural network for real-time speech enhancement. In: Interspeech, vol. 2018, pp. 3229\u20133233 (2018)","DOI":"10.21437\/Interspeech.2018-1405"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Tan, K., Wang, D.: Complex spectral mapping with a convolutional recurrent network for monaural speech enhancement. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6865\u20136869. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8682834"},{"key":"9_CR27","doi-asserted-by":"publisher","first-page":"1785","DOI":"10.1109\/TASLP.2021.3082282","volume":"29","author":"K Tan","year":"2021","unstructured":"Tan, K., Wang, D.: Towards model compression for deep learning based speech enhancement. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 1785\u20131794 (2021). https:\/\/doi.org\/10.1109\/TASLP.2021.3082282","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"9_CR28","first-page":"19","volume":"1","author":"V Tiwari","year":"2010","unstructured":"Tiwari, V.: MFCC and its applications in speaker recognition. Int. J. Emerg. Technol. 1(1), 19\u201322 (2010)","journal-title":"Int. J. Emerg. Technol."},{"issue":"4","key":"9_CR29","doi-asserted-by":"publisher","first-page":"1462","DOI":"10.1109\/TSA.2005.858005","volume":"14","author":"E Vincent","year":"2006","unstructured":"Vincent, E., Gribonval, R., F\u00e9votte, C.: Performance measurement in blind audio source separation. IEEE Trans. Audio Speech Lang. Process. 14(4), 1462\u20131469 (2006)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"9_CR30","doi-asserted-by":"crossref","unstructured":"Wang, J., et al.: Three-dimensional lip motion network for text-independent speaker recognition. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp. 3380\u20133387. IEEE (2021)","DOI":"10.1109\/ICPR48806.2021.9413218"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Wang, K., He, B., Zhu, W.P.: TSTNN: two-stage transformer based neural network for speech enhancement in the time domain. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7098\u20137102. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9413740"},{"key":"9_CR32","doi-asserted-by":"crossref","unstructured":"Wichern, G., et al.: Wham!: extending speech separation to noisy environments. arXiv preprint arXiv:1907.01160 (2019)","DOI":"10.21437\/Interspeech.2019-2821"},{"issue":"3","key":"9_CR33","doi-asserted-by":"publisher","first-page":"483","DOI":"10.1109\/TASLP.2015.2512042","volume":"24","author":"DS Williamson","year":"2015","unstructured":"Williamson, D.S., Wang, Y., Wang, D.: Complex ratio masking for monaural speech separation. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(3), 483\u2013492 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."}],"container-title":["Lecture Notes in Computer Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-30108-7_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T03:19:56Z","timestamp":1729221596000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-30108-7_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031301070","9783031301087"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-30108-7_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"13 April 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"New Delhi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 November 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 November 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/iconip2022.apnns.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easy Chair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"810","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"359","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"44% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.65","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"ICONIP 2022 consists of a two-volume set, LNCS & CCIS, which includes 146 and 213 papers","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}