{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T18:35:23Z","timestamp":1780079723117,"version":"3.54.0"},"reference-count":29,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2023,10,5]],"date-time":"2023-10-05T00:00:00Z","timestamp":1696464000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,5]],"date-time":"2023-10-05T00:00:00Z","timestamp":1696464000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16879-5","type":"journal-article","created":{"date-parts":[[2023,10,5]],"date-time":"2023-10-05T13:01:35Z","timestamp":1696510895000},"page":"38465-38479","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["MAuD: a multivariate audio database of samples collected from benchmark conferencing platforms"],"prefix":"10.1007","volume":"83","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0026-6426","authenticated-orcid":false,"given":"Tapas","family":"Chakraborty","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rudrajit","family":"Bhattacharyya","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nibaran","family":"Das","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Subhadip","family":"Basu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mita","family":"Nasipuri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,10,5]]},"reference":[{"key":"16879_CR1","doi-asserted-by":"crossref","unstructured":"Aronowitz H, Aronowitz V (2010) Efficient score normalization for speaker recognition. In: ICASSP, IEEE International conference on acoustics, speech and signal processing - proceedings. pp. 4402\u20134405","DOI":"10.1109\/ICASSP.2010.5495629"},{"key":"16879_CR2","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-019-07988-1","author":"I Bakkouri","year":"2020","unstructured":"Bakkouri I, Afdel K (2020) Computer-aided diagnosis (cad) system based on multi-layer feature fusion network for skin lesion recognition in dermoscopy images. Multimedia Tools and Applications. https:\/\/doi.org\/10.1007\/s11042-019-07988-1","journal-title":"Multimedia Tools and Applications"},{"key":"16879_CR3","doi-asserted-by":"crossref","unstructured":"Barai B, Chakraborty T, Das N, Basu S, Nasipuri M (2022) Closed-set speaker identification using vq and gmm based models. International Journal of Speech Technology, Springer","DOI":"10.1007\/s10772-021-09899-9"},{"issue":"5","key":"16879_CR4","doi-asserted-by":"publisher","first-page":"308","DOI":"10.1109\/LSP.2006.870086","volume":"13","author":"WM Campbell","year":"2006","unstructured":"Campbell WM, Sturim DE, Reynolds DA (2006) Support vector machines using gmm supervectors for speaker verification. IEEE signal processing letters 13(5):308\u2013311","journal-title":"IEEE signal processing letters"},{"key":"16879_CR5","volume-title":"Callhome american english speech","author":"A Canavan","year":"1997","unstructured":"Canavan A, David G, George Z (1997) Callhome american english speech. Linguistic Data Consortium, Philadelphia"},{"key":"16879_CR6","doi-asserted-by":"publisher","unstructured":"Chakraborty T (2021) Audio files recorded using different voice calling platforms. figshare. media. In: https:\/\/doi.org\/10.6084\/m9.figshare.14731629.v1","DOI":"10.6084\/m9.figshare.14731629.v1"},{"key":"16879_CR7","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1007\/978-981-15-1084-7_28","volume-title":"Intelligent Computing and Communication","author":"T Chakraborty","year":"2020","unstructured":"Chakraborty T, Barai B, Chatterjee B, Das N, Basu S, Nasipuri M (2020) Closed-set device-independent speaker identification using cnn. In: Bhateja V, Satapathy SC, Zhang YD, Aradhya VNM (eds) Intelligent Computing and Communication. Springer Singapore, Singapore, pp 291\u2013299"},{"key":"16879_CR8","doi-asserted-by":"crossref","unstructured":"Dieleman S, Schrauwen B (2014) End-to-end learning for music audio. In: 2014 IEEE International conference on acoustics, speech and signal processing (ICASSP). pp. 6964\u20136968. IEEE","DOI":"10.1109\/ICASSP.2014.6854950"},{"key":"16879_CR9","doi-asserted-by":"crossref","unstructured":"Esmaeilpour M, Cardinal P, Koerich AL (2022) From environmental sound representation to robustness of 2d cnn models against adversarial attacks. Applied Acoustics 195:108817. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0003682X22001918","DOI":"10.1016\/j.apacoust.2022.108817"},{"key":"16879_CR10","doi-asserted-by":"crossref","unstructured":"Fujihara H, Kitahara T, Goto M, Komatani K, Ogata T, Okuno HG (2006) Speaker identification under noisy environments by using harmonic structure extraction and reliable frame weighting. In: Ninth international conference on spoken language processing","DOI":"10.21437\/Interspeech.2006-180"},{"key":"16879_CR11","doi-asserted-by":"crossref","unstructured":"Gemmeke JF, Ellis DPW, Freedman D, Jansen A, Lawrence W, Moore RC, Plakal M, Ritter M (2017) Audio set: An ontology and human-labeled dataset for audio events. In: Proc. IEEE ICASSP 2017. New Orleans, LA","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"16879_CR12","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1016\/j.csl.2017.06.007","volume":"47","author":"O Ghahabi","year":"2018","unstructured":"Ghahabi O, Hernando J (2018) Restricted boltzmann machines for vector representation of speech in speaker recognition. Computer Speech & Language 47:16\u201329","journal-title":"Computer Speech & Language"},{"key":"16879_CR13","volume-title":"Switchboard-1 release 2","author":"JJ Godfrey","year":"1993","unstructured":"Godfrey JJ, Edward H (1993) Switchboard-1 release 2. Linguistic Data Consortium, Philadelphia"},{"key":"16879_CR14","doi-asserted-by":"crossref","unstructured":"Haris B, Pradhan G, Misra A, Shukla S, Sinha R, Prasanna S (2011) Multi-variability speech database for robust speaker recognition. In: Communications (NCC), 2011 national conference on. pp 1\u20135. IEEE","DOI":"10.1109\/NCC.2011.5734775"},{"key":"16879_CR15","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: 2016 IEEE Conference on computer vision and pattern recognition (CVPR). pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"16879_CR16","doi-asserted-by":"crossref","unstructured":"Hershey S, Chaudhuri S, Ellis DPW, Gemmeke JF, Jansen A, Moore C, Plakal M, Platt D, Saurous RA, Seybold B, Slaney M, Weiss R, Wilson K (2017) Cnn architectures for large-scale audio classification. In: International conference on acoustics, speech and signal processing (ICASSP). arxiv:1609.09430","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"16879_CR17","unstructured":"Jumelle M, Sakmeche T (2018) Speaker clustering with neural networks and audio processing. arXiv:1803.08276"},{"key":"16879_CR18","doi-asserted-by":"crossref","unstructured":"Madikeri S, Bourlard H (2015) Kl-hmm based speaker diarization system for meetings. In: 2015 IEEE International conference on acoustics, speech and signal processing (ICASSP). pp 4435\u20134439. IEEE","DOI":"10.1109\/ICASSP.2015.7178809"},{"key":"16879_CR19","doi-asserted-by":"crossref","unstructured":"McFee B, Raffel C, Liang D, Ellis DP, McVicar M, Battenberg E, Nieto O (2015) librosa: Audio and music signal analysis in python","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"16879_CR20","doi-asserted-by":"crossref","unstructured":"Mesaros A, Heittola T, Virtanen T (2016) TUT database for acoustic scene classification and sound event detection. In: 24th European Signal Processing Conference 2016 (EUSIPCO 2016). Budapest, Hungary","DOI":"10.1109\/EUSIPCO.2016.7760424"},{"key":"16879_CR21","doi-asserted-by":"publisher","unstructured":"Piczak KJ (2015) Esc: Dataset for environmental sound classification. https:\/\/doi.org\/10.7910\/DVN\/YDEPUT","DOI":"10.7910\/DVN\/YDEPUT"},{"key":"16879_CR22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-07130-5","volume-title":"Robust speaker recognition in noisy environments","author":"KS Rao","year":"2014","unstructured":"Rao KS, Sarkar S (2014) Robust speaker recognition in noisy environments. Springer"},{"key":"16879_CR23","doi-asserted-by":"crossref","unstructured":"Ren J, Hu Y, Tai YW, Wang C, Xu L, Sun W, Yan Q (2016) Look, listen and learn-a multimodal lstm for speaker identification. In: Thirtieth AAAI Conference on Artificial Intelligence","DOI":"10.1609\/aaai.v30i1.10471"},{"key":"16879_CR24","doi-asserted-by":"crossref","unstructured":"Robotham T, Singla A, Rummukainen OS, Raake A, Habets EAP (2022) Audiovisual database with 360$$\\circ $$ video and higher-order ambisonics audio for perception, cognition, behavior, and qoe evaluation research. In: 2022 14th International conference on quality of multimedia experience (QoMEX). pp 1\u20136","DOI":"10.1109\/QoMEX55416.2022.9900893"},{"issue":"2\u20133","key":"16879_CR25","doi-asserted-by":"publisher","first-page":"159","DOI":"10.1016\/j.csl.2005.07.003","volume":"20","author":"P Rose","year":"2006","unstructured":"Rose P (2006) Technical forensic speaker recognition: Evaluation, types and testing of evidence. Computer Speech & Language 20(2\u20133):159\u2013191","journal-title":"Computer Speech & Language"},{"key":"16879_CR26","doi-asserted-by":"publisher","first-page":"1041","DOI":"10.1145\/2647868.2655045","volume-title":"22nd ACM International Conference on Multimedia (ACM-MM\u201914)","author":"J Salamon","year":"2014","unstructured":"Salamon J, Jacoby C, Bello JP (2014) A dataset and taxonomy for urban sound research. 22nd ACM International Conference on Multimedia (ACM-MM\u201914). Orlando, FL, USA, pp 1041\u20131044"},{"key":"16879_CR27","first-page":"3122","volume":"38","author":"N Singh","year":"2012","unstructured":"Singh N, Khan R, Shree R (2012) Applications of speaker recognition. Procedia engineering 38:3122\u20133126","journal-title":"Applications of speaker recognition. Procedia engineering"},{"key":"16879_CR28","doi-asserted-by":"crossref","unstructured":"Yamada T, Wang L, Kai A (2013) Improvement of distant-talking speaker identification using bottleneck features of dnn. In: Interspeech. pp 3661\u20133664","DOI":"10.21437\/Interspeech.2013-686"},{"key":"16879_CR29","doi-asserted-by":"crossref","unstructured":"Zheng J (2022) Construction and application of music audio database based on collaborative filtering algorithm. Discrete Dynamics in Nature and Society, Hindawi","DOI":"10.1155\/2022\/1756357"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16879-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16879-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16879-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,3]],"date-time":"2024-04-03T10:34:04Z","timestamp":1712140444000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16879-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,5]]},"references-count":29,"journal-issue":{"issue":"13","published-online":{"date-parts":[[2024,4]]}},"alternative-id":["16879"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16879-5","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,5]]},"assertion":[{"value":"15 January 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 August 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 September 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 October 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors state that there are no conflicts of interests. For this research work, We have not received any funding from any of these conferencing application providers. Our aim was to include all the frequently used conferencing applications without having any bias towards any corporate entity. We were not able to include some of the conferencing apps (for example Whats-App) as they do not have built-in recording features.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}