{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,2]],"date-time":"2025-11-02T11:53:21Z","timestamp":1762084401220,"version":"build-2065373602"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"21","license":[{"start":{"date-parts":[[2023,3,4]],"date-time":"2023-03-04T00:00:00Z","timestamp":1677888000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,3,4]],"date-time":"2023-03-04T00:00:00Z","timestamp":1677888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 62172267"],"award-info":[{"award-number":["No. 62172267"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai","doi-asserted-by":"publisher","award":["No. 20ZR1420400"],"award-info":[{"award-number":["No. 20ZR1420400"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]},{"name":"State Key Program of National Nature Science Foundation of China","award":["No. 61936001"],"award-info":[{"award-number":["No. 61936001"]}]},{"name":"Fund Project of the Science and Technology on Near-Surface Detection Laboratory","award":["Grant No. 6142414210101"],"award-info":[{"award-number":["Grant No. 6142414210101"]}]},{"name":"Shanghai Pujiang Program","award":["Grant No. 21PJ1404200"],"award-info":[{"award-number":["Grant No. 21PJ1404200"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["No. 2021PE0AC02"],"award-info":[{"award-number":["No. 2021PE0AC02"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2023,9]]},"DOI":"10.1007\/s11042-023-14657-x","type":"journal-article","created":{"date-parts":[[2023,3,4]],"date-time":"2023-03-04T22:02:18Z","timestamp":1677967338000},"page":"33039-33061","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["ASTT: acoustic spatial-temporal transformer for short utterance speaker recognition"],"prefix":"10.1007","volume":"82","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5331-022X","authenticated-orcid":false,"given":"Xing","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruixuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Deng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xingyue","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianjia","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Ding","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,3,4]]},"reference":[{"doi-asserted-by":"publisher","unstructured":"Al-Kaltakchi MT, Abdullah MA, Woo WL, Dlay SS (2021) Closed-set speaker identification system based on mfcc and pncc features combination with different fusion strategies. In: Applied speech processing, pp 147\u2013173. https:\/\/doi.org\/10.1016\/B978-0-12-823898-1.00001-1","key":"14657_CR1","DOI":"10.1016\/B978-0-12-823898-1.00001-1"},{"issue":"14","key":"14657_CR2","doi-asserted-by":"publisher","first-page":"22231","DOI":"10.1007\/s11042-021-10767-6","volume":"80","author":"KA Al-Karawi","year":"2021","unstructured":"Al-Karawi KA, Mohammed DY (2021) Improving short utterance speaker verification by combining mfcc and entrocy in noisy conditions. Multimed Tools Appl 80(14):22231\u201322249. https:\/\/doi.org\/10.1109\/IWCMC48107.2020.9148102","journal-title":"Multimed Tools Appl"},{"doi-asserted-by":"publisher","unstructured":"Bhattacharya G, Alam MJ, Kenny P (2017) Deep speaker embeddings for short-duration speaker verification. In: INTERSPEECH, pp 1517\u20131521. https:\/\/doi.org\/10.21437\/Interspeech.2017-1575","key":"14657_CR3","DOI":"10.21437\/Interspeech.2017-1575"},{"doi-asserted-by":"publisher","unstructured":"Biswas M, Rahaman S, Ahmadian A, Subari K, Singh PK (2022) Automatic spoken language identification using MFCC based time series features. Multimed Tools Appl. https:\/\/doi.org\/10.1007\/s11042-021-11439-1","key":"14657_CR4","DOI":"10.1007\/s11042-021-11439-1"},{"issue":"1","key":"14657_CR5","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1007\/s10772-012-9160-6","volume":"16","author":"D Chakrabarty","year":"2013","unstructured":"Chakrabarty D, Prasanna SRM, Das RK (2013) Development and evaluation of online text-independent speaker verification system for remote person authentication. Int J Speech Technol 16(1):75\u201388. https:\/\/doi.org\/10.1007\/s10772-012-9160-6","journal-title":"Int J Speech Technol"},{"doi-asserted-by":"crossref","unstructured":"Chakroun R, Frikha M (2020) Robust text-independent speaker recognition with short utterances using gaussian mixture models. In: 2020 International wireless communications and mobile computing (IWCMC), pp 2204\u20132209","key":"14657_CR6","DOI":"10.1109\/IWCMC48107.2020.9148102"},{"doi-asserted-by":"publisher","unstructured":"Chung JS, Nagrani A, Zisserman A (2018) Voxceleb2: deep speaker recognition. In: INTERSPEECH, pp 1086\u20131090. https:\/\/doi.org\/10.21437\/Interspeech.2018-1929","key":"14657_CR7","DOI":"10.21437\/Interspeech.2018-1929"},{"doi-asserted-by":"publisher","unstructured":"Cong Y, Liao W, Ackermann H, Rosenhahn B, Yang MY (2021) Spatial-temporal transformer for dynamic scene graph generation. In: 2021 IEEE\/CVF international conference on computer vision (ICCV), pp 16352\u201316362. https:\/\/doi.org\/10.1109\/ICCV48922.2021.01606","key":"14657_CR8","DOI":"10.1109\/ICCV48922.2021.01606"},{"issue":"3","key":"14657_CR9","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1007\/s11265-016-1148-z","volume":"88","author":"RK Das","year":"2016","unstructured":"Das RK, Jelil S, Prasanna SRM (2016) Development of multi-level speech based person authentication system. J Signal Process Syst 88(3):259\u2013271. https:\/\/doi.org\/10.1007\/s11265-016-1148-z","journal-title":"J Signal Process Syst"},{"issue":"4","key":"14657_CR10","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N Dehak","year":"2010","unstructured":"Dehak N, Kenny PJ, Dehak R, Dumouchel P, Ouellet P (2010) Front-end factor analysis for speaker verification. IEEE Trans Audio Speech Lang Process 19(4):788\u2013798. https:\/\/doi.org\/10.1109\/TASL.2010.2064307","journal-title":"IEEE Trans Audio Speech Lang Process"},{"unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, Uszkoreit J, Houlsby N (2021) An image is worth 16x16 words: transformers for image recognition at scale. ICLR","key":"14657_CR11"},{"key":"14657_CR12","doi-asserted-by":"publisher","first-page":"108666","DOI":"10.1016\/j.patcog.2022.108666","volume":"128","author":"G Feng","year":"2022","unstructured":"Feng G, Meng J, Zhang L, Lu H (2022) Encoder deep interleaved network with multi-scale aggregation for rgb-d salient object detection. Pattern Recogn 128:108666. https:\/\/doi.org\/10.1016\/j.patcog.2022.108666","journal-title":"Pattern Recogn"},{"doi-asserted-by":"publisher","unstructured":"Gao Z, Song Y, McLoughlin I, Guo W, Dai L (2018) An improved deep embedding learning method for short duration speaker verification. In: INTERSPEECH, pp 3578\u20133582. https:\/\/doi.org\/10.21437\/Interspeech.2018-1515","key":"14657_CR13","DOI":"10.21437\/Interspeech.2018-1515"},{"doi-asserted-by":"publisher","unstructured":"Gemmeke JF, Ellis DP, Freedman D, Jansen A, Lawrence W, Moore RC, Plakal M, Ritter M (2017) Audio set: an ontology and human-labeled dataset for audio events. In: 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 776\u2013780. https:\/\/doi.org\/10.1109\/ICASSP.2017.7952261","key":"14657_CR14","DOI":"10.1109\/ICASSP.2017.7952261"},{"issue":"13","key":"14657_CR15","doi-asserted-by":"publisher","first-page":"18617","DOI":"10.1007\/s11042-022-12632-6","volume":"81","author":"T Grzywalski","year":"2022","unstructured":"Grzywalski T, Drgas S (2022) Speech enhancement using u-nets with wide-context units. Multimed Tools Appl 81(13):18617\u201318639. https:\/\/doi.org\/10.1007\/s11042-022-12632-6","journal-title":"Multimed Tools Appl"},{"doi-asserted-by":"publisher","unstructured":"Guo M, Yang J, Gao S (2021) Speaker recognition method for short utterance. In: Journal of physics: conference series, vol 1827, p 012158. https:\/\/doi.org\/10.1088\/1742-6596\/1827\/1\/012158","key":"14657_CR16","DOI":"10.1088\/1742-6596\/1827\/1\/012158"},{"doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp 770\u2013778","key":"14657_CR17","DOI":"10.1109\/CVPR.2016.90"},{"issue":"3","key":"14657_CR18","doi-asserted-by":"publisher","first-page":"3352","DOI":"10.1007\/s10489-021-02613-x","volume":"52","author":"\u00c1C Hidalgo","year":"2021","unstructured":"Hidalgo \u00c1C, Ger PM, Valent\u00edn LDLF (2021) Using meta-learning to predict student performance in virtual learning environments. Appl Intell 52(3):3352\u20133365. https:\/\/doi.org\/10.1007\/s10489-021-02613-x","journal-title":"Appl Intell"},{"issue":"8","key":"14657_CR19","doi-asserted-by":"publisher","first-page":"2011","DOI":"10.1109\/TPAMI.2019.2913372","volume":"42","author":"J Hu","year":"2020","unstructured":"Hu J, Shen L, Albanie S, Sun G, Wu E (2020) Squeeze-and-excitation networks. IEEE Trans Pattern Anal Mach Intell 42 (8):2011\u20132023. https:\/\/doi.org\/10.1109\/TPAMI.2019.2913372","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"doi-asserted-by":"publisher","unstructured":"Illa A, Ghosh PK (2019) Representation learning using convolution neural network for acoustic-to-articulatory inversion. In: ICASSP 2019 - 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 5931\u20135935. https:\/\/doi.org\/10.1109\/ICASSP.2019.8682506","key":"14657_CR20","DOI":"10.1109\/ICASSP.2019.8682506"},{"key":"14657_CR21","doi-asserted-by":"publisher","first-page":"175448","DOI":"10.1109\/ACCESS.2020.3025941","volume":"8","author":"Y Jung","year":"2020","unstructured":"Jung Y, Choi Y, Lim H, Kim H (2020) A unified deep learning framework for short-duration speaker verification in adverse environments. IEEE Access 8:175448\u2013175466. https:\/\/doi.org\/10.1109\/ACCESS.2020.3025941","journal-title":"IEEE Access"},{"doi-asserted-by":"publisher","unstructured":"Jung J-W, Heo H-S, Shim H-J, Yu H-J (2019) Short utterance compensation in speaker verification via cosine-based teacher-student learning of speaker embeddings. In: 2019 IEEE automatic speech recognition and understanding workshop (ASRU), pp 335\u2013341. https:\/\/doi.org\/10.1109\/ASRU46091.2019.9004029","key":"14657_CR22","DOI":"10.1109\/ASRU46091.2019.9004029"},{"doi-asserted-by":"publisher","unstructured":"Kanagasundaram A, Sridharan S, Ganapathy S, Singh P, Fookes C (2019) A study of x-vector based speaker recognition on short utterances. In: Proceedings of the 20th annual conference of the international speech communication association, INTERSPEECH 2019. vol 2019-September, pp 2943\u20132947. https:\/\/doi.org\/10.21437\/Interspeech.2019-1891","key":"14657_CR23","DOI":"10.21437\/Interspeech.2019-1891"},{"doi-asserted-by":"crossref","unstructured":"Kanagasundaram A, Vogt R, Dean D, Sridharan S (2012) Plda based speaker recognition on short utterances. In: Proceedings of the speaker and language recognition workshop: odyssey 2012, pp 28\u201333","key":"14657_CR24","DOI":"10.21437\/Interspeech.2011-58"},{"doi-asserted-by":"crossref","unstructured":"Kanagasundaram A, Vogt R, Dean D, Sridharan S, Mason M (2011) I-vector based speaker recognition on short utterances. In: Proceedings of the 12th annual conference of the international speech communication association, pp 2341\u20132344","key":"14657_CR25","DOI":"10.21437\/Interspeech.2011-58"},{"doi-asserted-by":"publisher","unstructured":"Kye SM, Jung Y, Lee HB, Hwang SJ, Kim H (2020) Meta-learning for short utterance speaker recognition with imbalance length pairs. In: INTERSPEECH. https:\/\/doi.org\/10.21437\/Interspeech.2020-1283","key":"14657_CR26","DOI":"10.21437\/Interspeech.2020-1283"},{"unstructured":"Lee KA, Larcher A, Thai H, Ma B, Li H (2011) Joint application of speech and speaker recognition for automation and security in smart home. In: Annual conference of the international speech communication association","key":"14657_CR27"},{"doi-asserted-by":"publisher","unstructured":"Li L, Wang D, Zhang X, Zheng TF, Jin P (2016) System combination for short utterance speaker recognition. In: 2016 Asia-pacific signal and information processing association annual summit and conference (APSIPA), pp 1\u20135. https:\/\/doi.org\/10.1109\/APSIPA.2016.7820903","key":"14657_CR28","DOI":"10.1109\/APSIPA.2016.7820903"},{"issue":"7","key":"14657_CR29","doi-asserted-by":"publisher","first-page":"3244","DOI":"10.1109\/TII.2018.2799928","volume":"14","author":"Z Liu","year":"2018","unstructured":"Liu Z, Wu Z, Li T, Li J, Shen C (2018) Gmm and cnn hybrid method for short utterance speaker recognition. IEEE Trans Industr Inf 14(7):3244\u20133252. https:\/\/doi.org\/10.1109\/TII.2018.2799928","journal-title":"IEEE Trans Industr Inf"},{"issue":"6","key":"14657_CR30","doi-asserted-by":"publisher","first-page":"6441","DOI":"10.1007\/s11042-018-6256-2","volume":"78","author":"A Mansour","year":"2018","unstructured":"Mansour A, Chenchah F, Lachiri Z (2018) Emotional speaker recognition in real life conditions using multiple descriptors and i-vector speaker modeling technique. Multimed Tools Appl 78(6):6441\u20136458. https:\/\/doi.org\/10.1007\/s11042-018-6256-2","journal-title":"Multimed Tools Appl"},{"doi-asserted-by":"publisher","unstructured":"Nagrani A, Chung JS, Zisserman A (2017) Voxceleb: a large-scale speaker identification dataset. In: INTERSPEECH. https:\/\/doi.org\/10.21437\/Interspeech.2017-950","key":"14657_CR31","DOI":"10.21437\/Interspeech.2017-950"},{"doi-asserted-by":"publisher","unstructured":"Nj MSM, Umesh S, Katta SV (2021) S-vectors and tesa: speaker embeddings and a speaker authenticator based on transformer encoder. IEEE\/ACM Trans Audio Speech Lang Process. https:\/\/doi.org\/10.1109\/TASLP.2021.3134566","key":"14657_CR32","DOI":"10.1109\/TASLP.2021.3134566"},{"issue":"39-40","key":"14657_CR33","doi-asserted-by":"publisher","first-page":"28859","DOI":"10.1007\/s11042-020-09353-z","volume":"79","author":"BK P","year":"2020","unstructured":"P BK, M RK (2020) ELM Speaker identification for limited dataset using multitaper based MFCC and PNCC features with fusion score. Multimed Tools Appl 79(39-40):28859\u201328883. https:\/\/doi.org\/10.1007\/s11042-020-09353-z","journal-title":"Multimed Tools Appl"},{"doi-asserted-by":"publisher","unstructured":"Plizzari C, Cannici M, Matteucci M (2021) Spatial temporal transformer network for skeleton-based action recognition. In: International conference on pattern recognition, pp 694\u2013701. https:\/\/doi.org\/10.1007\/978-3-030-68796-0_50","key":"14657_CR34","DOI":"10.1007\/978-3-030-68796-0_50"},{"doi-asserted-by":"publisher","unstructured":"Rakhmanenko I, Kostyuchenko E, Choynzonov E, Balatskaya L, Shelupanov A (2020) Score normalization of x-vector speaker verification system for short-duration speaker verification challenge. In: Speech and computer, pp 457\u2013466. https:\/\/doi.org\/10.1007\/978-3-030-60276-5_44","key":"14657_CR35","DOI":"10.1007\/978-3-030-60276-5_44"},{"issue":"1-3","key":"14657_CR36","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1006\/dspr.1999.0361","volume":"10","author":"DA Reynolds","year":"2000","unstructured":"Reynolds DA, Quatieri TF, Dunn RB (2000) Speaker verification using adapted gaussian mixture models. Digital Signal Process 10(1-3):19\u201341. https:\/\/doi.org\/10.1006\/dspr.1999.0361","journal-title":"Digital Signal Process"},{"doi-asserted-by":"publisher","unstructured":"Sahidullah M, Kumar Sarkar A, Vestman V, Liu X, Serizel R, Kinnunen T, Tan Z-H, Vincent E (2021) Uiai system for short-duration speaker verification challenge 2020. In: 2021 IEEE spoken language technology workshop (SLT), pp 323\u2013329. https:\/\/doi.org\/10.1109\/SLT48900.2021.9383596","key":"14657_CR37","DOI":"10.1109\/SLT48900.2021.9383596"},{"doi-asserted-by":"publisher","unstructured":"Seo S, Rim DJ, Lim M, Lee D, Park H, Oh J, Kim C, Kim J-H (2019) Shortcut connections based deep speaker embeddings for end-to-end speaker verification system. In: INTERSPEECH, pp 2928\u20132932. https:\/\/doi.org\/10.21437\/Interspeech.2019-2195","key":"14657_CR38","DOI":"10.21437\/Interspeech.2019-2195"},{"doi-asserted-by":"publisher","unstructured":"Snyder D, Garcia-Romero D, Sell G, Povey D, Khudanpur S (2018) X-vectors: robust dnn embeddings for speaker recognition. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 5329\u20135333. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461375","key":"14657_CR39","DOI":"10.1109\/ICASSP.2018.8461375"},{"doi-asserted-by":"publisher","unstructured":"Srinivasu PN, JayaLakshmi G, Jhaveri RH, Praveen SP (2022) Ambient assistive living for monitoring the physical activity of diabetic adults through body area networks. Mob Inf Syst, vol 2022. https:\/\/doi.org\/10.1155\/2022\/3169927","key":"14657_CR40","DOI":"10.1155\/2022\/3169927"},{"unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Advances in neural information processing systems, pp 5998\u20136008","key":"14657_CR41"},{"issue":"3","key":"14657_CR42","doi-asserted-by":"publisher","first-page":"328","DOI":"10.1109\/29.21701","volume":"37","author":"A Waibel","year":"1989","unstructured":"Waibel A, Hanazawa T, Hinton G, Shikano K, Lang KJ (1989) Phoneme recognition using time-delay neural networks. IEEE Trans Acoustics Speech Signal Process 37(3):328\u2013339. https:\/\/doi.org\/10.1109\/29.21701","journal-title":"IEEE Trans Acoustics Speech Signal Process"},{"unstructured":"Ward R, Wu X, Bottou L (2019) Adagrad stepsizes: sharp convergence over nonconvex landscapes. In: International conference on machine learning, pp 6677\u20136686","key":"14657_CR43"},{"key":"14657_CR44","doi-asserted-by":"publisher","first-page":"385","DOI":"10.1016\/j.ins.2022.02.006","volume":"593","author":"X Wu","year":"2022","unstructured":"Wu X, Chen C, Li P, Zhong M, Wang J, Qian Q, Ding P, Yao J, Guo Y (2022) Ftap: feature transferring autonomous machine learning pipeline. Inf Sci 593:385\u2013397. https:\/\/doi.org\/10.1016\/j.ins.2022.02.006","journal-title":"Inf Sci"},{"issue":"4","key":"14657_CR45","doi-asserted-by":"publisher","first-page":"1548","DOI":"10.1007\/s10489-018-1342-8","volume":"49","author":"X Wu","year":"2018","unstructured":"Wu X, Dai S, Guo Y, Fujita H (2018) A machine learning attack against variable-length chinese character CAPTCHAs. Appl Intell 49(4):1548\u20131565. https:\/\/doi.org\/10.1007\/s10489-018-1342-8","journal-title":"Appl Intell"},{"issue":"1","key":"14657_CR46","doi-asserted-by":"publisher","first-page":"44","DOI":"10.1007\/s10489-018-1206-2","volume":"49","author":"X Wu","year":"2018","unstructured":"Wu X, Du Z, Guo Y, Fujita H (2018) Hierarchical attention based long short-term memory for chinese lyric generation. Appl Intell 49(1):44\u201352. https:\/\/doi.org\/10.1007\/s10489-018-1206-2","journal-title":"Appl Intell"},{"doi-asserted-by":"publisher","unstructured":"Wu X, Ji S, Wang J, Guo Y (2022) Speech synthesis with face embeddings. Appl Intell:1\u201314. https:\/\/doi.org\/10.1007\/s10489-022-03227-7","key":"14657_CR47","DOI":"10.1007\/s10489-022-03227-7"},{"doi-asserted-by":"publisher","unstructured":"Wu X, Jin Y, Wang J, Qian Q, Guo Y (2022) Mkd: mixup-based knowledge distillation for mandarin end-to-end speech recognition. Algorithms, vol 15(5). https:\/\/doi.org\/10.3390\/a15050160","key":"14657_CR48","DOI":"10.3390\/a15050160"},{"key":"14657_CR49","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1016\/j.ins.2019.08.059","volume":"508","author":"X Wu","year":"2020","unstructured":"Wu X, Zhong M, Guo Y, Fujita H (2020) The assessment of small bowel motility with attentive deformable neural network. Inf Sci 508:22\u201332. https:\/\/doi.org\/10.1016\/j.ins.2019.08.059","journal-title":"Inf Sci"},{"issue":"1","key":"14657_CR50","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s13636-022-00240-z","volume":"2022","author":"Y Xu","year":"2022","unstructured":"Xu Y, Wang W, Cui H, Xu M, Li M (2022) Paralinguistic singing attribute recognition using supervised machine learning for describing the classical tenor solo singing voice in vocal pedagogy. EURASIP J Audio Speech Music Process 2022 (1):1\u201316. https:\/\/doi.org\/10.1109\/ICASSP.2019.8682506","journal-title":"EURASIP J Audio Speech Music Process"},{"issue":"175","key":"14657_CR51","first-page":"12","volume":"3","author":"S Young","year":"2002","unstructured":"Young S, Evermann G, Gales M, Hain T, Kershaw D, Liu X, Moore G, Odell J, Ollason D, Povey D et al (2002) The htk book. Cambridge Univ Eng Department 3(175):12","journal-title":"Cambridge Univ Eng Department"},{"doi-asserted-by":"crossref","unstructured":"Zeng Y, Fu J, Chao H (2020) Learning joint spatial-temporal transformations for video inpainting. In: European conference on computer vision, pp 528\u2013543","key":"14657_CR52","DOI":"10.1007\/978-3-030-58517-4_31"},{"issue":"13","key":"14657_CR53","doi-asserted-by":"publisher","first-page":"20283","DOI":"10.1007\/s11042-021-10718-1","volume":"80","author":"Q Zheng","year":"2021","unstructured":"Zheng Q, Chen Y (2021) Feature pyramid of bi-directional stepped concatenation for small object detection. Multimed Tools Appl 80(13):20283\u201320305. https:\/\/doi.org\/10.1007\/s11042-021-10718-1","journal-title":"Multimed Tools Appl"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-14657-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-14657-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-14657-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,31]],"date-time":"2023-08-31T09:37:33Z","timestamp":1693474653000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-14657-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,4]]},"references-count":53,"journal-issue":{"issue":"21","published-print":{"date-parts":[[2023,9]]}},"alternative-id":["14657"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-14657-x","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"type":"print","value":"1380-7501"},{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2023,3,4]]},"assertion":[{"value":"14 May 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 August 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 February 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 March 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}