{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:43Z","timestamp":1779228403998,"version":"3.51.4"},"reference-count":23,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T00:00:00Z","timestamp":1772582400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100014440","name":"Gobierno de Espa\u00f1a Ministerio de Ciencia e Innovaci\u00f3n","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100014440","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101967","type":"journal-article","created":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T16:23:14Z","timestamp":1772641394000},"page":"101967","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Exploring efficient attention strategies in conformer-based sound event detection"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8519-0549","authenticated-orcid":false,"given":"Sara","family":"Barahona","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Juan Ignacio","family":"Alvarez-Trejos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alicia","family":"Lozano-Diez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daniel","family":"Ramos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Doroteo T.","family":"Toledano","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101967_b1","series-title":"IberSPEECH 2024","first-page":"111","article-title":"Towards efficient conformer-based sound event detection","author":"Barahona","year":"2024"},{"key":"10.1016\/j.csl.2026.101967_b2","unstructured":"Barahona, Sara, Benito-Gorron, Diego de, Segovia, Sergio, Ramos, Daniel, Toledano, Doroteo T., 2023. Multi-Resolution Conformer for Sound Event Detection: Analysis and Optimization. In: Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023). pp. 11\u201315."},{"key":"10.1016\/j.csl.2026.101967_b3","series-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing","first-page":"3896","article-title":"Enhancing conformer-based sound event detection using frequency dynamic convolutions and BEATs audio embeddings","author":"Barahona","year":"2024"},{"key":"10.1016\/j.csl.2026.101967_b4","doi-asserted-by":"crossref","unstructured":"Bilen, Cagdas, Ferroni, Giacomo, Tuveri, Francesco, Azcarreta, Juan, Krstulovic, Sacha, 2020. A Framework for the Robust Evaluation of Sound Event Detection. In: 2020 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 61\u201365.","DOI":"10.1109\/ICASSP40776.2020.9052995"},{"key":"10.1016\/j.csl.2026.101967_b5","doi-asserted-by":"crossref","unstructured":"Burchi, Maxime, Vielzeuf, Valentin, 2021. Efficient Conformer: Progressive Downsampling and Grouped Attention for Automatic Speech Recognition. In: 2021 IEEE Automatic Speech Recognition and UnderstandIng Workshop. ASRU, pp. 8\u201315.","DOI":"10.1109\/ASRU51503.2021.9687874"},{"key":"10.1016\/j.csl.2026.101967_b6","doi-asserted-by":"crossref","unstructured":"Dai, Zihang, Yang, Zhilin, Yang, Yiming, Carbonell, Jaime G., Le, Quoc Viet, Salakhutdinov, Ruslan, 2019. Transformer-XL: Attentive Language Models beyond a Fixed-Length Context. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics. pp. 2978\u20132988.","DOI":"10.18653\/v1\/P19-1285"},{"key":"10.1016\/j.csl.2026.101967_b7","unstructured":"Dekkers, Gert, Lauwereins, Steven, Thoen, Bart, Adhana, Mulu Weldegebreal, Brouckxon, Henk, Waterschoot, Toon van, Vanrumste, Bart, Verhelst, Marian, Karsmakers, Peter, 2017. The SINS Database for Detection of Daily Activities in a Home Environment Using an Acoustic Sensor Network. In: Proceedings of the Detection and Classification of Acoustic Scenes and Events 2017 Workshop (DCASE2017). pp. 32\u201336."},{"key":"10.1016\/j.csl.2026.101967_b8","doi-asserted-by":"crossref","unstructured":"Ebbers, Janek, Haeb-Umbach, Reinhold, Serizel, Romain, 2022. Threshold Independent Evaluation of Sound Event Detection Scores. In: 2022 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 1021\u20131025.","DOI":"10.1109\/ICASSP43922.2022.9747556"},{"key":"10.1016\/j.csl.2026.101967_b9","unstructured":"Fonseca, Eduardo, Pons, Jordi, Favory, Xavier, Font, Frederic, Bogdanov, Dmitry, Ferraro, Andres, Oramas, Sergio, Porter, Alastair, Serra, Xavier, 2017. Freesound Datasets: A Platform for the Creation of Open Audio Datasets. In: Proceedings of the 18th International Society for Music Information Retrieval Conference (ISMIR 2017). Suzhou, China, pp. 486\u2013493."},{"key":"10.1016\/j.csl.2026.101967_b10","doi-asserted-by":"crossref","unstructured":"Gemmeke, Jort F., Ellis, Daniel P. W., Freedman, Dylan, Jansen, Aren, Lawrence, Wade, Channing Moore, R., Plakal, Manoj, Ritter, Marvin, 2017. Audio Set: An ontology and human-labeled dataset for audio events. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 776\u2013780.","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"10.1016\/j.csl.2026.101967_b11","series-title":"Interspeech 2020","first-page":"5036","article-title":"Pang conformer: Convolution-augmented transformer for speech recognition","author":"Gulati","year":"2020"},{"key":"10.1016\/j.csl.2026.101967_b12","unstructured":"Lu, JiaKai, 2018. Mean Teacher Convolution System for DCASE 2018 Task 4, DCASE2018 Challenge. Technical Report."},{"key":"10.1016\/j.csl.2026.101967_b13","doi-asserted-by":"crossref","unstructured":"Mesaros, Annamaria, Heittola, Toni, Virtanen, Tuomas, 2016. TUT Database for Acoustic Scene Classification and Sound Event Detection. In: 2016 24th European Signal Processing Conference. EUSIPCO, pp. 1128\u20131132.","DOI":"10.1109\/EUSIPCO.2016.7760424"},{"key":"10.1016\/j.csl.2026.101967_b14","unstructured":"Miyazaki, Koichi, Komatsu, Tatsuya, Hayashi, Tomoki, Watanabe, Shinji, Toda, Tomoki, Takeda, Kazuya, 2020. Conformer-Based Sound Event Detection with Semi-Supervised Learning and Data Augmentation. In: Proceedings of the Detection and Classification of Acoustic Scenes and Events 2020 Workshop (DCASE2020). pp. 100\u2013104."},{"key":"10.1016\/j.csl.2026.101967_b15","doi-asserted-by":"crossref","unstructured":"Nam, Hyeonuk, Kim, Seong-Hu, Ko, Byeong-Yun, Park, Yong-Hwa, 2022b. Frequency Dynamic Convolution: Frequency-Adaptive Pattern Recognition for Sound Event Detection. In: Proc. Interspeech 2022. pp. 2763\u20132767.","DOI":"10.21437\/Interspeech.2022-10127"},{"key":"10.1016\/j.csl.2026.101967_b16","unstructured":"Nam, Hyeonuk, Kim, Seong-Hu, Min, Deokki, Park, Yong-Hwa, 2023. Frequency & Channel Attention for Computationally Efficient Sound Event Detection. In: Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023). pp. 136\u2013140."},{"key":"10.1016\/j.csl.2026.101967_b17","doi-asserted-by":"crossref","unstructured":"Nam, Hyeonuk, Kim, Seong-Hu, Park, Yong-Hwa, 2022a. Filteraugment: An Acoustic Environmental Data Augmentation Method. In: Proceedings of the 2022 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 4308\u20134312.","DOI":"10.1109\/ICASSP43922.2022.9747680"},{"key":"10.1016\/j.csl.2026.101967_b18","doi-asserted-by":"crossref","unstructured":"Salamon, Justin, MacConnell, Duncan, Cartwright, Mark, Li, Peter, Bello, Juan Pablo, 2017. Scaper: A library for soundscape synthesis and augmentation. In: 2017 IEEE Workshop on Applications of Signal Processing To Audio and Acoustics. WASPAA, pp. 344\u2013348.","DOI":"10.1109\/WASPAA.2017.8170052"},{"key":"10.1016\/j.csl.2026.101967_b19","unstructured":"Tarvainen, Antti, Valpola, Harri, 2017. Mean Teachers Are Better Role Models: Weight-Averaged Consistency Targets Improve Semi-Supervised Deep Learning Results. In: Proceedings of the 31st International Conference on Neural Information Processing Systems. pp. 1195\u20131204."},{"key":"10.1016\/j.csl.2026.101967_b20","doi-asserted-by":"crossref","unstructured":"Thienpondt, Jenthe, Desplanques, Brecht, Demuynck, Kris, 2021. Integrating Frequency Translational Invariance in TDNNs and Frequency Positional Information in 2D ResNets to Enhance Speaker Verification. In: Proc. Interspeech 2021. pp. 2302\u20132306.","DOI":"10.21437\/Interspeech.2021-1570"},{"key":"10.1016\/j.csl.2026.101967_b21","doi-asserted-by":"crossref","unstructured":"Turpault, Nicolas, Serizel, Romain, Shah, Ankit Parag, Salamon, Justin, 2019. Sound event detection in domestic environments with weakly labeled data and soundscape synthesis. In: Proceedings of the Detection and Classification of Acoustic Scenes and Events 2019 Workshop (DCASE2019). pp. 253\u2013257.","DOI":"10.33682\/006b-jx26"},{"key":"10.1016\/j.csl.2026.101967_b22","series-title":"Advances in Neural Information Processing Systems","first-page":"5998","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.csl.2026.101967_b23","unstructured":"Zhang, Hongyi, Ciss\u00e9, Moustapha, Dauphin, Yann N., Lopez-Paz, David, 2018. mixup: Beyond Empirical Risk Minimization. In: 6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 - May 3 2018, Conference Track Proceedings."}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000306?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000306?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:14:32Z","timestamp":1779225272000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000306"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":23,"alternative-id":["S0885230826000306"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101967","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Exploring efficient attention strategies in conformer-based sound event detection","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101967","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"101967"}}