{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T07:07:05Z","timestamp":1778828825835,"version":"3.51.4"},"reference-count":27,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100000055","name":"National Institute on Deafness and Other Communication Disorders","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000055","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Biomedical Signal Processing and Control"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.bspc.2026.110428","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T23:41:11Z","timestamp":1777592471000},"page":"110428","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["OtoSTVT: a diagnostic-driven frame selection method for otoscopy videos using spatio-temporal transformers"],"prefix":"10.1016","volume":"122","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6899-6065","authenticated-orcid":false,"given":"Seda","family":"Camalan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Debashis","family":"Gupta","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3779-4813","authenticated-orcid":false,"given":"Sean","family":"Catley","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aaron C.","family":"Moberly","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2421-8229","authenticated-orcid":false,"given":"Metin N.","family":"Gurcan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"issue":"5","key":"10.1016\/j.bspc.2026.110428_b0005","doi-asserted-by":"crossref","DOI":"10.1371\/journal.pone.0232776","article-title":"OtoMatch: content-based eardrum image retrieval using deep learning","volume":"15","author":"Camalan","year":"2020","journal-title":"Plos one"},{"issue":"1","key":"10.1016\/j.bspc.2026.110428_b0010","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1038\/s41746-019-0094-0","article-title":"Automated classification platform for the identification of otitis media using optical coherence tomography","volume":"2","author":"Monroy","year":"2019","journal-title":"NPJ Digital Med."},{"key":"10.1016\/j.bspc.2026.110428_b0015","unstructured":"S. Camalan et al., \u201cDigital Otoscopy With Computer\u2010Aided Composite Image Generation: Impact on the Correct Diagnosis, Confidence, and Time,\u201d Otolaryngology\u2013Head and Neck Surgery."},{"issue":"8","key":"10.1016\/j.bspc.2026.110428_b0020","doi-asserted-by":"crossref","first-page":"1891","DOI":"10.1002\/lary.27550","article-title":"Diagnostic accuracy and confidence for otoscopy: are medical students receiving sufficient training?","volume":"129","author":"Niermeyer","year":"2019","journal-title":"Laryngoscope"},{"issue":"6","key":"10.1016\/j.bspc.2026.110428_b0025","doi-asserted-by":"crossref","first-page":"993","DOI":"10.1542\/peds.109.6.993","article-title":"Pediatric residents\u2019 clinical diagnostic accuracy of otitis media","volume":"109","author":"Steinbach","year":"2002","journal-title":"Pediatrics"},{"issue":"14","key":"10.1016\/j.bspc.2026.110428_b0030","doi-asserted-by":"crossref","first-page":"12197","DOI":"10.1007\/s00521-022-07107-6","article-title":"OtoXNet\u2014Automated identification of eardrum diseases from otoscope videos: a deep learning study for video-representing images","volume":"34","author":"Binol","year":"2022","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.bspc.2026.110428_b0035","series-title":"International Workshop on Applications of Medical AI","first-page":"155","article-title":"Accessible otitis media screening with a deep learning-powered mobile otoscope","author":"Kovvali","year":"2023"},{"key":"10.1016\/j.bspc.2026.110428_b0040","doi-asserted-by":"crossref","unstructured":"H. Lu, S. Camalan, C. Elmaraghy, A. C. Moberly, and M. N. Gurcan, \u201cA video classification method for diagnosing ear diseases using otoscope imaging,\u201d in Medical Imaging 2025: Computer-Aided Diagnosis, 2025, vol. 13407: SPIE, pp. 758-766.","DOI":"10.1117\/12.3046822"},{"key":"10.1016\/j.bspc.2026.110428_b0045","doi-asserted-by":"crossref","unstructured":"H. Binol, M. K. K. Niazi, C. Elmaraghy, A. C. Moberly, and M. N. Gurcan, \u201cAutomated video summarization and label assignment for otoscopy videos using deep learning and natural language processing,\u201d in Medical Imaging 2021: Imaging Informatics for Healthcare, Research, and Applications, 2021, vol. 11601: SPIE, pp. 153\u2013158.","DOI":"10.1117\/12.2582009"},{"issue":"17","key":"10.1016\/j.bspc.2026.110428_b0050","doi-asserted-by":"crossref","first-page":"5894","DOI":"10.3390\/app10175894","article-title":"SelectStitch: automated frame segmentation and stitching to create composite images from otoscope video clips","volume":"10","author":"Binol","year":"2020","journal-title":"Appl. Sci."},{"key":"10.1016\/j.bspc.2026.110428_b0055","doi-asserted-by":"crossref","unstructured":"S. Camalan, M. K. K. Niazi, C. Elmaraghy, A. C. Moberly, and M. N. Gurcan, \u201cTympanic membrane segmentation of video frames to create composite images using SAM,\u201d in Medical Imaging 2024: Computer-Aided Diagnosis, 2024, vol. 12927: SPIE, pp. 766\u2013773.","DOI":"10.1117\/12.3006926"},{"key":"10.1016\/j.bspc.2026.110428_b0060","doi-asserted-by":"crossref","DOI":"10.3389\/fdgth.2021.810427","article-title":"Pediatric otoscopy video screening with shift contrastive anomaly detection","volume":"3","author":"Wang","year":"2022","journal-title":"Front. Digital Health"},{"key":"10.1016\/j.bspc.2026.110428_b0065","doi-asserted-by":"crossref","unstructured":"H. Lu et al., \u201cLUCID: Intelligent Informative Frame Selection in Otoscopy for Enhanced Diagnostic Utilitys,\u201d Research Square, pp. rs. 3. rs-7502743, 2025.","DOI":"10.21203\/rs.3.rs-7502743\/v1"},{"key":"10.1016\/j.bspc.2026.110428_b0070","doi-asserted-by":"crossref","first-page":"5612","DOI":"10.1109\/TIP.2020.2984879","article-title":"No-reference video quality assessment using natural spatiotemporal scene statistics","volume":"29","author":"Dendi","year":"2020","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.bspc.2026.110428_b0075","doi-asserted-by":"crossref","first-page":"8059","DOI":"10.1109\/TIP.2021.3112055","article-title":"ChipQA: no-reference video quality prediction via space-time chips","volume":"30","author":"Ebenezer","year":"2021","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.bspc.2026.110428_b0080","doi-asserted-by":"crossref","DOI":"10.1016\/j.compmedimag.2022.102121","article-title":"A neural network based framework for effective laparoscopic video quality assessment","volume":"101","author":"Khan","year":"2022","journal-title":"Computer. Med. Imag. Graphics"},{"key":"10.1016\/j.bspc.2026.110428_b0085","series-title":"in Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.bspc.2026.110428_b0090","series-title":"in Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"3431","article-title":"Fully convolutional networks for semantic segmentation","author":"Long","year":"2015"},{"key":"10.1016\/j.bspc.2026.110428_b0095","unstructured":"R. Rodrigues et al., \u201cObjective quality assessment of medical images and videos: Review and challenges,\u201d Multimedia Tools and Applications, pp. 1\u201334, 2024."},{"issue":"4","key":"10.1016\/j.bspc.2026.110428_b0100","doi-asserted-by":"crossref","first-page":"635","DOI":"10.1177\/01945998221083502","article-title":"Advances in artificial intelligence to diagnose otitis media: state of the art review","volume":"168","author":"Ngombu","year":"2023","journal-title":"Otolaryngol Head Neck Surg"},{"key":"10.1016\/j.bspc.2026.110428_b0105","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part VII 14","first-page":"766","article-title":"Video summarization with long short-term memory","author":"Zhang","year":"2016"},{"key":"10.1016\/j.bspc.2026.110428_b0110","doi-asserted-by":"crossref","unstructured":"K. Zhou, Y. Qiao, and T. Xiang, \u201cDeep reinforcement learning for unsupervised video summarization with diversity-representativeness reward,\u201d in Proceedings of the AAAI conference on artificial intelligence, 2018, vol. 32, no. 1.","DOI":"10.1609\/aaai.v32i1.12255"},{"key":"10.1016\/j.bspc.2026.110428_b0115","series-title":"in Proceedings of the IEEE conference on Computer Vision and Pattern Recognition","first-page":"202","article-title":"Unsupervised video summarization with adversarial lstm networks","author":"Mahasseni","year":"2017"},{"key":"10.1016\/j.bspc.2026.110428_b0120","doi-asserted-by":"crossref","first-page":"948","DOI":"10.1109\/TIP.2020.3039886","article-title":"Dsnet: a flexible detect-to-summarize network for video summarization","volume":"30","author":"Zhu","year":"2020","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.bspc.2026.110428_b0125","doi-asserted-by":"crossref","first-page":"3013","DOI":"10.1109\/TIP.2023.3275069","article-title":"Video summarization with spatiotemporal vision transformer","volume":"32","author":"Hsu","year":"2023","journal-title":"IEEE Trans. Image Process."},{"issue":"15","key":"10.1016\/j.bspc.2026.110428_b0130","doi-asserted-by":"crossref","first-page":"17864","DOI":"10.1007\/s10489-022-03451-1","article-title":"Video summarization with u-shaped transformer","volume":"52","author":"Chen","year":"2022","journal-title":"Appl. Intell."},{"key":"10.1016\/j.bspc.2026.110428_b0135","doi-asserted-by":"crossref","unstructured":"L. Lan, L. Jiang, T. Yu, X. Liu, and Z. He, \u201cFullTransNet: Full Transformer with Local-Global Attention for Video Summarization,\u201d arXiv preprint arXiv:2501.00882, 2025.","DOI":"10.1109\/TETCI.2025.3634718"}],"container-title":["Biomedical Signal Processing and Control"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1746809426009821?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1746809426009821?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T06:09:58Z","timestamp":1778825398000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1746809426009821"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":27,"alternative-id":["S1746809426009821"],"URL":"https:\/\/doi.org\/10.1016\/j.bspc.2026.110428","relation":{},"ISSN":["1746-8094"],"issn-type":[{"value":"1746-8094","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"OtoSTVT: a diagnostic-driven frame selection method for otoscopy videos using spatio-temporal transformers","name":"articletitle","label":"Article Title"},{"value":"Biomedical Signal Processing and Control","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.bspc.2026.110428","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"110428"}}