{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T22:32:55Z","timestamp":1777501975374,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100000923","name":"Australian Research Council","doi-asserted-by":"publisher","award":["CE200100005"],"award-info":[{"award-number":["CE200100005"]}],"id":[{"id":"10.13039\/501100000923","id-type":"DOI","asserted-by":"publisher"}]},{"name":"NSF-CSIRO","award":["2302968, 2302969, 2302970"],"award-info":[{"award-number":["2302968, 2302969, 2302970"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,7]]},"DOI":"10.1145\/3715071.3750404","type":"proceedings-article","created":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T16:20:02Z","timestamp":1759854002000},"page":"45-52","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["CoughViT: A Self-Supervised Vision Transformer for Cough Audio Representation Learning"],"prefix":"10.1145","author":[{"given":"Justin","family":"Luong","sequence":"first","affiliation":[{"name":"University of New South Wales, Sydney, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1700-9215","authenticated-orcid":false,"given":"Hao","family":"Xue","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1237-1664","authenticated-orcid":false,"given":"Flora D.","family":"Salim","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, University of New South Wales, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,7]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3097559"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISMS.2015.41"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2015.05.001"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.2196\/38439"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1183\/09031936.00057407"},{"key":"e_1_3_2_1_6_1","volume-title":"Exploring automatic diagnosis of COVID-19 from crowdsourced respiratory sound data. arXiv preprint arXiv:2006.05919","author":"Brown Chlo\u00eb","year":"2020","unstructured":"Chlo\u00eb Brown, Jagmohan Chauhan, Andreas Grammenos, Jing Han, Apinan Hasthanasombat, Dimitris Spathis, Tong Xia, Pietro Cicuta, and Cecilia Mascolo. 2020. Exploring automatic diagnosis of COVID-19 from crowdsourced respiratory sound data. arXiv preprint arXiv:2006.05919 (2020)."},{"key":"e_1_3_2_1_7_1","volume-title":"Alexander Titcomb, Richard Payne, David Hurley, Sabrina Egglestone, et al.","author":"Budd Jobie","year":"2022","unstructured":"Jobie Budd, Kieran Baker, Emma Karoune, Harry Coppock, Selina Patel, Ana Tendero Ca nadas, Alexander Titcomb, Richard Payne, David Hurley, Sabrina Egglestone, et al., 2022. A large-scale and PCR-referenced vocal audio dataset for COVID-19. arXiv preprint arXiv:2212.07738 (2022)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1798"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41746-021-00472-x"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/S2589-7500(21)00141-2"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10274"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compbiomed.2021.104944"},{"key":"e_1_3_2_1_13_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_14_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_15_1","volume-title":"Rethinking supervised pre-training for better downstream transferring. arXiv preprint arXiv:2110.06014","author":"Feng Yutong","year":"2021","unstructured":"Yutong Feng, Jianwen Jiang, Mingqian Tang, Rong Jin, and Yue Gao. 2021. Rethinking supervised pre-training for better downstream transferring. arXiv preprint arXiv:2110.06014 (2021)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_1_17_1","volume-title":"Ast: Audio spectrogram transformer. arXiv preprint arXiv:2104.01778","author":"Gong Yuan","year":"2021","unstructured":"Yuan Gong, Yu-An Chung, and James Glass. 2021. Ast: Audio spectrogram transformer. arXiv preprint arXiv:2104.01778 (2021)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21315"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458754"},{"key":"e_1_3_2_1_20_1","volume-title":"Computerized lung sound analysis as diagnostic aid for the detection of abnormal lung sounds: a systematic review and meta-analysis. Respiratory medicine","author":"Gurung Arati","year":"2011","unstructured":"Arati Gurung, Carolyn G Scrafford, James M Tielsch, Orin S Levine, and William Checkley. 2011. Computerized lung sound analysis as diagnostic aid for the detection of abnormal lung sounds: a systematic review and meta-analysis. Respiratory medicine, Vol. 105, 9 (2011), 1396-1403."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0220606"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Jing Han Tong Xia Dimitris Spathis Erika Bondareva Chlo\u00eb Brown Jagmohan Chauhan Ting Dang Andreas Grammenos Apinan Hasthanasombat Andres Floto et al. 2022. Sounds of COVID-19: exploring realistic performance of audio-based digital testing. NPJ digital medicine Vol. 5 1 (2022) 1-9.","DOI":"10.1038\/s41746-021-00553-x"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"e_1_3_2_1_24_1","first-page":"131","article-title":"CNN architectures for large-scale audio classification. In 2017 ieee international conference on acoustics, speech and signal processing (icassp)","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al., 2017. CNN architectures for large-scale audio classification. In 2017 ieee international conference on acoustics, speech and signal processing (icassp). IEEE, 131-135.","journal-title":"IEEE"},{"key":"e_1_3_2_1_25_1","first-page":"28708","article-title":"Masked autoencoders that listen","volume":"35","author":"Huang Po-Yao","year":"2022","unstructured":"Po-Yao Huang, Hu Xu, Juncheng Li, Alexei Baevski, Michael Auli, Wojciech Galuba, Florian Metze, and Christoph Feichtenhofer. 2022. Masked autoencoders that listen. Advances in Neural Information Processing Systems, Vol. 35 (2022), 28708-28720.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imu.2021.100832"},{"key":"e_1_3_2_1_27_1","volume-title":"Kamran Ali, Charles N John, MD Iftikhar Hussain, and Muhammad Nabeel.","author":"Imran Ali","year":"2020","unstructured":"Ali Imran, Iryna Posokhova, Haneya N Qureshi, Usama Masood, Muhammad Sajid Riaz, Kamran Ali, Charles N John, MD Iftikhar Hussain, and Muhammad Nabeel. 2020. AI4COVID-19: AI enabled preliminary diagnosis for COVID-19 from cough samples via an app. Informatics in medicine unlocked, Vol. 20 (2020), 100378."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1186\/s12890-022-01896-1"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054458"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/EMBC.2019.8856412"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-021-00937-4"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cmpb.2023.107743"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/EMBC40787.2023.10340413"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compbiomed.2021.104572"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/OJEMB.2020.3026468"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Paul Porter Udantha Abeyratne Vinayak Swarnkar Jamie Tan Ti-wan Ng Joanna M Brisbane Deirdre Speldewinde Jennifer Choveaux Roneel Sharan Keegan Kosasih et al. 2019. A prospective multicentre study testing the diagnostic accuracy of an automated cough sound centred analytic system for the identification of common respiratory disorders in children. Respiratory research Vol. 20 1 (2019) 1-10.","DOI":"10.1186\/s12931-019-1046-6"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2908700"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.4103\/1817-1737.160831"},{"key":"e_1_3_2_1_41_1","volume-title":"Prasanta Kumar Ghosh, Sriram Ganapathy, et al.","author":"Sharma Neeraj","year":"2020","unstructured":"Neeraj Sharma, Prashant Krishnan, Rohit Kumar, Shreyas Ramoji, Srikanth Raj Chetupalli, Prasanta Kumar Ghosh, Sriram Ganapathy, et al., 2020. Coswara-a database of breathing, cough, and voice sounds for COVID-19 diagnosis. arXiv preprint arXiv:2005.10548 (2020)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747188"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1080\/02770903.2019.1684516"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10439-013-0741-6"},{"key":"e_1_3_2_1_45_1","volume-title":"COVID-19 Sounds: A Large-Scale Audio Dataset for Digital Respiratory Screening. In Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2).","author":"Xia Tong","year":"2021","unstructured":"Tong Xia, Dimitris Spathis, J Ch, Andreas Grammenos, Jing Han, Apinan Hasthanasombat, Erika Bondareva, Ting Dang, Andres Floto, Pietro Cicuta, et al., 2021. COVID-19 Sounds: A Large-Scale Audio Dataset for Digital Respiratory Screening. In Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467263"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00309"}],"event":{"name":"UbiComp '25: The 2025 ACM International Joint Conference on Pervasive and Ubiquitous Computing \/ ISWC ACM International Symposium on Wearable Computers","location":"Espoo Finland","acronym":"UbiComp '25","sponsor":["SIGSPATIAL ACM Special Interest Group on Spatial Information","SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2025 ACM International Symposium on Wearable Computers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3715071.3750404","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T17:20:56Z","timestamp":1762536056000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3715071.3750404"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,7]]},"references-count":47,"alternative-id":["10.1145\/3715071.3750404","10.1145\/3715071"],"URL":"https:\/\/doi.org\/10.1145\/3715071.3750404","relation":{},"subject":[],"published":{"date-parts":[[2025,10,7]]},"assertion":[{"value":"2025-10-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}