{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T04:57:21Z","timestamp":1780635441820,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,19]]},"DOI":"10.1145\/3736425.3770100","type":"proceedings-article","created":{"date-parts":[[2025,11,11]],"date-time":"2025-11-11T12:21:55Z","timestamp":1762863715000},"page":"96-106","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["ViLA: Leveraging General-Purpose Audio for Training Vibration-Based Stadium Crowd Monitoring Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-1986-9485","authenticated-orcid":false,"given":"Yen Cheng","family":"Chang","sequence":"first","affiliation":[{"name":"University of Michigan-Ann Arbor, Ann Arbor, MI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8355-7186","authenticated-orcid":false,"given":"Jesse","family":"Codling","sequence":"additional","affiliation":[{"name":"University of Michigan, Standford, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7877-1783","authenticated-orcid":false,"given":"Yiwen","family":"Dong","sequence":"additional","affiliation":[{"name":"University of Illinois, Urbana-Champaign, Urbana, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0688-564X","authenticated-orcid":false,"given":"Jiale","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Michigan, Ann Arbor, Ann Arbor, MI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7998-3657","authenticated-orcid":false,"given":"Hae Young","family":"Noh","sequence":"additional","affiliation":[{"name":"Stanford University, Stanford, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8512-1615","authenticated-orcid":false,"given":"Pei","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Michigan, Ann Arbor, MI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,11]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675","author":"Abu-El-Haija Sami","year":"2016","unstructured":"Sami Abu-El-Haija, Nisarg Kothari, Joonseok Lee, Paul Natsev, George Toderici, Balakrishnan Varadarajan, and Sudheendra Vijayanarasimhan. 2016. Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675 (2016)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.5555\/1887984.1887990"},{"key":"e_1_3_2_1_3_1","volume-title":"2006 2nd International Conference on Information & Communication Technologies","volume":"1","author":"Majd","unstructured":"Majd Alwan et al. 2006. A Smart and Passive Floor-Vibration Based Fall Detector for Elderly. In 2006 2nd International Conference on Information & Communication Technologies, Vol. 1. IEEE."},{"key":"e_1_3_2_1_4_1","volume-title":"MRCNet: Crowd counting and density map estimation in aerial and ground imagery. arXiv preprint arXiv:1909.12743","author":"Bahmanyar Reza","year":"2019","unstructured":"Reza Bahmanyar, Elenora Vig, and Peter Reinartz. 2019. MRCNet: Crowd counting and density map estimation in aerial and ground imagery. arXiv preprint arXiv:1909.12743 (2019)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3412382.3458902"},{"key":"e_1_3_2_1_6_1","volume-title":"Smile: Sequence-to-Sequence Domain Adaptation with Minimizing Latent Entropy for Text Image Recognition. In 2022 IEEE International Conference on Image Processing (ICIP). IEEE.","author":"Yen-Cheng","unstructured":"Yen-Cheng Chang et al. 2022. Smile: Sequence-to-Sequence Domain Adaptation with Minimizing Latent Entropy for Text Image Recognition. In 2022 IEEE International Conference on Image Processing (ICIP). IEEE."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3671127.3698170"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1049\/ecej:19950106"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ymssp.2023.110756"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-024-60034-5"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.3390\/s24082496"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1017\/dce.2024.28"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600100.3623750"},{"key":"e_1_3_2_1_15_1","unstructured":"DSP Stack Exchange. 2016. The relationship between downsampling and frequency resolution. https:\/\/dsp.stackexchange.com\/questions\/30374\/the-relationship-between-downsampling-and-frequency-resolution. Available at https:\/\/dsp.stackexchange.com\/questions\/30374\/the-relationship-between-downsampling-and-frequency-resolution."},{"key":"e_1_3_2_1_16_1","volume-title":"Factors influencing experience in crowds-the participant perspective. Applied ergonomics 59","author":"Filingeri Victoria","year":"2017","unstructured":"Victoria Filingeri, Ken Eason, Patrick Waterson, and Roger Haslam. 2017. Factors influencing experience in crowds-the participant perspective. Applied ergonomics 59 (2017), 431\u2013441."},{"key":"e_1_3_2_1_17_1","volume-title":"Wave motion in elastic solids","author":"Graff Karl F","unstructured":"Karl F Graff. 2012. Wave motion in elastic solids. Courier Corporation."},{"key":"e_1_3_2_1_18_1","unstructured":"Stacey Gray. 2016. Always on: privacy implications of microphone-enabled devices. In Future of privacy forum. 1\u201310."},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","author":"Kaiming","unstructured":"Kaiming He et al. 2022. Masked autoencoders are scalable vision learners. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_20_1","volume-title":"Benefits of Training Audio-Visual Models with Temporally Synchronized Video. In ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4155\u20134159","author":"Hershey Shawn","year":"2021","unstructured":"Shawn Hershey, Martin Z. Shou, Daniel P. W. Ellis, and Ashok Chandrashekar. 2021. Benefits of Training Audio-Visual Models with Temporally Synchronized Video. In ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4155\u20134159."},{"key":"e_1_3_2_1_21_1","first-page":"28708","article-title":"Masked autoencoders that listen","volume":"35","author":"Huang P. Y.","year":"2022","unstructured":"P. Y. Huang, H. Xu, J. Li, A. Baevski, M. Auli, W. Galuba, and C. Feichtenhofer. 2022. Masked autoencoders that listen. Advances in Neural Information Processing Systems 35 (2022), 28708\u201328720.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2021.3076257"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1177\/1550147719881608"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1080\/19361610.2014.913229"},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the International Conference on Acoustics, Speech, and Signal Processing (ICASSP). 671\u2013675","author":"Kong Qiuqiang","unstructured":"Qiuqiang Kong, Yong Xu, Wenwu Wang, and Mark D. Plumbley. 2020. SSIM-based Loss Function for Audio Spectrogram Enhancement. In Proceedings of the International Conference on Acoustics, Speech, and Signal Processing (ICASSP). 671\u2013675."},{"key":"e_1_3_2_1_26_1","volume-title":"et al. Kong","author":"Q.","year":"2020","unstructured":"Q. et al. Kong. 2020. PANNs: Large-Scale Pretrained Audio Neural Networks for Audio Pattern Recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing (2020)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","first-page":"1941","DOI":"10.1177\/14759217221081159","article-title":"HierMUD: Hierarchical Multi-Task Unsupervised Domain Adaptation Between Bridges for Drive-By Damage Diagnosis","volume":"22","author":"Jingxiao Liu","year":"2023","unstructured":"Jingxiao Liu et al. 2023. HierMUD: Hierarchical Multi-Task Unsupervised Domain Adaptation Between Bridges for Drive-By Damage Diagnosis. Structural Health Monitoring 22, 3 (2023), 1941\u20131968.","journal-title":"Structural Health Monitoring"},{"key":"e_1_3_2_1_28_1","unstructured":"Fredrik Ljunggren. 2006. Floor vibration: dynamic properties and subjective perception. Ph. D. Dissertation. Lule\u00e5 tekniska universitet."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2905055.2905258"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.2196\/45297"},{"key":"e_1_3_2_1_31_1","volume-title":"Signal Processing for Music Analysis","author":"M\u00fcller Meinard","unstructured":"Meinard M\u00fcller. 2011. Signal Processing for Music Analysis. Springer, Berlin, Heidelberg."},{"key":"e_1_3_2_1_32_1","volume-title":"Representation learning with contrastive predictive coding. arXiv preprint arXiv.1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv.1807.03748 (2018)."},{"key":"e_1_3_2_1_33_1","first-page":"1","article-title":"FootprintID: Indoor Pedestrian Identification through Ambient Structural Vibration Sensing","volume":"1","author":"Shijia Pan","year":"2017","unstructured":"Shijia Pan et al. 2017. FootprintID: Indoor Pedestrian Identification through Ambient Structural Vibration Sensing. Proceedings of the ACM on Interactive, Mobile, Wearable and Ubiquitous Technologies 1, 3 (2017), 1\u201331.","journal-title":"Proceedings of the ACM on Interactive, Mobile, Wearable and Ubiquitous Technologies"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2699343.2699364"},{"key":"e_1_3_2_1_35_1","volume-title":"et al. Purwins","author":"H.","year":"2019","unstructured":"H. et al. Purwins. 2019. Deep learning for audio signal processing. IEEE Journal of Selected Topics in Signal Processing (2019)."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the 35th IMAC, A Conference and Exposition on Structural Dynamics","author":"Yves Reuland","year":"2017","unstructured":"Yves Reuland et al. 2017. Vibration-Based Occupant Detection Using a Multiple-Model Approach. In Dynamics of Civil Structures, Volume 2: Proceedings of the 35th IMAC, A Conference and Exposition on Structural Dynamics. Springer International Publishing."},{"key":"e_1_3_2_1_37_1","volume-title":"2020 IEEE\/ACM Fifth International Conference on Internet-of-Things Design and Implementation. 40\u201352","author":"Ruiz Carlos","year":"2020","unstructured":"Carlos Ruiz, Shijia Pan, Adeola Bannis, Ming-Po Chang, Hae Young Noh, and Pei Zhang. 2020. IDIoT: Towards Ubiquitous Identification of IoT Devices through Visual and Inertial Orientation Matching During Human Activity. In 2020 IEEE\/ACM Fifth International Conference on Internet-of-Things Design and Implementation. 40\u201352."},{"key":"e_1_3_2_1_38_1","volume-title":"Data Driven Inverse Design of Optical Metamaterials. Ph. D. Dissertation","author":"Sarkar Sulagna","unstructured":"Sulagna Sarkar. 2024. Data Driven Inverse Design of Optical Metamaterials. Ph. D. Dissertation. Carnegie Mellon University."},{"key":"e_1_3_2_1_39_1","volume-title":"Lin (Eds.)","volume":"33","author":"Tian Yonglong","year":"2020","unstructured":"Yonglong Tian, Chen Sun, Ben Poole, Dilip Krishnan, Cordelia Schmid, and Phillip Isola. 2020. What Makes for Good Views for Contrastive Learning?. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 6827\u20136839. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/4c2e5eaae9152079b9e95845750bb9ab-Paper.pdf"},{"key":"e_1_3_2_1_40_1","volume-title":"Classification of human induced floor vibrations. Building acoustics 13, 3","author":"Toratti Tomi","year":"2006","unstructured":"Tomi Toratti and Asko Talja. 2006. Classification of human induced floor vibrations. Building acoustics 13, 3 (2006), 211\u2013221."},{"key":"e_1_3_2_1_41_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_42_1","volume-title":"Sheikh","author":"Wang Zhou","year":"2017","unstructured":"Zhou Wang, Alan C. Bovik, and Hamid R. Sheikh. 2017. Structural Similarity Based Image Quality Assessment. In Digital Video Image Quality and Perceptual Coding. CRC Press, 225\u2013242."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2003.819861"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/WETICE.2012.26"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","first-page":"7361597","DOI":"10.1155\/2018\/7361597","article-title":"Managing crowds with wireless and mobile technologies","volume":"2018","author":"Yamin Mohammad","year":"2018","unstructured":"Mohammad Yamin, Abdullah M Basahel, and Adnan A Abi Sen. 2018. Managing crowds with wireless and mobile technologies. Wireless Communications and Mobile Computing 2018, 1 (2018), 7361597.","journal-title":"Wireless Communications and Mobile Computing"},{"key":"e_1_3_2_1_46_1","volume-title":"VIME: Extending the Success of Self- and Semi-supervised Learning to Tabular Domain. In Advances in Neural Information Processing Systems","author":"Yoon Jinsung","year":"2020","unstructured":"Jinsung Yoon, Yao Zhang, James Jordon, and Mihaela van der Schaar. 2020. VIME: Extending the Success of Self- and Semi-supervised Learning to Tabular Domain. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 11033\u201311043. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/7d97667a3e056acab9aaf653807b4a03-Paper.pdf"},{"key":"e_1_3_2_1_47_1","volume-title":"Crowd behavior at mass gatherings: a literature review. Prehospital and disaster medicine 24, 1","author":"Zeitz Kathryn M","year":"2009","unstructured":"Kathryn M Zeitz, Heather M Tan, M Grief, PC Couns, and Christopher J Zeitz. 2009. Crowd behavior at mass gatherings: a literature review. Prehospital and disaster medicine 24, 1 (2009), 32\u201338."},{"key":"e_1_3_2_1_48_1","volume-title":"Julia Gersey, Pei Zhang, Hae Young Noh, and Yiwen Dong.","author":"Zhang Jiale","year":"2025","unstructured":"Jiale Zhang, Yuyan Wu, Jesse R Codling, Yen Cheng Chang, Julia Gersey, Pei Zhang, Hae Young Noh, and Yiwen Dong. 2025. WeVibe: Weight Change Estimation ThroughAudio-Induced Shelf Vibrations In Autonomous Stores. arXiv:2502.12093 [eess.SP] https:\/\/arxiv.org\/abs\/2502.12093"}],"event":{"name":"BUILDSYS '25: 12th ACM International Conference on Systems for Energy-Efficient Buildings, Cities, and Transportation","location":"Colorado School of Mines Golden CO USA","acronym":"BUILDSYS '25","sponsor":["SIGEnergy ACM Special Interest Group on Energy Systems and Informatics"]},"container-title":["Proceedings of the 12th ACM International Conference on Systems for Energy-Efficient Buildings, Cities, and Transportation"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3736425.3770100","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,11]],"date-time":"2025-11-11T12:23:01Z","timestamp":1762863781000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3736425.3770100"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,11]]},"references-count":48,"alternative-id":["10.1145\/3736425.3770100","10.1145\/3736425"],"URL":"https:\/\/doi.org\/10.1145\/3736425.3770100","relation":{},"subject":[],"published":{"date-parts":[[2025,11,11]]},"assertion":[{"value":"2025-11-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}