{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T16:39:46Z","timestamp":1785602386596,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Natural Science Foundation of China","award":["No. 62201524, No. 61971383, No. 62271455"],"award-info":[{"award-number":["No. 62201524, No. 61971383, No. 62271455"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3613765","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"8525-8533","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["RD-FGFS: A Rule-Data Hybrid Framework for Fine-Grained Footstep Sound Synthesis from Visual Guidance"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6806-667X","authenticated-orcid":false,"given":"Qiutang","family":"Qi","sequence":"first","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3407-4318","authenticated-orcid":false,"given":"Haonan","family":"Cheng","sequence":"additional","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8933-3433","authenticated-orcid":false,"given":"Yang","family":"Wang","sequence":"additional","affiliation":[{"name":"North China Institute of Science and Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3562-5612","authenticated-orcid":false,"given":"Long","family":"Ye","sequence":"additional","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5106-3246","authenticated-orcid":false,"given":"Shaobin","family":"Li","sequence":"additional","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3241394"},{"key":"e_1_3_2_2_2_1","volume-title":"Proceedings of the ACM SIGGRAPH Symposium on Computer Animation. 349--356","author":"Cardle Marc","year":"2003","unstructured":"Marc Cardle, Stephen Brooks, Ziv Bar-Joseph, and Peter Robinson. 2003. Sound-by-numbers: Motion-driven sound synthesis. In Proceedings of the ACM SIGGRAPH Symposium on Computer Animation. 349--356."},{"key":"e_1_3_2_2_3_1","volume-title":"Proceedings of the European Conference on Computer Vision Workshops. 0--0.","author":"Chen Kan","year":"2018","unstructured":"Kan Chen, Chuanxi Zhang, Chen Fang, Zhaowen Wang, Trung Bui, and Ram Nevatia. 2018. Visually indicated sound generation by perceptually optimized classification. In Proceedings of the European Conference on Computer Vision Workshops. 0--0."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126686.3126723"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3009820"},{"key":"e_1_3_2_2_6_1","volume-title":"Physically informed sonic modeling (phism): Synthesis of percussive sounds. Computer Music Journal","author":"Cook Perry R","year":"1997","unstructured":"Perry R Cook. 1997. Physically informed sonic modeling (phism): Synthesis of percussive sounds. Computer Music Journal (1997)."},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the Audio Engineering Society Conference on Virtual, Synthetic, and Entertainment Audio.","author":"Cook Perry R","year":"2002","unstructured":"Perry R Cook. 2002. Modeling Bill's gait: Analysis and parametric synthesis of walking sounds. In Proceedings of the Audio Engineering Society Conference on Virtual, Synthetic, and Entertainment Audio."},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of The Pure Data Convention.","author":"Farnell Andy James","year":"2007","unstructured":"Andy James Farnell and Obiwannabe Uk. 2007. Marching onwards: Procedural synthetic footsteps for video games and animation. In Proceedings of The Pure Data Convention."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the Colloquium on Musical Informatics. 109--114","author":"Fontana Federico","year":"2003","unstructured":"Federico Fontana and Roberto Bresin. 2003. Physics-based sound synthesis and control: Crushing, walking and running by crumpling sounds. In Proceedings of the Colloquium on Musical Informatics. 109--114."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_2_12_1","volume-title":"Autofoley: Artificial synthesis of synchronized sound tracks for silent videos with deep learning","author":"Ghose Sanchita","year":"2020","unstructured":"Sanchita Ghose and John Jeffrey Prevost. 2020. Autofoley: Artificial synthesis of synchronized sound tracks for silent videos with deep learning. IEEE Transactions on Multimedia (2020)."},{"key":"e_1_3_2_2_13_1","volume":"202","author":"Ghose Sanchita","unstructured":"Sanchita Ghose and John J Prevost. 2022. Foleygan: Visually guided generative adversarial network-based synchronous sound generation in silent videos. IEEE Transactions on Multimedia (2022).","journal-title":"John J Prevost."},{"key":"e_1_3_2_2_14_1","volume-title":"Generative adversarial networks. Commun. ACM","author":"Goodfellow Ian","year":"2020","unstructured":"Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2020. Generative adversarial networks. Commun. ACM (2020)."},{"key":"e_1_3_2_2_15_1","volume-title":"Long short-term memory. Neural Computation","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural Computation (1997)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.5244\/C.35.336"},{"key":"e_1_3_2_2_17_1","volume-title":"Proceedings of the International Conference on Machine Learning. 448--456","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy. 2015. Batch normalization: Accelerating deep network training by reducing internal covariate shift. In Proceedings of the International Conference on Machine Learning. 448--456."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"e_1_3_2_2_19_1","volume-title":"Audio in VR: Effects of a soundscape and movement-triggered step sounds on presence. Frontiers in Robotics and AI","author":"Kern Angelika C","year":"2020","unstructured":"Angelika C Kern and Wolfgang Ellermeier. 2020. Audio in VR: Effects of a soundscape and movement-triggered step sounds on presence. Frontiers in Robotics and AI (2020)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00530"},{"key":"e_1_3_2_2_21_1","volume-title":"Physically-based statistical simulation of rain sound. ACM Transactions on Graphics","author":"Liu Shiguang","year":"2019","unstructured":"Shiguang Liu, Haonan Cheng, and Yiying Tong. 2019. Physically-based statistical simulation of rain sound. ACM Transactions on Graphics (2019)."},{"key":"e_1_3_2_2_22_1","volume-title":"Automatic synthesis of explosion sound synchronized with animation. Virtual Reality","author":"Liu Shiguang","year":"2020","unstructured":"Shiguang Liu and Si Gao. 2020. Automatic synthesis of explosion sound synchronized with animation. Virtual Reality (2020)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3079897"},{"key":"e_1_3_2_2_24_1","volume-title":"Visually aligned sound generation via sound-producing motion parsing. Neurocomputing","author":"Ma Xin","year":"2022","unstructured":"Xin Ma, Wei Zhong, Long Ye, and Qin Zhang. 2022. Visually aligned sound generation via sound-producing motion parsing. Neurocomputing (2022)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2040532"},{"key":"e_1_3_2_2_26_1","volume-title":"Proceedings of the International Conference on Learning Representations.","author":"Mehri Soroush","year":"2017","unstructured":"Soroush Mehri, Kundan Kumar, Ishaan Gulrajani, Rithesh Kumar, Shubham Jain, Jose Sotelo, Aaron Courville, and Yoshua Bengio. 2017. SampleRNN: An unconditional end-to-end neural audio generation model. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2011.30"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.264"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.885924"},{"key":"e_1_3_2_2_30_1","volume-title":"Animating elastic rods with sound. ACM Transactions on Graphics","author":"Schweickart Eston","year":"2017","unstructured":"Eston Schweickart, Doug L James, and Steve Marschner. 2017. Animating elastic rods with sound. ACM Transactions on Graphics (2017)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_2_2_32_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2856400.2856419"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/133994.134063"},{"key":"e_1_3_2_2_35_1","volume-title":"Proceedings of the International Conference on Machine Learning. 6105--6114","author":"Tan Mingxing","year":"2019","unstructured":"Mingxing Tan and Quoc Le. 2019. Efficientnet: Rethinking model scaling for convolutional neural networks. In Proceedings of the International Conference on Machine Learning. 6105--6114."},{"key":"e_1_3_2_2_36_1","volume-title":"Footstep sounds synthesis: Design, implementation, and evaluation of foot--floor interactions, surface materials, shoe types, and walkers' features. Applied Acoustics","author":"Luca Turchet","year":"2016","unstructured":"Luca Turchet 2016. Footstep sounds synthesis: Design, implementation, and evaluation of foot--floor interactions, surface materials, shoe types, and walkers' features. Applied Acoustics (2016)."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15841-4_11"},{"key":"e_1_3_2_2_38_1","volume-title":"Proceedings of the 9th ISCA Speech Synthesis Workshop. 125--125","author":"van den Oord A\u00e4ron","year":"2016","unstructured":"A\u00e4ron van den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew Senior, and Koray Kavukcuoglu. 2016. Wavenet: A generative model for raw audio. In Proceedings of the 9th ISCA Speech Synthesis Workshop. 125--125."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413894"},{"key":"e_1_3_2_2_40_1","volume-title":"Foley: The art of footsteps, props, and cloth movement. In Practical Art of Motion Picture Sound","author":"Yewdall David Lewis","year":"2012","unstructured":"David Lewis Yewdall. 2012. Foley: The art of footsteps, props, and cloth movement. In Practical Art of Motion Picture Sound. Routledge, 402--439."},{"key":"e_1_3_2_2_41_1","volume-title":"Acoustic texture rendering for extended sources in complex scenes. ACM Transactions on Graphics","author":"Zhang Zechen","year":"2019","unstructured":"Zechen Zhang, Nikunj Raghuvanshi, John Snyder, and Steve Marschner. 2019. Acoustic texture rendering for extended sources in complex scenes. ACM Transactions on Graphics (2019)."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00374"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613765","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3613765","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:05:20Z","timestamp":1755821120000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613765"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":42,"alternative-id":["10.1145\/3581783.3613765","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3613765","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}