{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T23:16:05Z","timestamp":1751411765961},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,10,8]],"date-time":"2023-10-08T00:00:00Z","timestamp":1696723200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,10,8]],"date-time":"2023-10-08T00:00:00Z","timestamp":1696723200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,10,8]]},"DOI":"10.1109\/icip49359.2023.10222066","type":"proceedings-article","created":{"date-parts":[[2023,9,11]],"date-time":"2023-09-11T17:58:31Z","timestamp":1694455111000},"page":"2795-2799","source":"Crossref","is-referenced-by-count":2,"title":["Depth Estimation of Multi-Modal Scene Based on Multi-Scale Modulation"],"prefix":"10.1109","author":[{"given":"Anjie","family":"Wang","sequence":"first","affiliation":[{"name":"Peking University,School of Electronic and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhijun","family":"Fang","sequence":"additional","affiliation":[{"name":"Donghua University,School of Computer Science and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyan","family":"Jiang","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science,School of Electronic and Electrical Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongbin","family":"Gao","sequence":"additional","affiliation":[{"name":"Shanghai University of Engineering Science,School of Electronic and Electrical Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaofeng","family":"Cao","sequence":"additional","affiliation":[{"name":"Peking University,School of Electronic and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siwei","family":"Ma","sequence":"additional","affiliation":[{"name":"Peking University,School of Computer Science"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196934"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2019.00041"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58545-7_38"},{"journal-title":"The replica dataset: A digital replica of indoor spaces","year":"2019","author":"Straub","key":"ref4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_27"},{"article-title":"Batvision with gcc-phat features for better sound to vision predictions","volume-title":"CVPR Workshop","author":"Christensen","key":"ref7"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_37"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00817"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00840"},{"key":"ref11","article-title":"Depth map prediction from a single image using a multi-scale deep network","author":"Eigen","year":"2014","journal-title":"NIPS"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2015.2505283"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.700"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.699"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3080928"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2968751"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196934"},{"article-title":"Ambient soundprovides supervision for visual learning","volume-title":"ECCV","author":"Owens","key":"ref18"},{"key":"ref19","article-title":"Soundnet: Learning sound representations from unlabeled video","author":"Aytar","year":"2016","journal-title":"NeurIPS"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.304"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2019.00116"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.660"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01254"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2022.3189597"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_24"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-4012"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746476"}],"event":{"name":"2023 IEEE International Conference on Image Processing (ICIP)","start":{"date-parts":[[2023,10,8]]},"location":"Kuala Lumpur, Malaysia","end":{"date-parts":[[2023,10,11]]}},"container-title":["2023 IEEE International Conference on Image Processing (ICIP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10221937\/10221892\/10222066.pdf?arnumber=10222066","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T20:14:46Z","timestamp":1709324086000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10222066\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,8]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/icip49359.2023.10222066","relation":{},"subject":[],"published":{"date-parts":[[2023,10,8]]}}}