{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T10:16:11Z","timestamp":1777889771862,"version":"3.51.4"},"reference-count":105,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["RS-2023-00218601"],"award-info":[{"award-number":["RS-2023-00218601"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00734","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"7828-7840","source":"Crossref","is-referenced-by-count":0,"title":["Learning 3D Scene Analogies With Neural Contextual Scene Maps"],"prefix":"10.1109","author":[{"given":"Junho","family":"Kim","sequence":"first","affiliation":[{"name":"Seoul National University,Dept. of Electrical and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gwangtak","family":"Bae","sequence":"additional","affiliation":[{"name":"Seoul National University,Dept. of Electrical and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eun Sun","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University,Dept. of Electrical and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Young Min","family":"Kim","sequence":"additional","affiliation":[{"name":"Seoul National University,Dept. of Electrical and Computer Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","year":"2024","journal-title":"Moondream: A tiny vision model"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00576"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2003.1211500"},{"key":"ref4","article-title":"ARKitscenes - a diverse real-world dataset for 3 d indoor scene understanding using mobile RGB-d data","volume-title":"Proceedings of the Conference on Neural Information Processing Systems Datasets and Benchmarks Track (NeurIPS)","author":"Baruch"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1126\/science.177.4043.77"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/34.24792"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref8","article-title":"A simple framework for contrastive learning of visual representations","volume-title":"In Proceedings of the International Conference on Machine Learning (ICML)","author":"Chen"},{"key":"ref9","article-title":"Big self-supervised models are strong semi-supervised learners","volume-title":"Proceedings of the Conference on Neural Information Processing Systems (NeurIPS)","author":"Chen"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01590"},{"key":"ref12","article-title":"Open-television: Teleoperation with immersive active visual feedback","volume-title":"Proceedings of the Conference on Robot Learning (CoRL)","author":"Cheng"},{"key":"ref13","article-title":"Automated creation of digital cousins for robust policy learning","volume-title":"Proceedings of the Conference on Robot Learning (CoRL)","author":"Dai"},{"key":"ref14","article-title":"Vision transformers need registers","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Darcet"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01198"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"ref17","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Dosovitskiy"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/355759.355766"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02059"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/83.623193"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161393"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.264"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00444"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.2307\/2007474"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01075"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01534-z"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1207\/s15516709cog0702_3"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.3758\/BF03200458"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1111\/tops.12278"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160794"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1016\/j.actpsy.2022.103505"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00382"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1029\/JB076i008p01905"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TSSC.1968.300136"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0982"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_22"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58604-1_38"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72940-9_13"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01807"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR-Adjunct57072.2022.00111"},{"key":"ref41","article-title":"Mutual scene synthesis for mixed reality telepresence","volume-title":"ArXiv","volume":"abs\/2204.00161","author":"Keshavarzi","year":"2022"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729392"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR-Adjunct57072.2022.00140"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/s10055-022-00734-3"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.73"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.485"},{"key":"ref47","article-title":"Recurrent transformer networks for semantic correspondence","volume-title":"Proceedings of the Conference on Neural Information Processing Systems (NeurIPS)","author":"Kim"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00850"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/VR58804.2024.00099"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1145\/2010324.1964974"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1406.3269"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1016\/j.brainres.2010.11.080"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1002\/nav.20053"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00339"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.1999.770022"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00806"},{"key":"ref57","article-title":"Mimicgen: A data generation system for scalable robot learning using human demonstrations","volume-title":"Proceedings of the Conference on Robot Learning (CoRL)","author":"Mandlekar"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73242-3_8"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00349"},{"key":"ref61","article-title":"Spair-71k: A large-scale benchmark for semantic correspondence","volume":"abs\/1908.10543","author":"Min","year":"2019","journal-title":"ArXiv"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3233884"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00461"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.15005"},{"key":"ref65","article-title":"Diffusion model for dense matching","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Nam"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2007.09.009"},{"key":"ref67","article-title":"Dinov2: Learning robust visual features without supervision","author":"Oquab","year":"2023","journal-title":"Transactions on Machine Learning Research (TMLR)"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02168"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00085"},{"key":"ref70","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Radford"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.5244\/c.31.89"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1016\/j.jecp.2006.02.002"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02004"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00499"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676463"},{"key":"ref78","article-title":"Clip-fields: Weakly supervised semantic fields for robotic memory","volume-title":"In Proceedings of the Workshop on Language and Robotics at the Conference on Robot Learning (CoRL)","author":"Muhammad"},{"key":"ref79","article-title":"Distilled feature fields enable few-shot language-guided manipulation","volume-title":"In Proceedings of the Conference on Robot Learning (CoRL)","author":"Shen"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812146"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00688"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0068"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/34.88573"},{"key":"ref84","article-title":"Representation learning with contrastive predictive coding","volume":"abs\/1807.03748","author":"van den Oord","year":"2018","journal-title":"ArXiv"},{"key":"ref85","article-title":"Attention is all you need","volume-title":"Proceedings of the Conference on Neural Information Processing Systems (NeurIPS)","author":"Vaswani"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611970128"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00402"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_1"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01277"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/3DV57658.2022.00055"},{"key":"ref91","article-title":"Neus: Learning neural implicit surfaces by volume rendering for multi-view reconstruction","volume-title":"In Proceedings of the Conference on Neural Information Processing Systems","author":"Wang"},{"key":"ref92","article-title":"Neural attention field: Emerging point relevance in 3d scenes for one-shot dexterous grasping","volume-title":"Proceedings of the Conference on Robot Learning (CoRL)","author":"Wang"},{"key":"ref93","article-title":"SparseDFF: Sparse-view feature distillation for one-shot dexterous manipulation","volume-title":"In Proceedings of the International Conference on Learning Representations (ICLR)","author":"Wang"},{"key":"ref94","article-title":"$\\mathrm{D}^{3}$ fields: Dynamic 3d descriptor fields for zero-shot generalizable rearrangement","volume-title":"In Proceedings of the Conference on Robot Learning (CoRL)","author":"Wang"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01825"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_34"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.14505"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02683"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126300"},{"key":"ref100","article-title":"Multiview neural surface reconstruction by disentangling geometry and appearance","volume-title":"Proceedings of the Conference on Neural Information Processing Systems","author":"Yariv"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3130590"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00297"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01554"},{"key":"ref104","article-title":"Densematcher: Learning 3d semantic correspondence for category-level manipulation from a single demo","volume-title":"In Proceedings of the International Conference on Learning Representations (ICLR)","author":"Zhu"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01245"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11443498.pdf?arnumber=11443498","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:20:03Z","timestamp":1777612803000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11443498\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":105,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00734","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}