{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T18:10:04Z","timestamp":1748196604199,"version":"3.41.0"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031915741","type":"print"},{"value":"9783031915758","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91575-8_5","type":"book-chapter","created":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:57:29Z","timestamp":1748195849000},"page":"71-87","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["ROMEO: Revisiting Optimization Methods for\u00a0Reconstructing 3D Human-Object Interaction Models From Images"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3211-183X","authenticated-orcid":false,"given":"Alexey","family":"Gavryushin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daoji","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6433-6713","authenticated-orcid":false,"given":"Yen-Ling","family":"Kuo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Julien","family":"Valentin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3445-5711","authenticated-orcid":false,"given":"Luc","family":"van Gool","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5068-3474","authenticated-orcid":false,"given":"Otmar","family":"Hilliges","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5442-1116","authenticated-orcid":false,"given":"Xi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"5_CR1","unstructured":"Bhat, S.F., Birkl, R., Wofk, D., Wonka, P., M\u00fcller, M.: ZoeDepth: zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288 (2023)"},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Bhatnagar, B.L., Xie, X., Petrov, I.A., Sminchisescu, C., Theobalt, C., Pons-Moll, G.: BEHAVE: dataset and method for tracking human object interactions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15935\u201315946 (2022)","DOI":"10.1109\/CVPR52688.2022.01547"},{"key":"5_CR3","unstructured":"Chen, W., et al.: Learning to predict 3D objects with an interpolation-based differentiable renderer. In: NeurIPS, pp. 9605\u20139616 (2019)"},{"key":"5_CR4","unstructured":"Fuji\u00a0Tsang, C., et al.: Kaolin: a Pytorch Library for Accelerating 3D Deep Learning Research (2022). https:\/\/github.com\/NVIDIAGameWorks\/kaolin"},{"issue":"3","key":"5_CR5","doi-asserted-by":"publisher","first-page":"227","DOI":"10.2307\/1574154","volume":"11","author":"JJ Gibson","year":"1978","unstructured":"Gibson, J.J.: The ecological approach to the visual perception of pictures. Leonardo 11(3), 227\u2013235 (1978)","journal-title":"Leonardo"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Hassan, M., Choutas, V., Tzionas, D., Black, M.J.: Resolving 3D human pose ambiguities with 3D scene constraints. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2282\u20132292 (2019)","DOI":"10.1109\/ICCV.2019.00237"},{"key":"5_CR7","doi-asserted-by":"crossref","unstructured":"Huang, C.H.P., et al.: Capturing and inferring dense full-body human-scene contact. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13274\u201313285 (Jun 2022)","DOI":"10.1109\/CVPR52688.2022.01292"},{"key":"5_CR8","doi-asserted-by":"publisher","unstructured":"Huang, Y., Taheri, O., Black, M.J., Tzionas, D.: InterCap: joint markerless 3D tracking of humans and objects in interaction. In: German Conference on Pattern Recognition (GCPR). LNCS, vol. 13485, pp. 281\u2013299. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-16788-1_18","DOI":"10.1007\/978-3-031-16788-1_18"},{"key":"5_CR9","unstructured":"Jiang, N., et al.: CHAIRS: towards full-body articulated human-object interaction. arXiv preprint arXiv:2212.10621 (2022)"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Joo, H., Neverova, N., Vedaldi, A.: Exemplar fine-tuning for 3D human pose fitting towards in-the-wild 3D human pose estimation. In: International Conference on 3D Vision (2021)","DOI":"10.1109\/3DV53792.2021.00015"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Kato, H., Ushiku, Y., Harada, T.: Neural 3D mesh renderer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2018)","DOI":"10.1109\/CVPR.2018.00411"},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment Anything. arXiv:2304.02643 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Wu, Y., He, K., Girshick, R.: PointRend: image segmentation as rendering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9799\u20139808 (2020)","DOI":"10.1109\/CVPR42600.2020.00982"},{"key":"5_CR14","doi-asserted-by":"crossref","unstructured":"Kocabas, M., Huang, C.H.P., Hilliges, O., Black, M.J.: PARE: part attention regressor for 3D human body estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11127\u201311137 (2021)","DOI":"10.1109\/ICCV48922.2021.01094"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Liu, S., et\u00a0al.: Grounding DINO: marrying DINO with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Mirmohammadi, M., Saremi, P., Kuo, Y.L., Wang, X.: Reconstruction of 3D interaction models from images using shape prior. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2141\u20132147 (2023)","DOI":"10.1109\/ICCVW60793.2023.00228"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Mittal, P., Cheng, Y.C., Singh, M., Tulsiani, S.: AutoSDF: shape priors for 3D completion, reconstruction and generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 306\u2013315 (2022)","DOI":"10.1109\/CVPR52688.2022.00040"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Ranftl, R., Bochkovskiy, A., Koltun, V.: Vision transformers for dense prediction. arXiv:2103.13413 (2021)","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"5_CR19","unstructured":"Ranftl, R., Lasinger, K., Hafner, D., Schindler, K., Koltun, V.: Towards robust monocular depth estimation: mixing datasets for zero-shot cross-dataset transfer. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) (2020)"},{"key":"5_CR20","doi-asserted-by":"crossref","unstructured":"Rong, Y., Shiratori, T., Joo, H.: FrankMocap: a monocular 3D whole-body pose estimation system via regression and integration. In: IEEE International Conference on Computer Vision Workshops (2021)","DOI":"10.1109\/ICCVW54120.2021.00201"},{"issue":"2","key":"5_CR21","doi-asserted-by":"publisher","first-page":"2579","DOI":"10.1109\/LRA.2021.3062350","volume":"6","author":"I Shugurov","year":"2021","unstructured":"Shugurov, I., Pavlov, I., Zakharov, S., Ilic, S.: Multi-view object pose refinement with differentiable renderer. IEEE Robot. Autom. Lett. 6(2), 2579\u20132586 (2021)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Tripathi, S., Chatterjee, A., Passy, J.C., Yi, H., Tzionas, D., Black, M.J.: DECO: dense estimation of 3d human-scene contact in the wild. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 8001\u20138013 (October 2023)","DOI":"10.1109\/ICCV51070.2023.00735"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"Wang, G., Manhardt, F., Shao, J., Ji, X., Navab, N., Tombari, F.: Self6D: self-supervised monocular 6D object pose estimation. In: The European Conference on Computer Vision (August 2020)","DOI":"10.1007\/978-3-030-58452-8_7"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Wang, X., Li, G., Kuo, Y., Kocabas, M., Aksan, E., Hilliges, O.: Reconstructing action-conditioned human-object interactions using commonsense knowledge priors. In: 2022 International Conference on 3D Vision, pp. 353\u2013362 (2022)","DOI":"10.1109\/3DV57658.2022.00047"},{"key":"5_CR25","doi-asserted-by":"crossref","unstructured":"Weng, Z., Yeung, S.: Holistic 3D human and scene mesh estimation from single view images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, . 334\u2013343 (2021)","DOI":"10.1109\/CVPR46437.2021.00040"},{"key":"5_CR26","doi-asserted-by":"publisher","unstructured":"Xie, X., Bhatnagar, B.L., Pons-Moll, G.: CHORE: contact, human and object reconstruction from a single RGB image. In: European Conference on Computer Vision. Springer (October 2022). https:\/\/doi.org\/10.1007\/978-3-031-20086-1_8","DOI":"10.1007\/978-3-031-20086-1_8"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Xie, X., Bhatnagar, B.L., Pons-Moll, G.: visibility aware human-object interaction tracking from single RGB camera. In: IEEE Conference on Computer Vision and Pattern Recognition (June 2023)","DOI":"10.1109\/CVPR52729.2023.00461"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Xu, S., Li, Z., Wang, Y.X., Gui, L.Y.: InterDiff: generating 3D human-object interactions with physics-informed diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2023)","DOI":"10.1109\/ICCV51070.2023.01371"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Yi, H., et al.: Human-aware object placement for visual environment reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3959\u20133970 (2022)","DOI":"10.1109\/CVPR52688.2022.00393"},{"key":"5_CR30","unstructured":"Zhang, B., Nie\u00dfner, M., Wonka, P.: 3DILG: irregular latent grids for 3D generative modeling. arXiv preprint arXiv:2205.13914 (2022)"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Zhang, H., et al.: PyMAF: 3D human pose and shape regression with pyramidal mesh alignment feedback loop. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11446\u201311456 (2021)","DOI":"10.1109\/ICCV48922.2021.01125"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Zhang, J.Y., Pepose, S., Joo, H., Ramanan, D., Malik, J., Kanazawa, A.: Perceiving 3D human-object spatial arrangements from a single image in the wild. In: European Conference on Computer Vision (2020)","DOI":"10.1007\/978-3-030-58610-2_3"},{"key":"5_CR33","doi-asserted-by":"publisher","unstructured":"Zhang, J., et al.: NeuralDome: a neural modeling pipeline on multi-view human-object interactions (2022). https:\/\/doi.org\/10.48550\/ARXIV.2212.07626","DOI":"10.48550\/ARXIV.2212.07626"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91575-8_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:57:39Z","timestamp":1748195859000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91575-8_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031915741","9783031915758"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91575-8_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}