{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:53:22Z","timestamp":1784300002208,"version":"3.55.0"},"publisher-location":"Cham","reference-count":73,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031731150","type":"print"},{"value":"9783031731167","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73116-7_18","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T15:15:38Z","timestamp":1730301338000},"page":"306-324","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":42,"title":["Track2Act: Predicting Point Tracks from\u00a0Internet Videos Enables Generalizable Robot Manipulation"],"prefix":"10.1007","author":[{"given":"Homanga","family":"Bharadhwaj","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Roozbeh","family":"Mottaghi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abhinav","family":"Gupta","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shubham","family":"Tulsiani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"18_CR1","doi-asserted-by":"crossref","unstructured":"Baek, S., Kim, K.I., Kim, T.K.: Pushing the envelope for RGB-based dense 3d hand pose estimation via neural rendering. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00116"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Bahl, S., Gupta, A., Pathak, D.: Human-to-robot imitation in the wild. In: RSS (2022)","DOI":"10.15607\/RSS.2022.XVIII.026"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Bahl, S., Mendonca, R., Chen, L., Jain, U., Pathak, D.: Affordances from human videos as a versatile representation for robotics. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01324"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Bharadhwaj, H., Gupta, A., Kumar, V., Tulsiani, S.: Towards generalizable zero-shot manipulation via translating human interaction plans. In: 2024 IEEE International Conference on Robotics and Automation (ICRA) (2024)","DOI":"10.1109\/ICRA57147.2024.10610288"},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Bharadhwaj, H., Vakil, J., Sharma, M., Gupta, A., Tulsiani, S., Kumar, V.: Roboagent: generalization and efficiency in robot manipulation via semantic augmentations and action chunking. In: 2024 IEEE International Conference on Robotics and Automation (ICRA) (2024)","DOI":"10.1109\/ICRA57147.2024.10611293"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Boukhayma, A., Bem, R.d., Torr, P.H.: 3d hand shape and pose from images in the wild. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01110"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Brahmbhatt, S., Handa, A., Hays, J., Fox, D.: Contactgrasp: functional multi-finger grasp synthesis from contact. arXiv (2019)","DOI":"10.1109\/IROS40897.2019.8967960"},{"key":"18_CR8","unstructured":"Brohan, A., et\u00a0al.: Rt-1: robotics transformer for real-world control at scale. arXiv preprint arXiv:2212.06817 (2022)"},{"key":"18_CR9","doi-asserted-by":"crossref","unstructured":"Byravan, A., Fox, D.: Se3-nets: learning rigid body motion using deep neural networks. In: 2017 IEEE International Conference on Robotics and Automation (ICRA), pp. 173\u2013180. IEEE (2017)","DOI":"10.1109\/ICRA.2017.7989023"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Damen, D., et\u00a0al.: Scaling egocentric vision: the epic-kitchens dataset. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 720\u2013736 (2018)","DOI":"10.1007\/978-3-030-01225-0_44"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Das, P., Xu, C., Doell, R.F., Corso, J.J.: A thousand frames in just a few words: lingual description of videos through latent topics and sparse object stitching. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2634\u20132641 (2013)","DOI":"10.1109\/CVPR.2013.340"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"18_CR13","unstructured":"Doersch, C., et al.: Tap-vid: a benchmark for tracking any point in a video. Adv. Neural. Inf. Process. Syst. 35, 13610\u201313626 (2022)"},{"key":"18_CR14","unstructured":"Du, Y., et al.: Learning universal policies via text-guided video generation. Adv. Neural Inf. Process. Syst. 36 (2024)"},{"key":"18_CR15","doi-asserted-by":"crossref","unstructured":"Fang, H.S., et al.: Rh20t: a robotic dataset for learning diverse skills in one-shot. arXiv preprint arXiv:2307.00595 (2023)","DOI":"10.1109\/ICRA57147.2024.10611615"},{"key":"18_CR16","unstructured":"Finn, C., Yu, T., Zhang, T., Abbeel, P., Levine, S.: One-shot visual imitation learning via meta-learning. In: Conference on Robot Learning, pp. 357\u2013368. PMLR (2017)"},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"Fu, T.J., et al.: Tell me what happened: unifying text-guided video completion via multimodal masked video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10681\u201310692 (2023)","DOI":"10.1109\/CVPR52729.2023.01029"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Ge, L., et al.: 3d hand shape and pose estimation from a single RGB image. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01109"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Goyal, A., et al.: Ifor: iterative flow minimization for robotic object rearrangement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14787\u201314797 (2022)","DOI":"10.1109\/CVPR52688.2022.01437"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Goyal, M., Modi, S., Goyal, R., Gupta, S.: Human hands as probes for interactive object understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3293\u20133303 (2022)","DOI":"10.1109\/CVPR52688.2022.00329"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Goyal, R., et\u00a0al.: The \u201csomething something\u201d video database for learning and evaluating visual common sense. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"18_CR22","unstructured":"Grauman, K., et\u00a0al.: Ego4d: around the world in 3,000 hours of egocentric video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18995\u201319012 (2022)"},{"key":"18_CR23","unstructured":"Gupta, A., et al.: Photorealistic video generation with diffusion models. arXiv preprint arXiv:2312.06662 (2023)"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Hasson, Y., et al.: Learning joint reconstruction of hands and manipulated objects. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01208"},{"key":"18_CR25","doi-asserted-by":"crossref","unstructured":"He, Y., Sun, W., Huang, H., Liu, J., Fan, H., Sun, J.: Pvn3d: a deep point-wise 3d keypoints voting network for 6dof pose estimation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01165"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Hu, Y., Hugonot, J., Fua, P., Salzmann, M.: Segmentation-driven 6d object pose estimation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00350"},{"key":"18_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1007\/978-3-030-01252-6_8","volume-title":"Computer Vision \u2013 ECCV 2018","author":"U Iqbal","year":"2018","unstructured":"Iqbal, U., Molchanov, P., Breuel, T., Gall, J., Kautz, J.: Hand pose estimation via latent 2.5D heatmap regression. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11215, pp. 125\u2013143. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01252-6_8"},{"key":"18_CR28","unstructured":"Jang, E., et al.: Bc-z: zero-shot task generalization with robotic imitation learning. In: Conference on Robot Learning, pp. 991\u20131002. PMLR (2022)"},{"key":"18_CR29","doi-asserted-by":"crossref","unstructured":"Karaev, N., Rocco, I., Graham, B., Neverova, N., Vedaldi, A., Rupprecht, C.: Cotracker: it is better to track together. arXiv preprint arXiv:2307.07635 (2023)","DOI":"10.1007\/978-3-031-73033-7_2"},{"key":"18_CR30","unstructured":"Kay, W., et\u00a0al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"18_CR31","doi-asserted-by":"crossref","unstructured":"Kehl, W., Manhardt, F., Tombari, F., Ilic, S., Navab, N.: Ssd-6d: making rgb-based 3d detection and 6d pose estimation great again. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.169"},{"key":"18_CR32","doi-asserted-by":"crossref","unstructured":"Keselman, L., Iselin\u00a0Woodfill, J., Grunnet-Jepsen, A., Bhowmik, A.: Intel realsense stereoscopic depth cameras. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp. 1\u201310 (2017)","DOI":"10.1109\/CVPRW.2017.167"},{"key":"18_CR33","unstructured":"Ko, P.C., Mao, J., Du, Y., Sun, S.H., Tenenbaum, J.B.: Learning to act from actionless videos through dense correspondences. arXiv preprint arXiv:2310.08576 (2023)"},{"key":"18_CR34","doi-asserted-by":"crossref","unstructured":"Kulon, D., Guler, R.A., Kokkinos, I., Bronstein, M.M., Zafeiriou, S.: Weakly-supervised mesh-convolutional hand reconstruction in the wild. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00504"},{"key":"18_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"639","DOI":"10.1007\/978-3-030-01228-1_38","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Y Li","year":"2018","unstructured":"Li, Y., Liu, M., Rehg, J.M.: In the eye of beholder: joint learning of gaze and actions in first person video. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11209, pp. 639\u2013655. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01228-1_38"},{"key":"18_CR36","doi-asserted-by":"crossref","unstructured":"Liu, C., Yuen, J., Torralba, A., Sivic, J., Freeman, W.T.: Sift flow: dense correspondence across different scenes. In: Computer Vision\u2013ECCV 2008: 10th European Conference on Computer Vision, Marseille, 12\u201318 October 2008, Proceedings, Part III 10, pp. 28\u201342. Springer (2008)","DOI":"10.1007\/978-3-540-88690-7_3"},{"key":"18_CR37","doi-asserted-by":"crossref","unstructured":"Liu, S., Jiang, H., Xu, J., Liu, S., Wang, X.: Semi-supervised 3d hand-object poses estimation with interactions in time. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01445"},{"key":"18_CR38","doi-asserted-by":"crossref","unstructured":"Liu, S., Tripathi, S., Majumdar, S., Wang, X.: Joint hand motion and interaction hotspots prediction from egocentric videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3282\u20133292 (2022)","DOI":"10.1109\/CVPR52688.2022.00328"},{"key":"18_CR39","unstructured":"Ma, Y.J., Sodhani, S., Jayaraman, D., Bastani, O., Kumar, V., Zhang, A.: Vip: towards universal visual reward and representation via value-implicit pre-training. arXiv preprint arXiv:2210.00030 (2022)"},{"key":"18_CR40","unstructured":"Mahi\u00a0Shafiullah, N.M., et al.: On bringing robots home. arXiv e-prints, pp. arXiv\u20132311 (2023)"},{"key":"18_CR41","unstructured":"Majumdar, A., et\u00a0al.: Where are we in the search for an artificial visual cortex for embodied intelligence? arXiv preprint arXiv:2303.18240 (2023)"},{"key":"18_CR42","unstructured":"Mandlekar, A., et\u00a0al.: Roboturk: a crowdsourcing platform for robotic skill learning through imitation. In: Conference on Robot Learning, pp. 879\u2013893. PMLR (2018)"},{"key":"18_CR43","doi-asserted-by":"crossref","unstructured":"Mo, K., Guibas, L.J., Mukadam, M., Gupta, A., Tulsiani, S.: Where2act: from pixels to actions for articulated 3d objects. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6813\u20136823 (2021)","DOI":"10.1109\/ICCV48922.2021.00674"},{"key":"18_CR44","doi-asserted-by":"crossref","unstructured":"Nagarajan, T., Feichtenhofer, C., Grauman, K.: Grounded human-object interaction hotspots from video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8688\u20138697 (2019)","DOI":"10.1109\/ICCV.2019.00878"},{"key":"18_CR45","unstructured":"Nair, S., Rajeswaran, A., Kumar, V., Finn, C., Gupta, A.: R3m: a universal visual representation for robot manipulation. arXiv preprint arXiv:2203.12601 (2022)"},{"key":"18_CR46","unstructured":"Padalkar, A., et\u00a0al.: Open x-embodiment: robotic learning datasets and rt-x models. arXiv preprint arXiv:2310.08864 (2023)"},{"key":"18_CR47","unstructured":"Pan, C., Okorn, B., Zhang, H., Eisner, B., Held, D.: Tax-pose: task-specific cross-pose estimation for robot manipulation. In: Conference on Robot Learning, pp. 1783\u20131792. PMLR (2023)"},{"key":"18_CR48","unstructured":"Parisi, S., Rajeswaran, A., Purushwalkam, S., Gupta, A.: The unsurprising effectiveness of pre-trained vision models for control. arXiv preprint arXiv:2203.03580 (2022)"},{"key":"18_CR49","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"18_CR50","doi-asserted-by":"crossref","unstructured":"Qin, Y., et al.: Dexmv: imitation learning for dexterous manipulation from human videos. arXiv preprint arXiv:2108.05877 (2021)","DOI":"10.1007\/978-3-031-19842-7_33"},{"key":"18_CR51","doi-asserted-by":"crossref","unstructured":"Qin, Z., Fang, K., Zhu, Y., Fei-Fei, L., Savarese, S.: Keto: learning keypoint representations for tool manipulation. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 7278\u20137285. IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9196971"},{"key":"18_CR52","doi-asserted-by":"crossref","unstructured":"Rad, M., Lepetit, V.: Bb8: a scalable, accurate, robust to partial occlusion method for predicting the 3d poses of challenging objects without using depth. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.413"},{"key":"18_CR53","doi-asserted-by":"crossref","unstructured":"Rong, Y., Shiratori, T., Joo, H.: Frankmocap: fast monocular 3d hand and body motion capture by regression and integration. arXiv preprint arXiv:2008.08324 (2020)","DOI":"10.1109\/ICCVW54120.2021.00201"},{"key":"18_CR54","unstructured":"Seita, D., Wang, Y., Shetty, S.J., Li, E.Y., Erickson, Z., Held, D.: Toolflownet: robotic manipulation with tools via predicting tool flow from point clouds. In: Conference on Robot Learning, pp. 1038\u20131049. PMLR (2023)"},{"key":"18_CR55","doi-asserted-by":"crossref","unstructured":"Shan, D., Geng, J., Shu, M., Fouhey, D.: Understanding human hands in contact at internet scale. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00989"},{"key":"18_CR56","unstructured":"Shaw, K., Bahl, S., Pathak, D.: Videodex: learning dexterity from internet videos. In: 6th Annual Conference on Robot Learning"},{"key":"18_CR57","doi-asserted-by":"crossref","unstructured":"Smith, L., Dhawan, N., Zhang, M., Abbeel, P., Levine, S.: Avid: learning multi-stage tasks via pixel-level translation of human videos. arXiv (2019)","DOI":"10.15607\/RSS.2020.XVI.024"},{"key":"18_CR58","doi-asserted-by":"crossref","unstructured":"Spurr, A., Song, J., Park, S., Hilliges, O.: Cross-modal deep variational hand pose estimation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00017"},{"key":"18_CR59","unstructured":"De\u00a0la Torre, F., et al.: Guide to the Carnegie Mellon University Multimodal Activity (CMU-MMAC) Database (2009)"},{"key":"18_CR60","doi-asserted-by":"crossref","unstructured":"Vecerik, M., et al.: Robotap: tracking arbitrary points for few-shot visual imitation. arXiv preprint arXiv:2308.15975 (2023)","DOI":"10.1109\/ICRA57147.2024.10611409"},{"key":"18_CR61","unstructured":"Walke, H.R., et\u00a0al.: Bridgedata v2: a dataset for robot learning at scale. In: Conference on Robot Learning, pp. 1723\u20131736. PMLR (2023)"},{"key":"18_CR62","unstructured":"Wang, C., et al.: Mimicplay: long-horizon imitation learning by watching human play. arXiv preprint arXiv:2302.12422 (2023)"},{"key":"18_CR63","doi-asserted-by":"crossref","unstructured":"Wen, C., et al.: Any-point trajectory modeling for policy learning. arXiv preprint arXiv:2401.00025 (2023)","DOI":"10.15607\/RSS.2024.XX.092"},{"key":"18_CR64","doi-asserted-by":"crossref","unstructured":"Xiang, Y., Schmidt, T., Narayanan, V., Fox, D.: Posecnn: a convolutional neural network for 6d object pose estimation in cluttered scenes. arXiv (2018)","DOI":"10.15607\/RSS.2018.XIV.019"},{"key":"18_CR65","unstructured":"Xiao, T., Radosavovic, I., Darrell, T., Malik, J.: Masked visual pre-training for motor control. arXiv preprint arXiv:2203.06173 (2022)"},{"key":"18_CR66","doi-asserted-by":"crossref","unstructured":"Xiong, H., Li, Q., Chen, Y.C., Bharadhwaj, H., Sinha, S., Garg, A.: Learning by watching: physical imitation of manipulation skills from human videos. arXiv (2021)","DOI":"10.1109\/IROS51168.2021.9636080"},{"key":"18_CR67","doi-asserted-by":"crossref","unstructured":"Xu, H., et al.: Unifying flow, stereo and depth estimation. IEEE Trans. Pattern Anal. Mach. Intell. (2023)","DOI":"10.1109\/TPAMI.2023.3298645"},{"key":"18_CR68","unstructured":"Yan, W., Hafner, D., James, S., Abbeel, P.: Temporally consistent transformers for video generation. arXiv preprint arXiv:2210.02396 (2022)"},{"key":"18_CR69","unstructured":"Yang, M., Du, Y., Ghasemipour, K., Tompson, J., Schuurmans, D., Abbeel, P.: Learning interactive real-world simulators. arXiv preprint arXiv:2310.06114 (2023)"},{"key":"18_CR70","unstructured":"Young, S., Gandhi, D., Tulsiani, S., Gupta, A., Abbeel, P., Pinto, L.: Visual imitation made easy. In: Conference on Robot Learning (CoRL) (2020)"},{"key":"18_CR71","doi-asserted-by":"crossref","unstructured":"Zhao, T.Z., Kumar, V., Levine, S., Finn, C.: Learning fine-grained bimanual manipulation with low-cost hardware. arXiv preprint arXiv:2304.13705 (2023)","DOI":"10.15607\/RSS.2023.XIX.016"},{"key":"18_CR72","doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Brox, T.: Learning to estimate 3d hand pose from single RGB images. In: CVPR (2017)","DOI":"10.1109\/ICCV.2017.525"},{"key":"18_CR73","unstructured":"Zitkovich, B., et\u00a0al.: Rt-2: vision-language-action models transfer web knowledge to robotic control. In: Conference on Robot Learning, pp. 2165\u20132183. PMLR (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73116-7_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T14:14:58Z","timestamp":1732976098000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73116-7_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031731150","9783031731167"],"references-count":73,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73116-7_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}