{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T04:07:40Z","timestamp":1769746060885,"version":"3.49.0"},"publisher-location":"Cham","reference-count":70,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733369","type":"print"},{"value":"9783031733376","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73337-6_24","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:02:27Z","timestamp":1730329347000},"page":"422-441","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Chameleon: A Data-Efficient Generalist for\u00a0Dense Visual Prediction in\u00a0the\u00a0Wild"],"prefix":"10.1007","author":[{"given":"Donggyun","family":"Kim","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Seongwoong","family":"Cho","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Semin","family":"Kim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chong","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Seunghoon","family":"Hong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"24_CR1","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Advances in Neural Information Processing Systems, vol. 35, pp. 23716\u201323736 (2022)"},{"key":"24_CR2","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., Schiele, B.: 2D human pose estimation: new benchmark and state of the art analysis. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2014 (2014)","DOI":"10.1109\/CVPR.2014.471"},{"key":"24_CR3","unstructured":"Awais, M., et al.: Foundational models defining a new era in vision: a survey and outlook. arXiv preprint arXiv:2307.13721 (2023)"},{"key":"24_CR4","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: BEit: BERT pre-training of image transformers. In: International Conference on Learning Representations (2022). https:\/\/openreview.net\/forum?id=p-BhZSz59o4"},{"key":"24_CR5","doi-asserted-by":"crossref","unstructured":"Bateni, P., Goyal, R., Masrani, V., Wood, F., Sigal, L.: Improved few-shot visual classification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14493\u201314502 (2020)","DOI":"10.1109\/CVPR42600.2020.01450"},{"issue":"3","key":"24_CR6","doi-asserted-by":"publisher","first-page":"346","DOI":"10.1016\/j.cviu.2007.09.014","volume":"110","author":"H Bay","year":"2008","unstructured":"Bay, H., Ess, A., Tuytelaars, T., Van Gool, L.: Speeded-up robust features (SURF). Comput. Vis. Image Underst. 110(3), 346\u2013359 (2008)","journal-title":"Comput. Vis. Image Underst."},{"key":"24_CR7","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Caesar, H., Uijlings, J., Ferrari, V.: COCO-stuff: thing and stuff classes in context. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1209\u20131218 (2018)","DOI":"10.1109\/CVPR.2018.00132"},{"key":"24_CR9","unstructured":"Chang, L., Yujie, Z., Andrew, Z., Weidi, X.: CounTR: transformer-based generalised visual counting. In: British Machine Vision Conference (BMVC) (2022)"},{"key":"24_CR10","unstructured":"Chen, L.C., Papandreou, G., Schroff, F., Adam, H.: Rethinking atrous convolution for semantic image segmentation. arXiv preprint arXiv:1706.05587 (2017)"},{"key":"24_CR11","doi-asserted-by":"crossref","unstructured":"Chen, T., Li, L., Saxena, S., Hinton, G., Fleet, D.J.: A generalist framework for panoptic segmentation of images and videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 909\u2013919 (2023)","DOI":"10.1109\/ICCV51070.2023.00090"},{"key":"24_CR12","doi-asserted-by":"publisher","unstructured":"Cheng, H.K., Schwing, A.G.: XMem: long-term video object segmentation with an Atkinson-Shiffrin memory model. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision, ECCV 2022, Part XXVIII. LNCS, vol. 13688, pp. 640\u2013658. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_37","DOI":"10.1007\/978-3-031-19815-1_37"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"Djukic, N., Lukezic, A., Zavrtanik, V., Kristan, M.: A low-shot object counting network with iterative prototype adaptation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 18872\u201318881 (2023)","DOI":"10.1109\/ICCV51070.2023.01730"},{"key":"24_CR14","unstructured":"Dosovitskiy, A., et al.: An image is worth 16$$\\times $$16 words: transformers for image recognition at scale. In: International Conference on Learning Representations (2021). https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"24_CR15","doi-asserted-by":"crossref","unstructured":"Fan, Q., Zhuo, W., Tang, C.K., Tai, Y.W.: Few-shot object detection with attention-RPN and multi-relation detector. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4013\u20134022 (2020)","DOI":"10.1109\/CVPR42600.2020.00407"},{"key":"24_CR16","doi-asserted-by":"crossref","unstructured":"Fonder, M., Droogenbroeck, M.V.: Mid-Air: a multi-modal dataset for extremely low altitude drone flights. In: Conference on Computer Vision and Pattern Recognition Workshop (CVPRW), June 2019 (2019)","DOI":"10.1109\/CVPRW.2019.00081"},{"key":"24_CR17","doi-asserted-by":"crossref","unstructured":"Geng, Z., et\u00a0al.: InstructDiffusion: a generalist modeling interface for vision tasks. arXiv preprint arXiv:2309.03895 (2023)","DOI":"10.1109\/CVPR52733.2024.01208"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Han, G., Ma, J., Huang, S., Chen, L., Chang, S.F.: Few-shot object detection with fully cross-transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5321\u20135330 (2022)","DOI":"10.1109\/CVPR52688.2022.00525"},{"key":"24_CR19","doi-asserted-by":"publisher","first-page":"102357","DOI":"10.1016\/j.media.2022.102357","volume":"77","author":"X He","year":"2022","unstructured":"He, X., Tan, E.L., Bi, H., Zhang, X., Zhao, S., Lei, B.: Fully transformer network for skin lesion analysis. Med. Image Anal. 77, 102357 (2022)","journal-title":"Med. Image Anal."},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Hinterstoisser, S., et al.: Multimodal templates for real-time detection of texture-less objects in heavily cluttered scenes. In: 2011 International Conference on Computer Vision, pp. 858\u2013865. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126326"},{"key":"24_CR21","doi-asserted-by":"publisher","unstructured":"Hong, S., Cho, S., Nam, J., Lin, S., Kim, S.: Cost aggregation with 4D convolutional Swin Transformer for few-shot segmentation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision, ECCV 2022. LNCS, vol. 13689, pp. 108\u2013126. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19818-2_7","DOI":"10.1007\/978-3-031-19818-2_7"},{"key":"24_CR22","unstructured":"Ibarz, B., et\u00a0al.: A generalist neural algorithmic learner. In: Learning on Graphs Conference, pp.\u00a02\u20131. PMLR (2022)"},{"key":"24_CR23","first-page":"358","volume":"23","author":"N Kanopoulos","year":"1988","unstructured":"Kanopoulos, N., Vasanthavada, N., Baker, R.L.: Design of an image edge detection filter using the Sobel operator. JSSC 23, 358\u2013367 (1988)","journal-title":"JSSC"},{"key":"24_CR24","unstructured":"Kim, D., Kim, J., Cho, S., Luo, C., Hong, S.: Universal few-shot learning of dense prediction tasks with visual token matching. In: The Eleventh International Conference on Learning Representations (2023)"},{"key":"24_CR25","unstructured":"Kolesnikov, A., Susano Pinto, A., Beyer, L., Zhai, X., Harmsen, J., Houlsby, N.: UViM: a unified modeling approach for vision with learned guiding codes. In: Advances in Neural Information Processing Systems, vol. 35, pp. 26295\u201326308 (2022)"},{"key":"24_CR26","doi-asserted-by":"publisher","first-page":"155","DOI":"10.1007\/s11263-008-0152-6","volume":"81","author":"V Lepetit","year":"2009","unstructured":"Lepetit, V., Moreno-Noguer, F., Fua, P.: EPnP: an accurate O(n) solution to the PnP problem. Int. J. Comput. Vis. 81, 155\u2013166 (2009). https:\/\/doi.org\/10.1007\/s11263-008-0152-6","journal-title":"Int. J. Comput. Vis."},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Li, H., et\u00a0al.: Uni-Perceiver v2: a generalist model for large-scale vision and vision-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2691\u20132700 (2023)","DOI":"10.1109\/CVPR52729.2023.00264"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, G., Ji, X.: CDPN: coordinates-based disentangled pose network for real-time RGB-based 6-DoF object pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7678\u20137687 (2019)","DOI":"10.1109\/ICCV.2019.00777"},{"key":"24_CR29","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2117\u20132125 (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"24_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014, Part V 13. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"24_CR31","unstructured":"Liu, L., Hamilton, W.L., Long, G., Jiang, J., Larochelle, H.: A universal representation transformer layer for few-shot image classification. In: International Conference on Learning Representations (2020)"},{"key":"24_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Z., Luo, P., Qiu, S., Wang, X., Tang, X.: DeepFashion: powering robust clothes recognition and retrieval with rich annotations. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2016 (2016)","DOI":"10.1109\/CVPR.2016.124"},{"key":"24_CR33","unstructured":"Lu, J., Clark, C., Zellers, R., Mottaghi, R., Kembhavi, A.: UNIFIED-IO: a unified model for vision, language, and multi-modal tasks. In: The Eleventh International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=E01k9048soZ"},{"key":"24_CR34","unstructured":"Mahdavi, S., Swersky, K., Kipf, T., Hashemi, M., Thrampoulidis, C., Liao, R.: Towards better out-of-distribution generalization of neural algorithmic reasoning tasks. Trans. Mach. Learn. Res. (2022)"},{"key":"24_CR35","unstructured":"Milton, M.A.A.: Automated skin lesion classification using ensemble of deep neural networks in ISIC 2018: skin lesion analysis towards melanoma detection challenge. arXiv preprint arXiv:1901.10802 (2019)"},{"key":"24_CR36","doi-asserted-by":"crossref","unstructured":"Min, J., Kang, D., Cho, M.: Hypercorrelation squeeze for few-shot segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6941\u20136952 (2021)","DOI":"10.1109\/ICCV48922.2021.00686"},{"key":"24_CR37","unstructured":"OpenAI, et al.: GPT-4 technical report. arXiv arxiv:2303.08774. View in Article 2 (2023)"},{"key":"24_CR38","unstructured":"Oquab, M., et\u00a0al.: DINOv2: learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"24_CR39","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In: Advances in Neural Information Processing Systems, vol. 35, pp. 27730\u201327744 (2022)"},{"key":"24_CR40","unstructured":"Peng, Z., Dong, L., Bao, H., Ye, Q., Wei, F.: BEiT v2: masked image modeling with vector-quantized visual tokenizers. arXiv preprint arXiv:2208.06366 (2022)"},{"key":"24_CR41","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Van Gool, L.: The 2017 DAVIS challenge on video object segmentation. arXiv arXiv:1704.00675 (2017)"},{"key":"24_CR42","doi-asserted-by":"crossref","unstructured":"Rad, M., Lepetit, V.: BB8: a scalable, accurate, robust to partial occlusion method for predicting the 3D poses of challenging objects without using depth. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3828\u20133836 (2017)","DOI":"10.1109\/ICCV.2017.413"},{"issue":"1","key":"24_CR43","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(1), 5485\u20135551 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"24_CR44","doi-asserted-by":"crossref","unstructured":"Ranftl, R., Bochkovskiy, A., Koltun, V.: Vision transformers for dense prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12179\u201312188 (2021)","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"24_CR45","doi-asserted-by":"crossref","unstructured":"Ranjan, V., Sharma, U., Nguyen, T., Hoai, M.: Learning to count everything. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition pp. 3394\u20133403 (2021)","DOI":"10.1109\/CVPR46437.2021.00340"},{"key":"24_CR46","unstructured":"Reed, S., et al.: A generalist agent. Transactions on Machine Learning Research (2022). https:\/\/openreview.net\/forum?id=1ikK0kHjvj, featured Certification"},{"key":"24_CR47","unstructured":"Rodionov, G., Prokhorenkova, L.: Neural algorithmic reasoning without intermediate supervision. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"24_CR48","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1007\/978-3-030-00934-2_30","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2018","author":"U Schmidt","year":"2018","unstructured":"Schmidt, U., Weigert, M., Broaddus, C., Myers, G.: Cell detection with star-convex polygons. In: Frangi, A.F., Schnabel, J.A., Davatzikos, C., Alberola-L\u00f3pez, C., Fichtinger, G. (eds.) MICCAI 2018, Part II 11. LNCS, vol. 11071, pp. 265\u2013273. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-00934-2_30"},{"key":"24_CR49","unstructured":"Schubert, I., et al.: A generalist dynamics model for control. arXiv preprint arXiv:2305.10912 (2023)"},{"key":"24_CR50","doi-asserted-by":"crossref","unstructured":"Shaban, A., Bansal, S., Liu, Z., Essa, I., Boots, B.: One-shot learning for semantic segmentation. In: BMVC (2017)","DOI":"10.5244\/C.31.167"},{"key":"24_CR51","unstructured":"Shridhar, M., Manuelli, L., Fox, D.: Perceiver-actor: a multi-task transformer for robotic manipulation. In: Conference on Robot Learning, pp. 785\u2013799. PMLR (2023)"},{"key":"24_CR52","unstructured":"Snell, J., Swersky, K., Zemel, R.: Prototypical networks for few-shot learning. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"24_CR53","unstructured":"Steder, B., Rusu, R.B., Konolige, K., Burgard, W.: NARF: 3D range image features for object recognition. In: Workshop on Defining and Solving Realistic Perception Problems in Personal Robotics at the IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), vol.\u00a044, p.\u00a02. Citeseer (2010)"},{"issue":"1","key":"24_CR54","doi-asserted-by":"publisher","first-page":"100","DOI":"10.1038\/s41592-020-01018-x","volume":"18","author":"C Stringer","year":"2021","unstructured":"Stringer, C., Wang, T., Michaelos, M., Pachitariu, M.: Cellpose: a generalist algorithm for cellular segmentation. Nat. Meth. 18(1), 100\u2013106 (2021)","journal-title":"Nat. Meth."},{"key":"24_CR55","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"24_CR56","doi-asserted-by":"publisher","unstructured":"Valanarasu, J.M.J., Patel, V.M.: UNeXt: MLP-based rapid medical image segmentation network. In: Wang, L., Dou, Q., Fletcher, P.T., Speidel, S., Li, S. (eds.) Medical Image Computing and Computer Assisted Intervention, MICCAI 2022. LNCS, vol. 13435, pp. 23\u201333. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-16443-9_3","DOI":"10.1007\/978-3-031-16443-9_3"},{"key":"24_CR57","unstructured":"Vaswani, A., et al: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"24_CR58","unstructured":"Vinyals, O., Blundell, C., Lillicrap, T., Wierstra, D., et\u00a0al.: Matching networks for one shot learning. In: Advances in Neural Information Processing Systems, vol. 29 (2016)"},{"key":"24_CR59","doi-asserted-by":"crossref","unstructured":"Wang, J., et al.: Look before you match: Instance understanding matters in video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2268\u20132278 (2023)","DOI":"10.1109\/CVPR52729.2023.00225"},{"key":"24_CR60","unstructured":"Wang, X., Huang, T., Gonzalez, J., Darrell, T., Yu, F.: Frustratingly simple few-shot object detection. In: International Conference on Machine Learning, pp. 9919\u20139928. PMLR (2020)"},{"key":"24_CR61","doi-asserted-by":"crossref","unstructured":"Wang, X., Wang, W., Cao, Y., Shen, C., Huang, T.: Images speak in images: a generalist painter for in-context visual learning. arXiv preprint arXiv:2212.02499 (2022)","DOI":"10.1109\/CVPR52729.2023.00660"},{"key":"24_CR62","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, X., Cao, Y., Wang, W., Shen, C., Huang, T.: SegGPT: towards segmenting everything in context. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1130\u20131140 (2023)","DOI":"10.1109\/ICCV51070.2023.00110"},{"key":"24_CR63","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"472","DOI":"10.1007\/978-3-030-01231-1_29","volume-title":"Computer Vision \u2013 ECCV 2018","author":"B Xiao","year":"2018","unstructured":"Xiao, B., Wu, H., Wei, Y.: Simple baselines for human pose estimation and tracking. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11210, pp. 472\u2013487. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01231-1_29"},{"key":"24_CR64","doi-asserted-by":"crossref","unstructured":"D Ye, H., Xu, D.: TaskPrompter: spatial-channel multi-task prompting for dense scene understanding. In: The Eleventh International Conference on Learning Representations (2022)","DOI":"10.1007\/978-3-031-19812-0_30"},{"key":"24_CR65","unstructured":"Yu, H., Xu, Y., Zhang, J., Zhao, W., Guan, Z., Tao, D.: AP-10K: a benchmark for animal pose estimation in the wild. arXiv preprint arXiv:2108.12617 (2021)"},{"key":"24_CR66","unstructured":"Zaken, E.B., Goldberg, Y., Ravfogel, S.: BitFit: simple parameter-efficient fine-tuning for transformer-based masked language-models. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), pp.\u00a01\u20139 (2022)"},{"key":"24_CR67","doi-asserted-by":"crossref","unstructured":"Zakharov, S., Shugurov, I., Ilic, S.: DPOD: 6D pose object detector and refiner. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1941\u20131950 (2019)","DOI":"10.1109\/ICCV.2019.00203"},{"key":"24_CR68","doi-asserted-by":"crossref","unstructured":"Zamir, A.R., Sax, A., Shen, W., Guibas, L.J., Malik, J., Savarese, S.: Taskonomy: disentangling task transfer learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3712\u20133722 (2018)","DOI":"10.1109\/CVPR.2018.00391"},{"key":"24_CR69","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., Torralba, A.: Scene parsing through ADE20K dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 633\u2013641 (2017)","DOI":"10.1109\/CVPR.2017.544"},{"key":"24_CR70","doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Ceylan, D., Yang, J., Russel, B., Argus, M., Brox, T.: FreiHAND: a dataset for markerless capture of hand pose and shape from single RGB images. In: IEEE International Conference on Computer Vision (ICCV) (2019). https:\/\/lmb.informatik.uni-freiburg.de\/projects\/freihand\/","DOI":"10.1109\/ICCV.2019.00090"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73337-6_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:07:32Z","timestamp":1730329652000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73337-6_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031733369","9783031733376"],"references-count":70,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73337-6_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}