{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T14:00:33Z","timestamp":1784210433278,"version":"3.55.0"},"publisher-location":"Cham","reference-count":69,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726903","type":"print"},{"value":"9783031726910","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72691-0_16","type":"book-chapter","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T18:05:12Z","timestamp":1730570712000},"page":"275-292","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["SPIN: Hierarchical Segmentation with\u00a0Subpart Granularity in\u00a0Natural Images"],"prefix":"10.1007","author":[{"given":"Josh","family":"Myers-Dean","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jarek","family":"Reynolds","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Brian","family":"Price","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifei","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Danna","family":"Gurari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,3]]},"reference":[{"key":"16_CR1","unstructured":"Explore images. https:\/\/support.apple.com\/guide\/iphone\/use-voiceover-for-images-and-videos-iph37e6b3844\/ios"},{"key":"16_CR2","unstructured":"Achiam, J., et\u00a0al.: GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"16_CR3","unstructured":"Berglund, L., et al.: The reversal curse: LLMs trained on \u201ca is b\u201d fail to learn \u201cb is a\u201d. arXiv preprint arXiv:2309.12288 (2023)"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Cai, M., et al.: Making large multimodal models understand arbitrary visual prompts. In: IEEE Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.01227"},{"key":"16_CR5","unstructured":"Chang, A.X., et\u00a0al.: ShapeNet: an information-rich 3D moel repository. arXiv preprint arXiv:1512.03012 (2015)"},{"key":"16_CR6","unstructured":"Chen, K., Zhang, Z., Zeng, W., Zhang, R., Zhu, F., Zhao, R.: Shikra: unleashing multimodal LLM\u2019s referential dialogue magic. arXiv preprint arXiv:2306.15195 (2023)"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Chen, X., Mottaghi, R., Liu, X., Fidler, S., Urtasun, R., Yuille, A.: Detect what you can: detecting and representing objects using holistic models and body parts. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1971\u20131978 (2014)","DOI":"10.1109\/CVPR.2014.254"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Deitke, M., et al.: RoboTHOR: an open simulation-to-real embodied AI platform. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00323"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Deng, B., Genova, K., Yazdani, S., Bouaziz, S., Hinton, G., Tagliasacchi, A.: CVXNet: learnable convex decomposition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 31\u201344 (2020)","DOI":"10.1109\/CVPR42600.2020.00011"},{"key":"16_CR10","unstructured":"Desai, K., Nickel, M., Rajpurohit, T., Johnson, J., Vedantam, S.R.: Hyperbolic image-text representations. In: International Conference on Machine Learning, pp. 7694\u20137731. PMLR (2023)"},{"key":"16_CR11","doi-asserted-by":"crossref","unstructured":"Ding, M., et al.: Visual dependency transformers: Dependency tree emerges from reversed attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14528\u201314539 (2023)","DOI":"10.1109\/CVPR52729.2023.01396"},{"key":"16_CR12","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16$$\\times $$16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"16_CR13","doi-asserted-by":"crossref","unstructured":"Geng, H., et al.: GAPartNet: cross-category domain-generalizable object perception and manipulation via generalizable and actionable parts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7081\u20137091 (2023)","DOI":"10.1109\/CVPR52729.2023.00684"},{"key":"16_CR14","doi-asserted-by":"crossref","unstructured":"de\u00a0Geus, D., Meletis, P., Lu, C., Wen, X., Dubbelman, G.: Part-aware panoptic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5485\u20135494 (2021)","DOI":"10.1109\/CVPR46437.2021.00544"},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Gong, K., Liang, X., Li, Y., Chen, Y., Yang, M., Lin, L.: Instance-level human parsing via part grouping network. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 770\u2013785 (2018)","DOI":"10.1007\/978-3-030-01225-0_47"},{"key":"16_CR16","doi-asserted-by":"crossref","unstructured":"He, J., Chen, J., Lin, M.X., Yu, Q., Yuille, A.L.: Compositor: bottom-up clustering and compositing for robust part and object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11259\u201311268 (2023)","DOI":"10.1109\/CVPR52729.2023.01083"},{"key":"16_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"128","DOI":"10.1007\/978-3-031-20074-8_8","volume-title":"Computer Vision \u2013 ECCV 2022","author":"J He","year":"2022","unstructured":"He, J., et al.: PartImageNet: a large, high-quality dataset of parts. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13668, pp. 128\u2013145. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20074-8_8"},{"key":"16_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"16_CR19","doi-asserted-by":"crossref","unstructured":"Hong, Y., Li, Q., Zhu, S.C., Huang, S.: VLGrammar: grounded grammar induction of vision and language. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1665\u20131674 (2021)","DOI":"10.1109\/ICCV48922.2021.00169"},{"key":"16_CR20","first-page":"17427","volume":"34","author":"Y Hong","year":"2021","unstructured":"Hong, Y., Yi, L., Tenenbaum, J., Torralba, A., Gan, C.: PTR: a benchmark for part-based conceptual, relational, and physical reasoning. Adv. Neural. Inf. Process. Syst. 34, 17427\u201317440 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"316","DOI":"10.1007\/978-3-030-58452-8_19","volume-title":"Computer Vision \u2013 ECCV 2020","author":"M Jia","year":"2020","unstructured":"Jia, M., et al.: Fashionpedia: ontology, segmentation, and an attribute localization dataset. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020, Part I. LNCS, vol. 12346, pp. 316\u2013332. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_19"},{"key":"16_CR22","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Koo, J., Huang, I., Achlioptas, P., Guibas, L.J., Sung, M.: PartGlot: learning shape part segmentation from language reference games. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16505\u201316514 (2022)","DOI":"10.1109\/CVPR52688.2022.01601"},{"key":"16_CR24","doi-asserted-by":"crossref","unstructured":"Lai, X., et al.: LISA: reasoning segmentation via large language model. arXiv preprint arXiv:2308.00692 (2023)","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"16_CR25","doi-asserted-by":"publisher","unstructured":"Lee, J., Peng, Y.H., Herskovitz, J., Guo, A.: Image explorer: multi-layered touch exploration to make images accessible. In: Proceedings of the 23rd International ACM SIGACCESS Conference on Computers and Accessibility, ASSETS 2021. Association for Computing Machinery, New York (2021). https:\/\/doi.org\/10.1145\/3441852.3476548","DOI":"10.1145\/3441852.3476548"},{"key":"16_CR26","unstructured":"Li, F., et al.: Semantic-SAM: segment and recognize anything at any granularity. arXiv preprint arXiv:2307.04767 (2023)"},{"key":"16_CR27","doi-asserted-by":"crossref","unstructured":"Li, L., Zhou, T., Wang, W., Li, J., Yang, Y.: Deep hierarchical semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1246\u20131257 (2022)","DOI":"10.1109\/CVPR52688.2022.00131"},{"key":"16_CR28","doi-asserted-by":"crossref","unstructured":"Li, T., Gupta, V., Mehta, M., Srikumar, V.: A logic-driven framework for consistency of neural models. arXiv preprint arXiv:1909.00126 (2019)","DOI":"10.18653\/v1\/D19-1405"},{"key":"16_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"729","DOI":"10.1007\/978-3-031-19812-0_42","volume-title":"Computer Vision \u2013 ECCV 2022","author":"X Li","year":"2022","unstructured":"Li, X., Xu, S., Yang, Y., Cheng, G., Tong, Y., Tao, D.: Panoptic-partformer: learning a unified model for panoptic part segmentation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13687, pp. 729\u2013747. Springer, Cham (2022)"},{"key":"16_CR30","doi-asserted-by":"crossref","unstructured":"Li, X., et al.: Panoptic-PartFormer++: a unified and decoupled view for panoptic part segmentation. arXiv preprint arXiv:2301.00954 (2023)","DOI":"10.1109\/TPAMI.2024.3453916"},{"issue":"4","key":"16_CR31","doi-asserted-by":"publisher","first-page":"871","DOI":"10.1109\/TPAMI.2018.2820063","volume":"41","author":"X Liang","year":"2018","unstructured":"Liang, X., Gong, K., Shen, X., Lin, L.: Look into person: joint body parsing & pose estimation network and a new benchmark. IEEE Trans. Pattern Anal. Mach. Intell. 41(4), 871\u2013885 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"16_CR32","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1007\/978-3-319-46448-0_8","volume-title":"Computer Vision \u2013 ECCV 2016","author":"X Liang","year":"2016","unstructured":"Liang, X., Shen, X., Feng, J., Lin, L., Yan, S.: Semantic object parsing with graph LSTM. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016, Part I. LNCS, vol. 9905, pp. 125\u2013143. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_8"},{"key":"16_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014, Part V. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"16_CR34","unstructured":"Liu, J., Min, S., Zettlemoyer, L., Choi, Y., Hajishirzi, H.: Infini-gram: scaling unbounded n-gram language models to a trillion tokens. arXiv preprint arXiv:2401.17377 (2024)"},{"key":"16_CR35","unstructured":"Liu, Q., et al.: CGPart: a part segmentation dataset based on 3D computer graphics models. arXiv preprint arXiv:2103.14098 (2021)"},{"key":"16_CR36","doi-asserted-by":"crossref","unstructured":"Martin, D., Fowlkes, C., Tal, D., Malik, J.: A database of human segmented natural images and its application to evaluating segmentation algorithms and measuring ecological statistics. In: Proceedings Eighth IEEE International Conference on Computer Vision, ICCV 2001, vol.\u00a02, pp. 416\u2013423. IEEE (2001)","DOI":"10.1109\/ICCV.2001.937655"},{"issue":"11","key":"16_CR37","doi-asserted-by":"publisher","first-page":"2797","DOI":"10.1007\/s11263-022-01671-z","volume":"130","author":"U Michieli","year":"2022","unstructured":"Michieli, U., Zanuttigh, P.: Edge-aware graph matching network for part-based semantic segmentation. Int. J. Comput. Vision 130(11), 2797\u20132821 (2022)","journal-title":"Int. J. Comput. Vision"},{"issue":"11","key":"16_CR38","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1145\/219717.219748","volume":"38","author":"GA Miller","year":"1995","unstructured":"Miller, G.A.: WordNet: a lexical database for English. Commun. ACM 38(11), 39\u201341 (1995). https:\/\/doi.org\/10.1145\/219717.219748","journal-title":"Commun. ACM"},{"key":"16_CR39","doi-asserted-by":"crossref","unstructured":"Mo, K., et al.: PartNet: a large-scale benchmark for fine-grained and hierarchical part-level 3d object understanding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 909\u2013918 (2019)","DOI":"10.1109\/CVPR.2019.00100"},{"key":"16_CR40","doi-asserted-by":"crossref","unstructured":"Mo, K., et al.: PartNet: a large-scale benchmark for fine-grained and hierarchical part-level 3D object understanding. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00100"},{"key":"16_CR41","doi-asserted-by":"crossref","unstructured":"Myers-Dean, J., Fan, Y., Price, B., Chan, W., Gurari, D.: Interactive segmentation for diverse gesture types without context. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 7198\u20137208 (2024)","DOI":"10.1109\/WACV57701.2024.00703"},{"key":"16_CR42","doi-asserted-by":"publisher","unstructured":"Nair, V., Zhu, H.H., Smith, B.A.: ImageAssist: tools for enhancing touchscreen-based image exploration systems for blind and low vision users. In: Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems, CHI 2023. Association for Computing Machinery, New York (2023). https:\/\/doi.org\/10.1145\/3544548.3581302","DOI":"10.1145\/3544548.3581302"},{"key":"16_CR43","unstructured":"Peng, Z., et al.: Kosmos-2: grounding multimodal large language models to the world. arXiv preprint arXiv:2306.14824 (2023)"},{"key":"16_CR44","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"16_CR45","doi-asserted-by":"crossref","unstructured":"Ramanathan, V., et\u00a0al.: PACO: parts and attributes of common objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7141\u20137151 (2023)","DOI":"10.1109\/CVPR52729.2023.00690"},{"key":"16_CR46","doi-asserted-by":"crossref","unstructured":"Rasheed, H., et al.: GLaMM: pixel grounding large multimodal model. arXiv preprint arXiv:2311.03356 (2023)","DOI":"10.1109\/CVPR52733.2024.01236"},{"key":"16_CR47","doi-asserted-by":"crossref","unstructured":"Raymond, W., Gibbs, J., Matlock, T.: Psycholinguistics and mental representations (2000)","DOI":"10.1515\/cogl.2000.003"},{"key":"16_CR48","doi-asserted-by":"crossref","unstructured":"Ren, Z., et al.: PixeLLM: pixel reasoning with large multimodal model (2023)","DOI":"10.1109\/CVPR52733.2024.02491"},{"key":"16_CR49","doi-asserted-by":"crossref","unstructured":"Sennrich, R., Haddow, B., Birch, A.: Neural machine translation of rare words with subword units. arXiv preprint arXiv:1508.07909 (2015)","DOI":"10.18653\/v1\/P16-1162"},{"key":"16_CR50","doi-asserted-by":"crossref","unstructured":"Song, X., et al.: ApolloCar3D: a large 3D car instance understanding benchmark for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5452\u20135462 (2019)","DOI":"10.1109\/CVPR.2019.00560"},{"key":"16_CR51","doi-asserted-by":"crossref","unstructured":"Sun, P., et al.: Going denser with open-vocabulary part segmentation. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 15407\u201315419 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258762519","DOI":"10.1109\/ICCV51070.2023.01417"},{"key":"16_CR52","doi-asserted-by":"crossref","unstructured":"Tang, C., Xie, L., Zhang, X., Hu, X., Tian, Q.: Visual recognition by request. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15265\u201315274 (2023)","DOI":"10.1109\/CVPR52729.2023.01465"},{"key":"16_CR53","doi-asserted-by":"publisher","first-page":"104471","DOI":"10.1016\/j.imavis.2022.104471","volume":"123","author":"K Tong","year":"2022","unstructured":"Tong, K., Wu, Y.: Deep learning-based detection from the perspective of small or tiny objects: a survey. Image Vis. Comput. 123, 104471 (2022)","journal-title":"Image Vis. Comput."},{"key":"16_CR54","unstructured":"Touvron, H., et al.: LLaMA: open and efficient foundation language models (2023)"},{"key":"16_CR55","unstructured":"Tsogkas, S., Kokkinos, I., Papandreou, G., Vedaldi, A.: Deep learning for semantic part segmentation with high-level guidance. arXiv preprint arXiv:1505.02438 (2015)"},{"key":"16_CR56","unstructured":"Wah, C., Branson, S., Welinder, P., Perona, P., Belongie, S.: The caltech-UCSD birds-200-2011 dataset (2011)"},{"key":"16_CR57","doi-asserted-by":"crossref","unstructured":"Wang, J., Yuille, A.L.: Semantic part segmentation using compositional model combining shape and appearance. In: Proceedings of the IEEE Conference on Computer Vision And Pattern Recognition, pp. 1788\u20131797 (2015)","DOI":"10.1109\/CVPR.2015.7298788"},{"key":"16_CR58","doi-asserted-by":"crossref","unstructured":"Wang, P., Shen, X., Lin, Z., Cohen, S., Price, B., Yuille, A.L.: Joint object and part segmentation using deep learned potentials. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1573\u20131581 (2015)","DOI":"10.1109\/ICCV.2015.184"},{"key":"16_CR59","unstructured":"Wang, W., et\u00a0al.: CogVLM: visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)"},{"key":"16_CR60","unstructured":"Wang, X., Li, S., Kallidromitis, K., Kato, Y., Kozuka, K., Darrell, T.: Hierarchical open-vocabulary universal image segmentation. Adv. Neural Inf. Process. Syst. 36 (2024)"},{"key":"16_CR61","unstructured":"Wei, M., Yue, X., Zhang, W., Kong, S., Liu, X., Pang, J.: OV-PARTS: towards open-vocabulary part segmentation. Adv. Neural Inf. Process. Syst 36 (2024)"},{"key":"16_CR62","doi-asserted-by":"crossref","unstructured":"Wu, T.H., et al.: See, say, and segment: teaching LMMs to overcome false premises. arXiv preprint arXiv:2312.08366 (2023)","DOI":"10.1109\/CVPR52733.2024.01278"},{"key":"16_CR63","doi-asserted-by":"crossref","unstructured":"Xiang, F., et al.: SAPIEN: a simulated part-based interactive environment. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01111"},{"key":"16_CR64","unstructured":"You, H., et al.: FERRET: refer and ground anything anywhere at any granularity. arXiv preprint arXiv:2310.07704 (2023)"},{"key":"16_CR65","doi-asserted-by":"crossref","unstructured":"Yu, F., Liu, K., Zhang, Y., Zhu, C., Xu, K.: PartNet: a recursive part decomposition network for fine-grained and hierarchical shape segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9491\u20139500 (2019)","DOI":"10.1109\/CVPR.2019.00972"},{"key":"16_CR66","doi-asserted-by":"crossref","unstructured":"Yuan, Y., et al.: Osprey: pixel understanding with visual instruction tuning (2023)","DOI":"10.1109\/CVPR52733.2024.02664"},{"key":"16_CR67","doi-asserted-by":"crossref","unstructured":"Zhao, J., Li, J., Cheng, Y., Sim, T., Yan, S., Feng, J.: Understanding humans in crowded scenes: deep nested adversarial learning and a new benchmark for multi-human parsing. In: Proceedings of the 26th ACM international conference on Multimedia, pp. 792\u2013800 (2018)","DOI":"10.1145\/3240508.3240509"},{"key":"16_CR68","doi-asserted-by":"crossref","unstructured":"Zheng, S., Yang, F., Kiapour, M.H., Piramuthu, R.: ModaNet: a large-scale street fashion dataset with polygon annotations. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 1670\u20131678 (2018)","DOI":"10.1145\/3240508.3240652"},{"key":"16_CR69","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., Torralba, A.: Scene parsing through ade20k dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 633\u2013641 (2017)","DOI":"10.1109\/CVPR.2017.544"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72691-0_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T18:08:22Z","timestamp":1730570902000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72691-0_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,3]]},"ISBN":["9783031726903","9783031726910"],"references-count":69,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72691-0_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,3]]},"assertion":[{"value":"3 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}