{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T17:25:42Z","timestamp":1783099542913,"version":"3.54.6"},"publisher-location":"Cham","reference-count":52,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031919060","type":"print"},{"value":"9783031919077","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91907-7_24","type":"book-chapter","created":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T11:46:24Z","timestamp":1748346384000},"page":"401-418","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["TaskCLIP: Extend Large Vision-Language Model for\u00a0Task Oriented Object Detection"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1956-5135","authenticated-orcid":false,"given":"Hanning","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenjun","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Ni","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sanggeon","family":"Yun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yezi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei","family":"Wen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alvaro","family":"Velasquez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hugo","family":"Latapie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5761-0622","authenticated-orcid":false,"given":"Mohsen","family":"Imani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"24_CR1","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Advances in Neural Information Processing Systems, vol. 35, pp. 23716\u201323736 (2022)"},{"key":"24_CR2","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"24_CR3","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"24_CR4","doi-asserted-by":"crossref","unstructured":"Cherti, M., et al.: Reproducible scaling laws for contrastive language-image learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132829 (2023)","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"24_CR5","doi-asserted-by":"crossref","unstructured":"Cordts, M., et al.: The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3213\u20133223 (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"24_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Yang, Z., Chen, T., Zhou, W., Li, H.: TransVG: end-to-end visual grounding with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1769\u20131779 (2021)","DOI":"10.1109\/ICCV48922.2021.00179"},{"key":"24_CR7","unstructured":"Du, D., et\u00a0al.: Visdrone-det2019: the vision meets drone object detection in image challenge results. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops (2019)"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Dvornik, N., Mairal, J., Schmid, C.: Modeling visual context is key to augmenting object detection datasets. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 364\u2013380 (2018)","DOI":"10.1007\/978-3-030-01258-8_23"},{"key":"24_CR9","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C.K., Winn, J., Zisserman, A.: The pascal visual object classes (VOC) challenge. Int. J. Comput. Vision 88, 303\u2013338 (2010)","journal-title":"Int. J. Comput. Vision"},{"key":"24_CR10","doi-asserted-by":"crossref","unstructured":"Gao, P., et al.: Clip-adapter: better vision-language models with feature adapters. Int. J. Comput. Vis. 1\u201315 (2023)","DOI":"10.1007\/s11263-023-01891-x"},{"issue":"2","key":"24_CR11","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao, P., et al.: Clip-adapter: better vision-language models with feature adapters. Int. J. Comput. Vision 132(2), 581\u2013595 (2024)","journal-title":"Int. J. Comput. Vision"},{"key":"24_CR12","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 580\u2013587 (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"24_CR14","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S.J., Tao, D.: Knowledge distillation: a survey. Int. J. Comput. Vision 129, 1789\u20131819 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"24_CR15","doi-asserted-by":"crossref","unstructured":"Goyal, S., Kumar, A., Garg, S., Kolter, Z., Raghunathan, A.: Finetune like you pretrain: improved finetuning of zero-shot vision models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19338\u201319347 (2023)","DOI":"10.1109\/CVPR52729.2023.01853"},{"key":"24_CR16","unstructured":"Huang, W., et al.: Ecosense: energy-efficient intelligent sensing for in-shore ship detection through edge-cloud collaboration. arXiv preprint arXiv:2403.14027 (2024)"},{"key":"24_CR17","unstructured":"Huang, W., et al.: A plug-in tiny AI module for intelligent and selective sensor data transmission. arXiv preprint arXiv:2402.02043 (2024)"},{"issue":"3","key":"24_CR18","doi-asserted-by":"publisher","first-page":"216","DOI":"10.1016\/j.compbiolchem.2009.04.004","volume":"33","author":"P Jain","year":"2009","unstructured":"Jain, P., Garibaldi, J.M., Hirst, J.D.: Supervised machine learning algorithms for protein structure classification. Comput. Biol. Chem. 33(3), 216\u2013223 (2009)","journal-title":"Comput. Biol. Chem."},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"James, W., Stein, C.: Estimation with quadratic loss. In: Breakthroughs in Statistics: Foundations and Basic Theory, pp. 443\u2013460. Springer (1992)","DOI":"10.1007\/978-1-4612-0919-5_30"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Jin, X., Pei, K., Won, J.Y., Lin, Z.: Symlm: predicting function names in stripped binaries via context-sensitive execution-aware code embeddings. In: 2022 ACM SIGSAC Conference on Computer and Communications Security. ACM (2022)","DOI":"10.1145\/3548606.3560612"},{"key":"24_CR21","unstructured":"Jin, X., Wang, Y.: Understand legal documents with contextualized large language models (2023)"},{"key":"24_CR22","unstructured":"Jocher, G., Chaurasia, A., Qiu, J.: Ultralytics YOLO (2023). https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"24_CR23","unstructured":"Jocher, G., et\u00a0al.: ultralytics\/yolov5: v3. 0. Zenodo (2020)"},{"key":"24_CR24","doi-asserted-by":"crossref","unstructured":"Kamath, A., Singh, M., LeCun, Y., Synnaeve, G., Misra, I., Carion, N.: Mdetr-modulated detection for end-to-end multi-modal understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1780\u20131790 (2021)","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"24_CR25","doi-asserted-by":"crossref","unstructured":"Li, P., Lin, Y., Schultz-Fellenz, E.: Contextual hourglass network for semantic segmentation of high resolution aerial imagery (2023)","DOI":"10.1109\/ICECAI62591.2024.10674750"},{"key":"24_CR26","unstructured":"Li, P., et al.: Toist: task oriented instance segmentation transformer with noun-pronoun distillation. In: Advances in Neural Information Processing Systems, vol. 35, pp. 17597\u201317611 (2022)"},{"key":"24_CR27","unstructured":"Li, Y., Tarlow, D., Brockschmidt, M., Zemel, R.: Gated graph sequence neural networks. arXiv preprint arXiv:1511.05493 (2015)"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Liang, F., et al.: Open-vocabulary semantic segmentation with mask-adapted clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7061\u20137070 (2023)","DOI":"10.1109\/CVPR52729.2023.00682"},{"key":"24_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"24_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/978-3-319-46448-0_2","volume-title":"Computer Vision \u2013 ECCV 2016","author":"W Liu","year":"2016","unstructured":"Liu, W., et al.: SSD: single shot multibox detector. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 21\u201337. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_2"},{"key":"24_CR31","unstructured":"Liu, Y., et al.: Roberta: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"24_CR32","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"issue":"3","key":"24_CR33","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1108\/LHTN-01-2023-0009","volume":"40","author":"BD Lund","year":"2023","unstructured":"Lund, B.D., Wang, T.: Chatting about ChatGPT: how may AI and GPT impact academia and libraries? Library Hi Tech News 40(3), 26\u201329 (2023)","journal-title":"Library Hi Tech News"},{"key":"24_CR34","unstructured":"Lyu, W., et al.: A multimodal transformer: fusing clinical notes with structured EHR data for interpretable in-hospital mortality prediction. In: AMIA Annual Symposium Proceedings (2022)"},{"key":"24_CR35","unstructured":"Mokady, R., Hertz, A., Bermano, A.H.: Clipcap: clip prefix for image captioning. arXiv preprint arXiv:2111.09734 (2021)"},{"key":"24_CR36","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In: Advances in Neural Information Processing Systems, vol. 35, pp. 27730\u201327744 (2022)"},{"key":"24_CR37","unstructured":"Paszke, A., et\u00a0al.: Pytorch: an imperative style, high-performance deep learning library. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"24_CR38","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"24_CR39","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 779\u2013788 (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"24_CR40","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"24_CR41","doi-asserted-by":"crossref","unstructured":"Sawatzky, J., Souri, Y., Grund, C., Gall, J.: What object should i use?-task driven object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7605\u20137614 (2019)","DOI":"10.1109\/CVPR.2019.00779"},{"key":"24_CR42","doi-asserted-by":"crossref","unstructured":"Sch\u00f6n, M., Buchholz, M., Dietmayer, K.: MGNet: monocular geometric scene understanding for autonomous driving. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15804\u201315815 (2021)","DOI":"10.1109\/ICCV48922.2021.01551"},{"key":"24_CR43","doi-asserted-by":"crossref","unstructured":"Shi, H., Hayat, M., Wu, Y., Cai, J.: Proposalclip: unsupervised open-category object proposal generation via exploiting clip cues. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9611\u20139620 (2022)","DOI":"10.1109\/CVPR52688.2022.00939"},{"key":"24_CR44","doi-asserted-by":"crossref","unstructured":"Tang, J., Zheng, G., Yu, J., Yang, S.: CoTDet: affordance knowledge prompting for task driven object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3068\u20133078 (2023)","DOI":"10.1109\/ICCV51070.2023.00285"},{"key":"24_CR45","unstructured":"Wang, M., Xing, J., Liu, Y.: Actionclip: a new paradigm for video action recognition. arXiv preprint arXiv:2109.08472 (2021)"},{"key":"24_CR46","doi-asserted-by":"crossref","unstructured":"Wang, Z., Li, T., Zheng, J.Q., Huang, B.: When CNN meet with ViT: towards semi-supervised learning for multi-class medical image semantic segmentation. In: ECCV 2022 Workshops. Springer Nature Switzerland (2023)","DOI":"10.1007\/978-3-031-25082-8_28"},{"key":"24_CR47","unstructured":"Wu, Y., Kirillov, A., Massa, F., Lo, W.Y., Girshick, R.: Detectron2 (2019). https:\/\/github.com\/facebookresearch\/detectron2"},{"key":"24_CR48","doi-asserted-by":"crossref","unstructured":"Xiao, T., Liu, Y., Zhou, B., Jiang, Y., Sun, J.: Unified perceptual parsing for scene understanding. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 418\u2013434 (2018)","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"24_CR49","doi-asserted-by":"crossref","unstructured":"Yun, S., et al.: Hypersense: hyperdimensional intelligent sensing for energy-efficient sparse data processing. Adv. Intell. Syst. 2400228 (2024)","DOI":"10.1002\/aisy.202400228"},{"key":"24_CR50","doi-asserted-by":"crossref","unstructured":"Zhong, Y., et\u00a0al.: Regionclip: region-based language-image pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16793\u201316803 (2022)","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"24_CR51","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Lei, Y., Zhang, B., Liu, L., Liu, Y.: Zegclip: towards adapting clip for zero-shot semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11175\u201311185 (2023)","DOI":"10.1109\/CVPR52729.2023.01075"},{"key":"24_CR52","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable detr: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91907-7_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T16:13:23Z","timestamp":1757175203000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91907-7_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031919060","9783031919077"],"references-count":52,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91907-7_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}