{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T11:25:07Z","timestamp":1762341907031,"version":"3.37.3"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2022,8,6]],"date-time":"2022-08-06T00:00:00Z","timestamp":1659744000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,8,6]],"date-time":"2022-08-06T00:00:00Z","timestamp":1659744000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"crossref","award":["201906280464"],"award-info":[{"award-number":["201906280464"]}],"id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["2018AAA0101501"],"award-info":[{"award-number":["2018AAA0101501"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61375040"],"award-info":[{"award-number":["61375040"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61772415"],"award-info":[{"award-number":["61772415"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2023,2]]},"DOI":"10.1007\/s11042-022-13592-7","type":"journal-article","created":{"date-parts":[[2022,8,6]],"date-time":"2022-08-06T06:03:51Z","timestamp":1659765831000},"page":"6557-6579","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Fine-grained label learning in object detection with weak supervision of captions"],"prefix":"10.1007","volume":"82","author":[{"given":"Xue","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1714-3433","authenticated-orcid":false,"given":"Youtian","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Suzan","family":"Verberne","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fons J.","family":"Verbeek","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,6]]},"reference":[{"issue":"2","key":"13592_CR1","doi-asserted-by":"publisher","first-page":"1143","DOI":"10.1007\/s42835-020-00650-z","volume":"16","author":"A Ahmed","year":"2021","unstructured":"Ahmed A, Jalal A, Kim K (2021) Multi-objects detection and segmentation for scene understanding based on texton forest and kernel sliding perceptron. J Electr Eng Technol 16(2):1143\u20131150","journal-title":"J Electr Eng Technol"},{"doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6077\u20136086","key":"13592_CR2","DOI":"10.1109\/CVPR.2018.00636"},{"doi-asserted-by":"crossref","unstructured":"Bengio Y, Louradour J, Collobert R, Weston J (2009) Curriculum learning. In: Proceedings of the 26th annual international conference on machine learning, pp 41\u201348","key":"13592_CR3","DOI":"10.1145\/1553374.1553380"},{"doi-asserted-by":"crossref","unstructured":"Bhujade S, Kamaleshwar T, Jaiswal S, Babu DV (2022) Deep learning application of image recognition based on self-driving vehicle. In: International conference on emerging technologies in computer engineering, Springer, pp 336\u2013344","key":"13592_CR4","DOI":"10.1007\/978-3-031-07012-9_29"},{"doi-asserted-by":"crossref","unstructured":"Bilen H, Vedaldi A (2016) Weakly supervised deep detection networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2846\u20132854","key":"13592_CR5","DOI":"10.1109\/CVPR.2016.311"},{"doi-asserted-by":"crossref","unstructured":"Buonviri A, York M, LeGrand K, Meub J (2019) Survey of challenges in labeled random finite set distributed multi-sensor multi-object tracking. In: 2019 IEEE Aerospace Conference, IEEE, pp 1\u201312","key":"13592_CR6","DOI":"10.1109\/AERO.2019.8742216"},{"doi-asserted-by":"publisher","unstructured":"Diba A, Sharma V, Pazandeh A, Pirsiavash H, Van Gool L (2017) Weakly supervised cascaded convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 914\u2013922. https:\/\/doi.org\/10.1109\/CVPR.2017.545","key":"13592_CR7","DOI":"10.1109\/CVPR.2017.545"},{"issue":"5","key":"13592_CR8","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1007\/s11265-018-1355-x","volume":"91","author":"W Du","year":"2019","unstructured":"Du W, Phlypo R, Adal\u0131 T (2019) Adaptive feature selection and feature fusion for semi-supervised classification. J Signal Process Syst 91(5):521\u2013537","journal-title":"J Signal Process Syst"},{"issue":"2","key":"13592_CR9","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham M, Van Gool L, Williams C K, Winn J, Zisserman A (2010) The pascal visual object classes (voc) challenge. Int J Comput Vis 88(2):303\u2013338","journal-title":"Int J Comput Vis"},{"doi-asserted-by":"crossref","unstructured":"Fang H, Gupta S, Iandola F, Srivastava RK, Deng L, Doll\u00e1r P, Gao J, He X, Mitchell M, Platt JC, et al. (2015) From captions to visual concepts and back. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1473\u20131482","key":"13592_CR10","DOI":"10.1109\/CVPR.2015.7298754"},{"doi-asserted-by":"crossref","unstructured":"Ge W, Yang S, Yu Y (2018) Multi-evidence filtering and fusion for multi-label classification, object detection and semantic segmentation based on weakly supervised learning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1277\u20131286","key":"13592_CR11","DOI":"10.1109\/CVPR.2018.00139"},{"doi-asserted-by":"crossref","unstructured":"Guo S, Huang W, Zhang H, Zhuang C, Dong D, Scott MR, Huang D (2018) CurriculumNet: weakly supervised learning from large-scale web images. In: Proceedings of the european conference on computer vision (ECCV), pp 135\u2013150","key":"13592_CR12","DOI":"10.1007\/978-3-030-01249-6_9"},{"unstructured":"Hacohen G, Weinshall D (2019) On the power of curriculum learning in training deep networks. arXiv:190403626","key":"13592_CR13"},{"unstructured":"Jerbi A, Herzig R, Berant J, Chechik G, Globerson A (2020) Learning object detection from captions via textual scene attributes. arXiv:200914558","key":"13592_CR14"},{"doi-asserted-by":"crossref","unstructured":"Kantorov V, Oquab M, Cho M, Laptev I (2016) ContextLocNet: context-aware deep network models for weakly supervised localization. In: European conference on computer vision, Springer, pp 350\u2013365","key":"13592_CR15","DOI":"10.1007\/978-3-319-46454-1_22"},{"doi-asserted-by":"crossref","unstructured":"Krause J, Johnson J, Krishna R, Fei-Fei L (2017) A hierarchical approach for generating descriptive image paragraphs. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 317\u2013325","key":"13592_CR16","DOI":"10.1109\/CVPR.2017.356"},{"issue":"1","key":"13592_CR17","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, Johnson J, Hata K, Kravitz J, Chen S, Kalantidis Y, Li L J, Shamma D A, et al. (2017) Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"doi-asserted-by":"crossref","unstructured":"Li C, Ma T, Zhou Y, Cheng J, Xu B (2017) Measuring word semantic similarity based on transferred vectors. In: International conference on neural information processing, Springer, pp 326\u2013335","key":"13592_CR18","DOI":"10.1007\/978-3-319-70093-9_34"},{"doi-asserted-by":"crossref","unstructured":"Lin TY, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: common objects in context. In: European conference on computer vision, Springer, pp 740\u2013755","key":"13592_CR19","DOI":"10.1007\/978-3-319-10602-1_48"},{"doi-asserted-by":"crossref","unstructured":"Manning CD, Surdeanu M, Bauer J, Finkel JR, Bethard S, McClosky D (2014) The Stanford CoreNLP natural language processing toolkit. In: Proceedings of 52nd annual meeting of the association for computational linguistics: system demonstrations, pp 55\u201360","key":"13592_CR20","DOI":"10.3115\/v1\/P14-5010"},{"unstructured":"Mikolov T, Chen K, Corrado G, Dean J (2013a) Efficient estimation of word representations in vector space. arXiv:13013781","key":"13592_CR21"},{"unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS, Dean J (2013b) Distributed representations of words and phrases and their compositionality. In: Advances in neural information processing systems, pp 3111\u20133119","key":"13592_CR22"},{"doi-asserted-by":"crossref","unstructured":"Misra I, Lawrence Zitnick C, Mitchell M, Girshick R (2016) Seeing through the human reporting bias: visual classifiers from noisy human-centric labels. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2930\u20132939","key":"13592_CR23","DOI":"10.1109\/CVPR.2016.320"},{"doi-asserted-by":"crossref","unstructured":"Oquab M, Bottou L, Laptev I, Sivic J (2015) Is object localization for free?-Weakly-supervised learning with convolutional neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 685\u2013694","key":"13592_CR24","DOI":"10.1109\/CVPR.2015.7298668"},{"doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R, Farhadi A (2016) You only look once: Unified, real-time object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 779\u2013788","key":"13592_CR25","DOI":"10.1109\/CVPR.2016.91"},{"unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster R-CNN: Towards real-time object detection with region proposal networks. In: Advances in neural information processing systems, pp 91\u201399","key":"13592_CR26"},{"doi-asserted-by":"crossref","unstructured":"Song Y, Soleymani M (2019) Polysemous visual-semantic embedding for cross-modal retrieval. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1979\u20131988","key":"13592_CR27","DOI":"10.1109\/CVPR.2019.00208"},{"doi-asserted-by":"crossref","unstructured":"Tang P, Wang X, Bai X, Liu W (2017) Multiple instance detection network with online instance classifier refinement. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2843\u20132851","key":"13592_CR28","DOI":"10.1109\/CVPR.2017.326"},{"issue":"1","key":"13592_CR29","doi-asserted-by":"publisher","first-page":"176","DOI":"10.1109\/TPAMI.2018.2876304","volume":"42","author":"P Tang","year":"2018","unstructured":"Tang P, Wang X, Bai S, Shen W, Bai X, Liu W, Yuille A (2018) PCL: proposal cluster learning for weakly supervised object detection. IEEE Trans Pattern Anal Mach Intell 42(1):176\u2013191","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"doi-asserted-by":"crossref","unstructured":"Teney D, Anderson P, He X, Van Den Hengel A (2018) Tips and tricks for visual question answering: learnings from the 2017 challenge. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4223\u20134232","key":"13592_CR30","DOI":"10.1109\/CVPR.2018.00444"},{"unstructured":"Thomas C, Kovashka A (2019) Predicting the politics of an image using webly supervised data. In: Advances in neural information processing systems, pp 3630\u20133642","key":"13592_CR31"},{"issue":"06","key":"13592_CR32","first-page":"602","volume":"28","author":"Jl Tian","year":"2010","unstructured":"Tian Jl, Zhao W (2010) Words similarity algorithm based on Tongyici Cilin in semantic web adaptive learning system. J Jilin University (Inf Sci Ed) 28 (06):602\u2013608","journal-title":"J Jilin University (Inf Sci Ed)"},{"doi-asserted-by":"crossref","unstructured":"Wan F, Wei P, Jiao J, Han Z, Ye Q (2018) Min-entropy latent model for weakly supervised object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1297\u20131306","key":"13592_CR33","DOI":"10.1109\/CVPR.2018.00141"},{"doi-asserted-by":"crossref","unstructured":"Wang J, Wang X, Liu W (2018) Weakly- and semi-supervised Faster R-CNN with curriculum learning. In: 2018 24th International Conference on Pattern Recognition (ICPR), IEEE, pp 2416\u20132421","key":"13592_CR34","DOI":"10.1109\/ICPR.2018.8546088"},{"doi-asserted-by":"crossref","unstructured":"Wei Y, Shen Z, Cheng B, Shi H, Xiong J, Feng J, Huang T (2018) TS2C: tight box mining with surrounding segmentation context for weakly supervised object detection. In: Proceedings of the european conference on computer vision (ECCV), pp 434\u2013450","key":"13592_CR35","DOI":"10.1007\/978-3-030-01252-6_27"},{"doi-asserted-by":"crossref","unstructured":"Ye K, Zhang M, Kovashka A, Li W, Qin D, Berent J (2019) Cap2Det: learning to amplify weak caption supervision for object detection. In: Proceedings of the IEEE international conference on computer vision, pp 9686\u20139695","key":"13592_CR36","DOI":"10.1109\/ICCV.2019.00978"},{"issue":"18","key":"13592_CR37","doi-asserted-by":"publisher","first-page":"27423","DOI":"10.1007\/s11042-021-11038-0","volume":"80","author":"J Zakraoui","year":"2021","unstructured":"Zakraoui J, Saleh M, Al-Maadeed S, Jaam JM (2021) Improving text-to-image generation with object layout guidance. Multimed Tools Appl 80(18):27423\u201327443","journal-title":"Multimed Tools Appl"},{"unstructured":"Zhang M, Hwa R, Kovashka A (2018) Equal but not the same: understanding the implicit relationship between persuasive images and text. arXiv:180708205","key":"13592_CR38"},{"doi-asserted-by":"crossref","unstructured":"Zhang X, Wei Y, Feng J, Yang Y, Huang TS (2018) Adversarial complementary learning for weakly supervised object localization. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1325\u20131334","key":"13592_CR39","DOI":"10.1109\/CVPR.2018.00144"},{"doi-asserted-by":"crossref","unstructured":"Zhou B, Khosla A, Lapedriza A, Oliva A, Torralba A (2016) Learning deep features for discriminative localization. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2921\u20132929","key":"13592_CR40","DOI":"10.1109\/CVPR.2016.319"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-13592-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-022-13592-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-13592-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,26]],"date-time":"2023-01-26T08:11:25Z","timestamp":1674720685000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-022-13592-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,6]]},"references-count":40,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2023,2]]}},"alternative-id":["13592"],"URL":"https:\/\/doi.org\/10.1007\/s11042-022-13592-7","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"type":"print","value":"1380-7501"},{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2022,8,6]]},"assertion":[{"value":"19 April 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 July 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 August 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interests regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}