{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:24:20Z","timestamp":1740122660603,"version":"3.37.3"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T00:00:00Z","timestamp":1667347200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T00:00:00Z","timestamp":1667347200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100002860","name":"China Sponsorship Council","doi-asserted-by":"publisher","award":["201906280464"],"award-info":[{"award-number":["201906280464"]}],"id":[{"id":"10.13039\/501100002860","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the National R&D Program of China","award":["2018AAA0101501"],"award-info":[{"award-number":["2018AAA0101501"]}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61772415"],"award-info":[{"award-number":["61772415"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s10489-022-04259-9","type":"journal-article","created":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T19:12:35Z","timestamp":1667416355000},"page":"14690-14702","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Improving weakly supervised phrase grounding via visual representation contextualization with contrastive learning"],"prefix":"10.1007","volume":"53","author":[{"given":"Xue","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1714-3433","authenticated-orcid":false,"given":"Youtian","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Suzan","family":"Verberne","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fons J.","family":"Verbeek","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,11,2]]},"reference":[{"key":"4259_CR1","unstructured":"Chen X, Fang H, Lin TY et al (2015) Microsoft COCO captions: data collection and evaluation server arXiv:150400325"},{"key":"4259_CR2","doi-asserted-by":"crossref","unstructured":"Antol S, Agrawal A, Lu J et al (2015) VQA: visual question answering. In: Proceedings of the IEEE international conference on computer vision, pp 2425\u20132433","DOI":"10.1109\/ICCV.2015.279"},{"key":"4259_CR3","doi-asserted-by":"crossref","unstructured":"Suhr A, Zhou S, Zhang A et al (2018) A corpus for reasoning about natural language grounded in photographs. arXiv:181100491","DOI":"10.18653\/v1\/P19-1644"},{"key":"4259_CR4","doi-asserted-by":"crossref","unstructured":"Zellers R, Bisk Y, Farhadi A et al (2019) From recognition to cognition: visual commonsense reasoning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6720\u20136731","DOI":"10.1109\/CVPR.2019.00688"},{"key":"4259_CR5","doi-asserted-by":"crossref","unstructured":"Plummer BA, Wang L, Cervantes CM et al (2015) Flickr30k Entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE international conference on computer vision, pp 2641\u20132649","DOI":"10.1109\/ICCV.2015.303"},{"key":"4259_CR6","doi-asserted-by":"crossref","unstructured":"Plummer BA, Mallya A, Cervantes CM et al (2017) Phrase localization and visual relationship detection with comprehensive image-language cues. In: Proceedings of the IEEE international conference on computer vision, pp 1928\u20131937","DOI":"10.1109\/ICCV.2017.213"},{"key":"4259_CR7","doi-asserted-by":"crossref","unstructured":"Fukui A, Park DH, Yang D et al (2016) Multimodal compact bilinear pooling for visual question answering and visual grounding. arXiv:160601847","DOI":"10.18653\/v1\/D16-1044"},{"issue":"2","key":"4259_CR8","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang L, Li Y, Huang J, et al. (2018) Learning two-branch neural networks for image-text matching tasks. IEEE Trans Pattern Anal Mach Intell 41(2):394\u2013407","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"4259_CR9","doi-asserted-by":"crossref","unstructured":"Datta S, Sikka K, Roy A et al (2019) Align2Ground: weakly supervised phrase grounding guided by image-caption alignment. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 2601\u20132610","DOI":"10.1109\/ICCV.2019.00269"},{"key":"4259_CR10","unstructured":"Lai F, Xie N, Doran D et al (2019) Contextual grounding of natural language entities in images. arXiv:191102133"},{"key":"4259_CR11","unstructured":"Oord Avd, Li Y, Vinyals O (2018) Representation learning with contrastive predictive coding. arXiv:180703748"},{"key":"4259_CR12","doi-asserted-by":"crossref","unstructured":"Gupta T, Vahdat A, Chechik G et al (2020) Contrastive learning for weakly supervised phrase grounding. In: Proceedings of the European conference on computer vision, Springer, pp 752\u2013768","DOI":"10.1007\/978-3-030-58580-8_44"},{"key":"4259_CR13","doi-asserted-by":"crossref","unstructured":"Yu T, Hui T, Yu Z et al (2020) Cross-modal omni interaction modeling for phrase grounding. In: Proceedings of the 28th ACM international conference on multimedia, pp 1725\u20131734","DOI":"10.1145\/3394171.3413846"},{"key":"4259_CR14","doi-asserted-by":"crossref","unstructured":"Bajaj M, Wang L, Sigal L (2019) G3raphGround: graph-based language grounding. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4281\u20134290","DOI":"10.1109\/ICCV.2019.00438"},{"key":"4259_CR15","doi-asserted-by":"crossref","unstructured":"Chen K, Gao J, Nevatia R (2018) Knowledge aided consistency for weakly supervised phrase grounding. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4042\u20134050","DOI":"10.1109\/CVPR.2018.00425"},{"key":"4259_CR16","doi-asserted-by":"crossref","unstructured":"Akbari H, Karaman S, Bhargava S et al (2019) Multi-level multimodal common semantic space for image-phrase grounding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12,476\u201312,486","DOI":"10.1109\/CVPR.2019.01276"},{"key":"4259_CR17","unstructured":"Vaswani A, Shazeer N, Parmar N et al (2017) Attention is all you need. arXiv:170603762"},{"key":"4259_CR18","doi-asserted-by":"crossref","unstructured":"Yu T, Yang Y, Li Y et al (2021) Heterogeneous attention network for effective and efficient cross-modal retrieval. In: Proceedings of the 44th international ACM SIGIR conference on research and development in information retrieval, pp 1146\u20131156","DOI":"10.1145\/3404835.3462924"},{"key":"4259_CR19","doi-asserted-by":"publisher","first-page":"107,138","DOI":"10.1016\/j.knosys.2021.107138","volume":"226","author":"X Dong","year":"2021","unstructured":"Dong X, Zhang H, Dong X, et al. (2021) Iterative graph attention memory network for cross-modal retrieval. Knowl-Based Syst 226:107,138","journal-title":"Knowl-Based Syst"},{"issue":"12","key":"4259_CR20","doi-asserted-by":"publisher","first-page":"5412","DOI":"10.1109\/TNNLS.2020.2967597","volume":"31","author":"X Xu","year":"2020","unstructured":"Xu X, Wang T, Yang Y, et al. (2020) Cross-modal attention with semantic consistence for image\u2013text matching. IEEE Trans Neural Netw Learning Syst 31(12):5412\u20135425","journal-title":"IEEE Trans Neural Netw Learning Syst"},{"key":"4259_CR21","doi-asserted-by":"crossref","unstructured":"Neubeck A, Van Gool L (2006) Efficient non-maximum suppression. In: 18th international conference on pattern recognition (ICPR\u201906), pp 850\u2013855","DOI":"10.1109\/ICPR.2006.479"},{"key":"4259_CR22","unstructured":"Ren S, He K, Girshick R et al (2015) Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in neural information processing systems, pp 91\u201399"},{"key":"4259_CR23","doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R et al (2016) You only look once: unified, real-time object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"key":"4259_CR24","doi-asserted-by":"crossref","unstructured":"Zitnick CL, Doll\u00e1r P (2014) Edge boxes: locating object proposals from edges. In: European conference on computer vision. Springer, pp 391\u2013405","DOI":"10.1007\/978-3-319-10602-1_26"},{"key":"4259_CR25","doi-asserted-by":"crossref","unstructured":"Bodla N, Singh B, Chellappa R et al (2017) Soft-NMS \u2013 improving object detection with one line of code. In: Proceedings of the IEEE international conference on computer vision (ICCV), pp 5561\u20135569","DOI":"10.1109\/ICCV.2017.593"},{"key":"4259_CR26","doi-asserted-by":"crossref","unstructured":"He Y, Zhang X, Savvides M et al (2018) Softer-NMS: rethinking bounding box regression for accurate object detection, vol 2(3) arXiv:180908545","DOI":"10.1109\/CVPR.2019.00300"},{"key":"4259_CR27","doi-asserted-by":"crossref","unstructured":"Chen L, Ma W, Xiao J et al (2021) Ref-NMS: Breaking proposal bottlenecks in two-stage referring expression grounding. In: Proceedings of the AAAI conference on artificial intelligence, pp 1036\u20131044","DOI":"10.1609\/aaai.v35i2.16188"},{"key":"4259_CR28","doi-asserted-by":"crossref","unstructured":"He K, Fan H, Wu Y et al (2020) Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9729\u20139738","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"4259_CR29","unstructured":"Chen T, Kornblith S, Norouzi M et al (2020) A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, PMLR, pp 1597\u20131607"},{"key":"4259_CR30","doi-asserted-by":"crossref","unstructured":"Wu Z, Xiong Y, Yu SX et al (2018) Unsupervised feature learning via non-parametric instance discrimination. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3733\u20133742","DOI":"10.1109\/CVPR.2018.00393"},{"key":"4259_CR31","doi-asserted-by":"crossref","unstructured":"Zhang H, Koh JY, Baldridge J et al (2021) Cross-modal contrastive learning for text-to-image generation. arXiv:210104702","DOI":"10.1109\/CVPR46437.2021.00089"},{"key":"4259_CR32","unstructured":"Dai B, Lin D (2017) Contrastive learning for image captioning. arXiv:171002534"},{"key":"4259_CR33","doi-asserted-by":"crossref","unstructured":"Li Z, Tran Q, Mai L et al (2020) Context-aware group captioning via self-attention and contrastive features. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3440\u20133450","DOI":"10.1109\/CVPR42600.2020.00350"},{"key":"4259_CR34","doi-asserted-by":"crossref","unstructured":"Huang X, Peng Y (2017) Cross-modal deep metric learning with multi-task regularization. In: 2017 IEEE international conference on multimedia and expo (ICME). IEEE, pp 943\u2013948","DOI":"10.1109\/ICME.2017.8019340"},{"key":"4259_CR35","unstructured":"Devlin J, Chang M W, Lee K, et al. (2019) BERT: pre-training Of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: human language technologies, vol 1, (Long and short papers), pp 4171\u20134186"},{"issue":"1","key":"4259_CR36","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O et al (2017) Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"key":"4259_CR37","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C et al (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"4259_CR38","doi-asserted-by":"crossref","unstructured":"Liu Y, Wan B, Ma L et al (2021) Relation-aware instance refinement for weakly supervised visual grounding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5612\u20135621","DOI":"10.1109\/CVPR46437.2021.00556"},{"key":"4259_CR39","doi-asserted-by":"crossref","unstructured":"Rohrbach A, Rohrbach M, Hu R et al (2016) Grounding of textual phrases in images by reconstruction. In: European conference on computer vision. Springer, pp 817\u2013834","DOI":"10.1007\/978-3-319-46448-0_49"},{"key":"4259_CR40","doi-asserted-by":"crossref","unstructured":"Wang L, Huang J, Li Y et al (2021) Improving weakly supervised visual grounding by contrastive knowledge distillation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14,090\u201314,100","DOI":"10.1109\/CVPR46437.2021.01387"},{"key":"4259_CR41","doi-asserted-by":"crossref","unstructured":"Fang H, Gupta S, Iandola F et al (2015) From captions to visual concepts and back. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1473\u20131482","DOI":"10.1109\/CVPR.2015.7298754"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-04259-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-022-04259-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-04259-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,31]],"date-time":"2023-05-31T21:04:53Z","timestamp":1685567093000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-022-04259-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,2]]},"references-count":41,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["4259"],"URL":"https:\/\/doi.org\/10.1007\/s10489-022-04259-9","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2022,11,2]]},"assertion":[{"value":"10 October 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 November 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interests regarding the publication of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Competing interests"}}]}}