{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T01:44:47Z","timestamp":1773193487736,"version":"3.50.1"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2021,3,17]],"date-time":"2021-03-17T00:00:00Z","timestamp":1615939200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,3,17]],"date-time":"2021-03-17T00:00:00Z","timestamp":1615939200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2021,11]]},"DOI":"10.1007\/s10489-020-02115-2","type":"journal-article","created":{"date-parts":[[2021,3,17]],"date-time":"2021-03-17T06:02:21Z","timestamp":1615960941000},"page":"7781-7793","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Scene graph generation by multi-level semantic tasks"],"prefix":"10.1007","volume":"51","author":[{"given":"Peng","family":"Tian","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongwei","family":"Mo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Laihao","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,3,17]]},"reference":[{"key":"2115_CR1","doi-asserted-by":"crossref","unstructured":"Fang Y, Kuan K, Lin J et al (2017) Object detection meets knowledge graphs. In: Proceedings of the 26th International Joint Conference on Artificial Intelligence. IEEE, Sydney, pp 1661\u20131667","DOI":"10.24963\/ijcai.2017\/230"},{"key":"2115_CR2","doi-asserted-by":"crossref","unstructured":"Marino K, Salakhutdinov R, Gupta A (2017) The more you know: using knowledge graphs for image classification. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 2673\u20132681","DOI":"10.1109\/CVPR.2017.10"},{"key":"2115_CR3","doi-asserted-by":"crossref","unstructured":"Chao Y-W, Wang A, He Y et al (2015) Hico: A benchmark for recognizing human-object interactions in images. In: Proceedings of the 2015 IEEE International Conference on Computer Vision. IEEE. Santiago, pp 1017\u20131025","DOI":"10.1109\/ICCV.2015.122"},{"key":"2115_CR4","doi-asserted-by":"crossref","unstructured":"Lu C, Krishna R, Bernstein M, Li F-F (2016) Visual relationship detection with language priors. In: European Conference on Computer Vision, IEEE, Amsterdam, pp 852\u2013869","DOI":"10.1007\/978-3-319-46448-0_51"},{"key":"2115_CR5","doi-asserted-by":"crossref","unstructured":"Donahue J, Hendricks LA, Rohrbach M et al (2017) Long-term Recurrent Convolutional Networks for Visual Recognition and Description. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 677\u2013691","DOI":"10.1109\/TPAMI.2016.2599174"},{"key":"2115_CR6","first-page":"664","volume-title":"Deep Visual-Semantic Alignments for Generating Image Descriptions [J]","author":"A Karpathy","year":"2016","unstructured":"Karpathy A, Li F-F (2016) Deep Visual-Semantic Alignments for Generating Image Descriptions [J]. IEEE Trans Pattern Anal Mach Intell. IEEE, Boston, USA, pp 664\u2013676"},{"key":"2115_CR7","first-page":"2048","volume-title":"Show, attend and tell: Neural image caption generation with visual attention. In: Proceedings of the 32nd International Conference on International Conference on Machine Learning","author":"K Xu","year":"2015","unstructured":"Xu K, Ba J, Kiros R et al (2015) Show, attend and tell: Neural image caption generation with visual attention. In: Proceedings of the 32nd International Conference on International Conference on Machine Learning. IEEE, Lille, pp 2048\u20132057"},{"key":"2115_CR8","doi-asserted-by":"crossref","unstructured":"Johnson J, Krishna R, Stark M et al (2015) Image retrieval using scene graphs. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Boston, 3668\u20133678","DOI":"10.1109\/CVPR.2015.7298990"},{"issue":"11","key":"2115_CR9","doi-asserted-by":"publisher","first-page":"5298","DOI":"10.1109\/TIP.2017.2735182","volume":"26","author":"W Shen","year":"2017","unstructured":"Shen W, Zhao K, Jiang Y, Wang Y, Bai X, Yuille A (2017) DeepSkeleton: learning multi-task scale-associated deep side outputs for object skeleton extraction in natural images[J]. IEEE Trans Image Process 26(11):5298\u20135311","journal-title":"IEEE Trans Image Process"},{"key":"2115_CR10","first-page":"5534","volume-title":"IEEE computer vision and pattern recognition","author":"M Yatskar","year":"2016","unstructured":"Yatskar M, Zettlemoyer L, Farhadi A et al (2016) Situation recognition: visual semantic role labeling for image understanding. In: IEEE computer vision and pattern recognition. IEEE, Las Vegas, pp 5534\u20135542"},{"key":"2115_CR11","first-page":"1681","volume-title":"Learning the visual interpretation of sentences","author":"C Zitnick","year":"2013","unstructured":"Zitnick C, Parikh VL (2013) Learning the visual interpretation of sentences. In International Conference on Computer Vision. IEEE, Sydney, pp 1681\u20131688"},{"issue":"17","key":"2115_CR12","doi-asserted-by":"publisher","first-page":"22159","DOI":"10.1007\/s11042-018-5704-3","volume":"77","author":"Y Liu","year":"2018","unstructured":"Liu Y, Yu J, Han Y et al (2018) Understanding the effective receptive field in semantic image segmentation[J]. Multimed Tools Appl 77(17):22159\u201322171","journal-title":"Multimed Tools Appl"},{"key":"2115_CR13","volume-title":"International Conference on Computer Vision & Graphics: Part I","author":"A Przelaskowski","year":"2010","unstructured":"Przelaskowski A (2010) The role of sparse data representation in semantic image understanding. In: International Conference on Computer Vision & Graphics: Part I. Springer, Berlin Heidelberg"},{"key":"2115_CR14","first-page":"580","volume-title":"IEEE Conference on Computer Vision and Pattern Recognition","author":"R Girshick","year":"2014","unstructured":"Girshick R, Donahue J, Darrell T et al (2014) Rich feature hierarchies for accurate object detection and semantic segmentation. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Columbus, pp 580\u2013587"},{"key":"2115_CR15","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1016\/j.image.2019.06.004","volume":"78","author":"Y Cao","year":"2019","unstructured":"Cao Y, Fu G, Yang J, Cao Y, Yang MY (2019) Accurate salient object detection via dense recurrent connections and residual-based hierarchical feature integration[J]. Signal Process Image Commun 78:103\u2013112","journal-title":"Signal Process Image Commun"},{"key":"2115_CR16","first-page":"1","volume-title":"Gated graph sequence neural networks[J]. In: IEEE International Conference on Learning Representations","author":"Y Li","year":"2016","unstructured":"Li Y, Tarlow D, Brockschmidt M et al (2016) Gated graph sequence neural networks[J]. In: IEEE International Conference on Learning Representations. IEEE, San Juan, pp 1\u201320"},{"issue":"6","key":"2115_CR17","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren S, He K, Girshick R, Sun J (2017) Faster R-CNN: towards real-time object detection with region proposal networks[J]. IEEE Trans Pattern Anal Mach Intell 39(6):1137\u20131149","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2115_CR18","first-page":"10934","volume":"2004","author":"A Bochkovskiy","year":"2020","unstructured":"Bochkovskiy A, Wang C-Y, Liao H-Y (2020) YOLOv4: optimal speed and accuracy of object detection. arXivpreprint arXiv 2004:10934","journal-title":"arXivpreprint arXiv"},{"key":"2115_CR19","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D et al (2016) SSD: single shot MultiBox detector[C]. In: European Conference on Computer Vision. IEEE, Amsterdam, pp 21\u201337","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"2115_CR20","doi-asserted-by":"crossref","unstructured":"Li Y, Ou Y, Wang X et al (2017) Vip-cnn: visual phrase guided convolutional neural network. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 1347\u20131356","DOI":"10.1109\/CVPR.2017.766"},{"key":"2115_CR21","doi-asserted-by":"crossref","unstructured":"Xu D-F, Zhu Y-K et al (2017) Scene graph generation by iterative message passing. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 5410\u20135419","DOI":"10.1109\/CVPR.2017.330"},{"key":"2115_CR22","doi-asserted-by":"crossref","unstructured":"Elhoseiny M, Cohen S, Chang W et al (2015) Sherlock: Scalable fact learning in images. In Proceedings of the AAAI Conference on Artificial Intelligence, IEEE. San Francisco 31(1)","DOI":"10.1609\/aaai.v31i1.11214"},{"key":"2115_CR23","doi-asserted-by":"crossref","unstructured":"Anderson P, He X-D, Chris B et al (2018) Bottom-up and top-down attention for image captioning and visual question answering. In: In IEEE conference on computer vision and pattern recognition, vol 2018. IEEE, Salt Lake, pp 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"issue":"12","key":"2115_CR24","doi-asserted-by":"publisher","first-page":"2891","DOI":"10.1109\/TPAMI.2012.162","volume":"35","author":"G Kulkarni","year":"2013","unstructured":"Kulkarni G, Premraj V, Ordonez V, Dhar S, Li S, Choi Y, Berg AC, Berg TL (2013) Babytalk: understanding and generating simple image descriptions[J]. IEEE Trans Pattern Anal Mach Intell 35(12):2891\u20132903","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2115_CR25","unstructured":"Elliott D, Keller F (2013) Image description using visual dependency representations. In: Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing. IEEE, Seattle, Washington, DC, pp 1292\u20131302"},{"key":"2115_CR26","doi-asserted-by":"crossref","unstructured":"Verma Y, Gupta A, Mannem P et al (2013) Generating image descriptions using semantic similarities in the output space. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Portland, pp 288\u2013293","DOI":"10.1109\/CVPRW.2013.50"},{"key":"2115_CR27","doi-asserted-by":"crossref","unstructured":"Devlin J, Cheng H, Fang H et al (2015) Language models for image captioning: The quirks and what works. In: Annual Meeting of the Association for Computational Linguistics. IEEE, Beijing, pp 100\u2013105","DOI":"10.3115\/v1\/P15-2017"},{"key":"2115_CR28","doi-asserted-by":"crossref","unstructured":"Chen L, Zhang H, Xiao J et al (2017) SCA-CNN: spatial and channel-wise attention in convolutional networks for image captioning. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 6298\u20136306","DOI":"10.1109\/CVPR.2017.667"},{"key":"2115_CR29","unstructured":"Oriol V, Alexander T, Samy B et al (2015) Show and tell: a neural image caption generator. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Boston, pp 3156\u20133164"},{"key":"2115_CR30","unstructured":"Steven J, Etienne M, Mroueh Y et al (2017) Self-critical sequence training for image captioning. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 7008\u20137024"},{"key":"2115_CR31","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y-W, Li Y-H et al (2017) Boosting image captioning with attributes. In: IEEE International Conference on Computer Vision. IEEE, Venice, pp 4894\u20134902","DOI":"10.1109\/ICCV.2017.524"},{"key":"2115_CR32","first-page":"2048","volume-title":"International Conference on Machine Learning","author":"K Xu","year":"2015","unstructured":"Xu K, Ba J, Kiros R et al (2015) Show, attend and tell: neural image caption generation with visual attention. In: International Conference on Machine Learning. IEEE, Lille, pp 2048\u20132057"},{"key":"2115_CR33","first-page":"4651","volume-title":"IEEE Conference on Computer Vision and Pattern Recognition","author":"Q You","year":"2016","unstructured":"You Q, Jin H, Wang Z et al (2016) Image captioning with semantic attention. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Las Vegas, pp 4651\u20134659"},{"key":"2115_CR34","doi-asserted-by":"crossref","unstructured":"Lu J, Xiong C, Parikh D et al (2017) Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 375\u2013383","DOI":"10.1109\/CVPR.2017.345"},{"key":"2115_CR35","doi-asserted-by":"crossref","unstructured":"Li Y, Ouyang W, Wang X et al (2017) Vip-cnn: visual phrase guided convolutional neural network. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Honolulu, pp 1347\u20131356","DOI":"10.1109\/CVPR.2017.766"},{"key":"2115_CR36","first-page":"670","volume-title":"Proceedings of European Conference on Computer Vision","author":"J Yang","year":"2018","unstructured":"Yang J, Lu J, Lee S et al (2018) Graph RCNN for scene graph generation. In: Proceedings of European Conference on Computer Vision. IEEE, Munich, pp 670\u2013685"},{"key":"2115_CR37","first-page":"335","volume-title":"Proceedings of European Conference on Computer Vision","author":"Y Li","year":"2018","unstructured":"Li Y, Ouyang W, Zhou B et al (2018) Factorizable net: an efficient subgraph-based framework for scene graph generation. In: Proceedings of European Conference on Computer Vision. IEEE, Munich, pp 335\u2013351"},{"key":"2115_CR38","first-page":"5831","volume-title":"IEEE Conference on Computer Vision and Pattern Recognition","author":"R Zellers","year":"2017","unstructured":"Zellers R, Yatskar M, Thomson S et al (2017) Neural motifs: Scene graph parsing with global context. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Salt Lake, pp 5831\u20135840"},{"key":"2115_CR39","first-page":"5199","volume-title":"3D graph neural networks for RGBD semantic segmentation. In IEEE International Conference on Computer Vision","author":"X Qi","year":"2017","unstructured":"Qi X, Liao R, Jia J et al (2017) 3D graph neural networks for RGBD semantic segmentation. In IEEE International Conference on Computer Vision. IEEE, Venice, pp 5199\u20135208"},{"key":"2115_CR40","first-page":"04844","volume":"1612","author":"M Kenneth","year":"2017","unstructured":"Kenneth M, Ruslan S, Abhinav G (2017) The more you know: using knowledge graphs for image classification. arXiv preprint arXiv 1612:04844","journal-title":"arXiv preprint arXiv"},{"issue":"1","key":"2115_CR41","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, Johnson J, Hata K, Kravitz J, Chen S, Kalantidis Y, Li LJ, Shamma DA, Bernstein MS, Fei-Fei L (2017) Visual genome: connecting language and vision using Crowdsourced dense image annotations[J]. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"key":"2115_CR42","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S et al (2014) Microsoft coco: common objects in context. In: IEEE European conference on computer vision. IEEE, Zurich, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2115_CR43","doi-asserted-by":"crossref","unstructured":"Karpathy A, Li F-F (2015) Deep visual-semantic alignments for generating image descriptions. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Boston, pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"2115_CR44","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization[J]. arXiv preprint arXiv:1412.6980"},{"key":"2115_CR45","first-page":"1261","volume-title":"Scene graph generation from objects, phrases and region captions. In IEEE International Conference on Computer Vision","author":"Y Li","year":"2017","unstructured":"Li Y, Ouyang W, Zhou B et al (2017) Scene graph generation from objects, phrases and region captions. In IEEE International Conference on Computer Vision. IEEE, Venice, pp 1261\u20131270"},{"key":"2115_CR46","doi-asserted-by":"crossref","unstructured":"Wang W, Wang R, Shan S et al (2019) Exploring context and visual pattern of relationship for scene graph generation. In: IEEE Conference on Computer Vision and Pattern Recognition. IEEE, Long Beach, pp 8188\u20138197","DOI":"10.1109\/CVPR.2019.00838"},{"key":"2115_CR47","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T et al (2002) BLEU: a method for automatic evaluation of machine translation. In: IEEE 40th annual meeting on association for computational linguistics. Association for Computational Linguistics. IEEE, Philadelphia, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"2115_CR48","first-page":"65","volume-title":"Proceedings of the Acl Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization","author":"S Banerjee","year":"2005","unstructured":"Banerjee S, Lavie A (2005) METEOR: an automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the Acl Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization. IEEE, Michigan, pp 65\u201372"},{"key":"2115_CR49","first-page":"71","volume-title":"IEEE Human Language Technology Conference of the North American Chapter of the Association for Computational Linguistics","author":"C-Y Lin","year":"2003","unstructured":"Lin C-Y, Hovy E (2003) Automatic evaluation of summaries using n-gram co-occurrence statistics. In: IEEE Human Language Technology Conference of the North American Chapter of the Association for Computational Linguistics. IEEE, Edmonton, pp 71\u201378"},{"key":"2115_CR50","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence Zitnick C, Parikh D (2015) Cider: Consensus-based image description evaluation. In: IEEE International Conference on Computer Vision. IEEE, Boston, pp 4566\u20134575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"2115_CR51","first-page":"382","volume-title":"European Conference on Computer Vision","author":"P Anderson","year":"2016","unstructured":"Anderson P, Fernando B, Johnson M et al (2016) Spice: Semantic propositional image caption evaluation. In: European Conference on Computer Vision. Springer, Cham, IEEE, Amsterdam, pp 382\u2013398"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-020-02115-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-020-02115-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-020-02115-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,21]],"date-time":"2022-12-21T18:43:58Z","timestamp":1671648238000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-020-02115-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,3,17]]},"references-count":51,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2021,11]]}},"alternative-id":["2115"],"URL":"https:\/\/doi.org\/10.1007\/s10489-020-02115-2","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,3,17]]},"assertion":[{"value":"2 December 2020","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 March 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}