{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T16:24:53Z","timestamp":1771950293637,"version":"3.50.1"},"publisher-location":"Cham","reference-count":62,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729065","type":"print"},{"value":"9783031729072","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72907-2_12","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T15:22:17Z","timestamp":1730301737000},"page":"195-213","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Scene-Graph ViT: End-to-End Open-Vocabulary Visual Relationship Detection"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7305-2667","authenticated-orcid":false,"given":"Tim","family":"Salzmann","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2203-2946","authenticated-orcid":false,"given":"Markus","family":"Ryll","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8428-9264","authenticated-orcid":false,"given":"Alex","family":"Bewley","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6428-8256","authenticated-orcid":false,"given":"Matthias","family":"Minderer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"12_CR1","doi-asserted-by":"crossref","unstructured":"Amiri, S., Chandan, K., Zhang, S.: Reasoning with scene graphs for robot planning under partial observability. IEEE Robot. Autom. Lett. 7(2), 5560\u20135567 (2022)","DOI":"10.1109\/LRA.2022.3157567"},{"key":"12_CR2","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Conference on Computer Vision (ECCV), pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Chao, Y.W., Liu, Y., Liu, X., Zeng, H., Deng, J.: Learning to detect human-object interactions. In: IEEE Winter Conference on Applications of Computer Vision (2018)","DOI":"10.1109\/WACV.2018.00048"},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Chen, B., et al.: Open-vocabulary queryable scene representations for real world planning. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 11509\u201311522. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10161534"},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Chen, T., Yu, W., Chen, R., Lin, L.: Knowledge-embedded routing network for scene graph generation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6163\u20136171 (2019)","DOI":"10.1109\/CVPR.2019.00632"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Cong, Y., Yang, M.Y., Rosenhahn, B.: Reltr: Relation transformer for scene graph generation. IEEE Trans. Pattern Anal. Mach. Intell. 45(9), 11169\u201311183 (2023)","DOI":"10.1109\/TPAMI.2023.3268066"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Desai, A., Wu, T.Y., Tripathi, S., Vasconcelos, N.: Learning of visual relations: the devil is in the tails. In: IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 15404\u201315413 (2021)","DOI":"10.1109\/ICCV48922.2021.01512"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Dong, X., Gan, T., Song, X., Wu, J., Cheng, Y., Nie, L.: Stacked hybrid-attention and group collaborative learning for unbiased scene graph generation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 19427\u201319436 (2022)","DOI":"10.1109\/CVPR52688.2022.01882"},{"key":"12_CR10","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Gu, Q., et\u00a0al.: Conceptgraphs: open-vocabulary 3d scene graphs for perception and planning. arXiv preprint arXiv:2309.16650 (2023)","DOI":"10.1109\/ICRA57147.2024.10610243"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Gupta, A., Dollar, P., Girshick, R.: LVIS: a dataset for large vocabulary instance segmentation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00550"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"He, T., Gao, L., Song, J., Li, Y.F.: Towards open-vocabulary scene graph generation with prompt-based finetuning. In: European Conference on Computer Vision (ECCV) (2022)","DOI":"10.1007\/978-3-031-19815-1_4"},{"key":"12_CR14","unstructured":"Hendrycks, D., Gimpel, K.: Gaussian error linear units (GELUS). arXiv preprint arXiv:1606.08415 (2016)"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Herzig, R., et al.: Incorporating structured representations into pretrained vision& language models using scene graphs. In: The 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.870"},{"key":"12_CR16","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: GQA: a new dataset for real-world visual reasoning and compositional question answering. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Hughes, N., Chang, Y., Carlone, L.: Hydra: a real-time spatial perception system for 3d scene graph construction and optimization. Robotics: Science and Systems (RSS) (2022)","DOI":"10.15607\/RSS.2022.XVIII.050"},{"key":"12_CR18","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"12_CR19","doi-asserted-by":"crossref","unstructured":"Johnson, J., et al.: Image retrieval using scene graphs. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), June 2015","DOI":"10.1109\/CVPR.2015.7298990"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Kim, B., Lee, J., Kang, J., Kim, E.S., Kim, H.J.: Hotr: end-to-end human-object interaction detection with transformers. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 74\u201383 (2021)","DOI":"10.1109\/CVPR46437.2021.00014"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Kim, K., et al.: LLM4SGG: large language model for weakly supervised scene graph generation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2024)","DOI":"10.1109\/CVPR52733.2024.02674"},{"key":"12_CR22","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"12_CR23","unstructured":"Knyazev, B., de\u00a0Vries, H., Cangea, C., Taylor, G.W., Courville, A., Belilovsky, E.: Graph density-aware losses for novel compositions in scene graph generation. In: British Machine Vision Conference (BMVC) (2020)"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"Kolesnikov, A., et al.: Big transfer (bit): general visual representation learning. In: European Conference on Computer Vision (ECCV), pp. 491\u2013507 (2020)","DOI":"10.1007\/978-3-030-58558-7_29"},{"key":"12_CR25","unstructured":"Koner, R., Shit, S., Tresp, V.: Relation transformer network. arXiv preprint arXiv:2004.06193 (2020)"},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Krishna, R., et\u00a0al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vis. 123, 32\u201373 (2017)","DOI":"10.1007\/s11263-016-0981-7"},{"key":"12_CR27","doi-asserted-by":"publisher","unstructured":"Li, H., et al.: Scene graph generation: a comprehensive survey. Neurocomputing (2024). https:\/\/doi.org\/10.1016\/j.neucom.2023.127052","DOI":"10.1016\/j.neucom.2023.127052"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Li, L., Chen, G., Xiao, J., Yang, Y., Wang, C., Chen, L.: Compositional feature augmentation for unbiased scene graph generation. In: IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 21685\u201321695 (2023)","DOI":"10.1109\/ICCV51070.2023.01982"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Li, R., Zhang, S., He, X.: SGTR: end-to-end scene graph generation with transformer. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 19486\u201319496 (2022)","DOI":"10.1109\/CVPR52688.2022.01888"},{"key":"12_CR30","doi-asserted-by":"crossref","unstructured":"Li, Y., Ouyang, W., Zhou, B., Wang, K., Wang, X.: Scene graph generation from objects, phrases and region captions. In: IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1261\u20131270 (2017)","DOI":"10.1109\/ICCV.2017.142"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Lin, X., Ding, C., Zeng, J., Tao, D.: Gps-net: graph property sensing network for scene graph generation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3746\u20133753 (2020)","DOI":"10.1109\/CVPR42600.2020.00380"},{"key":"12_CR32","doi-asserted-by":"crossref","unstructured":"Lu, C., Krishna, R., Bernstein, M., Fei-Fei, L.: Visual relationship detection with language priors. In: European Conference on Computer Vision (ECCV), pp. 852\u2013869. Springer (2016)","DOI":"10.1007\/978-3-319-46448-0_51"},{"key":"12_CR33","unstructured":"Minderer, M., Gritsenko, A., Houlsby, N.: Scaling open-vocabulary object detection. In: Advances in Neural Information Processing Systems vol. 36 (2024)"},{"key":"12_CR34","doi-asserted-by":"crossref","unstructured":"Minderer, M., et\u00a0al.: Simple open-vocabulary object detection. In: European Conference on Computer Vision (ECCV), pp. 728\u2013755. Springer (2022)","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"12_CR35","unstructured":"Open X-embodiment collaboration: Open X-embodiment: robotic learning datasets and RT-X models (2023)"},{"key":"12_CR36","unstructured":"Peng, Z., et al.: Grounding multimodal large language models to the world. In: International Conference on Learning Representations (2024)"},{"key":"12_CR37","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"12_CR38","unstructured":"Rana, K., Haviland, J., Garg, S., Abou-Chakra, J., Reid, I., Suenderhauf, N.: SayPlan: grounding large language models using 3d scene graphs for scalable robot task planning. In: 7th Annual Conference on Robot Learning (2023)"},{"key":"12_CR39","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"12_CR40","doi-asserted-by":"crossref","unstructured":"Shao, S., et al.: Objects365: a large-scale, high-quality dataset for object detection. In: IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 8430\u20138439 (2019)","DOI":"10.1109\/ICCV.2019.00852"},{"key":"12_CR41","doi-asserted-by":"crossref","unstructured":"Shi, H., Hayat, M., Cai, J.: Open-vocabulary object detection via scene graph discovery. In: 31st ACM International Conference on Multimedia, pp. 4012\u20134021 (2023)","DOI":"10.1145\/3581783.3612407"},{"key":"12_CR42","doi-asserted-by":"crossref","unstructured":"Singh, K.P., Salvador, J., Weihs, L., Kembhavi, A.: Scene graph contrastive learning for embodied navigation. In: IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 10884\u201310894 (2023)","DOI":"10.1109\/ICCV51070.2023.00999"},{"key":"12_CR43","unstructured":"Song, H., et al.: ViDT: an efficient and effective fully transformer-based object detector. In: International Conference on Learning Representations (2022)"},{"key":"12_CR44","doi-asserted-by":"crossref","unstructured":"Suhail, M., et al.: Energy-based learning for scene graph generation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13936\u201313945 (2021)","DOI":"10.1109\/CVPR46437.2021.01372"},{"key":"12_CR45","doi-asserted-by":"crossref","unstructured":"Tang, K., Niu, Y., Huang, J., Shi, J., Zhang, H.: Unbiased scene graph generation from biased training. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3716\u20133725 (2020)","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"12_CR46","doi-asserted-by":"crossref","unstructured":"Tang, K., Zhang, H., Wu, B., Luo, W., Liu, W.: Learning to compose dynamic tree structures for visual contexts. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6619\u20136628 (2019)","DOI":"10.1109\/CVPR.2019.00678"},{"key":"12_CR47","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"12_CR48","doi-asserted-by":"crossref","unstructured":"Xu, D., Zhu, Y., Choy, C.B., Fei-Fei, L.: Scene graph generation by iterative message passing. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5410\u20135419 (2017)","DOI":"10.1109\/CVPR.2017.330"},{"key":"12_CR49","doi-asserted-by":"crossref","unstructured":"Yang, J., Lu, J., Lee, S., Batra, D., Parikh, D.: Graph r-cnn for scene graph generation. In: European Conference on Computer Vision (ECCV), pp. 670\u2013685 (2018)","DOI":"10.1007\/978-3-030-01246-5_41"},{"key":"12_CR50","unstructured":"Yao, Z., Ai, J., Li, B., Zhang, C.: Efficient DETR: improving end-to-end object detector with dense prior. arXiv preprint arXiv:2104.01318 (2021)"},{"key":"12_CR51","unstructured":"Yu, J., Wang, Z., Vasudevan, V., Yeung, L., Seyedhosseini, M., Wu, Y.: CoCa: contrastive captioners are image-text foundation models. Transactions on Machine Learning Research (2022)"},{"key":"12_CR52","doi-asserted-by":"crossref","unstructured":"Yu, J., Chai, Y., Wang, Y., Hu, Y., Wu, Q.: CogTree: cognition tree loss for unbiased scene graph generation. In: International Joint Conference on Artificial Intelligence, IJCAI-21. International Joint Conferences on Artificial Intelligence Organization, August 2021","DOI":"10.24963\/ijcai.2021\/176"},{"key":"12_CR53","unstructured":"Yuan, H., et al.: Rlip: relational language-image pre-training for human-object interaction detection. Adv. Neural Inf. Process. Syst. 35, 37416\u201337431 (2022)"},{"key":"12_CR54","doi-asserted-by":"crossref","unstructured":"Yuan, H., et al.: RLIPv2: fast scaling of relational language-image pre-training. In: IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 21649\u201321661 (2023)","DOI":"10.1109\/ICCV51070.2023.01979"},{"key":"12_CR55","doi-asserted-by":"crossref","unstructured":"Zellers, R., Yatskar, M., Thomson, S., Choi, Y.: Neural motifs: Scene graph parsing with global context. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5831\u20135840 (2018)","DOI":"10.1109\/CVPR.2018.00611"},{"key":"12_CR56","doi-asserted-by":"crossref","unstructured":"Zhai, X., et al.: Lit: Zero-shot transfer with locked-image text tuning. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18123\u201318133 (2022)","DOI":"10.1109\/CVPR52688.2022.01759"},{"key":"12_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, A., et al.: Fine-grained scene graph generation with data transfer. In: European Conference on Computer Vision (ECCV), pp. 409\u2013424. Springer (2022)","DOI":"10.1007\/978-3-031-19812-0_24"},{"key":"12_CR58","unstructured":"Zhang, H., et al.: Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605 (2022)"},{"key":"12_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Ma, Z., Gao, X., Shakiah, S., Gao, Q., Chai, J.: GROUNDHOG: grounding large language models to holistic segmentation. In: Conference on Computer Vision and Pattern Recognition 2024 (2024). https:\/\/openreview.net\/forum?id=PWVW3ux1Ju","DOI":"10.1109\/CVPR52733.2024.01349"},{"key":"12_CR60","doi-asserted-by":"crossref","unstructured":"Zhao, L., et al.: Unified visual relationship detection with vision and language models. In: IEEE\/CVF International Conference on Computer Vision (ICCV) (2023)","DOI":"10.1109\/ICCV51070.2023.00641"},{"key":"12_CR61","doi-asserted-by":"crossref","unstructured":"Zhou, X., Girdhar, R., Joulin, A., Kr\u00e4henb\u00fchl, P., Misra, I.: Detecting twenty-thousand classes using image-level supervision. In: European Conference on Computer Vision (ECCV), pp. 350\u2013368 (2022)","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"12_CR62","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. In: International Conference on Learning Representations (2021)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72907-2_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T15:27:49Z","timestamp":1730302069000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72907-2_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031729065","9783031729072"],"references-count":62,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72907-2_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}