{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,27]],"date-time":"2026-05-27T18:28:17Z","timestamp":1779906497265,"version":"3.53.1"},"publisher-location":"Cham","reference-count":46,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031732225","type":"print"},{"value":"9783031732232","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,8]],"date-time":"2024-11-08T00:00:00Z","timestamp":1731024000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,8]],"date-time":"2024-11-08T00:00:00Z","timestamp":1731024000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73223-2_10","type":"book-chapter","created":{"date-parts":[[2024,11,7]],"date-time":"2024-11-07T18:48:02Z","timestamp":1731005282000},"page":"159-175","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Towards Scene Graph Anticipation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4705-8129","authenticated-orcid":false,"given":"Rohith","family":"Peddi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5320-393X","authenticated-orcid":false,"given":"Saksham","family":"Singh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4763-8872","authenticated-orcid":false,"family":"Saurabh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9190-9794","authenticated-orcid":false,"given":"Parag","family":"Singla","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6459-7358","authenticated-orcid":false,"given":"Vibhav","family":"Gogate","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,8]]},"reference":[{"key":"10_CR1","unstructured":"Agia, C., et al.: Taskography: Evaluating robot task planning over large 3d scene graphs (2022)"},{"key":"10_CR2","unstructured":"Bar, A., et al.: Compositional video synthesis with action graphs. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 662\u2013673. PMLR (2021). http:\/\/proceedings.mlr.press\/v139\/bar21a.html"},{"key":"10_CR3","unstructured":"Chen, R.T.Q., Rubanova, Y., Bettencourt, J., Duvenaud, D.: Neural ordinary differential equations. Neural Information Processing Systems (2018). null"},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Chen, T., Yu, W., Chen, R., Lin, L.: Knowledge-embedded routing network for scene graph generation (2019)","DOI":"10.1109\/CVPR.2019.00632"},{"key":"10_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Z., Wu, J., Lei, Z., Zhang, Z., Chen, C.: Expanding scene graph boundaries: Fully open-vocabulary scene graph generation via visual-concept alignment and retention (2023)","DOI":"10.1007\/978-3-031-72848-8_7"},{"key":"10_CR6","doi-asserted-by":"crossref","unstructured":"Cherian, A., Hori, C., Marks, T.K., Roux, J.L.: (2.5+1)d spatio-temporal scene graphs for video question answering (2022)","DOI":"10.1609\/aaai.v36i1.19922"},{"key":"10_CR7","doi-asserted-by":"publisher","unstructured":"Cong, Y., Liao, W., Ackermann, H., Yang, M., Rosenhahn, B.: Spatial-temporal transformer for dynamic scene graph generation. In: IEEE International Conference on Computer Vision (2021). https:\/\/doi.org\/10.1109\/iccv48922.2021.01606","DOI":"10.1109\/iccv48922.2021.01606"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Dhamo, H., et al.: Semantic image manipulation using scene graphs. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00526"},{"key":"10_CR9","doi-asserted-by":"publisher","unstructured":"Feng, S., Mostafa, H., Nassar, M., Majumdar, S., Tripathi, S.: Exploiting long-term dependencies for generating dynamic scene graphs. In: IEEE Workshop\/Winter Conference on Applications of Computer Vision (2021). https:\/\/doi.org\/10.1109\/wacv56688.2023.00510","DOI":"10.1109\/wacv56688.2023.00510"},{"key":"10_CR10","unstructured":"Finn, C., Goodfellow, I., Levine, S.: Unsupervised learning for physical interaction through video prediction (2016)"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Grauman, K.: Anticipative Video Transformer. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01325"},{"key":"10_CR12","unstructured":"Huang, Z., Sun, Y., Wang, W.: Learning continuous system dynamics from irregularly-sampled partial observations (2020)"},{"key":"10_CR13","doi-asserted-by":"publisher","unstructured":"Huang, Z., Sun, Y., Wang, W.: Coupled graph ode for learning interacting system dynamics. In: Proceedings of the 27th ACM SIGKDD Conference on Knowledge Discovery & Data Mining, KDD 2021, p. 705-715. Association for Computing Machinery, New York (2021). https:\/\/doi.org\/10.1145\/3447548.3467385, https:\/\/doi-org.libproxy.utdallas.edu\/10.1145\/3447548.3467385","DOI":"10.1145\/3447548.3467385"},{"key":"10_CR14","unstructured":"H\u00f6ppe, T., Mehrjou, A., Bauer, S., Nielsen, D., Dittadi, A.: Diffusion models for video prediction and infilling (2022)"},{"key":"10_CR15","doi-asserted-by":"publisher","unstructured":"Ji, J., Krishna, R., Fei-Fei, L., Li, F.F., Niebles, J.C.: Action genome: Actions as composition of spatio-temporal scene graphs. In: Computer Computer Vision and Pattern Recognition (2019). https:\/\/doi.org\/10.1109\/cvpr42600.2020.01025","DOI":"10.1109\/cvpr42600.2020.01025"},{"key":"10_CR16","unstructured":"Khandelwal, A.: Correlation debiasing for unbiased scene graph generation in videos (2023)"},{"key":"10_CR17","unstructured":"Kidger, P., Foster, J., Li, X., Lyons, T.: Efficient and accurate gradients for neural sdes (2021)"},{"key":"10_CR18","doi-asserted-by":"crossref","unstructured":"l Kim, K., et al.: Llm4sgg: Large language model for weakly supervised scene graph generation (2023)","DOI":"10.1109\/CVPR52733.2024.02674"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"Kim, U.H., Park, J.M., jin Song, T., Kim, J.H.: 3-d scene graph: a sparse and semantic representation of physical environments for intelligent agents. IEEE Trans. Cybernet. 50, 4921\u20134933 (2019). https:\/\/api.semanticscholar.org\/CorpusID:199577350","DOI":"10.1109\/TCYB.2019.2931042"},{"issue":"1","key":"10_CR20","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/S11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vis. 123(1), 32\u201373 (2017). https:\/\/doi.org\/10.1007\/S11263-016-0981-7","journal-title":"Int. J. Comput. Vis."},{"key":"10_CR21","unstructured":"Kurenkov, A., et al.: Modeling dynamic environments with scene graph memory (2023)"},{"key":"10_CR22","unstructured":"Lee, A.X., Zhang, R., Ebert, F., Abbeel, P., Finn, C., Levine, S.: Stochastic adversarial video prediction (2018)"},{"key":"10_CR23","unstructured":"Li, L., Xiao, J., Chen, G., Shao, J., Zhuang, Y., Chen, L.: Zero-shot visual relation detection via composite visual cues from large language models (2023). https:\/\/arxiv.org\/abs\/2305.12476"},{"key":"10_CR24","doi-asserted-by":"publisher","unstructured":"Li, R., Zhang, S., He, X.: Sgtr: End-to-end scene graph generation with transformer. In: Computer Vision and Pattern Recognition (2021). https:\/\/doi.org\/10.1109\/cvpr52688.2022.01888","DOI":"10.1109\/cvpr52688.2022.01888"},{"key":"10_CR25","doi-asserted-by":"publisher","unstructured":"Liang, Y., Ouyang, K., Yan, H., Wang, Y., Tong, Z., Zimmermann, R.: Modeling trajectories with neural ordinary differential equations. In: Zhou, Z.H. (ed.) Proceedings of the Thirtieth International Joint Conference on Artificial Intelligence, IJCAI 2021, pp. 1498\u20131504. International Joint Conferences on Artificial Intelligence Organization (Oct 2021).https:\/\/doi.org\/10.24963\/ijcai.2021\/207, main Track","DOI":"10.24963\/ijcai.2021\/207"},{"key":"10_CR26","unstructured":"Liu, Z., Shojaee, P., Reddy, C.K.: Graph-based multi-ode neural networks for spatio-temporal traffic forecasting (2023)"},{"key":"10_CR27","doi-asserted-by":"publisher","unstructured":"Lu, J., Chen, L., Song, Y., Lin, S., Wang, C., He, G.: Prior knowledge-driven dynamic scene graph generation with causal inference. In: Proceedings of the 31st ACM International Conference on Multimedia, MM 2023, pp. 4877-4885. Association for Computing Machinery, New York, (2023). https:\/\/doi.org\/10.1145\/3581783.3612249","DOI":"10.1145\/3581783.3612249"},{"key":"10_CR28","unstructured":"Mi, L., Ou, Y., Chen, Z.: Visual relationship forecasting in videos. arXiv.org (2021)"},{"key":"10_CR29","doi-asserted-by":"crossref","unstructured":"Nag, S., Min, K., Tripathi, S., Roy-Chowdhury, A.K.: Unbiased scene graph generation in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22803\u201322813 (2023)","DOI":"10.1109\/CVPR52729.2023.02184"},{"key":"10_CR30","doi-asserted-by":"crossref","unstructured":"Park, S., Kim, K., Lee, J., Choo, J., Lee, J., Kim, S., Choi, E.: Vid-ode: Continuous-time video generation with neural ordinary differential equation (2021)","DOI":"10.1609\/aaai.v35i3.16342"},{"key":"10_CR31","unstructured":"Poli, M., Massaroli, S., Park, J., Yamashita, A., Asama, H., Park, J.: Graph neural ordinary differential equations (2021)"},{"key":"10_CR32","unstructured":"Qi, H., Wang, X., Pathak, D., Ma, Y., Malik, J.: Learning long-term visual dynamics with region proposal interaction networks (2021)"},{"key":"10_CR33","doi-asserted-by":"crossref","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: Towards real-time object detection with region proposal networks (2016)","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"10_CR34","doi-asserted-by":"publisher","unstructured":"Shang, X., Ren, T., Guo, J., Zhang, H., Chua, T.S.: Video visual relation detection. In: Proceedings of the 25th ACM International Conference on Multimedia, MM 2017, pp. 1300-1308. Association for Computing Machinery, New York (2017). https:\/\/doi.org\/10.1145\/3123266.3123380, https:\/\/doi-org.libproxy.utdallas.edu\/10.1145\/3123266.3123380","DOI":"10.1145\/3123266.3123380"},{"key":"10_CR35","doi-asserted-by":"crossref","unstructured":"Shit, S., et al.: Relationformer: A unified framework for image-to-graph generation (2022)","DOI":"10.1007\/978-3-031-19836-6_24"},{"key":"10_CR36","unstructured":"Thickstun, J., Hall, D., Donahue, C., Liang, P.: Anticipatory music transformer (2023)"},{"key":"10_CR37","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Guyon, I., Luxburg, U.V., Bengio, S., Wallach, H., Fergus, R., Vishwanathan, S., Garnett, R. (eds.) Advances in Neural Information Processing Systems, vol.\u00a030. Curran Associates, Inc. (2017). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf"},{"key":"10_CR38","unstructured":"Voleti, V., Jolicoeur-Martineau, A., Pal, C.: Mcvd: Masked conditional video diffusion for prediction, generation, and interpolation (2022)"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Walker, J., Doersch, C., Gupta, A., Hebert, M.: An uncertain future: Forecasting from static images using variational autoencoders (2016)","DOI":"10.1007\/978-3-319-46478-7_51"},{"key":"10_CR40","doi-asserted-by":"publisher","unstructured":"Wu, X., Zhao, J., Wang, R.: Anticipating future relations via graph growing for action prediction. In: Proceedings of the AAAI Conference on Artificial Intelligence 35(4), 2952\u20132960 (2021). https:\/\/doi.org\/10.1609\/aaai.v35i4.16402","DOI":"10.1609\/aaai.v35i4.16402"},{"key":"10_CR41","unstructured":"Wu, Z., Dvornik, N., Greff, K., Kipf, T., Garg, A.: Slotformer: Unsupervised visual dynamics simulation with object-centric models (2023)"},{"key":"10_CR42","doi-asserted-by":"crossref","unstructured":"Yu, W., Chen, W., Yin, S., Easterbrook, S., Garg, A.: Modular action concept grounding in semantic video prediction (2022)","DOI":"10.1109\/CVPR52688.2022.00359"},{"key":"10_CR43","unstructured":"Zhao, S., Xu, H.: Less is more: Toward zero-shot local scene graph generation via foundation models (2023)"},{"key":"10_CR44","unstructured":"Zhou, Z., Shi, M., Caesar, H.: Vlprompt: Vision-language prompting for panoptic scene graph generation (2023)"},{"key":"10_CR45","unstructured":"Zhu, G., et al.: Scene graph generation: A comprehensive survey (2022)"},{"key":"10_CR46","unstructured":"\u00d8ksendal, B.: Stochastic Differential Equations: An Introduction with Applications. Springer, 6th edn. (2003), corr. 4th printing, 2007"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73223-2_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T23:20:34Z","timestamp":1733008834000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73223-2_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,8]]},"ISBN":["9783031732225","9783031732232"],"references-count":46,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73223-2_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,8]]},"assertion":[{"value":"8 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}