{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T18:07:39Z","timestamp":1775326059576,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":71,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T00:00:00Z","timestamp":1752969600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Beijing Natural Science Foundation","award":["No. 1232009"],"award-info":[{"award-number":["No. 1232009"]}]},{"name":"Strategic Priority Research Program of the Chinese Academy of Sciences","award":["No. XDB0620103"],"award-info":[{"award-number":["No. XDB0620103"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 92270118, No. 62276269"],"award-info":[{"award-number":["No. 92270118, No. 62276269"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["No. 202230265, No. E2EG2202X2"],"award-info":[{"award-number":["No. 202230265, No. E2EG2202X2"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,20]]},"DOI":"10.1145\/3690624.3709168","type":"proceedings-article","created":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T18:42:22Z","timestamp":1743792142000},"page":"659-670","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Reasoning-Enhanced Object-Centric Learning for Videos"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0685-0861","authenticated-orcid":false,"given":"Jian","family":"Li","sequence":"first","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6354-385X","authenticated-orcid":false,"given":"Pu","family":"Ren","sequence":"additional","affiliation":[{"name":"Northeastern University, Boston, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0127-4030","authenticated-orcid":false,"given":"Yang","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5145-3259","authenticated-orcid":false,"given":"Hao","family":"Sun","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,20]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Danilo Jimenez Rezende, et al","author":"Battaglia Peter","year":"2016","unstructured":"Peter Battaglia, Razvan Pascanu, Matthew Lai, Danilo Jimenez Rezende, et al. 2016. Interaction networks for learning about objects, relations and physics. Advances in neural information processing systems, Vol. 29 (2016)."},{"key":"e_1_3_2_2_2_1","first-page":"6027","article-title":"Learning physical graph representations from visual scenes","volume":"33","author":"Bear Daniel","year":"2020","unstructured":"Daniel Bear, Chaofei Fan, Damian Mrowca, Yunzhu Li, Seth Alter, Aran Nayebi, Jeremy Schwartz, Li F Fei-Fei, Jiajun Wu, Josh Tenenbaum, et al. 2020. Learning physical graph representations from visual scenes. Advances in Neural Information Processing Systems, Vol. 33 (2020), 6027--6039.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_3_1","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is space-time attention all you need for video understanding?. In ICML, Vol. 2. 4.","journal-title":"ICML"},{"key":"e_1_3_2_2_4_1","volume-title":"Monet: Unsupervised scene decomposition and representation. arXiv preprint arXiv:1901.11390","author":"Burgess Christopher P","year":"2019","unstructured":"Christopher P Burgess, Loic Matthey, Nicholas Watters, Rishabh Kabra, Irina Higgins, Matt Botvinick, and Alexander Lerchner. 2019. Monet: Unsupervised scene decomposition and representation. arXiv preprint arXiv:1901.11390 (2019)."},{"key":"e_1_3_2_2_5_1","volume-title":"A compositional object-based approach to learning physical dynamics. arXiv preprint arXiv:1612.00341","author":"Chang Michael B","year":"2016","unstructured":"Michael B Chang, Tomer Ullman, Antonio Torralba, and Joshua B Tenenbaum. 2016. A compositional object-based approach to learning physical dynamics. arXiv preprint arXiv:1612.00341 (2016)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2022.04.014"},{"key":"e_1_3_2_2_7_1","first-page":"11770","article-title":"Roots: Object-centric representation and rendering of 3d scenes","volume":"22","author":"Chen Chang","year":"2021","unstructured":"Chang Chen, Fei Deng, and Sungjin Ahn. 2021a. Roots: Object-centric representation and rendering of 3d scenes. The Journal of Machine Learning Research, Vol. 22, 1 (2021), 11770--11805.","journal-title":"The Journal of Machine Learning Research"},{"key":"e_1_3_2_2_8_1","volume-title":"Joshua B Tenenbaum, and Chuang Gan.","author":"Chen Zhenfang","year":"2021","unstructured":"Zhenfang Chen, Jiayuan Mao, Jiajun Wu, Kwan-Yee Kenneth Wong, Joshua B Tenenbaum, and Chuang Gan. 2021b. Grounding physical concepts of objects and events through dynamic visual reasoning. arXiv preprint arXiv:2103.16564 (2021)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i01.5371"},{"key":"e_1_3_2_2_10_1","volume-title":"Attention over learned object embeddings enables complex visual reasoning. Advances in neural information processing systems","author":"Ding David","year":"2021","unstructured":"David Ding, Felix Hill, Adam Santoro, Malcolm Reynolds, and Matt Botvinick. 2021b. Attention over learned object embeddings enables complex visual reasoning. Advances in neural information processing systems, Vol. 34 (2021), 9112--9124."},{"key":"e_1_3_2_2_11_1","first-page":"887","article-title":"Dynamic visual reasoning by learning differentiable physics models from video and language","volume":"34","author":"Ding Mingyu","year":"2021","unstructured":"Mingyu Ding, Zhenfang Chen, Tao Du, Ping Luo, Josh Tenenbaum, and Chuang Gan. 2021a. Dynamic visual reasoning by learning differentiable physics models from video and language. Advances In Neural Information Processing Systems, Vol. 34 (2021), 887--899.","journal-title":"Advances In Neural Information Processing Systems"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2020.04.110"},{"key":"e_1_3_2_2_13_1","volume-title":"Bernhard Sch\u00f6lkopf, Ole Winther, and Francesco Locatello.","author":"Dittadi Andrea","year":"2021","unstructured":"Andrea Dittadi, Samuele Papa, Michele De Vita, Bernhard Sch\u00f6lkopf, Ole Winther, and Francesco Locatello. 2021. Generalization and robustness implications in object-centric learning. arXiv preprint arXiv:2107.00637 (2021)."},{"key":"e_1_3_2_2_14_1","volume-title":"Conference on Robot Learning. PMLR, 1755--1768","author":"Driess Danny","year":"2023","unstructured":"Danny Driess, Zhiao Huang, Yunzhu Li, Russ Tedrake, and Marc Toussaint. 2023. Learning multi-object dynamics with compositional neural radiance fields. In Conference on Robot Learning. PMLR, 1755--1768."},{"key":"e_1_3_2_2_15_1","first-page":"28940","article-title":"Savi: Towards end-to-end object-centric learning from real-world videos","volume":"35","author":"Elsayed Gamaleldin","year":"2022","unstructured":"Gamaleldin Elsayed, Aravindh Mahendran, Sjoerd van Steenkiste, Klaus Greff, Michael C Mozer, and Thomas Kipf. 2022. Savi: Towards end-to-end object-centric learning from real-world videos. Advances in Neural Information Processing Systems, Vol. 35 (2022), 28940--28954.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00373"},{"key":"e_1_3_2_2_17_1","volume-title":"Visual attention methods in deep learning: An in-depth survey. arXiv preprint arXiv:2204.07756","author":"Hassanin Mohammed","year":"2022","unstructured":"Mohammed Hassanin, Saeed Anwar, Ibrahim Radwan, Fahad S Khan, and Ajmal Mian. 2022. Visual attention methods in deep learning: An in-depth survey. arXiv preprint arXiv:2204.07756 (2022)."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.567"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF01908075"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10777"},{"key":"e_1_3_2_2_21_1","volume-title":"Reasoning About Physical Interactions with Object-Oriented Prediction and Planning. In International Conference on Learning Representations.","author":"Janner Michael","year":"2019","unstructured":"Michael Janner, Sergey Levine, William T. Freeman, Joshua B. Tenenbaum, Chelsea Finn, and Jiajun Wu. 2019. Reasoning About Physical Interactions with Object-Oriented Prediction and Planning. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_22_1","volume-title":"Gerard De Melo, and Sungjin Ahn","author":"Jiang Jindong","year":"2019","unstructured":"Jindong Jiang, Sepehr Janghorbani, Gerard De Melo, and Sungjin Ahn. 2019. Scalor: Generative world models with scalable object representations. arXiv preprint arXiv:1910.02384 (2019)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.physrep.2021.10.005"},{"key":"e_1_3_2_2_24_1","first-page":"20146","article-title":"Simone: View-invariant, temporally-abstracted object representations via unsupervised video decomposition","volume":"34","author":"Kabra Rishabh","year":"2021","unstructured":"Rishabh Kabra, Daniel Zoran, Goker Erdogan, Loic Matthey, Antonia Creswell, Matt Botvinick, Alexander Lerchner, and Chris Burgess. 2021. Simone: View-invariant, temporally-abstracted object representations via unsupervised video decomposition. Advances in Neural Information Processing Systems, Vol. 34 (2021), 20146--20159.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_25_1","volume-title":"The reviewing of object files: Object-specific integration of information. Cognitive psychology","author":"Kahneman Daniel","year":"1992","unstructured":"Daniel Kahneman, Anne Treisman, and Brian J Gibbs. 1992. The reviewing of object files: Object-specific integration of information. Cognitive psychology, Vol. 24, 2 (1992), 175--219."},{"key":"e_1_3_2_2_26_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_2_27_1","volume-title":"International Conference on Learning Representations.","author":"Kipf Thomas","year":"2021","unstructured":"Thomas Kipf, Gamaleldin Fathy Elsayed, Aravindh Mahendran, Austin Stone, Sara Sabour, Georg Heigold, Rico Jonschkowski, Alexey Dosovitskiy, and Klaus Greff. 2021. Conditional Object-Centric Learning from Video. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_28_1","volume-title":"Elise Van der Pol, and Max Welling","author":"Kipf Thomas","year":"2019","unstructured":"Thomas Kipf, Elise Van der Pol, and Max Welling. 2019. Contrastive learning of structured world models. arXiv preprint arXiv:1911.12247 (2019)."},{"key":"e_1_3_2_2_29_1","volume-title":"Yee Whye Teh, and Ingmar Posner","author":"Kosiorek Adam","year":"2018","unstructured":"Adam Kosiorek, Hyunjik Kim, Yee Whye Teh, and Ingmar Posner. 2018. Sequential attend, infer, repeat: Generative modelling of moving objects. Advances in Neural Information Processing Systems, Vol. 31 (2018)."},{"key":"e_1_3_2_2_30_1","volume-title":"Intuitive physics: Current research and controversies. Trends in cognitive sciences","author":"Kubricht James R","year":"2017","unstructured":"James R Kubricht, Keith J Holyoak, and Hongjing Lu. 2017. Intuitive physics: Current research and controversies. Trends in cognitive sciences, Vol. 21, 10 (2017), 749--759."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6322"},{"key":"e_1_3_2_2_32_1","volume-title":"Building machines that learn and think like people. Behavioral and brain sciences","author":"Lake Brenden M","year":"2017","unstructured":"Brenden M Lake, Tomer D Ullman, Joshua B Tenenbaum, and Samuel J Gershman. 2017. Building machines that learn and think like people. Behavioral and brain sciences, Vol. 40 (2017), e253."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.2965434"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/MITS.2021.3049404"},{"key":"e_1_3_2_2_35_1","volume-title":"International conference on machine learning. PMLR, 6140--6149","author":"Lin Zhixuan","year":"2020","unstructured":"Zhixuan Lin, Yi-Fu Wu, Skand Peri, Bofeng Fu, Jindong Jiang, and Sungjin Ahn. 2020. Improving generative imagination in object-centric world models. In International conference on machine learning. PMLR, 6140--6149."},{"key":"e_1_3_2_2_36_1","first-page":"11525","article-title":"Object-centric learning with slot attention","volume":"33","author":"Locatello Francesco","year":"2020","unstructured":"Francesco Locatello, Dirk Weissenborn, Thomas Unterthiner, Aravindh Mahendran, Georg Heigold, Jakob Uszkoreit, Alexey Dosovitskiy, and Thomas Kipf. 2020. Object-centric learning with slot attention. Advances in Neural Information Processing Systems, Vol. 33 (2020), 11525--11538.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_37_1","volume-title":"Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983","author":"Loshchilov Ilya","year":"2016","unstructured":"Ilya Loshchilov and Frank Hutter. 2016. Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449998"},{"key":"e_1_3_2_2_39_1","volume-title":"When physics meets machine learning: A survey of physics-informed machine learning. arXiv preprint arXiv:2203.16797","author":"Meng Chuizheng","year":"2022","unstructured":"Chuizheng Meng, Sungyong Seo, Defu Cao, Sam Griesemer, and Yan Liu. 2022. When physics meets machine learning: A survey of physics-informed machine learning. arXiv preprint arXiv:2203.16797 (2022)."},{"key":"e_1_3_2_2_40_1","volume-title":"When it all falls down: the relationship between intuitive physics and spatial cognition. Cognitive research: principles and implications","author":"Mitko Alex","year":"2020","unstructured":"Alex Mitko and Jason Fischer. 2020. When it all falls down: the relationship between intuitive physics and spatial cognition. Cognitive research: principles and implications, Vol. 5 (2020), 1--13."},{"key":"e_1_3_2_2_41_1","volume-title":"Intuitive physics learning in a deep-learning model inspired by developmental psychology. Nature human behaviour","author":"Piloto Luis S","year":"2022","unstructured":"Luis S Piloto, Ari Weinstein, Peter Battaglia, and Matthew Botvinick. 2022. Intuitive physics learning in a deep-learning model inspired by developmental psychology. Nature human behaviour, Vol. 6, 9 (2022), 1257--1267."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.1971.10482356"},{"key":"e_1_3_2_2_43_1","unstructured":"Google Research. 2020. Google Scanned Objects. https:\/\/app.ignitionrobotics.org\/GoogleResearch\/fuel\/collections\/Google%20Scanned%20Objects"},{"key":"e_1_3_2_2_44_1","volume-title":"Mathieu Bernard, Adam Lerer, Rob Fergus, V\u00e9ronique Izard, and Emmanuel Dupoux.","author":"Riochet Ronan","year":"2018","unstructured":"Ronan Riochet, Mario Ynocente Castro, Mathieu Bernard, Adam Lerer, Rob Fergus, V\u00e9ronique Izard, and Emmanuel Dupoux. 2018. Intphys: A framework and benchmark for visual intuitive physics reasoning. arXiv preprint arXiv:1803.07616 (2018)."},{"key":"e_1_3_2_2_45_1","first-page":"18181","article-title":"Simple unsupervised object-centric learning for complex and naturalistic videos","volume":"35","author":"Singh Gautam","year":"2022","unstructured":"Gautam Singh, Yi-Fu Wu, and Sungjin Ahn. 2022. Simple unsupervised object-centric learning for complex and naturalistic videos. Advances in Neural Information Processing Systems, Vol. 35 (2022), 18181--18196.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_46_1","volume-title":"The promise of artificial intelligence: reckoning and judgment","author":"Smith Brian Cantwell","unstructured":"Brian Cantwell Smith. 2019. The promise of artificial intelligence: reckoning and judgment. Mit Press."},{"key":"e_1_3_2_2_47_1","volume-title":"R-sqair: Relational sequential attend, infer, repeat. arXiv preprint arXiv:1910.05231","author":"Stanic Aleksandar","year":"2019","unstructured":"Aleksandar Stanic and J\u00fcrgen Schmidhuber. 2019. R-sqair: Relational sequential attend, infer, repeat. arXiv preprint arXiv:1910.05231 (2019)."},{"key":"e_1_3_2_2_48_1","volume-title":"Graphical models for visual object recognition and tracking. Ph.,D. Dissertation","author":"Sudderth Erik Blaine","unstructured":"Erik Blaine Sudderth. 2006. Graphical models for visual object recognition and tracking. Ph.,D. Dissertation. Massachusetts Institute of Technology."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02227"},{"key":"e_1_3_2_2_50_1","volume-title":"Mind games: Game engines as an architecture for intuitive physics. Trends in cognitive sciences","author":"Ullman Tomer D","year":"2017","unstructured":"Tomer D Ullman, Elizabeth Spelke, Peter Battaglia, and Joshua B Tenenbaum. 2017. Mind games: Game engines as an architecture for intuitive physics. Trends in cognitive sciences, Vol. 21, 9 (2017), 649--665."},{"key":"e_1_3_2_2_51_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_52_1","volume-title":"Conference on Robot Learning. PMLR, 1439--1456","author":"Veerapaneni Rishi","year":"2020","unstructured":"Rishi Veerapaneni, John D Co-Reyes, Michael Chang, Michael Janner, Chelsea Finn, Jiajun Wu, Joshua Tenenbaum, and Sergey Levine. 2020. Entity abstraction in visual model-based reinforcement learning. In Conference on Robot Learning. PMLR, 1439--1456."},{"key":"e_1_3_2_2_53_1","volume-title":"Object-Centric Video Prediction Via Decoupling of Object Dynamics and Interactions. In 2023 IEEE International Conference on Image Processing (ICIP). IEEE, 570--574","author":"Villar-Corrales Angel","year":"2023","unstructured":"Angel Villar-Corrales, Ismail Wahdan, and Sven Behnke. 2023. Object-Centric Video Prediction Via Decoupling of Object Dynamics and Interactions. In 2023 IEEE International Conference on Image Processing (ICIP). IEEE, 570--574."},{"key":"e_1_3_2_2_54_1","volume-title":"Slot-VAE: Object-Centric Scene Generation with Slot Attention. arXiv preprint arXiv:2306.06997","author":"Wang Yanbo","year":"2023","unstructured":"Yanbo Wang, Letao Liu, and Justin Dauwels. 2023. Slot-VAE: Object-Centric Scene Generation with Slot Attention. arXiv preprint arXiv:2306.06997 (2023)."},{"key":"e_1_3_2_2_55_1","volume-title":"Image quality assessment: from error visibility to structural similarity","author":"Wang Zhou","year":"2004","unstructured":"Zhou Wang, Alan C Bovik, Hamid R Sheikh, and Eero P Simoncelli. 2004. Image quality assessment: from error visibility to structural similarity. IEEE transactions on image processing, Vol. 13, 4 (2004), 600--612."},{"key":"e_1_3_2_2_56_1","volume-title":"Spatial broadcast decoder: A simple architecture for learning disentangled representations in vaes. arXiv preprint arXiv:1901.07017","author":"Watters Nicholas","year":"2019","unstructured":"Nicholas Watters, Loic Matthey, Christopher P Burgess, and Alexander Lerchner. 2019. Spatial broadcast decoder: A simple architecture for learning disentangled representations in vaes. arXiv preprint arXiv:1901.07017 (2019)."},{"key":"e_1_3_2_2_57_1","volume-title":"Visual interaction networks: Learning a physics simulator from video. Advances in neural information processing systems","author":"Watters Nicholas","year":"2017","unstructured":"Nicholas Watters, Daniel Zoran, Theophane Weber, Peter Battaglia, Razvan Pascanu, and Andrea Tacchetti. 2017. Visual interaction networks: Learning a physics simulator from video. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_58_1","volume-title":"Unmasking the inductive biases of unsupervised object representations for video sequences. arXiv preprint arXiv:2006.07034","author":"Weis Marissa A","year":"2020","unstructured":"Marissa A Weis, Kashyap Chitta, Yash Sharma, Wieland Brendel, Matthias Bethge, Andreas Geiger, and Alexander S Ecker. 2020. Unmasking the inductive biases of unsupervised object representations for video sequences. arXiv preprint arXiv:2006.07034, Vol. 2 (2020)."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3547138"},{"key":"e_1_3_2_2_60_1","volume-title":"SlotFormer: Unsupervised Visual Dynamics Simulation with Object-Centric Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=TFbwV6I0VLg","author":"Wu Ziyi","year":"2023","unstructured":"Ziyi Wu, Nikita Dvornik, Klaus Greff, Thomas Kipf, and Animesh Garg. 2023. SlotFormer: Unsupervised Visual Dynamics Simulation with Object-Centric Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=TFbwV6I0VLg"},{"key":"e_1_3_2_2_61_1","first-page":"28023","article-title":"Segmenting moving objects via an object-centric layered representation","volume":"35","author":"Xie Junyu","year":"2022","unstructured":"Junyu Xie, Weidi Xie, and Andrew Zisserman. 2022. Segmenting moving objects via an object-centric layered representation. Advances in Neural Information Processing Systems, Vol. 35 (2022), 28023--28036.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_62_1","volume-title":"End-to-end slot alignment and recognition for cross-lingual NLU. arXiv preprint arXiv:2004.14353","author":"Xu Weijia","year":"2020","unstructured":"Weijia Xu, Batool Haider, and Saab Mansour. 2020. End-to-end slot alignment and recognition for cross-lingual NLU. arXiv preprint arXiv:2004.14353 (2020)."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00709"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2022.3162719"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449939"},{"key":"e_1_3_2_2_66_1","volume-title":"Clevrer: Collision events for video representation and reasoning. arXiv preprint arXiv:1910.01442","author":"Yi Kexin","year":"2019","unstructured":"Kexin Yi, Chuang Gan, Yunzhu Li, Pushmeet Kohli, Jiajun Wu, Antonio Torralba, and Joshua B Tenenbaum. 2019. Clevrer: Collision events for video representation and reasoning. arXiv preprint arXiv:1910.01442 (2019)."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2020.2984443"},{"key":"e_1_3_2_2_68_1","volume-title":"Proceedings of the Asian Conference on Computer Vision. 1976--1994","author":"Zhang Chuhan","year":"2022","unstructured":"Chuhan Zhang, Ankush Gupta, and Andrew Zisserman. 2022. Is an Object-Centric Video Representation Beneficial for Transfer?. In Proceedings of the Asian Conference on Computer Vision. 1976--1994."},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artmed.2020.101856"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01027"}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3690624.3709168","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3690624.3709168","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,16]],"date-time":"2025-08-16T15:38:55Z","timestamp":1755358735000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3690624.3709168"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,20]]},"references-count":71,"alternative-id":["10.1145\/3690624.3709168","10.1145\/3690624"],"URL":"https:\/\/doi.org\/10.1145\/3690624.3709168","relation":{},"subject":[],"published":{"date-parts":[[2025,7,20]]},"assertion":[{"value":"2025-07-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}