{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:13:04Z","timestamp":1763190784582,"version":"3.45.0"},"reference-count":45,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11227165","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["ORSA-T: Multi-View Object-Centric Scene Representation Learning with Slot Attention and Transformer"],"prefix":"10.1109","author":[{"given":"Henri","family":"Placek","sequence":"first","affiliation":[{"name":"University of London,Department of Computer Science City St George&#x2019;s,United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chris","family":"Child","sequence":"additional","affiliation":[{"name":"University of London,Department of Computer Science City St George&#x2019;s,United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tillman","family":"Weyde","sequence":"additional","affiliation":[{"name":"University of London,Department of Computer Science City St George&#x2019;s,United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/0010-0285(92)90007-O"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1111\/j.1551-6709.2010.01110.x"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1111\/j.1551-6709.2010.01127.x"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3286184"},{"key":"ref5","first-page":"5","article-title":"Building machines that learn and think like people","author":"Tenenbaum","journal-title":"AAMAS 2018"},{"key":"ref6","article-title":"On the binding problem in artificial neural networks","author":"Greff","year":"2020","journal-title":"CoRR"},{"article-title":"Grounding physical concepts of objects and events through dynamic visual reasoning","volume-title":"ICLR 2021","author":"Chen","key":"ref7"},{"key":"ref8","article-title":"COBRA: data-efficient model-based RL through unsupervised object discovery and curiosity-driven exploration","volume-title":"CoRR","author":"Watters","year":"2019"},{"key":"ref9","article-title":"Strategic object oriented reinforcement learning","author":"Keramati","year":"2018","journal-title":"CoRR"},{"key":"ref10","article-title":"Monet: Unsupervised scene decomposition and representation","author":"Burgess","year":"2019","journal-title":"CoRR"},{"key":"ref11","first-page":"2424","article-title":"Multi-object representation learning with iterative variational inference","volume-title":"ICML 2019","volume":"97","author":"Greff"},{"article-title":"SPACE: unsupervised object-oriented scene representation via spatial attention and decomposition","volume-title":"ICLR 2020","author":"Lin","key":"ref12"},{"article-title":"Object-centric learning with slot attention","volume-title":"NeurIPS 2020","author":"Locatello","key":"ref13"},{"key":"ref14","first-page":"2507","article-title":"Invariant slot attention: Object discovery with slot-centric reference frames","volume-title":"ICML 2023","volume":"202","author":"Biza"},{"key":"ref15","first-page":"32 694","article-title":"Object representations as fixed points: Training iterative refinement algorithms with implicit differentiation","volume-title":"NeurIPS 2022","volume":"35","author":"Chang"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6170"},{"key":"ref17","article-title":"Scene representation networks: Continuous 3d-structure-aware neural scene representations","volume-title":"NeurIPS 2019","volume":"32","author":"Sitzmann"},{"key":"ref18","first-page":"5742","article-title":"Nerf-vae: A geometry aware 3d scene generative model","volume-title":"ICML 2021","volume":"139","author":"Kosiorek"},{"key":"ref19","article-title":"Decomposing 3d scenes into objects via unsupervised volume segmentation","author":"Stelzner","year":"2021","journal-title":"CoRR"},{"key":"ref20","first-page":"20 146","article-title":"Simone: View-invariant, temporally-abstracted object representations via unsupervised video decomposition","volume-title":"NeurIPS 2021","volume":"34","author":"Kabra"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20880"},{"key":"ref22","article-title":"Learning to infer 3d object models from images","author":"Chen","year":"2020","journal-title":"CoRR"},{"key":"ref23","first-page":"9512","article-title":"Object scene representation transformer","volume-title":"NeurIPS 2022","volume":"35","author":"Sajjadi"},{"key":"ref24","first-page":"5656","article-title":"Learning object-centric representations of multi-object scenes from multiple views","volume-title":"NeurIPS 2020","volume":"33","author":"Li"},{"article-title":"Conditional object-centric learning from video","volume-title":"ICLR 2022","author":"Kipf","key":"ref25"},{"key":"ref26","first-page":"18 181","article-title":"Simple unsupervised object-centric learning for complex and naturalistic videos","volume-title":"NeurIPS 2022","volume":"35","author":"Singh"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref28","article-title":"Attend, infer, repeat: Fast scene understanding with generative models","volume-title":"NeurIPS 2016","volume":"29","author":"Eslami"},{"article-title":"GENESIS: generative scene inference and sampling with object-centric latent representations","volume-title":"ICLR 2020","author":"Engelcke","key":"ref29"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013412"},{"key":"ref31","first-page":"2437","article-title":"Deepvoxels: Learning persistent 3d feature embeddings","volume-title":"CVPR 2019","author":"Sitzmann"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_24"},{"key":"ref33","article-title":"Sequential attend, infer, repeat: Generative modelling of moving objects","volume-title":"NeurIPS 2018","volume":"31","author":"Kosiorek"},{"key":"ref34","first-page":"183:1","article-title":"Benchmarking unsupervised object representations for video sequences","volume":"22","author":"Weis","year":"2021","journal-title":"J. Mach. Learn. Res"},{"key":"ref35","first-page":"28 940","article-title":"Savi++: Towards end-to-end object-centric learning from real-world videos","volume-title":"NeurIPS 2022","volume":"35","author":"Elsayed"},{"article-title":"Slotformer: Unsupervised visual dynamics simulation with object-centric models","volume-title":"ICLR 2023","author":"Wu","key":"ref36"},{"article-title":"Unsupervised discovery of object radiance fields","volume-title":"ICLR 2022","author":"Yu","key":"ref37"},{"key":"ref38","article-title":"Unsupervised discovery and composition of object light fields","author":"Smith","year":"2023","journal-title":"Transactions on Machine Learning Research"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-46805-6_19"},{"article-title":"Improving object-centric learning with query optimization","volume-title":"ICLR 2023","author":"Jia","key":"ref40"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"ICLR 2021","author":"Dosovitskiy","key":"ref41"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.215"},{"article-title":"Adam: A method for stochastic optimization","volume-title":"ICLR 2015","author":"Kingma","key":"ref44"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1002\/nav.3800020109"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11227165.pdf?arnumber=11227165","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:09:17Z","timestamp":1763190557000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11227165\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":45,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11227165","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}