{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:47:31Z","timestamp":1778082451570,"version":"3.51.4"},"publisher-location":"Cham","reference-count":75,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729942","type":"print"},{"value":"9783031729959","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72995-9_13","type":"book-chapter","created":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T19:16:10Z","timestamp":1732389370000},"page":"215-233","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Betrayed by\u00a0Attention: A Simple yet\u00a0Effective Approach for\u00a0Self-supervised Video Object Segmentation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7033-774X","authenticated-orcid":false,"given":"Shuangrui","family":"Ding","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0378-6438","authenticated-orcid":false,"given":"Rui","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4715-1338","authenticated-orcid":false,"given":"Haohang","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8865-7896","authenticated-orcid":false,"given":"Dahua","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4552-0029","authenticated-orcid":false,"given":"Hongkai","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,24]]},"reference":[{"key":"13_CR1","unstructured":"Aydemir, G., Xie, W., Guney, F.: Self-supervised object-centric learning for videos. In: Thirty-seventh Conference on Neural Information Processing Systems (2023). https:\/\/openreview.net\/forum?id=919tWtJPXe"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Bao, L., Wu, B., Liu, W.: Cnn in mrf: video object segmentation via inference in a cnn-based higher-order spatio-temporal mrf. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5977\u20135986 (2018)","DOI":"10.1109\/CVPR.2018.00626"},{"key":"13_CR3","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML, vol.\u00a02, p.\u00a04 (2021)"},{"key":"13_CR4","doi-asserted-by":"crossref","unstructured":"Bian, Z., Jabri, A., Efros, A.A., Owens, A.: Learning pixel trajectories with multiscale contrastive random walks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6508\u20136519 (2022)","DOI":"10.1109\/CVPR52688.2022.00640"},{"key":"13_CR5","first-page":"33371","volume":"35","author":"A Bielski","year":"2022","unstructured":"Bielski, A., Favaro, P.: Move: unsupervised movable object segmentation and detection. Adv. Neural. Inf. Process. Syst. 35, 33371\u201333386 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Caelles, S., Maninis, K.K., Pont-Tuset, J., Leal-Taix\u00e9, L., Cremers, D., Van\u00a0Gool, L.: One-shot video object segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 221\u2013230 (2017)","DOI":"10.1109\/CVPR.2017.565"},{"key":"13_CR7","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"13_CR8","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"13_CR9","doi-asserted-by":"publisher","unstructured":"Chen, Y., et al.: Sdae: self-distillated masked autoencoder. In: European Conference on Computer Vision, pp. 108\u2013124. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-20056-4_7","DOI":"10.1007\/978-3-031-20056-4_7"},{"key":"13_CR10","unstructured":"Choudhury, S., Karazija, L., Laina, I., Vedaldi, A., Rupprecht, C.: Guess what moves: unsupervised video and image segmentation by anticipating motion. In: British Machine Vision Conference (BMVC) (2022)"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Ding, S., et al.: Motion-aware contrastive video representation learning via foreground-background merging. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9716\u20139726 (2022)","DOI":"10.1109\/CVPR52688.2022.00949"},{"key":"13_CR12","doi-asserted-by":"crossref","unstructured":"Ding, S., Qian, R., Xiong, H.: Dual contrastive learning for spatio-temporal representation. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 5649\u20135658 (2022)","DOI":"10.1145\/3503161.3547783"},{"key":"13_CR13","unstructured":"Ding, S., et al.: Motion-inductive self-supervised object discovery in videos. arXiv preprint arXiv:2210.00221 (2022)"},{"key":"13_CR14","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations (2021). https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Dutt\u00a0Jain, S., Xiong, B., Grauman, K.: Fusionseg: learning to combine motion and appearance for fully automatic segmentation of generic objects in videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3664\u20133673 (2017)","DOI":"10.1109\/CVPR.2017.228"},{"key":"13_CR16","first-page":"28940","volume":"35","author":"G Elsayed","year":"2022","unstructured":"Elsayed, G., Mahendran, A., van Steenkiste, S., Greff, K., Mozer, M.C., Kipf, T.: Savi++: towards end-to-end object-centric learning from real-world videos. Adv. Neural. Inf. Process. Syst. 35, 28940\u201328954 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Fan, D.P., Wang, W., Cheng, M.M., Shen, J.: Shifting more attention to video salient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8554\u20138564 (2019)","DOI":"10.1109\/CVPR.2019.00875"},{"issue":"3","key":"13_CR18","doi-asserted-by":"publisher","first-page":"1341","DOI":"10.1109\/TITS.2020.2972974","volume":"22","author":"D Feng","year":"2020","unstructured":"Feng, D., et al.: Deep multi-modal object detection and semantic segmentation for autonomous driving: Datasets, methods, and challenges. IEEE Trans. Intell. Transp. Syst. 22(3), 1341\u20131360 (2020)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"13_CR19","doi-asserted-by":"crossref","unstructured":"Greff, K., et al.: Kubric: a scalable dataset generator (2022)","DOI":"10.1109\/CVPR52688.2022.00373"},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Guo, Y., Yang, C., Rao, A., Agrawala, M., Lin, D., Dai, B.: Sparsectrl: adding sparse controls to text-to-video diffusion models. arXiv preprint arXiv:2311.16933 (2023)","DOI":"10.1007\/978-3-031-72946-1_19"},{"key":"13_CR21","unstructured":"Guo, Y., et al.: Animatediff: animate your personalized text-to-image diffusion models without specific tuning. In: The Twelfth International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=Fx2SbBgcte"},{"key":"13_CR22","unstructured":"Hamilton, M., Zhang, Z., Hariharan, B., Snavely, N., Freeman, W.T.: Unsupervised semantic segmentation by distilling feature correspondences. In: International Conference on Learning Representations (2022). https:\/\/openreview.net\/forum?id=SaKO6z6Hl0c"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9729\u20139738 (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"13_CR24","doi-asserted-by":"publisher","first-page":"97","DOI":"10.1007\/978-3-031-19821-2_6","volume-title":"Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXI","author":"Y Hu","year":"2022","unstructured":"Hu, Y., Wang, R., Zhang, K., Gao, Y.: Semantic-aware fine-grained correspondence. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXI, pp. 97\u2013115. Springer Nature Switzerland, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19821-2_6"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Hu, Y.T., Huang, J.B., Schwing, A.G.: Videomatch: matching based video object segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 54\u201370 (2018)","DOI":"10.1007\/978-3-030-01237-3_4"},{"key":"13_CR26","first-page":"19545","volume":"33","author":"A Jabri","year":"2020","unstructured":"Jabri, A., Owens, A., Efros, A.: Space-time correspondence as a contrastive random walk. Adv. Neural. Inf. Process. Syst. 33, 19545\u201319560 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"13_CR27","doi-asserted-by":"crossref","unstructured":"Johnander, J., Danelljan, M., Brissman, E., Khan, F.S., Felsberg, M.: A generative appearance model for end-to-end video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8953\u20138962 (2019)","DOI":"10.1109\/CVPR.2019.00916"},{"issue":"3","key":"13_CR28","doi-asserted-by":"publisher","first-page":"241","DOI":"10.1007\/BF02289588","volume":"32","author":"SC Johnson","year":"1967","unstructured":"Johnson, S.C.: Hierarchical clustering schemes. Psychometrika 32(3), 241\u2013254 (1967)","journal-title":"Psychometrika"},{"key":"13_CR29","unstructured":"Kipf, T., et al.: Conditional object-centric learning from video. In: International Conference on Learning Representations (2022). https:\/\/openreview.net\/forum?id=aD7uesX1GF_"},{"key":"13_CR30","unstructured":"Lai, Z., Xie, W.: Self-supervised learning for video correspondence flow. In: BMVC (2019)"},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Lai, Z., Lu, E., Xie, W.: Mast: a memory-augmented self-supervised tracker. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6479\u20136488 (2020)","DOI":"10.1109\/CVPR42600.2020.00651"},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Li, L., Wang, W., Zhou, T., Li, J., Yang, Y.: Unified mask embedding and correspondence learning for self-supervised video segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18706\u201318716 (2023)","DOI":"10.1109\/CVPR52729.2023.01794"},{"key":"13_CR33","doi-asserted-by":"crossref","unstructured":"Li, X., Loy, C.C.: Video object segmentation with joint re-identification and attention-aware mask propagation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 90\u2013105 (2018)","DOI":"10.1007\/978-3-030-01219-9_6"},{"key":"13_CR34","unstructured":"Li, X., Liu, S., De\u00a0Mello, S., Wang, X., Kautz, J., Yang, M.H.: Joint-task self-supervised learning for temporal correspondence. In: Wallach, H., Larochelle, H., Beygelzimer, A., d\u2019Alch\u00e9-Buc, F., Fox, E., Garnett, R. (eds.) Advances in Neural Information Processing Systems. vol.\u00a032. Curran Associates, Inc. (2019). https:\/\/proceedings.neurips.cc\/paper\/2019\/file\/140f6969d5213fd0ece03148e62e461e-Paper.pdf"},{"key":"13_CR35","doi-asserted-by":"crossref","unstructured":"Lian, L., Wu, Z., Yu, S.X.: Bootstrapping objectness from videos by relaxed common fate and visual grouping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14582\u201314591 (2023)","DOI":"10.1109\/CVPR52729.2023.01401"},{"key":"13_CR36","unstructured":"Liu, R., Wu, Z., Yu, S., Lin, S.: The emergence of objectness: Learning zero-shot segmentation from videos. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems. vol.\u00a034, pp. 13137\u201313152. Curran Associates, Inc. (2021). https:\/\/proceedings.neurips.cc\/paper\/2021\/file\/6d9cb7de5e8ac30bd5e8734bc96a35c1-Paper.pdf"},{"key":"13_CR37","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: Promoting semantic connectivity: dual nearest neighbors contrastive learning for unsupervised domain generalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3510\u20133519 (June 2023)","DOI":"10.1109\/CVPR52729.2023.00342"},{"key":"13_CR38","first-page":"11525","volume":"33","author":"F Locatello","year":"2020","unstructured":"Locatello, F., et al.: Object-centric learning with slot attention. Adv. Neural. Inf. Process. Syst. 33, 11525\u201311538 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"13_CR39","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (2018)"},{"key":"13_CR40","doi-asserted-by":"crossref","unstructured":"Melas-Kyriazi, L., Rupprecht, C., Laina, I., Vedaldi, A.: Deep spectral methods: a surprisingly strong baseline for unsupervised semantic segmentation and localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8364\u20138375 (2022)","DOI":"10.1109\/CVPR52688.2022.00818"},{"key":"13_CR41","unstructured":"Oquab, M., et\u00a0al.: Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"13_CR42","doi-asserted-by":"publisher","first-page":"7889","DOI":"10.1109\/TIP.2021.3108405","volume":"30","author":"PW Patil","year":"2021","unstructured":"Patil, P.W., Dudhane, A., Kulkarni, A., Murala, S., Gonde, A.B., Gupta, S.: An unified recurrent video object segmentation framework for various surveillance environments. IEEE Trans. Image Process. 30, 7889\u20137902 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"13_CR43","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Khoreva, A., Benenson, R., Schiele, B., Sorkine-Hornung, A.: Learning video object segmentation from static images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2663\u20132672 (2017)","DOI":"10.1109\/CVPR.2017.372"},{"key":"13_CR44","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Pont-Tuset, J., McWilliams, B., Van\u00a0Gool, L., Gross, M., Sorkine-Hornung, A.: A benchmark dataset and evaluation methodology for video object segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 724\u2013732 (2016)","DOI":"10.1109\/CVPR.2016.85"},{"key":"13_CR45","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Van\u00a0Gool, L.: The 2017 davis challenge on video object segmentation. arXiv preprint arXiv:1704.00675 (2017)"},{"key":"13_CR46","doi-asserted-by":"publisher","unstructured":"Qian, R., Ding, S., Liu, X., Lin, D.: Static and dynamic concepts for self-supervised video representation learning. In: European Conference on Computer Vision, pp. 145\u2013164. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19809-0_9","DOI":"10.1007\/978-3-031-19809-0_9"},{"key":"13_CR47","doi-asserted-by":"crossref","unstructured":"Qian, R., Ding, S., Liu, X., Lin, D.: Semantics meets temporal correspondence: Self-supervised object-centric learning in videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16675\u201316687 (2023)","DOI":"10.1109\/ICCV51070.2023.01529"},{"key":"13_CR48","unstructured":"Qian, R., et al.: Streaming long video understanding with large language models. arXiv preprint arXiv:2405.16009 (2024)"},{"key":"13_CR49","doi-asserted-by":"crossref","unstructured":"Qian, R., et al.: Enhancing self-supervised video representation learning via multi-level feature optimization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7990\u20138001 (2021)","DOI":"10.1109\/ICCV48922.2021.00789"},{"key":"13_CR50","doi-asserted-by":"crossref","unstructured":"Rambhatla, S.S., Misra, I., Chellappa, R., Shrivastava, A.: Most: multiple object localization with self-supervised transformers for object discovery. arXiv preprint arXiv:2304.05387 (2023)","DOI":"10.1109\/ICCV51070.2023.01450"},{"key":"13_CR51","doi-asserted-by":"crossref","unstructured":"Salehi, M., Gavves, E., Snoek, C.G., Asano, Y.M.: Time does tell: Self-supervised time-tuning of dense image representations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16536\u201316547 (2023)","DOI":"10.1109\/ICCV51070.2023.01516"},{"issue":"8","key":"13_CR52","doi-asserted-by":"publisher","first-page":"888","DOI":"10.1109\/34.868688","volume":"22","author":"J Shi","year":"2000","unstructured":"Shi, J., Malik, J.: Normalized cuts and image segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 22(8), 888\u2013905 (2000)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"13_CR53","unstructured":"Sim\u00e9oni, O., et al.: Localizing objects with self-supervised transformers and no labels. In: BMVC 2021-32nd British Machine Vision Conference (2021)"},{"key":"13_CR54","first-page":"18181","volume":"35","author":"G Singh","year":"2022","unstructured":"Singh, G., Wu, Y.F., Ahn, S.: Simple unsupervised object-centric learning for complex and naturalistic videos. Adv. Neural. Inf. Process. Syst. 35, 18181\u201318196 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"13_CR55","doi-asserted-by":"crossref","unstructured":"Tian, J., Aggarwal, L., Colaco, A., Kira, Z., Gonzalez-Franco, M.: Diffuse, attend, and segment: Unsupervised zero-shot segmentation using stable diffusion. arXiv preprint arXiv:2308.12469 (2023)","DOI":"10.1109\/CVPR52733.2024.00341"},{"key":"13_CR56","doi-asserted-by":"crossref","unstructured":"Vondrick, C., Shrivastava, A., Fathi, A., Guadarrama, S., Murphy, K.: Tracking emerges by colorizing videos. In: Proceedings of the European conference on computer vision (ECCV), pp. 391\u2013408 (2018)","DOI":"10.1007\/978-3-030-01261-8_24"},{"key":"13_CR57","doi-asserted-by":"crossref","unstructured":"Wang, N., Song, Y., Ma, C., Zhou, W., Liu, W., Li, H.: Unsupervised deep tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1308\u20131317 (2019)","DOI":"10.1109\/CVPR.2019.00140"},{"key":"13_CR58","doi-asserted-by":"crossref","unstructured":"Wang, X., Jabri, A., Efros, A.A.: Learning correspondence from the cycle-consistency of time. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2566\u20132576 (2019)","DOI":"10.1109\/CVPR.2019.00267"},{"key":"13_CR59","doi-asserted-by":"crossref","unstructured":"Wang, X., Girdhar, R., Yu, S.X., Misra, I.: Cut and learn for unsupervised object detection and instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3124\u20133134 (June 2023)","DOI":"10.1109\/CVPR52729.2023.00305"},{"key":"13_CR60","doi-asserted-by":"crossref","unstructured":"Wang, X., Misra, I., Zeng, Z., Girdhar, R., Darrell, T.: Videocutler: Surprisingly simple unsupervised video instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22755\u201322764 (2024)","DOI":"10.1109\/CVPR52733.2024.02147"},{"key":"13_CR61","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Tokencut: segmenting objects in images and videos with self-supervised transformer and normalized cut. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)","DOI":"10.1109\/TPAMI.2023.3305122"},{"key":"13_CR62","unstructured":"Wang, Y., et al.: BarleRIa: an efficient tuning framework for referring image segmentation. In: The Twelfth International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=wHLDHRkmEu"},{"key":"13_CR63","doi-asserted-by":"crossref","unstructured":"Xie, C., Xiang, Y., Harchaoui, Z., Fox, D.: Object discovery in videos as foreground motion clustering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9994\u201310003 (2019)","DOI":"10.1109\/CVPR.2019.01023"},{"key":"13_CR64","unstructured":"Xie, J., Xie, W., Zisserman, A.: Segmenting moving objects via an object-centric layered representation. In: Advances in Neural Information Processing Systems (2022)"},{"key":"13_CR65","unstructured":"Xu, H., Ding, S., Zhang, X., Xiong, H., Tian, Q.: Masked autoencoders are robust data augmentors. arXiv preprint arXiv:2206.04846 (2022)"},{"key":"13_CR66","doi-asserted-by":"crossref","unstructured":"Xu, J., Wang, X.: Rethinking self-supervised correspondence learning: a video frame-level similarity perspective. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10075\u201310085 (2021)","DOI":"10.1109\/ICCV48922.2021.00992"},{"key":"13_CR67","doi-asserted-by":"crossref","unstructured":"Yang, C., Lamdouar, H., Lu, E., Zisserman, A., Xie, W.: Self-supervised video object segmentation by motion grouping. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 7177\u20137188 (October 2021)","DOI":"10.1109\/ICCV48922.2021.00709"},{"key":"13_CR68","doi-asserted-by":"crossref","unstructured":"Yang, L., Fan, Y., Xu, N.: Video instance segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00529"},{"key":"13_CR69","doi-asserted-by":"crossref","unstructured":"Yang, Y., Lai, B., Soatto, S.: Dystab: unsupervised object segmentation via dynamic-static bootstrapping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2826\u20132836 (June 2021)","DOI":"10.1109\/CVPR46437.2021.00285"},{"key":"13_CR70","doi-asserted-by":"crossref","unstructured":"Yang, Y., Loquercio, A., Scaramuzza, D., Soatto, S.: Unsupervised moving object detection via contextual information separation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (June 2019)","DOI":"10.1109\/CVPR.2019.00097"},{"key":"13_CR71","doi-asserted-by":"crossref","unstructured":"Ye, V., Li, Z., Tucker, R., Kanazawa, A., Snavely, N.: Deformable sprites for unsupervised video decomposition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2657\u20132666 (2022)","DOI":"10.1109\/CVPR52688.2022.00268"},{"key":"13_CR72","unstructured":"Zadaianchuk, A., Kleindessner, M., Zhu, Y., Locatello, F., Brox, T.: Unsupervised semantic segmentation with self-supervised object-centric representations. In: The Eleventh International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=1_jFneF07YC"},{"key":"13_CR73","unstructured":"Zadaianchuk, A., Seitzer, M., Martius, G.: Object-centric learning for real-world videos by predicting temporal feature similarities. In: Thirty-seventh Conference on Neural Information Processing Systems (NeurIPS 2023) (2023)"},{"key":"13_CR74","doi-asserted-by":"crossref","unstructured":"Zhou, J., Pang, Z., Wang, Y.X.: Rmem: restricted memory banks improve video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18602\u201318611 (2024)","DOI":"10.1109\/CVPR52733.2024.01760"},{"key":"13_CR75","doi-asserted-by":"crossref","unstructured":"Ziegler, A., Asano, Y.M.: Self-supervised learning of object parts for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14502\u201314511 (2022)","DOI":"10.1109\/CVPR52688.2022.01410"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72995-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T20:05:19Z","timestamp":1732392319000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72995-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,24]]},"ISBN":["9783031729942","9783031729959"],"references-count":75,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72995-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,24]]},"assertion":[{"value":"24 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}