{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T02:39:00Z","timestamp":1784169540566,"version":"3.55.0"},"publisher-location":"Cham","reference-count":93,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729324","type":"print"},{"value":"9783031729331","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:00:00Z","timestamp":1727913600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:00:00Z","timestamp":1727913600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72933-1_14","type":"book-chapter","created":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T12:02:53Z","timestamp":1727870573000},"page":"238-256","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Appearance-Based Refinement for\u00a0Object-Centric Motion Segmentation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1123-493X","authenticated-orcid":false,"given":"Junyu","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8609-6826","authenticated-orcid":false,"given":"Weidi","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8945-8573","authenticated-orcid":false,"given":"Andrew","family":"Zisserman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,3]]},"reference":[{"key":"14_CR1","unstructured":"Aydemir, G., Xie, W., G\u00fcney, F.: Self-supervised object-centric learning for videos. In: NeurIPS (2023)"},{"key":"14_CR2","doi-asserted-by":"crossref","unstructured":"Bao, Z., Tokmakov, P., Jabri, A., Wang, Y.X., Gaidon, A., Hebert, M.: Discovering objects that can move. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01149"},{"key":"14_CR3","unstructured":"Bear, D., et\u00a0al.: Learning physical graph representations from visual scenes. In: NeurIPS (2020)"},{"key":"14_CR4","doi-asserted-by":"crossref","unstructured":"Bekuzarov, M., Bermudez, A., Lee, J.Y., Li, H.: Xmem++: production-level video segmentation from few annotated frames. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00065"},{"key":"14_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"433","DOI":"10.1007\/978-3-319-46484-8_26","volume-title":"Computer Vision \u2013 ECCV 2016","author":"P Bideau","year":"2016","unstructured":"Bideau, P., Learned-Miller, E.: It\u2019s moving! a probabilistic model for causal motion segmentation in moving camera videos. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9912, pp. 433\u2013449. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46484-8_26"},{"key":"14_CR6","doi-asserted-by":"crossref","unstructured":"Bideau, P., RoyChowdhury, A., Menon, R.R., Learned-Miller, E.: The best of both worlds: combining CNNs and geometric constraints for hierarchical motion segmentation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00060"},{"key":"14_CR7","doi-asserted-by":"crossref","unstructured":"Buch, S., Eyzaguirre, C., Gaidon, A., Wu, J., Fei-Fei, L., Niebles, J.C.: Revisiting the \u201cvideo\u201d in video-language understanding. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"14_CR8","unstructured":"Burgess, C.P., et al.: Monet: unsupervised scene decomposition and representation. arXiv preprint arXiv:1901.11390 (2019)"},{"key":"14_CR9","unstructured":"Caelles, S., Pont-Tuset, J., Perazzi, F., Montes, A., Maninis, K.K., Van Gool, L.: The 2019 davis challenge on vos: Unsupervised multi-object segmentation. arXiv preprint arXiv:1905.00737 (2019)"},{"key":"14_CR10","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., Joulin, A.: Emerging properties in self-supervised vision transformers. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"14_CR11","unstructured":"Cheng, B., Schwing, A.G., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. In: NeurIPS (2021)"},{"key":"14_CR12","unstructured":"Cho, D., Hong, S., Kang, S., Kim, J.: Key instance selection for unsupervised video object segmentation. arXiv preprint arXiv:1906.07851 (2019)"},{"key":"14_CR13","doi-asserted-by":"crossref","unstructured":"Cho, S., Lee, M., Lee, S., Lee, D., Lee, S.: Dual prototype attention for unsupervised video object segmentation. arXiv preprint arXiv:2211.12036 (2023)","DOI":"10.1109\/CVPR52733.2024.01820"},{"key":"14_CR14","unstructured":"Choudhury, S., Karazija, L., Laina, I., Vedaldi, A., Rupprecht, C.: Guess what moves: unsupervised video and image segmentation by anticipating motion. In: BMVC (2022)"},{"key":"14_CR15","doi-asserted-by":"crossref","unstructured":"Dave, A., Tokmakov, P., Ramanan, D.: Towards segmenting anything that moves. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00187"},{"key":"14_CR16","doi-asserted-by":"crossref","unstructured":"Delatolas, T., Kalogeiton, V., Papadopoulos, D.P.: Learning the what and how of annotation in video object segmentation. In: WACV (2024)","DOI":"10.1109\/WACV57701.2024.00680"},{"key":"14_CR17","unstructured":"Elsayed, G.F., Mahendran, A., van Steenkiste, S., Greff, K., Mozer, M.C., Kipf, T.: SAVi++: towards end-to-end object-centric learning from real-world videos. arXiv preprint arXiv:2206.07764 (2022)"},{"key":"14_CR18","unstructured":"Engelcke, M., Kosiorek, A.R., Jones, O.P., , Posner, I.: Genesis: generative scene inference and sampling with object-centric latent representations. In: ICLR (2020)"},{"key":"14_CR19","unstructured":"Eslami, S.M.A., et al.: Attend, infer, repeat: fast scene understanding with generative models. In: NeurIPS (2016)"},{"key":"14_CR20","doi-asserted-by":"crossref","unstructured":"Faktor, A., Irani, M.: Video segmentation by non-local consensus voting. In: BMVC (2014)","DOI":"10.5244\/C.28.21"},{"key":"14_CR21","doi-asserted-by":"crossref","unstructured":"Fan, K., et al.: Unsupervised open-vocabulary object localization in videos. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01264"},{"key":"14_CR22","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Convolutional two-stream network fusion for video action recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.213"},{"issue":"1","key":"14_CR24","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1016\/0166-2236(92)90344-8","volume":"15","author":"MA Goodale","year":"1992","unstructured":"Goodale, M.A., Milner, A.: Separate visual pathways for perception and action. Trends Neurosci. 15(1), 20\u201325 (1992)","journal-title":"Trends Neurosci."},{"key":"14_CR25","unstructured":"Greff, K., et al.: Multi-object representation learning with iterative variational inference. In: ICML (2019)"},{"key":"14_CR26","unstructured":"Greff, K., Rasmus, A., Berglund, M., Hao, T., Valpola, H., Schmidhuber, J.: Tagger: deep unsupervised perceptual grouping. In: NeurIPS (2016)"},{"key":"14_CR27","doi-asserted-by":"crossref","unstructured":"Griffin, B.A., Corso, J.J.: Bubblenets: Learning to select the guidance frame in video object segmentation by deep sorting frames. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00912"},{"key":"14_CR28","doi-asserted-by":"crossref","unstructured":"He, Z., Li, J., Liu, D., He, H., Barber, D.: Tracking by animation: Unsupervised learning of multi-object attentive trackers. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00141"},{"key":"14_CR29","unstructured":"Jabri, A., Owens, A., Efros, A.A.: Space-time correspondence as a contrastive random walk. In: NeurIPS (2020)"},{"key":"14_CR30","doi-asserted-by":"crossref","unstructured":"Jain, S.D., Xiong, B., Grauman, K.: Fusionseg: Learning to combine motion and appearance for fully automatic segmentation of generic objects in videos. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.228"},{"key":"14_CR31","unstructured":"Jiang, J., Janghorbani, S., de Melo, G., Ahn, S.: Scalor: generative world models with scalable object representations. In: ICLR (2020)"},{"key":"14_CR32","unstructured":"Kabra, R., et al.: Simone: view-invariant, temporally-abstracted object representations via unsupervised video decomposition. arXiv preprint arxiv2106.03849 (2021)"},{"key":"14_CR33","doi-asserted-by":"crossref","unstructured":"Keuper, M., Andres, B., Brox, T.: Motion trajectory segmentation via minimum cost multicuts. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.374"},{"key":"14_CR34","unstructured":"Kipf, T., et al.: Conditional object-centric learning from video. In: ICLR (2022)"},{"key":"14_CR35","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything. arXiv:2304.02643 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"14_CR36","unstructured":"Kosiorek, A., Kim, H., Teh, Y.W., Posner, I.: Sequential attend, infer, repeat: Generative modelling of moving objects. In: NeurIPS (2018)"},{"key":"14_CR37","doi-asserted-by":"crossref","unstructured":"Kuznetsova, A., Talati, A., Luo, Y., Simmons, K., Ferrari, V.: Efficient video annotation with visual interpolation and frame selection guidance. In: WACV (2021)","DOI":"10.1109\/WACV48630.2021.00311"},{"key":"14_CR38","doi-asserted-by":"crossref","unstructured":"Lai, Z., Lu, E., Xie, W.: MAST: a memory-augmented self-supervised tracker. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00651"},{"key":"14_CR39","unstructured":"Lai, Z., Xie, W.: Self-supervised learning for video correspondence flow. In: BMVC (2019)"},{"key":"14_CR40","unstructured":"Lamdouar, H., Xie, W., Zisserman, A.: Segmenting invisible moving objects. In: BMVC (2021)"},{"key":"14_CR41","doi-asserted-by":"crossref","unstructured":"Li, F., Kim, T., Humayun, A., Tsai, D., Rehg, J.M.: Video segmentation by tracking many figure-ground segments. In: ICCV (2013)","DOI":"10.1109\/ICCV.2013.273"},{"key":"14_CR42","doi-asserted-by":"crossref","unstructured":"Li, S., Seybold, B., Vorobyov, A., Fathi, A., Huang, Q., Kuo, C.C.J.: Instance embedding transfer to unsupervised video object segmentation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00683"},{"key":"14_CR43","unstructured":"Li, X., Liu, S., De\u00a0Mello, S., Wang, X., Kautz, J., Yang, M.H.: Joint-task self-supervised learning for temporal correspondence. In: NeurIPS (2019)"},{"key":"14_CR44","doi-asserted-by":"crossref","unstructured":"Liu, Q., Wu, J., Jiang, Y., Bai, X., Yuille, A., Bai, S.: Instmove: instance motion for object-centric video segmentation. arXiv preprint arXiv:2303.08132 (2023)","DOI":"10.1109\/CVPR52729.2023.00614"},{"key":"14_CR45","unstructured":"Locatello, F., et al.: Object-centric learning with slot attention. In: NeurIPS (2020)"},{"key":"14_CR46","doi-asserted-by":"crossref","unstructured":"Lu, X., Wang, W., Ma, C., Shen, J., Shao, L., Porikli, F.: See more, know more: unsupervised video object segmentation with co-attention siamese networks. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00374"},{"key":"14_CR47","doi-asserted-by":"crossref","unstructured":"Luiten, J., Zulfikar, I.E., Leibe, B.: Unovost: unsupervised offline video object segmentation and tracking. In: WACV (2020)","DOI":"10.1109\/WACV45572.2020.9093285"},{"key":"14_CR48","doi-asserted-by":"publisher","first-page":"76","DOI":"10.1016\/j.image.2018.09.003","volume":"71","author":"CY Ma","year":"2018","unstructured":"Ma, C.Y., Chen, M.H., Kira, Z., AlRegib, G.: TS-LSTM and temporal-inception: exploiting spatiotemporal dynamics for activity recognition. Sig. Process. Image Commun. 71, 76\u201387 (2018)","journal-title":"Sig. Process. Image Commun."},{"key":"14_CR49","doi-asserted-by":"crossref","unstructured":"Mahendran, A., Thewlis, J., Vedaldi, A.: Self-supervised segmentation by grouping optical-flow. In: ECCV (2018)","DOI":"10.1007\/978-3-030-11021-5_31"},{"key":"14_CR50","doi-asserted-by":"crossref","unstructured":"Martin, D., Fowlkes, C., Malik, J.: Learning to detect natural image boundaries using local brightness, color, and texture cues. TPAMI (2004)","DOI":"10.1109\/TPAMI.2004.1273918"},{"key":"14_CR51","doi-asserted-by":"crossref","unstructured":"Melas-Kyriazi, L., Laina, I., Rupprecht, C., Vedaldi, A.: Deep spectral methods: a surprisingly strong baseline for unsupervised semantic segmentation and localization. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00818"},{"key":"14_CR52","doi-asserted-by":"crossref","unstructured":"Meunier, E., Badoual, A., Bouthemy, P.: Em-driven unsupervised learning for efficient motion segmentation. TPAMI (2023)","DOI":"10.1109\/TPAMI.2022.3198480"},{"key":"14_CR53","doi-asserted-by":"crossref","unstructured":"Miao, B., Bennamoun, M., Gao, Y., Mian, A.: Self-supervised video object segmentation by motion-aware mask propagation. In: ICME (2022)","DOI":"10.1109\/ICME52920.2022.9859966"},{"key":"14_CR54","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1017\/S0140525X0200002X","volume":"25","author":"J Norman","year":"2002","unstructured":"Norman, J.: Two visual systems and two theories of perception: an attempt to reconcile the constructivist and ecological approaches. Behav. Brain Sci. 25, 73\u201396 (2002)","journal-title":"Behav. Brain Sci."},{"key":"14_CR55","doi-asserted-by":"crossref","unstructured":"Ochs, P., Malik, J., Brox, T.: Segmentation of moving objects by long term video analysis. TPAMI (2014)","DOI":"10.1109\/TPAMI.2013.242"},{"key":"14_CR56","doi-asserted-by":"crossref","unstructured":"Ochs, P., Brox, T.: Object segmentation in video: a hierarchical variational approach for turning point trajectories into dense regions. In: ICCV (2011)","DOI":"10.1109\/ICCV.2011.6126418"},{"key":"14_CR57","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Pont-Tuset, J., McWilliams, B., Van\u00a0Gool, L., Gross, M., Sorkine-Hornung, A.: A benchmark dataset and evaluation methodology for video object segmentation. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.85"},{"key":"14_CR58","doi-asserted-by":"crossref","unstructured":"Ponimatkin, G., Samet, N., Xiao, Y., Du, Y., Marlet, R., Lepetit, V.: A simple and powerful global optimization for unsupervised video object segmentation. In: WACV (2023)","DOI":"10.1109\/WACV56688.2023.00584"},{"key":"14_CR59","doi-asserted-by":"crossref","unstructured":"Price, W., Damen, D.: Play fair: frame attributions in video models. In: ACCV (2020)","DOI":"10.1007\/978-3-030-69541-5_29"},{"key":"14_CR60","doi-asserted-by":"crossref","unstructured":"Raptis, M., Sigal, L.: Poselet key-framing: a model for human activity recognition. In: CVPR (2013)","DOI":"10.1109\/CVPR.2013.342"},{"key":"14_CR61","doi-asserted-by":"crossref","unstructured":"Safadoust, S., G\u00fcney, F.: Multi-object discovery by low-dimensional object motion. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00074"},{"issue":"3870","key":"14_CR62","doi-asserted-by":"publisher","first-page":"895","DOI":"10.1126\/science.163.3870.895","volume":"163","author":"GE Schneider","year":"1969","unstructured":"Schneider, G.E.: Two visual systems. Science 163(3870), 895\u2013902 (1969)","journal-title":"Science"},{"key":"14_CR63","unstructured":"Seitzer, M., et al.: Bridging the gap to real-world object-centric learning. In: ICLR (2023)"},{"key":"14_CR64","doi-asserted-by":"crossref","unstructured":"Shin, G., Albanie, S., Wie, W.: Unsupervised salient object detection with spectral cluster voting. In: CVPR Workshops (2022)","DOI":"10.1109\/CVPRW56347.2022.00442"},{"key":"14_CR65","unstructured":"Sim\u00e9oni, O., et al.: Localizing objects with self-supervised transformers and no labels. In: BMVC (2021)"},{"key":"14_CR66","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. In: NeurIPS (2014)"},{"key":"14_CR67","unstructured":"Singh, S., Deshmukh, S., Sarkar, M., Krishnamurthy, B.: Locate: self-supervised object discovery via flow-guided graph-cut and bootstrapped self-training. In: BMVC (2023)"},{"key":"14_CR68","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/978-3-642-15555-0_21","volume-title":"Computer Vision \u2013 ECCV 2010","author":"T Brox","year":"2010","unstructured":"Brox, T., Malik, J.: Object segmentation by long term analysis of point trajectories. In: Daniilidis, K., Maragos, P., Paragios, N. (eds.) ECCV 2010. LNCS, vol. 6315, pp. 282\u2013295. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15555-0_21"},{"key":"14_CR69","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"402","DOI":"10.1007\/978-3-030-58536-5_24","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Z Teed","year":"2020","unstructured":"Teed, Z., Deng, J.: RAFT: recurrent all-pairs field transforms for optical flow. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12347, pp. 402\u2013419. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58536-5_24"},{"key":"14_CR70","doi-asserted-by":"crossref","unstructured":"Tokmakov, P., Alahari, K., Schmid, C.: Learning motion patterns in videos. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.64"},{"key":"14_CR71","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/s11263-018-1122-2","volume":"127","author":"P Tokmakov","year":"2019","unstructured":"Tokmakov, P., Schmid, C., Alahari, K.: Learning to segment moving objects. IJCV 127, 282\u2013301 (2019)","journal-title":"IJCV"},{"key":"14_CR72","unstructured":"Veerapaneni, R., et al.: Entity abstraction in visual model-based reinforcement learning. In: CoRL (2019)"},{"key":"14_CR73","doi-asserted-by":"crossref","unstructured":"Ventura, C., Bellver, M., Girbau, A., Salvador, A., Marques, F., Giro-i Nieto, X.: RVOS: end-to-end recurrent network for video object segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00542"},{"key":"14_CR74","doi-asserted-by":"crossref","unstructured":"Vondrick, C., Shrivastava, A., Fathi, A., Guadarrama, S., Murphy, K.: Tracking emerges by colorizing videos. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01261-8_24"},{"key":"14_CR75","unstructured":"Wang, R., Sun, Y., Gandelsman, Y., Chen, X., Efros, A.A., Wang, X.: Test-time training on video streams. arXiv:2307.05014 (2023)"},{"key":"14_CR76","doi-asserted-by":"crossref","unstructured":"Wang, X., Jabri, A., Efros, A.A.: Learning correspondence from the cycle-consistency of time. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00267"},{"key":"14_CR77","doi-asserted-by":"crossref","unstructured":"Wang, X., Misra, I., Zeng, Z., Girdhar, R., Darrell, T.: Videocutler: surprisingly simple unsupervised video instance segmentation. arXiv preprint arXiv:2308.14710 (2023)","DOI":"10.1109\/CVPR52733.2024.02147"},{"key":"14_CR78","doi-asserted-by":"crossref","unstructured":"Wang, Y., Shen, X., Hu, S.X., Yuan, Y., Crowley, J.L., Vaufreydaz, D.: Self-supervised transformers for unsupervised object discovery using normalized cut. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01414"},{"key":"14_CR79","unstructured":"Weis, M.A., et al.: Benchmarking unsupervised object representations for video sequences. In: JMLR (2021)"},{"key":"14_CR80","doi-asserted-by":"crossref","unstructured":"Wu, R., Lin, H., Qi, X., Jia, J.: Memory selection network for video propagation. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58555-6_11"},{"key":"14_CR81","doi-asserted-by":"crossref","unstructured":"Xie, C., Xiang, Y., Harchaoui, Z., Fox, D.: Object discovery in videos as foreground motion clustering. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01023"},{"key":"14_CR82","unstructured":"Xie, J., Xie, W., Zisserman, A.: Segmenting moving objects via an object-centric layered representation. In: NeurIPS (2022)"},{"key":"14_CR83","doi-asserted-by":"crossref","unstructured":"Xu, N., et al.: Youtube-vos: a large-scale video object segmentation benchmark. CoRR (2018)","DOI":"10.1007\/978-3-030-01228-1_36"},{"key":"14_CR84","doi-asserted-by":"crossref","unstructured":"Yang, C., Lamdouar, H., Lu, E., Zisserman, A., Xie, W.: Self-supervised video object segmentation by motion grouping. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00709"},{"key":"14_CR85","doi-asserted-by":"crossref","unstructured":"Yang, Y., Lai, B., Soatto, S.: Dystab: unsupervised object segmentation via dynamic-static bootstrapping. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00285"},{"key":"14_CR86","doi-asserted-by":"crossref","unstructured":"Yang, Y., Loquercio, A., Scaramuzza, D., Soatto, S.: Unsupervised moving object detection via contextual information separation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00097"},{"key":"14_CR87","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, Q., Bertinetto, L., Bai, S., Hu, W., Torr, P.H.: Anchor diffusion for unsupervised video object segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00102"},{"key":"14_CR88","doi-asserted-by":"crossref","unstructured":"Ye, V., Li, Z., Tucker, R., Kanazawa, A., Snavely, N.: Deformable sprites for unsupervised video decomposition. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00268"},{"key":"14_CR89","doi-asserted-by":"crossref","unstructured":"Yin, Z., Zheng, J., Luo, W., Qian, S., Zhang, H., Gao, S.: Learning to recommend frame for interactive video object segmentation in the wild. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01519"},{"key":"14_CR90","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"612","DOI":"10.1007\/978-3-031-19818-2_35","volume-title":"ECCV 2022","author":"Y Yu","year":"2022","unstructured":"Yu, Y., Yuan, J., Mittal, G., Fuxin, L., Chen, M.: Batman: bilateral attention transformer in motion-appearance neighboring space for video object segmentation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13689, pp. 612\u2013629. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19818-2_35"},{"key":"14_CR91","doi-asserted-by":"crossref","unstructured":"Zhao, S., Zhu, L., Wang, X., Yang, Y.: Centerclip: token clustering for efficient text-video retrieval. In: The 45th International ACM SIGIR Conference on Research and Development in Information Retrieval (2022)","DOI":"10.1145\/3477495.3531950"},{"key":"14_CR92","doi-asserted-by":"crossref","unstructured":"Zhou, T., Wang, S., Zhou, Y., Yao, Y., Li, J., Shao, L.: Motion-attentive transition for zero-shot video object segmentation. In: AAAI (2020)","DOI":"10.1109\/TIP.2020.3013162"},{"key":"14_CR93","doi-asserted-by":"crossref","unstructured":"Zhu, W., Hu, J., Sun, G., Cao, X., Qiao, Y.: A key volume mining deep framework for action recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.219"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72933-1_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T12:35:42Z","timestamp":1727872542000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72933-1_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,3]]},"ISBN":["9783031729324","9783031729331"],"references-count":93,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72933-1_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,3]]},"assertion":[{"value":"3 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}