{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T18:36:10Z","timestamp":1783967770377,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":63,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819609710","type":"print"},{"value":"9789819609727","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,10]],"date-time":"2024-12-10T00:00:00Z","timestamp":1733788800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,10]],"date-time":"2024-12-10T00:00:00Z","timestamp":1733788800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-0972-7_17","type":"book-chapter","created":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T08:11:24Z","timestamp":1733731884000},"page":"291-308","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":17,"title":["Moving Object Segmentation: All You Need is SAM (and Flow)"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1123-493X","authenticated-orcid":false,"given":"Junyu","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7044-1901","authenticated-orcid":false,"given":"Charig","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8609-6826","authenticated-orcid":false,"given":"Weidi","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8945-8573","authenticated-orcid":false,"given":"Andrew","family":"Zisserman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,10]]},"reference":[{"key":"17_CR1","doi-asserted-by":"crossref","unstructured":"Baker, S., Roth, S., Scharstein, D., Black, M.J., Lewis, J., Szeliski, R.: A database and evaluation methodology for optical flow. In: ICCV (2007)","DOI":"10.1109\/ICCV.2007.4408903"},{"key":"17_CR2","doi-asserted-by":"crossref","unstructured":"Bideau, P., Learned-Miller, E.: It\u2019s moving! a probabilistic model for causal motion segmentation in moving camera videos. In: ECCV (2016)","DOI":"10.1007\/978-3-319-46484-8_26"},{"key":"17_CR3","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., Joulin, A.: Emerging properties in self-supervised vision transformers. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"17_CR4","unstructured":"Cen, J., Fang, J., Yang, C., Xie, L., Zhang, X., Shen, W., Tian, Q.: Segment any 3d gaussians. arXiv preprint arXiv:2312.00860 (2023)"},{"key":"17_CR5","doi-asserted-by":"crossref","unstructured":"Chen, T., Zhu, L., Ding, C., Cao, R., Zhang, S., Wang, Y., Li, Z., Sun, L., Mao, P., Zang, Y.: Sam-adapter: Adapting segment anything in underperformed scenes. In: ICCV Workshop (2023)","DOI":"10.1109\/ICCVW60793.2023.00361"},{"key":"17_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Oh, S.W., Price, B., Schwing, A., Lee, J.Y.: Tracking anything with decoupled video segmentation. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00127"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Schwing, A.G.: XMem: Long-term video object segmentation with an atkinson-shiffrin memory model. In: ECCV (2022)","DOI":"10.1007\/978-3-031-19815-1_37"},{"key":"17_CR8","unstructured":"Cheng, Y., Li, L., Xu, Y., Li, X., Yang, Z., Wang, W., Yang, Y.: Segment and track anything. arXiv preprint arXiv:2305.06558 (2023)"},{"key":"17_CR9","unstructured":"Cho, D., Hong, S., Kang, S., Kim, J.: Key instance selection for unsupervised video object segmentation. arXiv preprint arXiv:1906.07851 (2019)"},{"key":"17_CR10","doi-asserted-by":"crossref","unstructured":"Cho, S., Lee, M., Lee, S., Park, C., Kim, D., Lee, S.: Treating motion as option to reduce motion dependency in unsupervised video object segmentation. In: WACV (2023)","DOI":"10.2139\/ssrn.4710755"},{"key":"17_CR11","unstructured":"Choudhury, S., Karazija, L., Laina, I., Vedaldi, A., Rupprecht, C.: Guess What Moves: Unsupervised Video and Image Segmentation by Anticipating Motion. In: BMVC (2022)"},{"key":"17_CR12","unstructured":"Jabri, A., Owens, A., Efros, A.A.: Space-time correspondence as a contrastive random walk. In: NeurIPS (2020)"},{"key":"17_CR13","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.Y., Dollar, P., Girshick, R.: Segment anything. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"17_CR14","doi-asserted-by":"crossref","unstructured":"Lai, Z., Lu, E., Xie, W.: Mast: A memory-augmented self-supervised tracker. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00651"},{"key":"17_CR15","unstructured":"Lai, Z., Xie, W.: Self-supervised learning for video correspondence flow. In: BMVC (2019)"},{"key":"17_CR16","unstructured":"Lamdouar, H., Xie, W., Zisserman, A.: Segmenting invisible moving objects. In: BMVC (2021)"},{"key":"17_CR17","doi-asserted-by":"crossref","unstructured":"Lamdouar, H., Yang, C., Xie, W., Zisserman, A.: Betrayed by motion: Camouflaged object discovery via motion segmentation. In: ACCV (2020)","DOI":"10.1007\/978-3-030-69532-3_30"},{"key":"17_CR18","doi-asserted-by":"crossref","unstructured":"Lee, M., Cho, S., Lee, S., Park, C., Lee, S.: Unsupervised video object segmentation via prototype memory network. In: WACV (2023)","DOI":"10.1109\/WACV56688.2023.00587"},{"key":"17_CR19","doi-asserted-by":"crossref","unstructured":"Li, F., Kim, T., Humayun, A., Tsai, D., Rehg, J.M.: Video segmentation by tracking many figure-ground segments. In: ICCV (2013)","DOI":"10.1109\/ICCV.2013.273"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Li, S., Seybold, B., Vorobyov, A., Fathi, A., Huang, Q., Kuo, C.C.J.: Instance embedding transfer to unsupervised video object segmentation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00683"},{"key":"17_CR21","doi-asserted-by":"crossref","unstructured":"Lin, H., Wu, R., Liu, S., Lu, J., Jia, J.: Video instance segmentation with a propose-reduce paradigm. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00176"},{"key":"17_CR22","doi-asserted-by":"crossref","unstructured":"Lu, X., Wang, W., Ma, C., Shen, J., Shao, L., Porikli, F.: See more, know more: Unsupervised video object segmentation with co-attention siamese networks. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00374"},{"key":"17_CR23","doi-asserted-by":"crossref","unstructured":"Luiten, J., Zulfikar, I.E., Leibe, B.: Unovost: Unsupervised offline video object segmentation and tracking. In: WACV (2020)","DOI":"10.1109\/WACV45572.2020.9093285"},{"key":"17_CR24","doi-asserted-by":"crossref","unstructured":"Ma, J., He, Y., Li, F., Han, L., You, C., Wang, B.: Segment anything in medical images. Nature Communications (2024)","DOI":"10.1038\/s41467-024-44824-z"},{"key":"17_CR25","doi-asserted-by":"crossref","unstructured":"Mahendran, A., Thewlis, J., Vedaldi, A.: Self-supervised segmentation by grouping optical-flow. In: ECCV (2018)","DOI":"10.1007\/978-3-030-11021-5_31"},{"key":"17_CR26","doi-asserted-by":"crossref","unstructured":"Meunier, E., Badoual, A., Bouthemy, P.: Em-driven unsupervised learning for efficient motion segmentation. IEEE TPAMI (2022)","DOI":"10.1109\/TPAMI.2022.3198480"},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Meunier, E., Bouthemy, P.: Unsupervised space-time network for temporally-consistent segmentation of multiple motions. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02120"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Miao, B., Bennamoun, M., Gao, Y., Mian, A.: Self-supervised video object segmentation by motion-aware mask propagation. In: ICME (2022)","DOI":"10.1109\/ICME52920.2022.9859966"},{"key":"17_CR29","doi-asserted-by":"crossref","unstructured":"Ochs, P., Malik, J., Brox, T.: Segmentation of moving objects by long term video analysis. IEEE TPAMI (2014)","DOI":"10.1109\/TPAMI.2013.242"},{"key":"17_CR30","doi-asserted-by":"crossref","unstructured":"Ochs, P., Brox, T.: Object segmentation in video: a hierarchical variational approach for turning point trajectories into dense regions. In: ICCV (2011)","DOI":"10.1109\/ICCV.2011.6126418"},{"key":"17_CR31","doi-asserted-by":"crossref","unstructured":"Oh, S.W., Lee, J.Y., Xu, N., Kim, S.J.: Video object segmentation using space-time memory networks. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00932"},{"key":"17_CR32","doi-asserted-by":"crossref","unstructured":"Pan, X., Li, P., Yang, Z., Zhou, H., Zhou, C., Yang, H., Zhou, J., Yang, Y.: In-n-out generative learning for dense unsupervised video segmentation. In: ACM MM (2022)","DOI":"10.1145\/3503161.3547909"},{"key":"17_CR33","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Pont-Tuset, J., McWilliams, B., Van\u00a0Gool, L., Gross, M., Sorkine-Hornung, A.: A benchmark dataset and evaluation methodology for video object segmentation. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.85"},{"key":"17_CR34","doi-asserted-by":"crossref","unstructured":"Ponimatkin, G., Samet, N., Xiao, Y., Du, Y., Marlet, R., Lepetit, V.: A simple and powerful global optimization for unsupervised video object segmentation. In: WACV (2023)","DOI":"10.1109\/WACV56688.2023.00584"},{"key":"17_CR35","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Gool, L.V.: The 2017 davis challenge on video object segmentation. arXiv preprint arXiv:1704.00675 (2017)"},{"key":"17_CR36","doi-asserted-by":"crossref","unstructured":"Ren, S., Luzi, F., Lahrichi, S., Kassaw, K., Collins, L.M., Bradbury, K., Malof, J.M.: Segment anything, from space? In: WACV (2024)","DOI":"10.1109\/WACV57701.2024.00817"},{"key":"17_CR37","doi-asserted-by":"crossref","unstructured":"Safadoust, S., G\u00fcney, F.: Multi-object discovery by low-dimensional object motion. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00074"},{"key":"17_CR38","doi-asserted-by":"crossref","unstructured":"Sun, Y., Chen, J., Zhang, S., Zhang, X., Chen, Q., Zhang, G., Ding, E., Wang, J., Li, Z.: Vrp-sam: Sam with visual reference prompt. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.02224"},{"key":"17_CR39","unstructured":"Tang, L., Xiao, H., Li, B.: Can sam segment anything? when sam meets camouflaged object detection. arXiv preprint arXiv:2304.04709 (2023)"},{"key":"17_CR40","doi-asserted-by":"crossref","unstructured":"Teed, Z., Deng, J.: Raft: Recurrent all-pairs field transforms for optical flow. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58536-5_24"},{"key":"17_CR41","doi-asserted-by":"crossref","unstructured":"Ventura, C., Bellver, M., Girbau, A., Salvador, A., Marques, F., Giro-i Nieto, X.: RVOS: End-to-end recurrent network for video object segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00542"},{"key":"17_CR42","doi-asserted-by":"crossref","unstructured":"Vondrick, C., Shrivastava, A., Fathi, A., Guadarrama, S., Murphy, K.: Tracking emerges by colorizing videos. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01261-8_24"},{"key":"17_CR43","doi-asserted-by":"crossref","unstructured":"Wang, X., Jabri, A., Efros, A.A.: Learning correspondence from the cycle-consistency of time. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00267"},{"key":"17_CR44","doi-asserted-by":"crossref","unstructured":"Wang, X., Misra, I., Zeng, Z., Girdhar, R., Darrell, T.: Videocutler: Surprisingly simple unsupervised video instance segmentation. arXiv preprint arXiv:2308.14710 (2023)","DOI":"10.1109\/CVPR52733.2024.02147"},{"key":"17_CR45","unstructured":"Wu, J., Ji, W., Liu, Y., Fu, H., Xu, M., Xu, Y., Jin, Y.: Medical sam adapter: Adapting segment anything model for medical image segmentation. arXiv preprint arXiv:2304.12620 (2023)"},{"key":"17_CR46","unstructured":"Xie, J., Xie, W., Zisserman, A.: Segmenting moving objects via an object-centric layered representation. In: NeurIPS (2022)"},{"key":"17_CR47","doi-asserted-by":"crossref","unstructured":"Xie, J., Xie, W., Zisserman, A.: Appearance-based refinement for object-centric motion segmentation. arXiv:2312.11463 (2023)","DOI":"10.1007\/978-3-031-72933-1_14"},{"key":"17_CR48","unstructured":"Xie, J., Yang, C., Xie, W., Zisserman, A.: Moving object segmentation: All you need is sam (and flow). arXiv preprint arXiv:2404.12389 (2024), https:\/\/arxiv.org\/abs\/2404.12389"},{"key":"17_CR49","doi-asserted-by":"crossref","unstructured":"Xiong, Y., Varadarajan, B., Wu, L., Xiang, X., Xiao, F., Zhu, C., Dai, X., Wang, D., Sun, F., Iandola, F., Krishnamoorthi, R., Chandra, V.: Efficientsam: Leveraged masked image pretraining for efficient segment anything. arXiv:2312.00863 (2023)","DOI":"10.1109\/CVPR52733.2024.01525"},{"key":"17_CR50","doi-asserted-by":"crossref","unstructured":"Xu, N., Yang, L., Fan, Y., Yue, D., Liang, Y., Yang, J., Huang, T.: Youtube-vos: A large-scale video object segmentation benchmark. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01228-1_36"},{"key":"17_CR51","doi-asserted-by":"crossref","unstructured":"Yang, C., Lamdouar, H., Lu, E., Zisserman, A., Xie, W.: Self-supervised video object segmentation by motion grouping. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00709"},{"key":"17_CR52","doi-asserted-by":"crossref","unstructured":"Yang, S., Zhang, L., Qi, J., Lu, H., Wang, S., Zhang, X.: Learning motion-appearance co-attention for zero-shot video object segmentation. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00159"},{"key":"17_CR53","doi-asserted-by":"crossref","unstructured":"Yang, Y., Lai, B., Soatto, S.: Dystab: Unsupervised object segmentation via dynamic-static bootstrapping. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00285"},{"key":"17_CR54","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, Q., Bertinetto, L., Bai, S., Hu, W., Torr, P.H.: Anchor diffusion for unsupervised video object segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00102"},{"key":"17_CR55","unstructured":"Yang, Z., Yang, Y.: Decoupling features in hierarchical propagation for video object segmentation. In: NeurIPS (2022)"},{"key":"17_CR56","unstructured":"Zhang, C., Han, D., Qiao, Y., Kim, J.U., Bae, S.H., Lee, S., Hong, C.S.: Faster segment anything: Towards lightweight sam for mobile applications. arXiv preprint arXiv:2306.14289 (2023)"},{"key":"17_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, K., Zhao, Z., Liu, D., Liu, Q., Liu, B.: Deep transport network for unsupervised video object segmentation. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00866"},{"key":"17_CR58","unstructured":"Zhang, X., Gu, C., Zhu, S.: Sam-helps-shadow:when segment anything model meet shadow removal. arXiv preprint arXiv:2306.06113 (2023)"},{"key":"17_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhang, S., Wei, Z., Dai, Z., Zhu, S.: Uvosam: A mask-free paradigm for unsupervised video object segmentation via segment anything model. arXiv preprint arXiv:2305.12659 (2024)","DOI":"10.2139\/ssrn.4729959"},{"key":"17_CR60","unstructured":"Zhao, X., Ding, W., An, Y., Du, Y., Yu, T., Li, M., Tang, M., Wang, J.: Fast segment anything. arXiv preprint arXiv:2306.12156 (2023)"},{"key":"17_CR61","unstructured":"Zheng, Z., Zhong, Y., Zhang, L., Ermon, S.: Segment any change. arXiv:2402.01188 (2024)"},{"key":"17_CR62","doi-asserted-by":"crossref","unstructured":"Zhou, T., Wang, S., Zhou, Y., Yao, Y., Li, J., Shao, L.: Motion-attentive transition for zero-shot video object segmentation. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i07.7008"},{"key":"17_CR63","unstructured":"Zou, X., Yang, J., Zhang, H., Li, F., Li, L., Gao, J., Lee, Y.J.: Segment everything everywhere all at once. arXiv preprint arXiv:2304.06718 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ACCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-0972-7_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T09:09:35Z","timestamp":1733735375000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-0972-7_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,10]]},"ISBN":["9789819609710","9789819609727"],"references-count":63,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-0972-7_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,10]]},"assertion":[{"value":"10 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ACCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hanoi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vietnam","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"accv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}