{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T06:45:56Z","timestamp":1785653156494,"version":"3.56.0"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_5","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:13Z","timestamp":1785649573000},"page":"64-79","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mamba-VOS: Efficient Video Object Segmentation with Selective State Space Models"],"prefix":"10.1007","author":[{"given":"Cheolhun","family":"Jang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wontae","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daehyun","family":"Ji","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nam Ik","family":"Cho","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"5_CR1","doi-asserted-by":"crossref","unstructured":"Caelles, S., Maninis, K.K., Pont-Tuset, J., Leal-Taix\u00e9, L., Cremers, D., Van Gool, L.: One-shot video object segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 221\u2013230 (2017)","DOI":"10.1109\/CVPR.2017.565"},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Chung, Y.W., Tai, Y.W., Tang, C.K.: Modular interactive video object segmentation: interaction-to-mask, propagation and difference-aware fusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5559\u20135568 (2021)","DOI":"10.1109\/CVPR46437.2021.00551"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Oh, S.W., Price, B., Lee, J.Y., Schwing, A.: Putting the object back into video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3151\u20133161 (2024)","DOI":"10.1109\/CVPR52733.2024.00304"},{"key":"5_CR4","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Oh, S.W., Price, B., Schwing, A., Lee, J.Y.: Tracking anything with decoupled video segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1316\u20131326 (2023)","DOI":"10.1109\/ICCV51070.2023.00127"},{"key":"5_CR5","doi-asserted-by":"publisher","unstructured":"Cheng, H.K., Schwing, A.G.: XMem: long-term video object segmentation with an Atkinson-Shiffrin memory model. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 640\u2013658. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_37","DOI":"10.1007\/978-3-031-19815-1_37"},{"key":"5_CR6","unstructured":"Cheng, H.K., Tai, Y.W., Tang, C.K.: Rethinking space-time networks with improved memory coverage for efficient video object segmentation. In: Advances in Neural Information Processing Systems, vol. 34, pp. 11781\u201311794 (2021)"},{"key":"5_CR7","unstructured":"Ding, H., Liu, C., He, S., Jiang, X., Torr, P.H., Bai, S.: Complex video object segmentation. In: European Conference on Computer Vision (ECCV), pp. 572\u2013589 (2022)"},{"key":"5_CR8","doi-asserted-by":"crossref","unstructured":"Duke, B., Ahmed, A., Wolf, C., Aarabi, P., Taylor, G.W.: SSTVOS: sparse spatiotemporal transformers for video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5912\u20135921 (2021)","DOI":"10.1109\/CVPR46437.2021.00585"},{"key":"5_CR9","unstructured":"Gu, A., Dao, T.: Mamba: linear-time sequence modeling with selective state spaces. In: First Conference on Language Modeling (2024)"},{"key":"5_CR10","unstructured":"Gu, A., Dao, T., Ermon, S., Rudra, A., R\u00e9, C.: Hippo: recurrent memory with optimal polynomial projections. In: Advances in Neural Information Processing Systems (NeurIPS), vol. 33, pp. 1474\u20131487 (2020)"},{"key":"5_CR11","unstructured":"Gu, A., Goel, K., R\u00e9, C.: Efficiently modeling long sequences with structured state spaces. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Hong, L., et al.: LVOS: a benchmark for long-term video object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13480\u201313492 (2023)","DOI":"10.1109\/ICCV51070.2023.01240"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Hu, Y.T., Huang, J.B., Schwing, A.G.: VideoMatch: matching based video object segmentation. In: European Conference on Computer Vision (ECCV), pp. 54\u201370 (2018)","DOI":"10.1007\/978-3-030-01237-3_4"},{"key":"5_CR14","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are RNNs: fast autoregressive transformers with linear attention. In: International Conference on Machine Learning (ICML), pp. 5156\u20135165 (2020)"},{"key":"5_CR15","doi-asserted-by":"publisher","unstructured":"Li, K., et al.: VideoMamba: state space model for efficient video understanding. In: Leonardis, A., Ricci, E., Roth, S., Russakovsky, O., Sattler, T., Varol, G. (eds.) European Conference on Computer Vision, pp. 237\u2013255. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-73347-5_14","DOI":"10.1007\/978-3-031-73347-5_14"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Li, M., Hu, L., Xiong, Z., Zhang, B., Pan, P., Liu, D.: Recurrent dynamic embedding for video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1332\u20131341 (2022)","DOI":"10.1109\/CVPR52688.2022.00139"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Liu, Q., et al.: LiVOS: light video object segmentation with gated linear matching. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 8668\u20138678 (2025)","DOI":"10.1109\/CVPR52734.2025.00810"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: VMamba: visual state space model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1\u201313 (2024)","DOI":"10.1109\/CVPR52733.2024.01557"},{"key":"5_CR19","doi-asserted-by":"crossref","unstructured":"Oh, S.W., Lee, J.Y., Xu, N., Kim, S.J.: Video object segmentation using space-time memory networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9226\u20139235 (2019)","DOI":"10.1109\/ICCV.2019.00932"},{"key":"5_CR20","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Van Gool, L.: The 2017 davis challenge on video object segmentation. arXiv preprint arXiv:1704.00675 (2017)"},{"key":"5_CR21","doi-asserted-by":"crossref","unstructured":"Seong, H., Hyun, J., Kim, E.: Kernelized memory network for video object segmentation. In: European Conference on Computer Vision (ECCV), pp. 629\u2013645 (2020)","DOI":"10.1007\/978-3-030-58542-6_38"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Seong, H., Oh, S.W., Lee, J.Y., Lee, S., Lee, S., Kim, E.: Hierarchical memory matching network for video object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 12889\u201312898 (2021)","DOI":"10.1109\/ICCV48922.2021.01265"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"Tokmakov, P., Alahari, K., Schmid, C.: Learning video object segmentation with visual memory. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4481\u20134490 (2017)","DOI":"10.1109\/ICCV.2017.480"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., Chai, Y., Schroff, F., Adam, H., Leibe, B., Chen, L.C.: FEELVOS: fast end-to-end embedding learning for video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9481\u20139490 (2019)","DOI":"10.1109\/CVPR.2019.00971"},{"key":"5_CR25","doi-asserted-by":"crossref","unstructured":"Wang, H., Jiang, X., Ren, H., Hu, Y., Bai, S.: SwiftNet: real-time video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1296\u20131305 (2021)","DOI":"10.1109\/CVPR46437.2021.00135"},{"key":"5_CR26","doi-asserted-by":"crossref","unstructured":"Wang, J., et al.: Look before you match: instance understanding matters in video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2397\u20132407 (2023)","DOI":"10.1109\/CVPR52729.2023.00225"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Wu, Q., Yang, T., Wu, W., Chan, A.: Scalable video object segmentation with simplified framework. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1274\u20131283 (2023)","DOI":"10.1109\/ICCV51070.2023.01276"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Xie, H., Yao, H., Zhou, S., Zhang, S., Sun, W.: Efficient regional memory network for video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1286\u20131295 (2021)","DOI":"10.1109\/CVPR46437.2021.00134"},{"key":"5_CR29","unstructured":"Xu, N., et al.: YouTube-VOS: a large-scale video object segmentation benchmark. arXiv preprint arXiv:1809.03327 (2018)"},{"key":"5_CR30","unstructured":"Yang, Y., Zhang, Z., Leach, M.: Vivim: a video vision mamba for medical video object segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention (MICCAI) (2024)"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wei, Y., Yang, Y.: Collaborative video object segmentation by foreground-background integration. In: European Conference on Computer Vision (ECCV), pp. 332\u2013348 (2020)","DOI":"10.1007\/978-3-030-58558-7_20"},{"key":"5_CR32","unstructured":"Yang, Z., Wei, Y., Yang, Y.: Associating objects with transformers for video object segmentation. In: Advances in Neural Information Processing Systems, vol. 34, pp. 2491\u20132502 (2021)"},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Yang, Z., Yang, Y.: Decoupling features in hierarchical propagation for video object segmentation. In: Advances in Neural Information Processing Systems, vol. 35, pp. 36324\u201336336 (2022)","DOI":"10.52202\/068431-2632"},{"key":"5_CR34","unstructured":"Zhu, L., Liao, B., Zhang, Q., Wang, X., Liu, W., Wang, X.: Vision mamba: efficient visual representation learning with bidirectional state space model. In: International Conference on Machine Learning (ICML) (2024)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:16Z","timestamp":1785649576000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}