{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T06:37:18Z","timestamp":1757313438977,"version":"3.37.3"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2022,8,13]],"date-time":"2022-08-13T00:00:00Z","timestamp":1660348800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,8,13]],"date-time":"2022-08-13T00:00:00Z","timestamp":1660348800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61802197","62072449"],"award-info":[{"award-number":["61802197","62072449"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61632003"],"award-info":[{"award-number":["61632003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s00371-022-02638-4","type":"journal-article","created":{"date-parts":[[2022,8,13]],"date-time":"2022-08-13T12:06:34Z","timestamp":1660392394000},"page":"4929-4942","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Spatio-temporal compression for semi-supervised video object segmentation"],"prefix":"10.1007","volume":"39","author":[{"given":"Chuanjun","family":"Ji","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4448-2617","authenticated-orcid":false,"given":"Yadang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi-Xin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Enhua","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,13]]},"reference":[{"key":"2638_CR1","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-021-02150-1","author":"Z Huang","year":"2021","unstructured":"Huang, Z., Zhao, H., Zhan, J., Li, H.: A multivariate intersection over union of siamrpn network for visual tracking. Vis. Comput. (2021). https:\/\/doi.org\/10.1007\/s00371-021-02150-1","journal-title":"Vis. Comput."},{"key":"2638_CR2","doi-asserted-by":"publisher","unstructured":"G\u00f6kstorp, B.T.P. S.G.E.: Temporal and non-temporal contextual saliency analysis for generalized wide-area search within unmanned aerial vehicle (uav) video. Visual Comput. (2021). https:\/\/doi.org\/10.1007\/s00371-021-02264-6","DOI":"10.1007\/s00371-021-02264-6"},{"key":"2638_CR3","doi-asserted-by":"publisher","unstructured":"Tschiedel, R.M.F.K.E.E.A.M.: Real-time limb tracking in single depth images based on circle matching and line fitting. Visual Comput. (2021). https:\/\/doi.org\/10.1007\/s00371-021-02138-x","DOI":"10.1007\/s00371-021-02138-x"},{"key":"2638_CR4","doi-asserted-by":"publisher","unstructured":"Li, W.Z.Y.X.E.A.Y.: Efficient convolutional hierarchical autoencoder for human motion prediction. Visual Comput. (2019). https:\/\/doi.org\/10.1007\/s00371-019-01692-9","DOI":"10.1007\/s00371-019-01692-9"},{"key":"2638_CR5","doi-asserted-by":"publisher","unstructured":"Xu, L.Z.C.Q. D.: Object-based illumination transferring and rendering for applications of mixed reality. Visual Comput. (2021). https:\/\/doi.org\/10.1007\/s00371-021-02292-2","DOI":"10.1007\/s00371-021-02292-2"},{"key":"2638_CR6","doi-asserted-by":"publisher","unstructured":"Caelles, S., Maninis, K-., Pont-Tuset, J., Leal-Taix\u00e9, L., Cremers, D., Van Gool, L.: One-shot video object segmentation. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5320\u20135329 (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.565","DOI":"10.1109\/CVPR.2017.565"},{"key":"2638_CR7","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., Leibe, B.: Online adaptation of convolutional neural networks for video object segmentation. CoRR abs\/1706.09364 (2017) arXiv:1706.09364","DOI":"10.5244\/C.31.116"},{"key":"2638_CR8","doi-asserted-by":"publisher","first-page":"1515","DOI":"10.1109\/TPAMI.2018.2838670","volume":"41","author":"K Maninis","year":"2019","unstructured":"Maninis, K., Caelles, S., Chen, Y., Pont-Tuset, J., Leal-Taix\u00e9, L., Cremers, D., Gool, L.V.: Video object segmentation without temporal information. IEEE Trans. Pattern Anal. Mach. Intell. 41, 1515\u20131530 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2638_CR9","first-page":"1","volume":"2","author":"A Khoreva","year":"2019","unstructured":"Khoreva, A., Benenson, R., Ilg, E., Brox, T., Schiele, B.: Lucid data dreaming for video object segmentation. Int. J. Comput. Vision 2, 1\u201323 (2019)","journal-title":"Int. J. Comput. Vision"},{"key":"2638_CR10","doi-asserted-by":"crossref","unstructured":"Li, X., Loy, C.C.: Video object segmentation with joint re-identification and attention-aware mask propagation. CoRR abs\/1803.04242 (2018) arXiv:1803.04242","DOI":"10.1007\/978-3-030-01219-9_6"},{"key":"2638_CR11","unstructured":"Luiten, J., Voigtlaender, P., Leibe, B.: Premvos: Proposal-generation, refinement and merging for video object segmentation. CoRR abs\/1807.09190 (2018) arXiv:1807.09190"},{"key":"2638_CR12","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Khoreva, A., Benenson, R., Schiele, B., Sorkine-Hornung, A.: Learning video object segmentation from static images. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3491\u20133500 (2017)","DOI":"10.1109\/CVPR.2017.372"},{"key":"2638_CR13","doi-asserted-by":"crossref","unstructured":"Yang, L., Wang, Y., Xiong, X., Yang, J., Katsaggelos, A.: Efficient video object segmentation via network modulation. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6499\u20136507 (2018)","DOI":"10.1109\/CVPR.2018.00680"},{"key":"2638_CR14","doi-asserted-by":"crossref","unstructured":"Oh, S., Lee, J.-Y., Sunkavalli, K., Kim, S.: Fast video object segmentation by reference-guided mask propagation. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7376\u20137385 (2018)","DOI":"10.1109\/CVPR.2018.00770"},{"key":"2638_CR15","doi-asserted-by":"crossref","unstructured":"Tsai, Y.-H., Yang, M.-H., Black, M.J.: Video segmentation via object flow. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3899\u20133908 (2016)","DOI":"10.1109\/CVPR.2016.423"},{"key":"2638_CR16","unstructured":"Hu, Y., Huang, J., Schwing, A.G.: Maskrnn: Instance level video object segmentation. CoRR abs\/1803.11187 (2018) arXiv:1803.11187"},{"key":"2638_CR17","doi-asserted-by":"crossref","unstructured":"Xiao, H., Feng, J., Lin, G., Liu, Y., Zhang, M.: Monet: Deep motion exploitation for video object segmentation. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1140\u20131148 (2018)","DOI":"10.1109\/CVPR.2018.00125"},{"issue":"4","key":"2638_CR18","doi-asserted-by":"publisher","first-page":"1607","DOI":"10.1109\/TCSVT.2020.3010293","volume":"31","author":"W Liu","year":"2021","unstructured":"Liu, W., Lin, G., Zhang, T., Liu, Z.: Guided co-segmentation network for fast video object segmentation. IEEE Trans. Circuits Syst. Video Technol. 31(4), 1607\u20131617 (2021). https:\/\/doi.org\/10.1109\/TCSVT.2020.3010293","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2638_CR19","doi-asserted-by":"publisher","unstructured":"Chen, Y., Pont-Tuset, J., Montes, A., Gool, L.V.: Blazingly fast video object segmentation with pixel-wise metric learning. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1189\u20131198 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00130","DOI":"10.1109\/CVPR.2018.00130"},{"key":"2638_CR20","doi-asserted-by":"crossref","unstructured":"Hu, Y., Huang, J., Schwing, A.G.: Videomatch: matching based video object segmentation. CoRR abs\/1809.01123 (2018) arXiv:1809.01123","DOI":"10.1007\/978-3-030-01237-3_4"},{"key":"2638_CR21","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., Chai, Y., Schroff, F., Adam, H., Leibe, B., Chen, L.-C.: Feelvos: Fast end-to-end embedding learning for video object segmentation. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9473\u20139482 (2019)","DOI":"10.1109\/CVPR.2019.00971"},{"key":"2638_CR22","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wei, Y., Yang, Y.: Collaborative video object segmentation by foreground-background integration. CoRR abs\/2003.08333 (2020) arXiv:2003.08333","DOI":"10.1007\/978-3-030-58558-7_20"},{"key":"2638_CR23","doi-asserted-by":"crossref","unstructured":"Oh, S., Lee, J.-Y., Xu, N., Kim, S.: Video object segmentation using space-time memory networks. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9225\u20139234 (2019)","DOI":"10.1109\/ICCV.2019.00932"},{"key":"2638_CR24","doi-asserted-by":"crossref","unstructured":"Li, Y., Shen, Z., Shan, Y.: Fast video object segmentation using the global context module. CoRR abs\/2001.11243 (2020) arXiv:2001.11243","DOI":"10.1007\/978-3-030-58607-2_43"},{"key":"2638_CR25","doi-asserted-by":"crossref","unstructured":"Seong, H., Hyun, J., Kim, E.: Kernelized memory network for video object segmentation. CoRR abs\/2007.08270 (2020) arXiv:2007.08270","DOI":"10.1007\/978-3-030-58542-6_38"},{"key":"2638_CR26","doi-asserted-by":"crossref","unstructured":"Lu, X., Wang, W., Danelljan, M., Zhou, T., Shen, J., Gool, L.V.: Video object segmentation with episodic graph memory networks. CoRR abs\/2007.07020 (2020) arXiv:2007.07020","DOI":"10.1007\/978-3-030-58580-8_39"},{"key":"2638_CR27","doi-asserted-by":"crossref","unstructured":"Bao, L., Wu, B., Liu, W.: Cnn in mrf: Video object segmentation via inference in a cnn-based higher-order spatio-temporal mrf. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5977\u20135986 (2018)","DOI":"10.1109\/CVPR.2018.00626"},{"issue":"6","key":"2638_CR28","first-page":"7789","volume":"54","author":"P Cunningham","year":"2021","unstructured":"Cunningham, P., Delany, S.J.: K-nearest neighbour classifiersl. ACM Comput. Surv. 54(6), 7789 (2021)","journal-title":"ACM Comput. Surv."},{"key":"2638_CR29","doi-asserted-by":"publisher","unstructured":"Zhang, Y., Wu, Z., Peng, H., Lin, S.: A transductive approach for video object segmentation. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6947\u20136956 (2020). https:\/\/doi.org\/10.1109\/CVPR42600.2020.00698","DOI":"10.1109\/CVPR42600.2020.00698"},{"key":"2638_CR30","doi-asserted-by":"crossref","unstructured":"Park, H., Yoo, J., Jeong, S., Venkatesh, G., Kwak, N.: Learning dynamic network using a reuse gate function in semi-supervised video object segmentation. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8401\u20138410 (2021)","DOI":"10.1109\/CVPR46437.2021.00830"},{"key":"2638_CR31","doi-asserted-by":"crossref","unstructured":"Xie, H., Yao, H., Zhou, S., Zhang, S., Sun, W.: Efficient regional memory network for video object segmentation. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00134"},{"key":"2638_CR32","doi-asserted-by":"publisher","unstructured":"Hu, L., Zhang, P., Zhang, B., Pan, P., Xu, Y., Jin, R.: Learning position and target consistency for memory-based video object segmentation. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4142\u20134152 (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.00413","DOI":"10.1109\/CVPR46437.2021.00413"},{"key":"2638_CR33","first-page":"3430","volume":"33","author":"Y Liang","year":"2020","unstructured":"Liang, Y., Li, X., Jafari, N., Chen, J.: Video object segmentation with adaptive feature bank and uncertain-region refinement. Adv. Neural. Inf. Process. Syst. 33, 3430\u20133441 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2638_CR34","doi-asserted-by":"crossref","unstructured":"Wang, H., Jiang, X., Ren, H., Hu, Y., Bai, S.: Swiftnet: Real-time video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1296\u20131305 (2021)","DOI":"10.1109\/CVPR46437.2021.00135"},{"key":"2638_CR35","volume-title":"Generating Long Sequences with Sparse Transformers","author":"R Child","year":"2019","unstructured":"Child, R., Gray, S., Radford, A., Sutskever, I.: Generating Long Sequences with Sparse Transformers. Springer, Berlin (2019)"},{"key":"2638_CR36","unstructured":"Kitaev, N., \u0141ukasz Kaiser, Levskaya, A.: Reformer: The Efficient Transformer (2020)"},{"key":"2638_CR37","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention (2020)"},{"key":"2638_CR38","unstructured":"Shen, Z., Zhang, M., Zhao, H., Yi, S., Li, H.: Efficient Attention: Attention with Linear Complexities (2020)"},{"key":"2638_CR39","unstructured":"Li, R., Su, J., Duan, C., Zheng, S.: Linear Attention Mechanism: An Efficient Attention for Semantic Segmentation (2020)"},{"key":"2638_CR40","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"2638_CR41","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Pont-Tuset, J., McWilliams, B., Gool, L., Gross, M., Sorkine-Hornung, A.: A benchmark dataset and evaluation methodology for video object segmentation. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 724\u2013732 (2016)","DOI":"10.1109\/CVPR.2016.85"},{"key":"2638_CR42","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbelaez, P., Sorkine-Hornung, A., Gool, L.V.: The 2017 DAVIS challenge on video object segmentation. CoRR abs\/1704.00675 (2017) arXiv:1704.00675"},{"key":"2638_CR43","unstructured":"Xu, N., Yang, L., Fan, Y., Yue, D., Liang, Y., Yang, J., Huang, T.S.: Youtube-vos: A large-scale video object segmentation benchmark. CoRR abs\/1809.03327 (2018) arXiv:1809.03327"},{"key":"2638_CR44","doi-asserted-by":"publisher","unstructured":"Johnander, J., Danelljan, M., Brissman, E., Khan, F.S., Felsberg, M.: A generative appearance model for end-to-end video object segmentation. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8945\u20138954 (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00916","DOI":"10.1109\/CVPR.2019.00916"},{"key":"2638_CR45","doi-asserted-by":"publisher","unstructured":"Wang, Z., Xu, J., Liu, L., Zhu, F., Shao, L.: Ranet: Ranking attention network for fast video object segmentation. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3977\u20133986 (2019). https:\/\/doi.org\/10.1109\/ICCV.2019.00408","DOI":"10.1109\/ICCV.2019.00408"},{"key":"2638_CR46","doi-asserted-by":"crossref","unstructured":"Seong, H., Oh, S.W., Lee, J.-Y., Lee, S., Lee, S., Kim, E.: Hierarchical memory matching network for video object segmentation. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 12869\u201312878 (2021)","DOI":"10.1109\/ICCV48922.2021.01265"},{"key":"2638_CR47","doi-asserted-by":"crossref","unstructured":"Cho, S., Lee, H., Kim, M., Jang, S., Lee, S.: Pixel-level bijective matching for video object segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 129\u2013138 (2022)","DOI":"10.1109\/WACV51458.2022.00152"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-022-02638-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-022-02638-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-022-02638-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,29]],"date-time":"2023-09-29T09:11:42Z","timestamp":1695978702000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-022-02638-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,13]]},"references-count":47,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["2638"],"URL":"https:\/\/doi.org\/10.1007\/s00371-022-02638-4","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2022,8,13]]},"assertion":[{"value":"28 July 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We declare that we do not have any commercial or associative interest that represents a conflict of interest in connection with the work submitted.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}