{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T17:04:58Z","timestamp":1780765498611,"version":"3.54.1"},"publisher-location":"Cham","reference-count":101,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030586201","type":"print"},{"value":"9783030586218","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-58621-8_10","type":"book-chapter","created":{"date-parts":[[2020,11,26]],"date-time":"2020-11-26T19:03:23Z","timestamp":1606417403000},"page":"158-177","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":104,"title":["STEm-Seg: Spatio-Temporal Embeddings for Instance Segmentation in Videos"],"prefix":"10.1007","author":[{"given":"Ali","family":"Athar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sabarinath","family":"Mahadevan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aljos\u0306a","family":"Os\u0306ep","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Laura","family":"Leal-Taix\u00e9","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bastian","family":"Leibe","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,11,27]]},"reference":[{"key":"10_CR1","unstructured":"Hu, A., Kendall, A., Cipolla, R.: Learning a spatio-temporal embedding for video instance segmentation. arxiv preprint arXiv:1912:08969v (2019)"},{"key":"10_CR2","doi-asserted-by":"crossref","unstructured":"Van den Bergh, M., Roig, G., Boix, X., Manen, S., Van Gool, L.: Online video seeds for temporal window objectness. In: ICCV (2013)","DOI":"10.1109\/ICCV.2013.54"},{"key":"10_CR3","unstructured":"Berman, M., Blaschko, M.B.: Optimization of the Jaccard index for image segmentation with the Lov\u00e1sz hinge. In: CVPR (2018)"},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Berman, M., Rannen Triki, A., Blaschko, M.B.: The Lov\u00e1sz-Softmax loss: a tractable surrogate for the optimization of the intersection-over-union measure in neural networks. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00464"},{"key":"10_CR5","first-page":"1:1","volume":"2008","author":"K Bernardin","year":"2008","unstructured":"Bernardin, K., Stiefelhagen, R.: Evaluating multiple object tracking performance: the CLEAR MOT metrics. JIVP 2008, 1:1\u20131:10 (2008)","journal-title":"JIVP"},{"key":"10_CR6","doi-asserted-by":"crossref","unstructured":"Bochinski, E., Eiselein, V., Sikora, T.: High-speed tracking-by-detection without using image information. In: AVSS (2017)","DOI":"10.1109\/AVSS.2017.8078516"},{"key":"10_CR7","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/978-3-642-15555-0_21","volume-title":"Computer Vision \u2013 ECCV 2010","author":"T Brox","year":"2010","unstructured":"Brox, T., Malik, J.: Object segmentation by long term analysis of point trajectories. In: Daniilidis, K., Maragos, P., Paragios, N. (eds.) ECCV 2010. LNCS, vol. 6315, pp. 282\u2013295. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15555-0_21"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Butt, A.A., Collins, R.T.: Multi-target tracking by Lagrangian relaxation to min-cost network flow. In: CVPR (2013)","DOI":"10.1109\/CVPR.2013.241"},{"key":"10_CR9","doi-asserted-by":"crossref","unstructured":"Caelles, S., Maninis, K.K., Pont-Tuset, J., Leal-Taix\u00e9, L., Cremers, D., Van Gool, L.: One-shot video object segmentation. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.565"},{"key":"10_CR10","unstructured":"Caelles, S., et al.: The 2018 DAVIS challenge on video object segmentation. arXiv preprint arXiv:1803.00557 (2018)"},{"key":"10_CR11","unstructured":"Caelles, S., Pont-Tuset, J., Perazzi, F., Montes, A., Maninis, K., Gool, L.V.: The 2019 DAVIS challenge on VOS: unsupervised multi-object segmentation. arXiv arXiv:1905.00737 (2019)"},{"key":"10_CR12","unstructured":"Chen, L., Papandreou, G., Schroff, F., Adam, H.: Rethinking atrous convolution for semantic image segmentation. arXiv preprint arXiv:1706.05587 (2017)"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Chen, X., Girshick, R., He, K., Doll\u00e1r, P.: TensorMask: a foundation for dense object segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00215"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Chen, Y., Pont-Tuset, J., Montes, A., Van Gool, L.: Blazingly fast video object segmentation with pixel-wise metric learning. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00130"},{"key":"10_CR15","unstructured":"Cho, D., Hong, S., Kim, J., Kang, S.: Key instance selection for unsupervised video object segmentation. In: The 2019 DAVIS Challenge on Video Object Segmentation - CVPR Workshops (2019)"},{"issue":"5","key":"10_CR16","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1109\/34.1000236","volume":"24","author":"D Comaniciu","year":"2002","unstructured":"Comaniciu, D., Meer, P.: Mean shift: a robust approach toward feature space analysis. PAMI 24(5), 603\u2013619 (2002)","journal-title":"PAMI"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Dave, A., Tokmakov, P., Ramanan, D.: Towards segmenting everything that moves. arXiv preprint arXiv:1902.03715 (2019)","DOI":"10.1109\/ICCVW.2019.00187"},{"key":"10_CR18","doi-asserted-by":"crossref","unstructured":"De Brabandere, B., Neven, D., Van Gool, L.: Semantic instance segmentation for autonomous driving. In: CVPR Workshops (2017)","DOI":"10.1109\/CVPRW.2017.66"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"De Brabandere, B., Neven, D., Van Gool, L.: Semantic instance segmentation with a discriminative loss function. arXiv preprint arXiv:1708.02551 (2017)","DOI":"10.1109\/CVPRW.2017.66"},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Dong, M., et al.: Temporal feature augmented network for video instance segmentation. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00091"},{"key":"10_CR21","doi-asserted-by":"crossref","unstructured":"Elich, C., Engelmann, F., Schult, J., Kontogianni, T., Leibe, B.: 3D-BEVIS: birds-eye-view instance segmentation. In: German Conference on Pattern Recognition (GCPR) (2019)","DOI":"10.1007\/978-3-030-33676-9_4"},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Engelmann, F., Bokeloh, M., Fathi, A., Leibe, B., Nie\u00dfner, M.: 3D-MPA: multi proposal aggregation for 3D semantic instance segmentation. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00905"},{"key":"10_CR23","unstructured":"Ester, M., Kriegel, H.P., Sander, J., Xu, X., et al.: A density-based algorithm for discovering clusters in large spatial databases with noise. In: ACM Conference on Knowledge Discovery and Data Mining (KDD) (1996)"},{"issue":"2","key":"10_CR24","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C., Winn, J., Zisserman, A.: The pascal visual object classes (VOC) challenge. IJCV 88(2), 303\u2013338 (2010)","journal-title":"IJCV"},{"key":"10_CR25","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Detect to track and track to detect. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.330"},{"key":"10_CR26","doi-asserted-by":"crossref","unstructured":"Feng, Q., Yang, Z., Li, P., Wei, Y., Yang, Y.: Dual embedding learning for video instance segmentation. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00090"},{"issue":"1","key":"10_CR27","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1109\/TIT.1975.1055330","volume":"21","author":"K Fukunaga","year":"1975","unstructured":"Fukunaga, K., Hostetler, L.: The estimation of the gradient of a density function, with applications in pattern recognition. IEEE Trans. Inf. Theory 21(1), 32\u201340 (1975)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"10_CR28","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Urtasun, R.: Are we ready for autonomous driving? The KITTI vision benchmark suite. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"10_CR29","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Malik, J.: Finding action tubes. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7298676"},{"key":"10_CR30","unstructured":"Gori, M., Monfardini, G., Scarselli, F.: A new model for learning in graph domains. In: IJCNN (2005)"},{"key":"10_CR31","unstructured":"Han, W., et al.: Seq-NMS for video object detection. arXiv preprint arXiv:1602.08465 (2016)"},{"key":"10_CR32","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"10_CR33","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"10_CR34","doi-asserted-by":"crossref","unstructured":"Hou, R., Chen, C., Shah, M.: Tube convolutional neural network (T-CNN) for action detection in videos. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.620"},{"key":"10_CR35","unstructured":"Hou, R., Chen, C., Sukthankar, R., Shah, M.: An efficient 3D CNN for action\/object segmentation in video. In: BMVC (2019)"},{"key":"10_CR36","unstructured":"Hu, Y., Huang, J., Schwing, A.: MaskRNN: instance level video object segmentation. In: NIPS (2017)"},{"key":"10_CR37","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1007\/978-3-540-88688-4_58","volume-title":"Computer Vision \u2013 ECCV 2008","author":"C Huang","year":"2008","unstructured":"Huang, C., Wu, B., Nevatia, R.: Robust object tracking by hierarchical association of detection responses. In: Forsyth, D., Torr, P., Zisserman, A. (eds.) ECCV 2008. LNCS, vol. 5303, pp. 788\u2013801. Springer, Heidelberg (2008). https:\/\/doi.org\/10.1007\/978-3-540-88688-4_58"},{"key":"10_CR38","doi-asserted-by":"publisher","first-page":"206","DOI":"10.1109\/TPAMI.1979.4766907","volume":"1","author":"R Jain","year":"1979","unstructured":"Jain, R., Nagel, H.H.: On the analysis of accumulative difference pictures from image sequences of real world scenes. PAMI 1, 206\u2013214 (1979)","journal-title":"PAMI"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Jain, S., Xiong, B., Grauman, K.: FusionSeg: learning to combine motion and appearance for fully automatic segmentation of generic objects in videos. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.228"},{"key":"10_CR40","doi-asserted-by":"crossref","unstructured":"Jiang, L., Zhao, H., Shi, S., Liu, S., Fu, C.W., Jia, J.: PointGroup: dual-set point grouping for 3D instance segmentation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00492"},{"key":"10_CR41","doi-asserted-by":"crossref","unstructured":"Kang, K., et al.: Object detection in videos with tubelet proposal networks. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.101"},{"key":"10_CR42","doi-asserted-by":"crossref","unstructured":"Kong, S., Fowlkes, C.C.: Recurrent pixel embedding for instance grouping. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00940"},{"key":"10_CR43","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1002\/nav.3800020109","volume":"2","author":"HW Kuhn","year":"1955","unstructured":"Kuhn, H.W., Yaw, B.: The Hungarian method for the assignment problem. Naval Res. Logist. Q. 2, 83\u201397 (1955)","journal-title":"Naval Res. Logist. Q."},{"key":"10_CR44","doi-asserted-by":"crossref","unstructured":"Kwak, S., Cho, M., Laptev, I., Ponce, J., Schmid, C.: Unsupervised object discovery and tracking in video collections. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.363"},{"key":"10_CR45","unstructured":"Leal-Taix\u00e9, L., Milan, A., Reid, I., Roth, S., Schindler, K.: MOTChallenge 2015: towards a benchmark for multi-target tracking. arXiv preprint arXiv:1504.01942 (2015)"},{"issue":"1\u20133","key":"10_CR46","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1007\/s11263-007-0095-3","volume":"77","author":"B Leibe","year":"2008","unstructured":"Leibe, B., Leonardis, A., Schiele, B.: Robust object detection with interleaved categorization and segmentation. IJCV 77(1\u20133), 259\u2013289 (2008)","journal-title":"IJCV"},{"issue":"10","key":"10_CR47","doi-asserted-by":"publisher","first-page":"1683","DOI":"10.1109\/TPAMI.2008.170","volume":"30","author":"B Leibe","year":"2008","unstructured":"Leibe, B., Schindler, K., Cornelis, N., Gool, L.V.: Coupled object detection and tracking from static cameras and moving vehicles. PAMI 30(10), 1683\u20131698 (2008)","journal-title":"PAMI"},{"key":"10_CR48","doi-asserted-by":"crossref","unstructured":"Li, S., Seybold, B., Vorobyov, A., Fathi, A., Huang, Q., Kuo, C.C.J.: Instance embedding transfer to unsupervised video object segmentation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00683"},{"key":"10_CR49","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"10_CR50","unstructured":"Liu, R., et al.: An intriguing failing of convolutional neural networks and the CoordConv solution. In: NIPS (2018)"},{"key":"10_CR51","doi-asserted-by":"crossref","unstructured":"Liu, X., Ye, T.: Spatio-temporal attention network for video instance segmentation. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00092"},{"issue":"2","key":"10_CR52","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1109\/TIT.1982.1056489","volume":"28","author":"S Lloyd","year":"1982","unstructured":"Lloyd, S.: Least squares quantization in PCM. IEEE Trans. Inf. Theory 28(2), 129\u2013137 (1982)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"10_CR53","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"565","DOI":"10.1007\/978-3-030-20870-7_35","volume-title":"Computer Vision \u2013 ACCV 2018","author":"J Luiten","year":"2019","unstructured":"Luiten, J., Voigtlaender, P., Leibe, B.: PReMVOS: proposal-generation, refinement and merging for video object segmentation. In: Jawahar, C.V., Li, H., Mori, G., Schindler, K. (eds.) ACCV 2018. LNCS, vol. 11364, pp. 565\u2013580. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-20870-7_35"},{"issue":"11","key":"10_CR54","doi-asserted-by":"publisher","first-page":"205","DOI":"10.21105\/joss.00205","volume":"2","author":"L McInnes","year":"2017","unstructured":"McInnes, L., Healy, J., Astels, S.: HDBSCAN: hierarchical density based clustering. J. Open Source Softw. 2(11), 205 (2017)","journal-title":"J. Open Source Softw."},{"key":"10_CR55","unstructured":"Milan, A., Leal-Taix\u00e9, L., Reid, I., Roth, S., Schindler, K.: MOT16: a benchmark for multi-object tracking. arXiv preprint arXiv:1603.00831 (2016)"},{"key":"10_CR56","doi-asserted-by":"crossref","unstructured":"Milan, A., Leal-Taix\u00e9, L., Schindler, K., Reid, I.: Joint tracking and segmentation of multiple targets. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7299178"},{"key":"10_CR57","doi-asserted-by":"crossref","unstructured":"Neven, D., Brabandere, B.D., Proesmans, M., Gool, L.V.: Instance segmentation by jointly optimizing spatial embeddings and clustering bandwidth. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00904"},{"key":"10_CR58","unstructured":"Newell, A., Huang, Z., Deng, J.: Associative embedding: end-to-end learning for joint detection and grouping. In: NIPS (2017)"},{"key":"10_CR59","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1007\/978-3-030-01246-5_6","volume-title":"Computer Vision \u2013 ECCV 2018","author":"D Novotny","year":"2018","unstructured":"Novotny, D., Albanie, S., Larlus, D., Vedaldi, A.: Semi-convolutional operators for instance segmentation. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11205, pp. 89\u2013105. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01246-5_6"},{"key":"10_CR60","doi-asserted-by":"crossref","unstructured":"Ochs, P., Brox, T.: Higher order motion models and spectral clustering. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6247728"},{"key":"10_CR61","doi-asserted-by":"crossref","unstructured":"Oh, S.W., Lee, J.Y., Xu, N., Kim, S.J.: Video object segmentation using space-time memory networks. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00932"},{"key":"10_CR62","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1007\/978-3-540-24670-1_3","volume-title":"Computer Vision - ECCV 2004","author":"K Okuma","year":"2004","unstructured":"Okuma, K., Taleghani, A., de Freitas, N., Little, J.J., Lowe, D.G.: A boosted particle filter: multitarget detection and tracking. In: Pajdla, T., Matas, J. (eds.) ECCV 2004. LNCS, vol. 3021, pp. 28\u201339. Springer, Heidelberg (2004). https:\/\/doi.org\/10.1007\/978-3-540-24670-1_3"},{"key":"10_CR63","doi-asserted-by":"crossref","unstructured":"O\u0161ep, A., Mehner, W., Voigtlaender, P., Leibe, B.: Track, then decide: category-agnostic vision-based multi-object tracking. In: ICRA (2018)","DOI":"10.1109\/ICRA.2018.8460975"},{"key":"10_CR64","doi-asserted-by":"crossref","unstructured":"O\u0161ep, A., Voigtlaender, P., Luiten, J., Breuers, S., Leibe, B.: Large-scale object mining for object discovery from unlabeled video (2019)","DOI":"10.1109\/ICRA.2019.8793683"},{"key":"10_CR65","doi-asserted-by":"crossref","unstructured":"O\u0161ep, A., Voigtlaender, P., Weber, M., Luiten, J., Leibe, B.: 4D generic video object proposals. In: ICRA (2020)","DOI":"10.1109\/ICRA40945.2020.9196949"},{"key":"10_CR66","doi-asserted-by":"crossref","unstructured":"Palmer, S.E.: Organizing objects and scenes. In: Foundations of Cognitive Psychology: Core Readings, pp. 189\u2013211 (2002)","DOI":"10.7551\/mitpress\/3080.003.0014"},{"key":"10_CR67","doi-asserted-by":"crossref","unstructured":"Palou, G., Salembier, P.: Hierarchical video representation with trajectory binary partition tree. In: CVPR (2013)","DOI":"10.1109\/CVPR.2013.273"},{"key":"10_CR68","doi-asserted-by":"publisher","first-page":"266","DOI":"10.1109\/34.841758","volume":"22","author":"N Paragios","year":"2000","unstructured":"Paragios, N., Deriche, R.: Geodesic active contours and level sets for the detection and tracking of moving objects. PAMI 22, 266\u2013280 (2000)","journal-title":"PAMI"},{"key":"10_CR69","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1007\/978-3-319-46448-0_5","volume-title":"Computer Vision \u2013 ECCV 2016","author":"PO Pinheiro","year":"2016","unstructured":"Pinheiro, P.O., Lin, T.-Y., Collobert, R., Doll\u00e1r, P.: Learning to refine object segments. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 75\u201391. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_5"},{"key":"10_CR70","unstructured":"Pinheiro, P., Collobert, R., Doll\u00e1r, P.: Learning to segment object candidates. In: NIPS (2015)"},{"key":"10_CR71","doi-asserted-by":"crossref","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Gool, L.V.: A benchmark dataset and evaluation methodology for video object segmentation. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.85"},{"key":"10_CR72","doi-asserted-by":"crossref","unstructured":"Qi, C.R., Litany, O., He, K., Guibas, L.J.: Deep Hough voting for 3D object detection in point clouds. In: CVPR (2019)","DOI":"10.1109\/ICCV.2019.00937"},{"key":"10_CR73","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: NIPS (2015)"},{"key":"10_CR74","doi-asserted-by":"crossref","unstructured":"Siam, M., et al.: Video segmentation using teacher-student adaptation in a human robot interaction (HRI) setting. In: ICRA (2018)","DOI":"10.1109\/ICRA.2019.8794254"},{"key":"10_CR75","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"744","DOI":"10.1007\/978-3-030-01252-6_44","volume-title":"Computer Vision \u2013 ECCV 2018","author":"H Song","year":"2018","unstructured":"Song, H., Wang, W., Zhao, S., Shen, J., Lam, K.-M.: Pyramid dilated deeper ConvLSTM for video salient object detection. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11215, pp. 744\u2013760. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01252-6_44"},{"key":"10_CR76","doi-asserted-by":"crossref","unstructured":"Teichman, A., Levinson, J., Thrun, S.: Towards 3D object recognition via classification of arbitrary object tracks. In: ICRA (2011)","DOI":"10.1109\/ICRA.2011.5979636"},{"key":"10_CR77","doi-asserted-by":"crossref","unstructured":"Tokmakov, P., Alahari, K., Schmid, C.: Learning video object segmentation with visual memory. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.480"},{"key":"10_CR78","doi-asserted-by":"crossref","unstructured":"Ventura, C., Bellver, M., Girbau, A., Salvador, A., Marqu\u00e9s, F., Gir\u2019o i Nieto, X.: RVOS: end-to-end recurrent network for video object segmentation. CVPR (2019)","DOI":"10.1109\/CVPR.2019.00542"},{"key":"10_CR79","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., Chai, Y., Schroff, F., Adam, H., Leibe, B., Chen., L.C.: FEELVOS: fast end-to-end embedding learning for video object segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00971"},{"key":"10_CR80","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., et al.: MOTS: multi-object tracking and segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00813"},{"key":"10_CR81","doi-asserted-by":"crossref","unstructured":"Wang, H., Luo, R., Maire, M., Shakhnarovich, G.: Pixel consensus voting for panoptic segmentation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00948"},{"key":"10_CR82","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1007\/978-3-319-10593-2_42","volume-title":"Computer Vision \u2013 ECCV 2014","author":"L Wang","year":"2014","unstructured":"Wang, L., Hua, G., Sukthankar, R., Xue, J., Zheng, N.: Video object discovery and co-segmentation with extremely weak supervision. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8692, pp. 640\u2013655. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10593-2_42"},{"key":"10_CR83","doi-asserted-by":"crossref","unstructured":"Wang, Q., He, Y., Yang, X., Yang, Z., Torr, P.: An empirical study of detection-based video instance segmentation. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00089"},{"key":"10_CR84","doi-asserted-by":"crossref","unstructured":"Wang, W., Lu, X., Shen, J., Crandall, D.J., Shao, L.: Zero-shot video object segmentation via attentive graph neural networks. In: The IEEE International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00933"},{"key":"10_CR85","unstructured":"Wojke, N., Bewley, A., Paulus., D.: Onboard contextual classification of 3D point clouds with learned high-order Markov random fields. In: ICIP (2017)"},{"key":"10_CR86","doi-asserted-by":"publisher","first-page":"780","DOI":"10.1109\/34.598236","volume":"19","author":"CR Wren","year":"1997","unstructured":"Wren, C.R., Azarbayejani, A., Darrell, T., Pentland, A.: Pfinder: real-time tracking of the human body. PAMI 19, 780\u2013785 (1997)","journal-title":"PAMI"},{"key":"10_CR87","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/978-3-030-01261-8_1","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Y Wu","year":"2018","unstructured":"Wu, Y., He, K.: Group normalization. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11217, pp. 3\u201319. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01261-8_1"},{"key":"10_CR88","unstructured":"Wu, Z., Shen, C., van den Hengel, A.: Wider or deeper: revisiting the ResNet model for visual recognition. arXiv preprint arXiv:1611.10080 (2016)"},{"key":"10_CR89","doi-asserted-by":"crossref","unstructured":"Wug Oh, S., Lee, J.Y., Sunkavalli, K., Joo Kim, S.: Fast video object segmentation by reference-guided mask propagation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00770"},{"key":"10_CR90","doi-asserted-by":"crossref","unstructured":"Xiao, F., Jae Lee, Y.: Track and segment: an iterative unsupervised approach for video object proposals. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.107"},{"key":"10_CR91","doi-asserted-by":"crossref","unstructured":"Xie, C., Xiang, Y., Harchaoui, Z., Fox, D.: Object discovery in videos as foreground motion clustering. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.01023"},{"key":"10_CR92","unstructured":"Xu, C.: Evaluation of super-voxel methods for early video processing. In: CVPR (2012)"},{"key":"10_CR93","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1007\/978-3-030-01228-1_36","volume-title":"Computer Vision \u2013 ECCV 2018","author":"N Xu","year":"2018","unstructured":"Xu, N., et al.: YouTube-VOS: sequence-to-sequence video object segmentation. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11209, pp. 603\u2013619. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01228-1_36"},{"key":"10_CR94","doi-asserted-by":"crossref","unstructured":"Yang, L., Fan, Y., Xu, N.: Video instance segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00529"},{"key":"10_CR95","doi-asserted-by":"crossref","unstructured":"Yang, L., Wang, Y., Xiong, X., Yang, J., Katsaggelos, A.K.: Efficient video object segmentation via network modulation. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00680"},{"key":"10_CR96","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, Q., Bertinetto, L., Hu, W., Bai, S., Torr, P.H.S.: Anchor diffusion for unsupervised video object segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00102"},{"key":"10_CR97","doi-asserted-by":"crossref","unstructured":"Jun Koh, Y., Kim, C.S.: Primary object segmentation in videos based on region augmentation and reduction. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.784"},{"key":"10_CR98","unstructured":"Yu, J., Blaschko, M.: Learning submodular losses with the Lov\u00e1sz hinge. In: International Conference on Machine Learning (ICML) (2015)"},{"key":"10_CR99","doi-asserted-by":"crossref","unstructured":"Zeng, X., Liao, R., Gu, L., Xiong, Y., Fidler, S., Urtasun, R.: DMM-Net: differentiable mask-matching network for video object segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00403"},{"key":"10_CR100","unstructured":"Zhang, D., Chun, J., Cha, S.K., Kim, Y.M.: Spatial semantic embedding network: fast 3D instance segmentation with deep metric learning. arXiv preprint arXiv:2007.03169 (2020)"},{"key":"10_CR101","unstructured":"Zulfikar, I.E., Luiten, J., Leibe, B.: UnOVOST: unsupervised offline video object segmentation and tracking for the 2019 unsupervised DAVIS challenge. In: The 2019 DAVIS Challenge on Video Object Segmentation - CVPR Workshops (2019)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2020"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-58621-8_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T00:04:19Z","timestamp":1732579459000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-58621-8_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030586201","9783030586218"],"references-count":101,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-58621-8_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"27 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Glasgow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2020.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OpenReview","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5025","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1360","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"27% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic. From the ECCV Workshops 249 full papers, 18 short papers, and 21 further contributions were published out of a total of 467 submissions.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}