{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,28]],"date-time":"2026-02-28T22:34:43Z","timestamp":1772318083633,"version":"3.50.1"},"publisher-location":"Cham","reference-count":150,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031200434","type":"print"},{"value":"9783031200441","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20044-1_5","type":"book-chapter","created":{"date-parts":[[2022,10,19]],"date-time":"2022-10-19T23:11:54Z","timestamp":1666221114000},"page":"76-98","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["Few-Shot Video Object Detection"],"prefix":"10.1007","author":[{"given":"Qi","family":"Fan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chi-Keung","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu-Wing","family":"Tai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,20]]},"reference":[{"key":"5_CR1","doi-asserted-by":"crossref","unstructured":"Bell, S., Lawrence Zitnick, C., Bala, K., Girshick, R.: Inside-outside net: detecting objects in context with skip pooling and recurrent neural networks. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.314"},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Bergmann, P., Meinhardt, T., Leal-Taixe, L.: Tracking without bells and whistles. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00103"},{"key":"5_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"342","DOI":"10.1007\/978-3-030-01258-8_21","volume-title":"Computer Vision \u2013 ECCV 2018","author":"G Bertasius","year":"2018","unstructured":"Bertasius, G., Torresani, L., Shi, J.: Object detection in video with spatiotemporal sampling networks. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11216, pp. 342\u2013357. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01258-8_21"},{"key":"5_CR4","unstructured":"Bertinetto, L., Henriques, J.F., Torr, P.H., Vedaldi, A.: Meta-learning with differentiable closed-form solvers. In: ICLR (2019)"},{"key":"5_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"850","DOI":"10.1007\/978-3-319-48881-3_56","volume-title":"Computer Vision \u2013 ECCV 2016 Workshops","author":"L Bertinetto","year":"2016","unstructured":"Bertinetto, L., Valmadre, J., Henriques, J.F., Vedaldi, A., Torr, P.H.S.: Fully-convolutional Siamese networks for object tracking. In: Hua, G., J\u00e9gou, H. (eds.) ECCV 2016. LNCS, vol. 9914, pp. 850\u2013865. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-48881-3_56"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Bhat, G., Danelljan, M., Gool, L.V., Timofte, R.: Learning discriminative model prediction for tracking. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00628"},{"key":"5_CR7","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"205","DOI":"10.1007\/978-3-030-58592-1_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"G Bhat","year":"2020","unstructured":"Bhat, G., Danelljan, M., Van Gool, L., Timofte, R.: Know your surroundings: exploiting scene information for object tracking. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12368, pp. 205\u2013221. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58592-1_13"},{"key":"5_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1007\/978-3-319-46493-0_22","volume-title":"Computer Vision \u2013 ECCV 2016","author":"Z Cai","year":"2016","unstructured":"Cai, Z., Fan, Q., Feris, R.S., Vasconcelos, N.: A unified multi-scale deep convolutional neural network for fast object detection. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 354\u2013370. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_22"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Cai, Z., Vasconcelos, N.: Cascade R-CNN: delving into high quality object detection. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00644"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Cao, K., Ji, J., Cao, Z., Chang, C.Y., Niebles, J.C.: Few-shot video classification via temporal alignment. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Chen, H., Wang, Y., Wang, G., Qiao, Y.: LSTD: a low-shot transfer detector for object detection. In: AAAI (2018)","DOI":"10.1609\/aaai.v32i1.11716"},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Y., Cao, Y., Hu, H., Wang, L.: Memory enhanced global-local aggregation for video object detection. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01035"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Chu, P., Ling, H.: FAMNet: joint learning of feature, affinity and multi-dimensional assignment for online multiple object tracking. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00627"},{"key":"5_CR14","unstructured":"Dai, J., Li, Y., He, K., Sun, J.: R-FCN: object detection via region-based fully convolutional networks. In: NeurIPS (2016)"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Dai, J., Qi, H., Xiong, Y., Li, Y., Zhang, G., Hu, H., Wei, Y.: Deformable convolutional networks. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.89"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Bhat, G., Khan, F.S., Felsberg, M.: ATOM: accurate tracking by overlap maximization. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00479"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Bhat, G., Shahbaz Khan, F., Felsberg, M.: Eco: Efficient convolution operators for tracking. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.733"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Gool, L.V., Timofte, R.: Probabilistic regression for visual tracking. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00721"},{"key":"5_CR19","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Hager, G., Shahbaz Khan, F., Felsberg, M.: Learning spatially regularized correlation filters for visual tracking. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.490"},{"key":"5_CR20","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Shahbaz Khan, F., Felsberg, M., Van de Weijer, J.: Adaptive color attributes for real-time visual tracking. In: CVPR (2014)","DOI":"10.1109\/CVPR.2014.143"},{"key":"5_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1007\/978-3-030-58558-7_26","volume-title":"Computer Vision \u2013 ECCV 2020","author":"A Dave","year":"2020","unstructured":"Dave, A., Khurana, T., Tokmakov, P., Schmid, C., Ramanan, D.: TAO: a large-scale benchmark for tracking any object. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12350, pp. 436\u2013454. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58558-7_26"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Deng, H., Hua, Y., Song, T., Zhang, Z., Xue, Z., Ma, R., Robertson, N., Guan, H.: Object guided external memory network for video object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00678"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Deng, J., Pan, Y., Yao, T., Zhou, W., Li, H., Mei, T.: Relation distillation networks for video object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00712"},{"key":"5_CR25","doi-asserted-by":"crossref","unstructured":"Doll\u00e1r, P., Wojek, C., Schiele, B., Perona, P.: Pedestrian detection: a benchmark. In: CVPR (2009)","DOI":"10.1109\/CVPRW.2009.5206631"},{"key":"5_CR26","unstructured":"Dong, N., Xing, E.P.: Few-shot semantic segmentation with prototype learning. In: BMVC (2018)"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Duan, K., Bai, S., Xie, L., Qi, H., Huang, Q., Tian, Q.: Centernet: keypoint triplets for object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00667"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Ess, A., Leibe, B., Schindler, K., Van Gool, L.: A mobile vision system for robust multi-person tracking. In: CVPR (2008)","DOI":"10.1109\/CVPR.2008.4587581"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Fan, H., et al.: LaSOT: a high-quality benchmark for large-scale single object tracking. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00552"},{"key":"5_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1007\/978-3-030-58598-3_23","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Q Fan","year":"2020","unstructured":"Fan, Q., Ke, L., Pei, W., Tang, C.-K., Tai, Y.-W.: Commonality-parsing network across shape and appearance for partially supervised instance segmentation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12353, pp. 379\u2013396. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58598-3_23"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Fan, Q., Zhuo, W., Tang, C.K., Tai, Y.W.: Few-shot object detection with attention-RPN and multi-relation detector. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00407"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Fan, Z., Ma, Y., Li, Z., Sun, J.: Generalized few-shot object detection without forgetting. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00450"},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Fang, K., Xiang, Y., Li, X., Savarese, S.: Recurrent autoregressive networks for online multi-object tracking. In: WACV (2018)","DOI":"10.1109\/WACV.2018.00057"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Detect to track and track to detect. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.330"},{"key":"5_CR35","unstructured":"Finn, C., Abbeel, P., Levine, S.: Model-agnostic meta-learning for fast adaptation of deep networks. In: ICML (2017)"},{"key":"5_CR36","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"5_CR37","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: CVPR (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"5_CR38","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"777","DOI":"10.1007\/978-3-030-58536-5_46","volume-title":"Computer Vision \u2013 ECCV 2020","author":"G Bhat","year":"2020","unstructured":"Bhat, G., et al.: Learning what to learn for video object segmentation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12347, pp. 777\u2013794. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58536-5_46"},{"key":"5_CR39","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"441","DOI":"10.1007\/978-3-030-01237-3_27","volume-title":"Computer Vision \u2013 ECCV 2018","author":"L-Y Gui","year":"2018","unstructured":"Gui, L.-Y., Wang, Y.-X., Ramanan, D., Moura, J.M.F.: Few-shot human motion prediction via meta-learning. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11212, pp. 441\u2013459. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01237-3_27"},{"key":"5_CR40","doi-asserted-by":"crossref","unstructured":"Guo, Q., Feng, W., Zhou, C., Huang, R., Wan, L., Wang, S.: Learning dynamic Siamese network for visual object tracking. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.196"},{"key":"5_CR41","doi-asserted-by":"crossref","unstructured":"Gupta, A., Dollar, P., Girshick, R.: LVIS: a dataset for large vocabulary instance segmentation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00550"},{"key":"5_CR42","doi-asserted-by":"crossref","unstructured":"Hariharan, B., Girshick, R.: Low-shot visual recognition by shrinking and hallucinating features. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.328"},{"key":"5_CR43","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"5_CR44","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"5_CR45","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"749","DOI":"10.1007\/978-3-319-46448-0_45","volume-title":"Computer Vision \u2013 ECCV 2016","author":"D Held","year":"2016","unstructured":"Held, D., Thrun, S., Savarese, S.: Learning to track at 100 fps with deep regression networks. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 749\u2013765. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_45"},{"key":"5_CR46","doi-asserted-by":"crossref","unstructured":"Henriques, J.F., Caseiro, R., Martins, P., Batista, J.: High-speed tracking with kernelized correlation filters. IEEE TPAMI (2014)","DOI":"10.1109\/TPAMI.2014.2345390"},{"key":"5_CR47","doi-asserted-by":"crossref","unstructured":"Hu, H., Bai, S., Li, A., Cui, J., Wang, L.: Dense relation distillation with context-aware aggregation for few-shot object detection. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01005"},{"key":"5_CR48","doi-asserted-by":"crossref","unstructured":"Hu, T., Pengwan, Zhang, C., Yu, G., Mu, Y., Snoek, C.G.M.: Attention-based multi-context guiding for few-shot semantic segmentation. In: AAAI (2019)","DOI":"10.1609\/aaai.v33i01.33018441"},{"key":"5_CR49","unstructured":"Huang, L., Zhao, X., Huang, K.: Got-10k: a large high-diversity benchmark for generic object tracking in the wild. IEEE TPAMI (2019)"},{"key":"5_CR50","doi-asserted-by":"crossref","unstructured":"Kang, B., Liu, Z., Wang, X., Yu, F., Feng, J., Darrell, T.: Few-shot object detection via feature reweighting. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00851"},{"key":"5_CR51","doi-asserted-by":"crossref","unstructured":"Kang, K., Li, H., Xiao, T., Ouyang, W., Yan, J., Liu, X., Wang, X.: Object detection in videos with tubelet proposal networks. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.101"},{"key":"5_CR52","doi-asserted-by":"crossref","unstructured":"Kang, K., Ouyang, W., Li, H., Wang, X.: Object detection from video tubelets with convolutional neural networks. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.95"},{"key":"5_CR53","doi-asserted-by":"crossref","unstructured":"Karlinsky, L., et al.: RepMet: representative-based metric learning for classification and few-shot object detection. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00534"},{"key":"5_CR54","doi-asserted-by":"crossref","unstructured":"Kemelmacher-Shlizerman, I., Seitz, S.M., Miller, D., Brossard, E.: The megaface benchmark: 1 million faces for recognition at scale. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.527"},{"key":"5_CR55","unstructured":"Khodadadeh, S., Boloni, L., Shah, M.: Unsupervised meta-learning for few-shot image classification. NeurIPS (2019)"},{"key":"5_CR56","doi-asserted-by":"crossref","unstructured":"Kim, C., Li, F., Ciptadi, A., Rehg, J.M.: Multiple hypothesis tracking revisited. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.533"},{"key":"5_CR57","doi-asserted-by":"crossref","unstructured":"Kim, D., Woo, S., Lee, J.Y., Kweon, I.S.: Video panoptic segmentation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00988"},{"key":"5_CR58","unstructured":"Koch, G., Zemel, R., Salakhutdinov, R.: Siamese neural networks for one-shot image recognition. In: ICMLW (2015)"},{"key":"5_CR59","doi-asserted-by":"crossref","unstructured":"Kong, T., Sun, F., Liu, H., Jiang, Y., Li, L., Shi, J.: FoveaBox: beyound anchor-based object detection. IEEE TIP (2020)","DOI":"10.1109\/TIP.2020.3002345"},{"key":"5_CR60","doi-asserted-by":"crossref","unstructured":"Kong, T., Sun, F., Yao, A., Liu, H., Lu, M., Chen, Y.: Ron: Reverse connection with objectness prior networks for object detection. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.557"},{"key":"5_CR61","unstructured":"Kristan, M., et al.: The visual object tracking vot2017 challenge results. In: ICCVW (2017)"},{"key":"5_CR62","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/978-3-030-11009-3_1","volume-title":"Computer Vision \u2013 ECCV 2018 Workshops","author":"M Kristan","year":"2019","unstructured":"Kristan, M., et al.: The sixth visual object tracking VOT2018 challenge results. In: Leal-Taix\u00e9, L., Roth, S. (eds.) ECCV 2018. LNCS, vol. 11129, pp. 3\u201353. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-11009-3_1"},{"key":"5_CR63","unstructured":"Kristan, M., et al.: The visual object tracking vot2015 challenge results. In: ICCVW (2015)"},{"key":"5_CR64","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"765","DOI":"10.1007\/978-3-030-01264-9_45","volume-title":"Computer Vision \u2013 ECCV 2018","author":"H Law","year":"2018","unstructured":"Law, H., Deng, J.: CornerNet: detecting objects as paired keypoints. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) Computer Vision \u2013 ECCV 2018. LNCS, vol. 11218, pp. 765\u2013781. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01264-9_45"},{"key":"5_CR65","unstructured":"Lee, Y., Choi, S.: Gradient-based meta-learning with learned layerwise metric and subspace. In: ICML (2018)"},{"key":"5_CR66","doi-asserted-by":"crossref","unstructured":"Li, A., Li, Z.: Transformation invariant few-shot object detection. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00311"},{"key":"5_CR67","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, W., Wang, Q., Zhang, F., Xing, J., Yan, J.: SiamRPN++: evolution of Siamese visual tracking with very deep networks. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00441"},{"key":"5_CR68","doi-asserted-by":"crossref","unstructured":"Li, B., Yan, J., Wu, W., Zhu, Z., Hu, X.: High performance visual tracking with Siamese region proposal network. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00935"},{"key":"5_CR69","doi-asserted-by":"crossref","unstructured":"Li, B., Yang, B., Liu, C., Liu, F., Ji, R., Ye, Q.: Beyond max-margin: class margin equilibrium for few-shot object detection. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00728"},{"key":"5_CR70","doi-asserted-by":"crossref","unstructured":"Li, X., Wei, T., Chen, Y.P., Tai, Y.W., Tang, C.K.: FSS-1000: a 1000-class dataset for few-shot segmentation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00294"},{"key":"5_CR71","doi-asserted-by":"crossref","unstructured":"Li, Y., Chen, Y., Wang, N., Zhang, Z.: Scale-aware trident networks for object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00615"},{"key":"5_CR72","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Few-shot object detection via classification refinement and distractor retreatment. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01514"},{"key":"5_CR73","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"5_CR74","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"5_CR75","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"5_CR76","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"404","DOI":"10.1007\/978-3-030-01252-6_24","volume-title":"Computer Vision \u2013 ECCV 2018","author":"S Liu","year":"2018","unstructured":"Liu, S., Huang, D., Wang, Y.: Receptive field block net for accurate and fast object detection. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11215, pp. 404\u2013419. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01252-6_24"},{"key":"5_CR77","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/978-3-319-46448-0_2","volume-title":"Computer Vision \u2013 ECCV 2016","author":"W Liu","year":"2016","unstructured":"Liu, W., et al.: SSD: single shot multibox detector. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 21\u201337. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_2"},{"key":"5_CR78","doi-asserted-by":"crossref","unstructured":"Liu, W., Liao, S., Ren, W., Hu, W., Yu, Y.: High-level semantic feature detection: a new perspective for pedestrian detection. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00533"},{"key":"5_CR79","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1007\/978-3-030-58545-7_9","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Y Liu","year":"2020","unstructured":"Liu, Y., Zhang, X., Zhang, S., He, X.: Part-aware prototype network for few-shot semantic segmentation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12354, pp. 142\u2013158. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58545-7_9"},{"key":"5_CR80","doi-asserted-by":"crossref","unstructured":"Lu, X., Li, B., Yue, Y., Li, Q., Yan, J.: Grid R-CNN. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00754"},{"key":"5_CR81","doi-asserted-by":"crossref","unstructured":"Lu, Z., Rathod, V., Votel, R., Huang, J.: RetinaTrack: online single stage joint detection and tracking. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01468"},{"key":"5_CR82","doi-asserted-by":"crossref","unstructured":"Luiten, J., et al.: HOTA: a higher order metric for evaluating multi-object tracking. IJCV (2021)","DOI":"10.1007\/s11263-020-01375-2"},{"key":"5_CR83","doi-asserted-by":"crossref","unstructured":"Mayer, C., Danelljan, M., Paudel, D.P., Gool, L.V.: Learning target candidate association to keep track of what not to track. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01319"},{"key":"5_CR84","unstructured":"Michaelis, C., Bethge, M., Ecker, A.S.: One-shot segmentation in clutter. In: ICML (2018)"},{"key":"5_CR85","unstructured":"Milan, A., Leal-Taix\u00e9, L., Reid, I., Roth, S., Schindler, K.: MOT16: a benchmark for multi-object tracking. arXiv preprint arXiv:1603.00831 (2016)"},{"key":"5_CR86","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"310","DOI":"10.1007\/978-3-030-01246-5_19","volume-title":"Computer Vision \u2013 ECCV 2018","author":"M M\u00fcller","year":"2018","unstructured":"M\u00fcller, M., Bibi, A., Giancola, S., Alsubaihi, S., Ghanem, B.: TrackingNet: a large-scale dataset and benchmark for object tracking in the wild. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11205, pp. 310\u2013327. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01246-5_19"},{"key":"5_CR87","unstructured":"M\u00fcller, R., Kornblith, S., Hinton, G.E.: When does label smoothing help? In: NeurIPS (2019)"},{"key":"5_CR88","doi-asserted-by":"crossref","unstructured":"Oh, S.W., Lee, J.Y., Xu, N., Kim, S.J.: Video object segmentation using space-time memory networks. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00932"},{"key":"5_CR89","doi-asserted-by":"crossref","unstructured":"Pang, B., Li, Y., Zhang, Y., Li, M., Lu, C.: TubeTK: adopting tubes to track multi-object in a one-step training model. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00634"},{"key":"5_CR90","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1007\/978-3-030-58548-8_9","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Peng","year":"2020","unstructured":"Peng, J., et al.: Chained-Tracker: chaining paired attentive regression results for end-to-end joint multiple-object detection and tracking. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12349, pp. 145\u2013161. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58548-8_9"},{"key":"5_CR91","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Pont-Tuset, J., McWilliams, B., Van Gool, L., Gross, M., Sorkine-Hornung, A.: A benchmark dataset and evaluation methodology for video object segmentation. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.85"},{"key":"5_CR92","doi-asserted-by":"crossref","unstructured":"Perez-Rua, J.M., Zhu, X., Hospedales, T.M., Xiang, T.: Incremental few-shot object detection. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01386"},{"key":"5_CR93","doi-asserted-by":"crossref","unstructured":"Real, E., Shlens, J., Mazzocchi, S., Pan, X., Vanhoucke, V.: Youtube-boundingboxes: a large high-precision human-annotated data set for object detection in video. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.789"},{"key":"5_CR94","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"5_CR95","doi-asserted-by":"crossref","unstructured":"Redmon, J., Farhadi, A.: YOLO9000: better, faster, stronger. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.690"},{"key":"5_CR96","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: NeurIPS (2015)"},{"key":"5_CR97","doi-asserted-by":"crossref","unstructured":"Sadeghian, A., Alahi, A., Savarese, S.: Tracking the untrackable: learning to track multiple cues with long-term dependencies. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.41"},{"key":"5_CR98","unstructured":"Santoro, A., Bartunov, S., Botvinick, M., Wierstra, D., Lillicrap, T.: Meta-learning with memory-augmented neural networks. In: ICML (2016)"},{"key":"5_CR99","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"597","DOI":"10.1007\/978-3-030-58517-4_35","volume-title":"Computer Vision \u2013 ECCV 2020","author":"O Sbai","year":"2020","unstructured":"Sbai, O., Couprie, C., Aubry, M.: Impact of base dataset design on few-shot image classification. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 597\u2013613. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_35"},{"key":"5_CR100","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1007\/978-3-030-01240-3_13","volume-title":"Computer Vision \u2013 ECCV 2018","author":"D Shao","year":"2018","unstructured":"Shao, D., Xiong, Yu., Zhao, Y., Huang, Q., Qiao, Yu., Lin, D.: Find and focus: retrieve and localize video events with natural language queries. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11213, pp. 202\u2013218. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01240-3_13"},{"key":"5_CR101","doi-asserted-by":"crossref","unstructured":"Shen, Z., Liu, Z., Li, J., Jiang, Y.G., Chen, Y., Xue, X.: DSOD: learning deeply supervised object detectors from scratch. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.212"},{"key":"5_CR102","doi-asserted-by":"crossref","unstructured":"Shvets, M., Liu, W., Berg, A.C.: Leveraging long-range temporal relationships between proposals for video object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00985"},{"key":"5_CR103","unstructured":"Singh, B., Najibi, M., Davis, L.S.: Sniper: Efficient multi-scale training. In: NeurIPS (2018)"},{"key":"5_CR104","unstructured":"Snell, J., Swersky, K., Zemel, R.: Prototypical networks for few-shot learning. In: NeurIPS (2017)"},{"key":"5_CR105","doi-asserted-by":"crossref","unstructured":"Sun, B., Li, B., Cai, S., Yuan, Y., Zhang, C.: FSCE: few-shot object detection via contrastive proposal encoding. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00727"},{"key":"5_CR106","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"5_CR107","doi-asserted-by":"crossref","unstructured":"Tang, P., Wang, C., Wang, X., Liu, W., Zeng, W., Wang, J.: Object detection in videos by high quality object linking. IEEE TPAMI (2019)","DOI":"10.1109\/TPAMI.2019.2910529"},{"key":"5_CR108","doi-asserted-by":"crossref","unstructured":"Tang, P., Wang, X., Bai, X., Liu, W.: Multiple instance detection network with online instance classifier refinement. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.326"},{"key":"5_CR109","doi-asserted-by":"crossref","unstructured":"Tao, R., Gavves, E., Smeulders, A.W.: Siamese instance search for tracking. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.158"},{"key":"5_CR110","doi-asserted-by":"crossref","unstructured":"Thulasidasan, S., Chennupati, G., Bilmes, J.A., Bhattacharya, T., Michalak, S.: On mixup training: improved calibration and predictive uncertainty for deep neural networks. In: NeurIPS (2019)","DOI":"10.2172\/1525811"},{"key":"5_CR111","doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., He, T.: FCOS: fully convolutional one-stage object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00972"},{"key":"5_CR112","doi-asserted-by":"crossref","unstructured":"Valmadre, J., Bertinetto, L., Henriques, J., Vedaldi, A., Torr, P.H.: End-to-end representation learning for correlation filter based tracking. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.531"},{"key":"5_CR113","unstructured":"Valmadre, J., Bewley, A., Huang, J., Sun, C., Sminchisescu, C., Schmid, C.: Local metrics for multi-object tracking. arXiv preprint arXiv:2104.02631 (2021)"},{"key":"5_CR114","unstructured":"Vinyals, O., et al.: Matching networks for one shot learning. In: NeurIPS (2016)"},{"key":"5_CR115","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"557","DOI":"10.1007\/978-3-030-01261-8_33","volume-title":"Computer Vision \u2013 ECCV 2018","author":"S Wang","year":"2018","unstructured":"Wang, S., Zhou, Y., Yan, J., Deng, Z.: Fully motion-aware network for video object detection. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11217, pp. 557\u2013573. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01261-8_33"},{"key":"5_CR116","unstructured":"Wang, X., Huang, T.E., Darrell, T., Gonzalez, J.E., Yu, F.: Frustratingly simple few-shot object detection. In: ICML (2020)"},{"key":"5_CR117","doi-asserted-by":"crossref","unstructured":"Wang, Y.X., Girshick, R., Hebert, M., Hariharan, B.: Low-shot learning from imaginary data. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00760"},{"key":"5_CR118","doi-asserted-by":"crossref","unstructured":"Wang, Y.X., Ramanan, D., Hebert, M.: Meta-learning to detect rare objects. In: CVPR (2019)","DOI":"10.1109\/ICCV.2019.01002"},{"key":"5_CR119","unstructured":"Woo, S., Kim, D., Cho, D., Kweon, I.S.: LinkNet: relational embedding for scene graph. In: NeurIPS (2018)"},{"key":"5_CR120","doi-asserted-by":"crossref","unstructured":"Wu, C.Y., Feichtenhofer, C., Fan, H., He, K., Krahenbuhl, P., Girshick, R.: Long-term feature banks for detailed video understanding. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00037"},{"key":"5_CR121","doi-asserted-by":"crossref","unstructured":"Wu, H., Chen, Y., Wang, N., Zhang, Z.: Sequence level semantics aggregation for video object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00931"},{"key":"5_CR122","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"456","DOI":"10.1007\/978-3-030-58517-4_27","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Wu","year":"2020","unstructured":"Wu, J., Liu, S., Huang, D., Wang, Y.: Multi-scale positive sample refinement for few-shot object detection. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 456\u2013472. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_27"},{"key":"5_CR123","doi-asserted-by":"crossref","unstructured":"Wu, Y., Lim, J., Yang, M.H.: Online object tracking: a benchmark. In: CVPR (2013)","DOI":"10.1109\/CVPR.2013.312"},{"key":"5_CR124","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"494","DOI":"10.1007\/978-3-030-01237-3_30","volume-title":"Computer Vision \u2013 ECCV 2018","author":"F Xiao","year":"2018","unstructured":"Xiao, F., Lee, Y.J.: Video object detection with an aligned spatial-temporal memory. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11212, pp. 494\u2013510. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01237-3_30"},{"key":"5_CR125","doi-asserted-by":"crossref","unstructured":"Xiao, T., Li, S., Wang, B., Lin, L., Wang, X.: Joint detection and identification feature learning for person search. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.360"},{"key":"5_CR126","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"192","DOI":"10.1007\/978-3-030-58520-4_12","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Y Xiao","year":"2020","unstructured":"Xiao, Y., Marlet, R.: Few-shot object detection and viewpoint estimation for objects in the wild. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12362, pp. 192\u2013210. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58520-4_12"},{"key":"5_CR127","doi-asserted-by":"crossref","unstructured":"Xu, J., Cao, Y., Zhang, Z., Hu, H.: Spatial-temporal relation networks for multi-object tracking. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00409"},{"key":"5_CR128","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1007\/978-3-030-01228-1_36","volume-title":"Computer Vision \u2013 ECCV 2018","author":"N Xu","year":"2018","unstructured":"Xu, N., et al.: YouTube-VOS: sequence-to-sequence video object segmentation. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11209, pp. 603\u2013619. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01228-1_36"},{"key":"5_CR129","doi-asserted-by":"crossref","unstructured":"Yan, X., Chen, Z., Xu, A., Wang, X., Liang, X., Lin, L.: Meta R-CNN: towards general solver for instance-level low-shot learning. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00967"},{"key":"5_CR130","unstructured":"Yang, F.S.Y., Zhang, L., Xiang, T., Torr, P.H., Hospedales, T.M.: Learning to compare: relation network for few-shot learning. In: CVPR (2018)"},{"key":"5_CR131","doi-asserted-by":"crossref","unstructured":"Yang, L., Fan, Y., Xu, N.: Video instance segmentation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00529"},{"key":"5_CR132","unstructured":"Yang, Y., Wei, F., Shi, M., Li, G.: Restoring negative information in few-shot object detection. In: NeurIPS (2020)"},{"key":"5_CR133","doi-asserted-by":"crossref","unstructured":"Yang, Z., Liu, S., Hu, H., Wang, L., Lin, S.: RepPoints: point set representation for object detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00975"},{"key":"5_CR134","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, Y., Chen, X., Liu, J., Qiao, Y.: Context-transformer: tackling object confusion for few-shot detection. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i07.6957"},{"key":"5_CR135","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1007\/978-3-319-48881-3_3","volume-title":"Computer Vision \u2013 ECCV 2016 Workshops","author":"F Yu","year":"2016","unstructured":"Yu, F., Li, W., Li, Q., Liu, Yu., Shi, X., Yan, J.: POI: multiple object tracking with high performance detection and appearance feature. In: Hua, G., J\u00e9gou, H. (eds.) ECCV 2016. LNCS, vol. 9914, pp. 36\u201342. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-48881-3_3"},{"key":"5_CR136","unstructured":"Zhan, Y., Wang, C., Wang, X., Zeng, W., Liu, W.: A simple baseline for multi-object tracking. IJCV (2021)"},{"key":"5_CR137","doi-asserted-by":"crossref","unstructured":"Zhang, C., Cai, Y., Lin, G., Shen, C.: DeepEMD: few-shot image classification with differentiable earth mover\u2019s distance and structured classifiers. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01222"},{"key":"5_CR138","doi-asserted-by":"crossref","unstructured":"Zhang, L., Zhou, S., Guan, J., Zhang, J.: Accurate few-shot object detection with support-query mutual guidance and hybrid loss. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01419"},{"key":"5_CR139","doi-asserted-by":"crossref","unstructured":"Zhang, S., Chi, C., Yao, Y., Lei, Z., Li, S.Z.: Bridging the gap between anchor-based and anchor-free detection via adaptive training sample selection. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00978"},{"key":"5_CR140","doi-asserted-by":"crossref","unstructured":"Zhang, W., Wang, Y.X.: Hallucination improves few-shot object detection. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01281"},{"key":"5_CR141","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Qiao, S., Xie, C., Shen, W., Wang, B., Yuille, A.L.: Single-shot object detection with enriched semantics. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00609"},{"key":"5_CR142","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"474","DOI":"10.1007\/978-3-030-58548-8_28","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Zhou","year":"2020","unstructured":"Zhou, X., Koltun, V., Kr\u00e4henb\u00fchl, P.: Tracking objects as points. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12349, pp. 474\u2013490. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58548-8_28"},{"key":"5_CR143","doi-asserted-by":"crossref","unstructured":"Zhou, X., Zhuo, J., Krahenbuhl, P.: Bottom-up object detection by grouping extreme and center points. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00094"},{"key":"5_CR144","doi-asserted-by":"crossref","unstructured":"Zhu, C., Chen, F., Ahmed, U., Savvides, M.: Semantic relation reasoning for shot-stable few-shot object detection. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00867"},{"key":"5_CR145","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1007\/978-3-030-01228-1_23","volume-title":"Computer Vision \u2013 ECCV 2018","author":"J Zhu","year":"2018","unstructured":"Zhu, J., Yang, H., Liu, N., Kim, M., Zhang, W., Yang, M.-H.: Online multi-object tracking with dual matching attention networks. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11209, pp. 379\u2013396. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01228-1_23"},{"key":"5_CR146","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"782","DOI":"10.1007\/978-3-030-01234-2_46","volume-title":"Computer Vision \u2013 ECCV 2018","author":"L Zhu","year":"2018","unstructured":"Zhu, L., Yang, Y.: Compound memory networks for few-shot video classification. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11211, pp. 782\u2013797. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_46"},{"key":"5_CR147","doi-asserted-by":"crossref","unstructured":"Zhu, R., Zhang, S., Wang, X., Wen, L., Shi, H., Bo, L., Mei, T.: ScratchDet: training single-shot object detectors from scratch. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00237"},{"key":"5_CR148","doi-asserted-by":"crossref","unstructured":"Zhu, X., Dai, J., Yuan, L., Wei, Y.: Towards high performance video object detection. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00753"},{"key":"5_CR149","doi-asserted-by":"crossref","unstructured":"Zhu, X., Wang, Y., Dai, J., Yuan, L., Wei, Y.: Flow-guided feature aggregation for video object detection. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.52"},{"key":"5_CR150","doi-asserted-by":"crossref","unstructured":"Zhu, X., Xiong, Y., Dai, J., Yuan, L., Wei, Y.: Deep feature flow for video recognition. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.441"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20044-1_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,13]],"date-time":"2024-03-13T19:44:08Z","timestamp":1710359048000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20044-1_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031200434","9783031200441"],"references-count":150,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20044-1_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"20 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}