{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:42:15Z","timestamp":1772120535061,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Natural Science Foun- dation of China","award":["61876098"],"award-info":[{"award-number":["61876098"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475342","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T20:00:05Z","timestamp":1634587205000},"page":"1884-1892","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":15,"title":["Learning Hierarchical Embedding for Video Instance Segmentation"],"prefix":"10.1145","author":[{"given":"Zheyun","family":"Qin","sequence":"first","affiliation":[{"name":"Shandong University, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiankai","family":"Lu","sequence":"additional","affiliation":[{"name":"Shandong University, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiushan","family":"Nie","sequence":"additional","affiliation":[{"name":"Shandong Jianzhu University, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiantong","family":"Zhen","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yilong","family":"Yin","sequence":"additional","affiliation":[{"name":"Shandong University, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Lynton Ardizzone Carsten L\u00fcth Jakob Kruse C. Rother and U. K\u00f6the. 2019. Guided Image Generation with Conditional Invertible Neural Networks. ArXiv Vol. abs\/1907.02392 (2019).  Lynton Ardizzone Carsten L\u00fcth Jakob Kruse C. Rother and U. K\u00f6the. 2019. Guided Image Generation with Conditional Invertible Neural Networks. ArXiv Vol. abs\/1907.02392 (2019)."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Ali Athar S. Mahadevan Aljosa Osep L. Leal-Taix\u00e9 and B. Leibe. 2020. STEm-Seg: Spatio-temporal Embeddings for Instance Segmentation in Videos. In ECCV.  Ali Athar S. Mahadevan Aljosa Osep L. Leal-Taix\u00e9 and B. Leibe. 2020. STEm-Seg: Spatio-temporal Embeddings for Instance Segmentation in Videos. In ECCV.","DOI":"10.1007\/978-3-030-58621-8_10"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Gedas Bertasius and Lorenzo Torresani. 2020. Classifying segmenting and tracking object instances in video with mask propagation. In CVPR.  Gedas Bertasius and Lorenzo Torresani. 2020. Classifying segmenting and tracking object instances in video with mask propagation. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00976"},{"key":"e_1_3_2_2_4_1","volume-title":"Federico Perazzi, and Jordi Pont-Tuset.","author":"Caelles Sergi","year":"2018"},{"key":"e_1_3_2_2_5_1","volume-title":"The 2019 DAVIS Challenge on VOS: Unsupervised Multi-Object Segmentation. arXiv:1905.00737","author":"Caelles Sergi","year":"2019"},{"key":"e_1_3_2_2_6_1","volume-title":"Hisham Cholakkal, Fahad Shahbaz Khan, Yanwei Pang, and Ling Shao.","author":"Cao Jiale","year":"2020"},{"key":"e_1_3_2_2_7_1","volume-title":"Key Instance Selection for Unsupervised Video Object Segmentation. arXiv:1906.07851","author":"Cho Donghyeon","year":"2019"},{"key":"e_1_3_2_2_8_1","volume-title":"Nice: Non-linear independent components estimation. arXiv preprint arXiv:1410.8516","author":"Dinh Laurent","year":"2014"},{"key":"e_1_3_2_2_9_1","volume-title":"Density estimation using real nvp. arXiv preprint arXiv:1605.08803","author":"Dinh Laurent","year":"2016"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Minghui Dong Jian Wang Yuanyuan Huang Dongdong Yu Kai Su Kaihui Zhou Jie Shao Shiping Wen and Changhu Wang. 2019. Temporal Feature Augmented Network for Video Instance Segmentation. In ICCV.  Minghui Dong Jian Wang Yuanyuan Huang Dongdong Yu Kai Su Kaihui Zhou Jie Shao Shiping Wen and Changhu Wang. 2019. Temporal Feature Augmented Network for Video Instance Segmentation. In ICCV.","DOI":"10.1109\/ICCVW.2019.00091"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"e_1_3_2_2_12_1","volume-title":"Dual Embedding Learning for Video Instance Segmentation. In ICCV Workshops.","author":"Feng Qianyu","year":"2019"},{"key":"e_1_3_2_2_13_1","volume-title":"Generative Adversarial Networks. arXiv:1406.2661","author":"Goodfellow Ian J.","year":"2014"},{"key":"e_1_3_2_2_14_1","volume-title":"Piotr Doll\u00e1 r, and Ross B. Girshick","author":"He Kaiming","year":"2017"},{"key":"e_1_3_2_2_15_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Dahun Kim Sanghyun Woo Joon-Young Lee and In So Kweon. 2020. Video Panoptic Segmentation. In CVPR.  Dahun Kim Sanghyun Woo Joon-Young Lee and In So Kweon. 2020. Video Panoptic Segmentation. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00988"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.5555\/3327546.3327685"},{"key":"e_1_3_2_2_18_1","volume-title":"Kingma and Max Welling","author":"Diederik","year":"2014"},{"key":"e_1_3_2_2_19_1","volume-title":"Rog\u00e9 rio Feris, and Linglin He","author":"Lin Chung-Ching","year":"2020"},{"key":"e_1_3_2_2_20_1","volume-title":"Piotr Doll\u00e1 r, and C. Lawrence Zitnick","author":"Lin Tsung-Yi","year":"2014"},{"key":"e_1_3_2_2_21_1","volume-title":"Spatio-Temporal Attention Network for Video Instance Segmentation. In ICCV Workshops.","author":"Liu Xiaoyu","year":"2019"},{"key":"e_1_3_2_2_22_1","volume-title":"2020 a. Deep Object Tracking with Shrinkage Loss","author":"Lu Xiankai","year":"2020"},{"key":"e_1_3_2_2_23_1","volume-title":"See More","author":"Lu Xiankai"},{"key":"e_1_3_2_2_24_1","unstructured":"Xiankai Lu Wenguan Wang Danelljan Martin Tianfei Zhou Jianbing Shen and Van Gool Luc. 2020 b. Video Object Segmentation with Episodic Graph Memory Networks. In ECCV.  Xiankai Lu Wenguan Wang Danelljan Martin Tianfei Zhou Jianbing Shen and Van Gool Luc. 2020 b. Video Object Segmentation with Episodic Graph Memory Networks. In ECCV."},{"key":"e_1_3_2_2_25_1","volume-title":"2020 c. Zero-Shot Video Object Segmentation with Co-Attention Siamese Networks","author":"Lu Xiankai","year":"2020"},{"key":"e_1_3_2_2_26_1","unstructured":"Xiankai Lu Wenguan Wang Jianbing Shen Yu-Wing Tai David J Crandall and Steven CH Hoi. 2020 d. Learning video object segmentation from unlabeled videos. In CVPR.  Xiankai Lu Wenguan Wang Jianbing Shen Yu-Wing Tai David J Crandall and Steven CH Hoi. 2020 d. Learning video object segmentation from unlabeled videos. In CVPR."},{"key":"e_1_3_2_2_27_1","volume-title":"Luc Van Gool, and Radu Timofte","author":"Lugmayr Andreas","year":"2020"},{"key":"e_1_3_2_2_28_1","volume-title":"Classification and Tracking. In ICCV Workshop.","author":"Luiten Jonathon","year":"2019"},{"key":"e_1_3_2_2_29_1","volume-title":"Senthil Kumar Yogamani, and Ahmad El Sallab","author":"Mohamed Eslam","year":"2020"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.5555\/1703775.1704135"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3455008"},{"key":"e_1_3_2_2_32_1","volume-title":"Markus H. Gross, and Alexander Sorkine-Hornung.","author":"Perazzi Federico","year":"2016"},{"key":"e_1_3_2_2_33_1","volume-title":"The 2017 DAVIS Challenge on Video Object Segmentation. arXiv:1704.00675","author":"Pont-Tuset Jordi","year":"2017"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","unstructured":"Albert Pumarola Stefan Popov Francesc Moreno-Noguer and Vittorio Ferrari. 2020. C-Flow: Conditional Generative Flow Models for Images and 3D Point Clouds. In CVPR.  Albert Pumarola Stefan Popov Francesc Moreno-Noguer and Vittorio Ferrari. 2020. C-Flow: Conditional Generative Flow Models for Images and 3D Point Clouds. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00797"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.5555\/3045118.3045281"},{"key":"e_1_3_2_2_36_1","volume-title":"Ferran Marqu\u00e9 s, and Xavier Gir\u00f3 -i-Nieto","author":"Ventura Carles","year":"2019"},{"key":"e_1_3_2_2_37_1","volume-title":"FEELVOS: Fast End-To-End Embedding Learning for Video Object Segmentation. In CVPR.","author":"Voigtlaender Paul","year":"2019"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"crossref","unstructured":"Limin Wang Yuanjun Xiong Zhe Wang Yu Qiao Dahua Lin Xiaoou Tang and Luc Van Gool. 2016. Temporal Segment Networks: Towards Good Practices for Deep Action Recognition. In ECCV.  Limin Wang Yuanjun Xiong Zhe Wang Yu Qiao Dahua Lin Xiaoou Tang and Luc Van Gool. 2016. Temporal Segment Networks: Towards Good Practices for Deep Action Recognition. In ECCV.","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Wenguan Wang Xiankai Lu Jianbing Shen David J. Crandall and Ling Shao. 2019 a. Zero-Shot Video Object Segmentation via Attentive Graph Neural Networks. In ICCV.  Wenguan Wang Xiankai Lu Jianbing Shen David J. Crandall and Ling Shao. 2019 a. Zero-Shot Video Object Segmentation via Attentive Graph Neural Networks. In ICCV.","DOI":"10.1109\/ICCV.2019.00933"},{"key":"e_1_3_2_2_40_1","volume-title":"2021 a. Paying attention to video object pattern understanding","author":"Wang Wenguan","year":"2021"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2819173"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2662005"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00318"},{"key":"e_1_3_2_2_44_1","unstructured":"Wenguan Wang Tianfei Zhou Fatih Porikli David Crandall and Luc Van Gool. 2021 b. A Survey on Deep Learning Technique for Video Segmentation. arxiv: 2107.01153  Wenguan Wang Tianfei Zhou Fatih Porikli David Crandall and Luc Van Gool. 2021 b. A Survey on Deep Learning Technique for Video Segmentation. arxiv: 2107.01153"},{"key":"e_1_3_2_2_45_1","volume-title":"Learning Likelihoods with Conditional Normalizing Flows. arXiv:1912.00042","author":"Winkler Christina","year":"2019"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Guandao Yang Xun Huang Zekun Hao Ming-Yu Liu Serge J. Belongie and Bharath Hariharan. 2019 b. PointFlow: 3D Point Cloud Generation With Continuous Normalizing Flows. In ICCV.  Guandao Yang Xun Huang Zekun Hao Ming-Yu Liu Serge J. Belongie and Bharath Hariharan. 2019 b. PointFlow: 3D Point Cloud Generation With Continuous Normalizing Flows. In ICCV.","DOI":"10.1109\/ICCV.2019.00464"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Linjie Yang Yuchen Fan and Ning Xu. 2019 a. Video Instance Segmentation. In ICCV.  Linjie Yang Yuchen Fan and Ning Xu. 2019 a. Video Instance Segmentation. In ICCV.","DOI":"10.1109\/ICCV.2019.00529"},{"key":"e_1_3_2_2_48_1","volume-title":"Katsaggelos","author":"Yang Linjie","year":"2018"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.5555\/3045118.3045291"},{"key":"e_1_3_2_2_50_1","volume-title":"Hongyi Xu","author":"Zanfir Andrei","year":"2020"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475342","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475342","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:49:18Z","timestamp":1750193358000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475342"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":50,"alternative-id":["10.1145\/3474085.3475342","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475342","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}