{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T12:07:52Z","timestamp":1777637272208,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,10,15]],"date-time":"2018-10-15T00:00:00Z","timestamp":1539561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,10,15]]},"DOI":"10.1145\/3240508.3240693","type":"proceedings-article","created":{"date-parts":[[2018,10,18]],"date-time":"2018-10-18T13:52:08Z","timestamp":1539870728000},"page":"1811-1819","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Semi-Supervised DFF"],"prefix":"10.1145","author":[{"given":"Guangxing","family":"Han","sequence":"first","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chongrong","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Aria Ahmadi and Ioannis Patras. 2016. Unsupervised convolutional neural networks for motion estimation. In ICIP. 1629--1633. Aria Ahmadi and Ioannis Patras. 2016. Unsupervised convolutional neural networks for motion estimation. In ICIP. 1629--1633.","DOI":"10.1109\/ICIP.2016.7532634"},{"key":"e_1_3_2_1_2_1","volume-title":"LSTD: A Low-Shot Transfer Detector for Object Detection. In AAAI.","author":"Chen Chao","year":"2018"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Dongdong Chen Jing Liao Lu Yuan Nenghai Yu and Gang Hua. 2017. Coherent Online Video Style Transfer. In ICCV. 1105--1114. Dongdong Chen Jing Liao Lu Yuan Nenghai Yu and Gang Hua. 2017. Coherent Online Video Style Transfer. In ICCV. 1105--1114.","DOI":"10.1109\/ICCV.2017.126"},{"key":"e_1_3_2_1_4_1","unstructured":"Guobin Chen Wongun Choi Xiang Yu Tony Han and Manmohan Chandraker. 2017. Learning Efficient Object Detection Models with Knowledge Distillation. In NIPS. 742--751. Guobin Chen Wongun Choi Xiang Yu Tony Han and Manmohan Chandraker. 2017. Learning Efficient Object Detection Models with Knowledge Distillation. In NIPS. 742--751."},{"key":"e_1_3_2_1_5_1","unstructured":"Tianqi Chen Yutian Li Mu Li Min Lin Naiyan Wang Minjie Wang Tianjun Xiao Bing Xu Chiyuan Zhang and Zheng Zhang. 2015. MXNet: A Flexible and Efficient Machine Learning Library for Heterogeneous Distributed Systems. In NIPSW. Tianqi Chen Yutian Li Mu Li Min Lin Naiyan Wang Minjie Wang Tianjun Xiao Bing Xu Chiyuan Zhang and Zheng Zhang. 2015. MXNet: A Flexible and Efficient Machine Learning Library for Heterogeneous Distributed Systems. In NIPSW."},{"key":"e_1_3_2_1_6_1","unstructured":"Jifeng Dai Yi Li Kaiming He and Jian Sun. 2016. R-FCN: Object Detection via Region-based Fully Convolutional Networks. In NIPS. 379--387. Jifeng Dai Yi Li Kaiming He and Jian Sun. 2016. R-FCN: Object Detection via Region-based Fully Convolutional Networks. In NIPS. 379--387."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.316"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-014-0733-5"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Christoph Feichtenhofer Axel Pinz and Andrew Zisserman. 2017. Detect to Track and Track to Detect. In ICCV. 3038--3046. Christoph Feichtenhofer Axel Pinz and Andrew Zisserman. 2017. Detect to Track and Track to Detect. In ICCV. 3038--3046.","DOI":"10.1109\/ICCV.2017.330"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Raghudeep Gadde Varun Jampani and Peter V. Gehler. 2017. Semantic Video CNNs Through Representation Warping. In ICCV. 4453--4462. Raghudeep Gadde Varun Jampani and Peter V. Gehler. 2017. Semantic Video CNNs Through Representation Warping. In ICCV. 4453--4462.","DOI":"10.1109\/ICCV.2017.477"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"key":"e_1_3_2_1_13_1","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative Adversarial Nets. In NIPS. 2672--2680. Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative Adversarial Nets. In NIPS. 2672--2680."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Agrim Gupta Justin Johnson Alexandre Alahi and Li Fei-Fei. 2017. Characterizing and Improving Stability in Neural Style Transfer. In ICCV. 4067--4076. Agrim Gupta Justin Johnson Alexandre Alahi and Li Fei-Fei. 2017. Characterizing and Improving Stability in Neural Style Transfer. In ICCV. 4067--4076.","DOI":"10.1109\/ICCV.2017.438"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Guangxing Han Xuan Zhang and Chongrong Li. 2017. Revisiting Faster R-CNN: A Deeper Look at Region Proposal Network. In ICONIP. 14--24. Guangxing Han Xuan Zhang and Chongrong Li. 2017. Revisiting Faster R-CNN: A Deeper Look at Region Proposal Network. In ICONIP. 14--24.","DOI":"10.1007\/978-3-319-70090-8_2"},{"key":"e_1_3_2_1_16_1","unstructured":"Guangxing Han Xuan Zhang and Chongrong Li. 2017. Single shot object detection with top-down refinement. In ICIP. 3360--3364. Guangxing Han Xuan Zhang and Chongrong Li. 2017. Single shot object detection with top-down refinement. In ICIP. 3360--3364."},{"key":"e_1_3_2_1_17_1","unstructured":"Kaiming He Georgia Gkioxari Piotr Doll\u00e1r and Ross Girshick. 2017. Mask RCNN. In ICCV. 2961--2969. Kaiming He Georgia Gkioxari Piotr Doll\u00e1r and Ross Girshick. 2017. Mask RCNN. In ICCV. 2961--2969."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. 770--778. Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. 770--778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_19_1","volume-title":"NIPS Deep Learning and Representation Learning Workshop.","author":"Hinton Geoffrey","year":"2015"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Gao Huang Zhuang Liu Laurens van der Maaten and Kilian Q. Weinberger. 2017. Densely Connected Convolutional Networks. In CVPR. 4700--4708. Gao Huang Zhuang Liu Laurens van der Maaten and Kilian Q. Weinberger. 2017. Densely Connected Convolutional Networks. In CVPR. 4700--4708.","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Jonathan Huang Vivek Rathod Chen Sun Menglong Zhu Anoop Korattikara Alireza Fathi Ian Fischer Zbigniew Wojna Yang Song Sergio Guadarrama and Kevin Murphy. 2017. Speed\/accuracy trade-offs for modern convolutional object detectors. In CVPR. 7310--7311. Jonathan Huang Vivek Rathod Chen Sun Menglong Zhu Anoop Korattikara Alireza Fathi Ian Fischer Zbigniew Wojna Yang Song Sergio Guadarrama and Kevin Murphy. 2017. Speed\/accuracy trade-offs for modern convolutional object detectors. In CVPR. 7310--7311.","DOI":"10.1109\/CVPR.2017.351"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Phillip Isola Jun-Yan Zhu Tinghui Zhou and Alexei A. Efros. 2017. ImageTo-Image Translation With Conditional Adversarial Networks. In CVPR. 1125--1134. Phillip Isola Jun-Yan Zhu Tinghui Zhou and Alexei A. Efros. 2017. ImageTo-Image Translation With Conditional Adversarial Networks. In CVPR. 1125--1134.","DOI":"10.1109\/CVPR.2017.632"},{"key":"e_1_3_2_1_23_1","unstructured":"Max Jaderberg Karen Simonyan Andrew Zisserman and koray kavukcuoglu. 2015. Spatial Transformer Networks. In NIPS. 2017--2025. Max Jaderberg Karen Simonyan Andrew Zisserman and koray kavukcuoglu. 2015. Spatial Transformer Networks. In NIPS. 2017--2025."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Dinesh Jayaraman and Kristen Grauman. 2016. Slow and Steady Feature Analysis: Higher Order Temporal Coherence in Video. In CVPR. 3852--3861. Dinesh Jayaraman and Kristen Grauman. 2016. Slow and Steady Feature Analysis: Higher Order Temporal Coherence in Video. In CVPR. 3852--3861.","DOI":"10.1109\/CVPR.2016.418"},{"key":"e_1_3_2_1_25_1","unstructured":"Xiaojie Jin Xin Li Huaxin Xiao Xiaohui Shen Zhe Lin Jimei Yang Yunpeng Chen Jian Dong Luoqi Liu Zequn Jie Jiashi Feng and Shuicheng Yan. 2017. Video Scene Parsing With Predictive Feature Learning. In ICCV. 5580--5588. Xiaojie Jin Xin Li Huaxin Xiao Xiaohui Shen Zhe Lin Jimei Yang Yunpeng Chen Jian Dong Luoqi Liu Zequn Jie Jiashi Feng and Shuicheng Yan. 2017. Video Scene Parsing With Predictive Feature Learning. In ICCV. 5580--5588."},{"key":"e_1_3_2_1_26_1","unstructured":"Xiaojie Jin Huaxin Xiao Xiaohui Shen Jimei Yang Zhe Lin Yunpeng Chen Zequn Jie Jiashi Feng and Shuicheng Yan. 2017. Predicting Scene Parsing and Motion Dynamics in the Future. In NIPS. 6915--6924. Xiaojie Jin Huaxin Xiao Xiaohui Shen Jimei Yang Zhe Lin Yunpeng Chen Zequn Jie Jiashi Feng and Shuicheng Yan. 2017. Predicting Scene Parsing and Motion Dynamics in the Future. In NIPS. 6915--6924."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Justin Johnson Alexandre Alahi and Li Fei-Fei. 2016. Perceptual Losses for RealTime Style Transfer and Super-Resolution. In ECCV. 694--711. Justin Johnson Alexandre Alahi and Li Fei-Fei. 2016. Perceptual Losses for RealTime Style Transfer and Super-Resolution. In ECCV. 694--711.","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Kai Kang Hongsheng Li Tong Xiao Wanli Ouyang Junjie Yan Xihui Liu and Xiaogang Wang. 2017. Object Detection in Videos With Tubelet Proposal Networks. In CVPR. 727--735. Kai Kang Hongsheng Li Tong Xiao Wanli Ouyang Junjie Yan Xihui Liu and Xiaogang Wang. 2017. Object Detection in Videos With Tubelet Proposal Networks. In CVPR. 727--735.","DOI":"10.1109\/CVPR.2017.101"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Kai Kang Wanli Ouyang Hongsheng Li and Xiaogang Wang. 2016. Object Detection From Video Tubelets With Convolutional Neural Networks. In CVPR. 817--825. Kai Kang Wanli Ouyang Hongsheng Li and Xiaogang Wang. 2016. Object Detection From Video Tubelets With Convolutional Neural Networks. In CVPR. 817--825.","DOI":"10.1109\/CVPR.2016.95"},{"key":"e_1_3_2_1_30_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In NIPS. 1097--1105. Alex Krizhevsky Ilya Sutskever and Geoffrey E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In NIPS. 1097--1105."},{"key":"e_1_3_2_1_31_1","volume-title":"Learning Without Forgetting. In European Conference on Computer Vision (ECCV). 614--629","author":"Li Zhizhong","year":"2016"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Xiaodan Liang Lisa Lee Wei Dai and Eric P. Xing. 2017. Dual Motion GAN for Future-Flow Embedded Video Prediction. In ICCV. 1744--1752. Xiaodan Liang Lisa Lee Wei Dai and Eric P. Xing. 2017. Dual Motion GAN for Future-Flow Embedded Video Prediction. In ICCV. 1744--1752.","DOI":"10.1109\/ICCV.2017.194"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Tsung-Yi Lin Michael Maire Serge J. Belongie Lubomir D. Bourdev Ross B. Girshick James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1r and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV. 740--755. Tsung-Yi Lin Michael Maire Serge J. Belongie Lubomir D. Bourdev Ross B. Girshick James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1r and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV. 740--755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_34_1","unstructured":"Tsung-Yi Lin Priya Goyal Ross Girshick Kaiming He and Piotr Doll\u00e1r. 2017. Focal Loss for Dense Object Detection. In ICCV. 2980--2988. Tsung-Yi Lin Priya Goyal Ross Girshick Kaiming He and Piotr Doll\u00e1r. 2017. Focal Loss for Dense Object Detection. In ICCV. 2980--2988."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Wei Liu Dragomir Anguelov Dumitru Erhan Christian Szegedy Scott Reed Cheng-Yang Fu and Alexander C. Berg. 2016. SSD: Single Shot MultiBox Detector. In ECCV. 21--37. Wei Liu Dragomir Anguelov Dumitru Erhan Christian Szegedy Scott Reed Cheng-Yang Fu and Alexander C. Berg. 2016. SSD: Single Shot MultiBox Detector. In ECCV. 21--37.","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Ziwei Liu Raymond A. Yeh Xiaoou Tang Yiming Liu and Aseem Agarwala. 2017. Video Frame Synthesis Using Deep Voxel Flow. In ICCV. 4463--4471. Ziwei Liu Raymond A. Yeh Xiaoou Tang Yiming Liu and Aseem Agarwala. 2017. Video Frame Synthesis Using Deep Voxel Flow. In ICCV. 4463--4471.","DOI":"10.1109\/ICCV.2017.478"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Jonathan Long Evan Shelhamer and Trevor Darrell. 2015. Fully Convolutional Networks for Semantic Segmentation. In CVPR. 3431--3440. Jonathan Long Evan Shelhamer and Trevor Darrell. 2015. Fully Convolutional Networks for Semantic Segmentation. In CVPR. 3431--3440.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Pauline Luc Natalia Neverova Camille Couprie Jakob Verbeek and Yann LeCun. 2017. Predicting Deeper Into the Future of Semantic Segmentation. In ICCV. 648--657. Pauline Luc Natalia Neverova Camille Couprie Jakob Verbeek and Yann LeCun. 2017. Predicting Deeper Into the Future of Semantic Segmentation. In ICCV. 648--657.","DOI":"10.1109\/ICCV.2017.77"},{"key":"e_1_3_2_1_39_1","unstructured":"Michael Mathieu Camille Couprie and Yann LeCun. 2016. Deep multi-scale video prediction beyond mean square error. In ICLR. Michael Mathieu Camille Couprie and Yann LeCun. 2016. Deep multi-scale video prediction beyond mean square error. In ICLR."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Esteban Real Jonathon Shlens Stefano Mazzocchi Xin Pan and Vincent Vanhoucke. 2017. YouTube-BoundingBoxes: A Large High-Precision HumanAnnotated Data Set for Object Detection in Video. In CVPR. 5296--5305. Esteban Real Jonathon Shlens Stefano Mazzocchi Xin Pan and Vincent Vanhoucke. 2017. YouTube-BoundingBoxes: A Large High-Precision HumanAnnotated Data Set for Object Detection in Video. In CVPR. 5296--5305.","DOI":"10.1109\/CVPR.2017.789"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Joseph Redmon and Ali Farhadi. 2017. YOLO9000: Better Faster Stronger. In CVPR. 7263--7271. Joseph Redmon and Ali Farhadi. 2017. YOLO9000: Better Faster Stronger. In CVPR. 7263--7271.","DOI":"10.1109\/CVPR.2017.690"},{"key":"e_1_3_2_1_42_1","unstructured":"S. Ren K. He R. Girshick and J. Sun. 2015. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. In NIPS. 91--99. S. Ren K. He R. Girshick and J. Sun. 2015. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. In NIPS. 91--99."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Zhe Ren Junchi Yan Bingbing Ni Bin Liu Xiaokang Yang and Hongyuan Zha. 2017. Unsupervised Deep Learning for Optical Flow Estimation. In AAAI. 1495--1501. Zhe Ren Junchi Yan Bingbing Ni Bin Liu Xiaokang Yang and Hongyuan Zha. 2017. Unsupervised Deep Learning for Optical Flow Estimation. In AAAI. 1495--1501.","DOI":"10.1609\/aaai.v31i1.10723"},{"key":"e_1_3_2_1_44_1","volume-title":"FitNets: Hints for Thin Deep Nets. In In Proceedings of ICLR.","author":"Romero Adriana","year":"2015"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Evan Shelhamer Kate Rakelly Judy Hoffman and Trevor Darrell. 2016. Clockwork Convnets for Video Semantic Segmentation. In ECCVW. 852--868. Evan Shelhamer Kate Rakelly Judy Hoffman and Trevor Darrell. 2016. Clockwork Convnets for Video Semantic Segmentation. In ECCVW. 852--868.","DOI":"10.1007\/978-3-319-49409-8_69"},{"key":"e_1_3_2_1_47_1","volume-title":"Incremental Learning of Object Detectors Without Catastrophic Forgetting. In The IEEE International Conference on Computer Vision (ICCV). 3400--3409","author":"Shmelkov Konstantin","year":"2017"},{"key":"e_1_3_2_1_48_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-Stream Convolutional Networks for Action Recognition in Videos. In NIPS. 568--576. Karen Simonyan and Andrew Zisserman. 2014. Two-Stream Convolutional Networks for Action Recognition in Videos. In NIPS. 568--576."},{"key":"e_1_3_2_1_49_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In ICLR. Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In ICLR."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going Deeper with Convolutions. In CVPR. 1--9. Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going Deeper with Convolutions. In CVPR. 1--9.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"crossref","unstructured":"Carl Vondrick Hamed Pirsiavash and Antonio Torralba. 2016. Anticipating Visual Representations From Unlabeled Video. In CVPR. 98--106. Carl Vondrick Hamed Pirsiavash and Antonio Torralba. 2016. Anticipating Visual Representations From Unlabeled Video. In CVPR. 98--106.","DOI":"10.1109\/CVPR.2016.18"},{"key":"e_1_3_2_1_52_1","unstructured":"Carl Vondrick Hamed Pirsiavash and Antonio Torralba. 2016. Generating Videos with Scene Dynamics. In NIPS. 613--621. Carl Vondrick Hamed Pirsiavash and Antonio Torralba. 2016. Generating Videos with Scene Dynamics. In NIPS. 613--621."},{"key":"e_1_3_2_1_53_1","unstructured":"Tuan-Hung Vu Wongun Choi Samuel Schulter and Manmohan Chandraker. 2018. Memory Warps for Learning Long-Term Online Video Representations. arXiv:1803.10861 (2018). Tuan-Hung Vu Wongun Choi Samuel Schulter and Manmohan Chandraker. 2018. Memory Warps for Learning Long-Term Online Video Representations. arXiv:1803.10861 (2018)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Xiaolong Wang Ali Farhadi and Abhinav Gupta. 2016. Actions Transformations. In CVPR. 2658--2667. Xiaolong Wang Ali Farhadi and Abhinav Gupta. 2016. Actions Transformations. In CVPR. 2658--2667.","DOI":"10.1109\/CVPR.2016.291"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.320"},{"key":"e_1_3_2_1_56_1","volume-title":"Computer Vision -- ECCV","author":"Wang Zhenyang","year":"2016"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1162\/089976602317318938"},{"key":"e_1_3_2_1_58_1","first-page":"3","article-title":"Back to Basics: Unsupervised Learning of Optical Flow via Brightness Constancy and Motion Smoothness. In ECCV 2016 Workshops","volume":"3","author":"Yu Jason J.","year":"2016","journal-title":"Part"},{"key":"e_1_3_2_1_59_1","unstructured":"Sergey Zagoruyko and Nikos Komodakis. 2017. Paying More Attention to Attention: Improving the Performance of Convolutional Neural Networks via Attention Transfer. In ICLR. Sergey Zagoruyko and Nikos Komodakis. 2017. Paying More Attention to Attention: Improving the Performance of Convolutional Neural Networks via Attention Transfer. In ICLR."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"crossref","unstructured":"Tinghui Zhou Shubham Tulsiani Weilun Sun Jitendra Malik and Alexei A. Efros. 2016. View Synthesis by Appearance Flow. In ECCV. 286--301. Tinghui Zhou Shubham Tulsiani Weilun Sun Jitendra Malik and Alexei A. Efros. 2016. View Synthesis by Appearance Flow. In ECCV. 286--301.","DOI":"10.1007\/978-3-319-46493-0_18"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"Xizhou Zhu Jifeng Dai Lu Yuan and Yichen Wei. 2018. Towards High Performance Video Object Detection. In CVPR. 7210--7218. Xizhou Zhu Jifeng Dai Lu Yuan and Yichen Wei. 2018. Towards High Performance Video Object Detection. In CVPR. 7210--7218.","DOI":"10.1109\/CVPR.2018.00753"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"crossref","unstructured":"Xizhou Zhu Yujie Wang Jifeng Dai Lu Yuan and Yichen Wei. 2017. FlowGuided Feature Aggregation for Video Object Detection. In ICCV. 408--417. Xizhou Zhu Yujie Wang Jifeng Dai Lu Yuan and Yichen Wei. 2017. FlowGuided Feature Aggregation for Video Object Detection. In ICCV. 408--417.","DOI":"10.1109\/ICCV.2017.52"},{"key":"e_1_3_2_1_63_1","unstructured":"Xizhou Zhu Yuwen Xiong Jifeng Dai Lu Yuan and Yichen Wei. 2017. Deep Feature Flow for Video Recognition. In CVPR. 2349--2358. Xizhou Zhu Yuwen Xiong Jifeng Dai Lu Yuan and Yichen Wei. 2017. Deep Feature Flow for Video Recognition. In CVPR. 2349--2358."}],"event":{"name":"MM '18: ACM Multimedia Conference","location":"Seoul Republic of Korea","acronym":"MM '18","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 26th ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240693","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3240508.3240693","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T20:41:31Z","timestamp":1775248891000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240693"}},"subtitle":["Decoupling Detection and Feature Flow for Video Object Detectors"],"short-title":[],"issued":{"date-parts":[[2018,10,15]]},"references-count":63,"alternative-id":["10.1145\/3240508.3240693","10.1145\/3240508"],"URL":"https:\/\/doi.org\/10.1145\/3240508.3240693","relation":{},"subject":[],"published":{"date-parts":[[2018,10,15]]},"assertion":[{"value":"2018-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}