{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:23:54Z","timestamp":1740122634989,"version":"3.37.3"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2020,7,15]],"date-time":"2020-07-15T00:00:00Z","timestamp":1594771200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,7,15]],"date-time":"2020-07-15T00:00:00Z","timestamp":1594771200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61671266","61836004"],"award-info":[{"award-number":["61671266","61836004"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Tsinghua-Guoqiang research program","award":["2019GQG0006"],"award-info":[{"award-number":["2019GQG0006"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2020,12]]},"DOI":"10.1007\/s10489-020-01750-z","type":"journal-article","created":{"date-parts":[[2020,7,15]],"date-time":"2020-07-15T14:03:50Z","timestamp":1594821830000},"page":"4261-4280","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Cycle representation-disentangling network: learning to completely disentangle spatial-temporal features in video"],"prefix":"10.1007","volume":"50","author":[{"given":"Pengfei","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Su","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shangqi","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4813-2494","authenticated-orcid":false,"given":"Feng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,7,15]]},"reference":[{"key":"1750_CR1","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE international conference on computer vision, pp 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"1750_CR2","doi-asserted-by":"crossref","unstructured":"Majd M, Safabakhsh R (2019) A motion-aware convlstm network for action recognition. Appl Intell, pp 1\u20137","DOI":"10.1007\/s10489-018-1395-8"},{"key":"1750_CR3","doi-asserted-by":"crossref","unstructured":"Wang B, Ma L, Zhang W, Liu W (2018) Reconstruction network for video captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7622\u20137631","DOI":"10.1109\/CVPR.2018.00795"},{"key":"1750_CR4","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1016\/j.patcog.2017.03.021","volume":"75","author":"J Song","year":"2018","unstructured":"Song J, Gao L, Li L, Zhu X, Sebe N (2018) Quantization-based hashing: a general framework for scalable image and video retrieval. Pattern Recogn 75:175\u2013187","journal-title":"Pattern Recogn"},{"key":"1750_CR5","doi-asserted-by":"crossref","unstructured":"Wang J, Cherian A, Porikli F, Gould S (2018) Video representation learning using discriminative pooling. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1149\u20131158","DOI":"10.1109\/CVPR.2018.00126"},{"key":"1750_CR6","unstructured":"Simonyan K, Zisserman A (2014) Two-stream convolutional networks for action recognition in videos. arXiv:1406.2199"},{"key":"1750_CR7","doi-asserted-by":"crossref","unstructured":"Liang X, Lee L, Dai W, Xing E P (2017) Dual motion gan for future-flow embedded video prediction. In: proceedings of the IEEE international conference on computer vision, pp 1744\u2013 1752","DOI":"10.1109\/ICCV.2017.194"},{"key":"1750_CR8","unstructured":"Villegas R, Yang J, Hong S, Lin X, Lee H (2017) Decomposing motion and content for natural video sequence prediction. In: International conference on learning representations"},{"key":"1750_CR9","unstructured":"Hsieh J-T, Liu B, Huang D-A, Fei-Fei LF, Niebles JC (2018) Learning to decompose and disentangle representations for video prediction. In: Advances in neural information processing systems, pp 517\u2013526"},{"key":"1750_CR10","unstructured":"Fraccaro M, Kamronn S, Paquet U, Winther O (2017) A disentangled recognition and nonlinear dynamics model for unsupervised learning. In: Advances in neural information processing systems, pp 3601\u20133610"},{"key":"1750_CR11","unstructured":"Li Y, Mandt S (2018) Disentangled sequential autoencoder"},{"key":"1750_CR12","doi-asserted-by":"crossref","unstructured":"Wang L, Xiong Y, Wang Z, Qiao Y u, Lin D, Tang X, Luc Van Gool. (2016) Temporal segment networks: towards good practices for deep action recognition. In: European conference on computer vision, pp 20\u201336","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1750_CR13","doi-asserted-by":"crossref","unstructured":"Ji S, Xu W, Yang M, Yu K (2010) 3d convolutional neural networks for human action recognition, vol 35, pp 221\u2013231","DOI":"10.1109\/TPAMI.2012.59"},{"key":"1750_CR14","doi-asserted-by":"crossref","unstructured":"P\u00e9rez-Hern\u00e1ndez F, Tabik S, Lamas A C, Olmos R, Fujita H, Herrera F (2020) Object detection binary classifiers methodology based on deep learning to identify small objects handled similarly: application in video surveillance. Knowledge Based Systems, pp 105590","DOI":"10.1016\/j.knosys.2020.105590"},{"key":"1750_CR15","doi-asserted-by":"crossref","unstructured":"Zhou B, Andonian A, Torralba A (2017) Temporal relational reasoning in videos. In: European conference on computer vision","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"1750_CR16","doi-asserted-by":"crossref","unstructured":"Ilg E, Mayer N, Saikia T, Keuper M, Dosovitskiy A, Brox T (2016) Flownet 2.0: evolution of optical flow estimation with deep networks. In: 2017 IEEE conference on computer vision and pattern recognition, pp 1647\u20131655","DOI":"10.1109\/CVPR.2017.179"},{"issue":"1412","key":"1750_CR17","doi-asserted-by":"publisher","first-page":"2315","DOI":"10.1098\/rspb.1998.0577","volume":"265","author":"JH van Hateren","year":"1998","unstructured":"van Hateren JH, Ruderman DL (1998) Independent component analysis of natural image sequences yields spatio-temporal filters similar to simple cells in primary visual cortex. Proceedings of the Royal Society of London. Series B: Biological Sciences 265(1412):2315\u20132320","journal-title":"Proceedings of the Royal Society of London. Series B: Biological Sciences"},{"issue":"3","key":"1750_CR18","doi-asserted-by":"publisher","first-page":"663","DOI":"10.1162\/089976603321192121","volume":"15","author":"J Hurri","year":"2003","unstructured":"Hurri J, Hyv\u00e4rinen A (2003) Simple-cell-like receptive fields maximize temporal coherence in natural video. Neural Comput 15(3):663\u2013691","journal-title":"Neural Comput"},{"key":"1750_CR19","doi-asserted-by":"crossref","unstructured":"Yin Z, Shi J (2018) Geonet: unsupervised learning of dense depth, optical flow and camera pose. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1983\u20131992","DOI":"10.1109\/CVPR.2018.00212"},{"key":"1750_CR20","unstructured":"Lotter W, Kreiman G, Cox D (2016) Deep predictive coding networks for video prediction and unsupervised learning. arXiv:1605.08104"},{"key":"1750_CR21","unstructured":"Radford A, Metz L, Chintala S (2016) Unsupervised representation learning with deep convolutional generative adversarial networks. In: International conference on learning representations"},{"key":"1750_CR22","doi-asserted-by":"crossref","unstructured":"Saito M, Matsumoto E, Saito S (2017) Temporal generative adversarial nets with singular value clipping. In: Proceedings of the IEEE international conference on computer vision, pp 2830\u20132839","DOI":"10.1109\/ICCV.2017.308"},{"key":"1750_CR23","unstructured":"Wang Y, Gao Z, Long M, Wang J, Philip Y. (2018) Predrnn++: towards a resolution of the deep-in-time dilemma in spatiotemporal predictive learning. In: Thirty-fifth international conference on machine learning, pp 5110\u20135119"},{"key":"1750_CR24","unstructured":"Wang Y, Jiang L, Yang M-H, Li L-J, Long M, Fei-Fei L (2019) Eidetic 3d lstm: a model for video prediction and beyond. In: International conference on learning representations"},{"key":"1750_CR25","doi-asserted-by":"crossref","unstructured":"Wang Y, Zhang J, Zhu H, Long M, Wang J, Yu PS (2019) Memory in memory: a predictive neural network for learning higher-order non-stationarity from spatiotemporal dynamics. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 9154\u20139162","DOI":"10.1109\/CVPR.2019.00937"},{"key":"1750_CR26","unstructured":"Kingma DP, Welling M (2014) Auto-encoding variational bayes"},{"key":"1750_CR27","unstructured":"Patraucean V, Handa A, Cipolla R (2015) Spatio-temporal video autoencoder with differentiable memory. arXiv:1803.02991"},{"key":"1750_CR28","doi-asserted-by":"crossref","unstructured":"Walker J, Doersch C, Gupta A, Hebert M (2016) An uncertain future: forecasting from static images using variational autoencoders. In: European conference on computer vision. Springer, pp 835\u2013851","DOI":"10.1007\/978-3-319-46478-7_51"},{"key":"1750_CR29","unstructured":"Denton E, Fergus R (2018) Stochastic video generation with a learned prior. In: International conference on machine learning, pp 1174\u20131183"},{"key":"1750_CR30","doi-asserted-by":"crossref","unstructured":"Sevilla-Lara L, Liao Y, G\u00fcney F, Jampani V, Geiger A, Black MJ (2018) On the integration of optical flow and action recognition. In: German conference on pattern recognition. Springer, pp 281\u2013297","DOI":"10.1007\/978-3-030-12939-2_20"},{"issue":"4","key":"1750_CR31","doi-asserted-by":"publisher","first-page":"715","DOI":"10.1162\/089976602317318938","volume":"14","author":"L Wiskott","year":"2002","unstructured":"Wiskott L, Sejnowski TJ (2002) Slow feature analysis: unsupervised learning of invariances. Neural Computation 14(4):715\u2013770","journal-title":"Neural Computation"},{"key":"1750_CR32","unstructured":"Kulkarni TD, Whitney WF, Kohli P, Tenenbaum J (2015) Deep convolutional inverse graphics network. In: Advances in neural information processing systems, pp 2539\u20132547"},{"key":"1750_CR33","unstructured":"Whitney WF, Chang M, Kulkarni T, Tenenbaum JB (2016) Understanding visual concepts with continuation learning. arXiv:1602.06822"},{"key":"1750_CR34","unstructured":"Denton EL, et al. (2017) Unsupervised learning of disentangled representations from video. In: Advances in neural information processing systems, pp 4414\u20134423"},{"key":"1750_CR35","doi-asserted-by":"crossref","unstructured":"Zhu J-Y, Park T, Isola P, Efros AA (2017) Unpaired image-to-image translation using cycle-consistent adversarial networks. In: Proceedings of the IEEE international conference on computer vision, pp 2223\u20132232","DOI":"10.1109\/ICCV.2017.244"},{"key":"1750_CR36","unstructured":"Karras T, Aila T, Laine S, Lehtinen J (2017) Progressive growing of gans for improved quality, stability, and variation. In: International conference on learning representations"},{"key":"1750_CR37","doi-asserted-by":"crossref","unstructured":"Yi Z, Zhang H, Tan P, Gong M (2017) Dualgan: unsupervised dual learning for image-to-image translation. In: Proceedings of the IEEE international conference on computer vision, pp 2849\u2013 2857","DOI":"10.1109\/ICCV.2017.310"},{"key":"1750_CR38","doi-asserted-by":"crossref","unstructured":"Isola P, Zhu J-Y, Zhou T, Efros AA (2016) Image-to-image translation with conditional adversarial networks. In: IEEE conference on computer vision and pattern recognition, pp 5967\u20135976","DOI":"10.1109\/CVPR.2017.632"},{"key":"1750_CR39","unstructured":"Royer A, Bousmalis K, Gouws S, Bertsch F, Mosseri I, Cole F, Murphy K (2017) Xgan: unsupervised image-to-image translation for many-to-many mappings. arXiv:1711.05139"},{"key":"1750_CR40","unstructured":"Sharma P, Mohan L, Pinto L, Gupta A (2018) Multiple interactions made easy (mime): large scale demonstrations data for imitation. In: Conference on robot learning"},{"key":"1750_CR41","unstructured":"Pont-Tuset J, Perazzi F, Caelles S, Arbel\u00e1ez P, Sorkine-Hornung A, van Gool L (2017) The 2017 davis challenge on video object segmentation. arXiv:1704.00675"},{"key":"1750_CR42","unstructured":"Srivastava N, Mansimov E, Salakhudinov R (2015) Unsupervised learning of video representations using lstms. In: International conference on machine learning, pp 843\u2013852"},{"key":"1750_CR43","doi-asserted-by":"crossref","unstructured":"Oliu M, Selva J, Escalera S (2018) Folded recurrent neural networks for future video prediction. In: Proceedings of the European conference on computer vision, pp 745\u2013761","DOI":"10.1007\/978-3-030-01264-9_44"},{"key":"1750_CR44","doi-asserted-by":"crossref","unstructured":"Liu YX, Gupta A, Abbeel P, Levine S (2018) Imitation from observation: learning to imitate behaviors from raw video via context translation. In: 2018 IEEE international conference on robotics and automation (ICRA). IEEE, pp 1118\u20131125","DOI":"10.1109\/ICRA.2018.8462901"},{"key":"1750_CR45","unstructured":"Ho J, Ermon S (2016) Generative adversarial imitation learning. In: Advances in neural information processing systems, pp 4565\u20134573"},{"key":"1750_CR46","unstructured":"Berndt DJ, Clifford J (1994) Using dynamic time warping to find patterns in time series. In: KDD workshop , vol 10, pp 359\u2013 370"},{"key":"1750_CR47","doi-asserted-by":"crossref","unstructured":"Isola P, Zhu J-Y, Zhou T, Efros A A (2017) Image-to-image translation with conditional adversarial networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1125\u20131134","DOI":"10.1109\/CVPR.2017.632"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-020-01750-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-020-01750-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-020-01750-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,7,15]],"date-time":"2021-07-15T00:06:40Z","timestamp":1626307600000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-020-01750-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,7,15]]},"references-count":47,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2020,12]]}},"alternative-id":["1750"],"URL":"https:\/\/doi.org\/10.1007\/s10489-020-01750-z","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2020,7,15]]},"assertion":[{"value":"15 July 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with Ethical Standards"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of interests"}}]}}