{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T22:14:25Z","timestamp":1781648065838,"version":"3.54.5"},"reference-count":106,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2019,10,23]],"date-time":"2019-10-23T00:00:00Z","timestamp":1571788800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,10,23]],"date-time":"2019-10-23T00:00:00Z","timestamp":1571788800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1007\/s11263-019-01222-z","type":"journal-article","created":{"date-parts":[[2019,10,23]],"date-time":"2019-10-23T16:32:44Z","timestamp":1571848364000},"page":"1505-1536","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Generating Human Action Videos by Coupling 3D Game Engines and Probabilistic Graphical Models"],"prefix":"10.1007","volume":"128","author":[{"given":"C\u00e9sar Roberto","family":"de Souza","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Adrien","family":"Gaidon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yohann","family":"Cabon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Naila","family":"Murray","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Antonio Manuel","family":"L\u00f3pez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,10,23]]},"reference":[{"issue":"11","key":"1222_CR1","doi-asserted-by":"publisher","first-page":"1949","DOI":"10.1109\/TMM.2015.2477680","volume":"17","author":"AH Abdulnabi","year":"2015","unstructured":"Abdulnabi, A. H., Wang, G., Lu, J., & Jia, K. (2015). Multi-task cnn model for attribute prediction. IEEE Transactions on Multimedia, 17(11), 1949\u20131959.","journal-title":"IEEE Transactions on Multimedia"},{"issue":"16","key":"1222_CR2","first-page":"1781","volume":"41","author":"JML Asensio","year":"2014","unstructured":"Asensio, J. M. L., Peralta, J., Arrabales, R., Bedia, M. G., Cortez, P., & L\u00f3pez, A. (2014). Artificial intelligence approaches for the generation and assessment of believable human-like behaviour in virtual characters. Expert Systems With Applications, 41(16), 1781\u20137290.","journal-title":"Expert Systems With Applications"},{"key":"1222_CR3","doi-asserted-by":"crossref","unstructured":"Aubry, M., & Russell, B. (2015). Understanding deep features with computer-generated imagery. In ICCV.","DOI":"10.1109\/ICCV.2015.329"},{"key":"1222_CR4","volume-title":"Pattern recognition and machine learning","author":"CM Bishop","year":"2006","unstructured":"Bishop, C. M. (2006). Pattern recognition and machine learning. Berlin: Springer."},{"issue":"20","key":"1222_CR5","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1016\/j.patrec.2008.04.005","volume":"30","author":"G Brostow","year":"2009","unstructured":"Brostow, G., Fauqueur, J., & Cipolla, R. (2009). Semantic object classes in video: A high-definition ground truth database. Pattern Recognition Letters, 30(20), 88\u201397.","journal-title":"Pattern Recognition Letters"},{"key":"1222_CR6","doi-asserted-by":"crossref","unstructured":"Butler, D., Wulff, J., Stanley, G., & Black, M. (2012). A naturalistic open source movie for optical flow evaluation. In ECCV.","DOI":"10.1007\/978-3-642-33783-3_44"},{"key":"1222_CR7","unstructured":"Carnegie Mellon Graphics Lab. (2016). Carnegie Mellon University motion capture database."},{"key":"1222_CR8","doi-asserted-by":"crossref","unstructured":"Carreira, J., & Zisserman, A. (2017). Quo vadis, action recognition? A new model and the Kinetics dataset. In CVPR.","DOI":"10.1109\/CVPR.2017.502"},{"key":"1222_CR9","volume-title":"Computer graphics: principles and practice","author":"MP Carter","year":"1997","unstructured":"Carter, M. P. (1997). Computer graphics: principles and practice (Vol. 22). Boston: Addison-Wesley Professional."},{"key":"1222_CR10","doi-asserted-by":"crossref","unstructured":"Chen, C., Seff, A., Kornhauser, A., & Xiao, J. (2015). DeepDriving: Learning affordance for direct perception in autonomous driving. In ICCV.","DOI":"10.1109\/ICCV.2015.312"},{"issue":"4","key":"1222_CR11","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2018","unstructured":"Chen, L. C., Papandreou, G., Kokkinos, I., Murphy, K., & Yuille, A. L. (2018). Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs. T-PAMI, 40(4), 834\u2013848.","journal-title":"T-PAMI"},{"key":"1222_CR12","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., et al. (2016). The cityscapes dataset for semantic urban scene understanding. In CVPR.","DOI":"10.1109\/CVPR.2016.350"},{"key":"1222_CR13","unstructured":"De Souza, C. R. (2014). The Accord.NET framework, a framework for scientific computing in .NET. http:\/\/accord-framework.net ."},{"key":"1222_CR14","doi-asserted-by":"crossref","unstructured":"De Souza, C. R., Gaidon, A., Vig, E., & L\u00f3pez, A. M. (2016). Sympathy for the details: Dense trajectories and hybrid classification architectures for action recognition. In ECCV.","DOI":"10.1007\/978-3-319-46478-7_43"},{"key":"1222_CR15","doi-asserted-by":"crossref","unstructured":"De Souza, C. R., Gaidon, A., Cabon, Y., & L\u00f3pez, A. M. (2017). Procedural generation of videos to train deep action recognition networks. In CVPR.","DOI":"10.1109\/CVPR.2017.278"},{"key":"1222_CR16","unstructured":"Dosovitskiy, A., Ros, G., Codevilla, F., Lopez, A., & Koltun, V. (2017). CARLA: An open urban driving simulator. In Proceedings of the 1st annual conference on robot learning."},{"key":"1222_CR17","unstructured":"Egges, A., Kamphuis, A., & Overmars, M. (Eds.). (2008). Motion in Games: First International Workshop, MIG 2008, Utrecht, The Netherlands, June 14\u201317, 2008, Revised Papers (Vol. 5277). Springer."},{"key":"1222_CR18","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., & Zisserman, A. (2016). Convolutional two-stream network fusion for video action recognition. In CVPR.","DOI":"10.1109\/CVPR.2016.213"},{"key":"1222_CR19","doi-asserted-by":"crossref","unstructured":"Fernando, B., Gavves, E., Oramas, M. J., Ghodrati, A., & Tuytelaars, T. (2015). Modeling video evolution for action recognition. In CVPR.","DOI":"10.1109\/CVPR.2015.7299176"},{"issue":"11","key":"1222_CR20","doi-asserted-by":"publisher","first-page":"2782","DOI":"10.1109\/TPAMI.2013.65","volume":"35","author":"A Gaidon","year":"2013","unstructured":"Gaidon, A., Harchaoui, Z., & Schmid, C. (2013). Temporal localization of actions with actoms. T-PAMI, 35(11), 2782\u20132795.","journal-title":"T-PAMI"},{"key":"1222_CR21","unstructured":"Gaidon, A., Wang, Q., Cabon, Y., & Vig, E. (2016). Virtual worlds as proxy for multi-object tracking analysis. In CVPR."},{"key":"1222_CR22","doi-asserted-by":"crossref","unstructured":"Galvane, Q., Christie, M., Lino, C., & Ronfard, R. (2015). Camera-on-rails: Automated computation of constrained camera paths. In SIGGRAPH.","DOI":"10.1145\/2822013.2822025"},{"key":"1222_CR23","doi-asserted-by":"crossref","unstructured":"Gatys, L. A., Ecker, A. S., & Bethge, M. (2016). Image style transfer using convolutional neural networks. In CVPR.","DOI":"10.1109\/CVPR.2016.265"},{"key":"1222_CR24","doi-asserted-by":"crossref","unstructured":"Gu, C., Sun, C., Ross, D., Vondrick, C., Pantofaru, C., Li, Y., et al. (2018). Ava: A video dataset of spatio-temporally localized atomic visual actions. In CVPR.","DOI":"10.1109\/CVPR.2018.00633"},{"key":"1222_CR25","unstructured":"Guay, M., Ronfard, R., Gleicher, M., Cani, M. P. (2015a). Adding dynamics to sketch-based character animations. In Sketch-based interfaces and modeling."},{"issue":"4","key":"1222_CR26","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1145\/2766893","volume":"34","author":"M Guay","year":"2015","unstructured":"Guay, M., Ronfard, R., Gleicher, M., & Cani, M. P. (2015b). Space-time sketching of character animation. ACM Transactions on Graphics, 34(4), 118.","journal-title":"ACM Transactions on Graphics"},{"key":"1222_CR27","doi-asserted-by":"crossref","unstructured":"Haeusler, R., & Kondermann, D. (2013). Synthesizing real world stereo challenges. In German conference on pattern recognition","DOI":"10.1007\/978-3-642-40602-7_17"},{"key":"1222_CR28","doi-asserted-by":"crossref","unstructured":"Haltakov, V., Unger, C., & Ilic, S. (2013). Framework for generation of synthetic ground truth data for driver assistance applications. In German conference on pattern recognition.","DOI":"10.1007\/978-3-642-40602-7_35"},{"key":"1222_CR29","unstructured":"Handa, A., Patraucean, V., Badrinarayanan, V., Stent, S., & Cipolla, R. (2015). SynthCam3D: Semantic understanding with synthetic indoor scenes. CoRR. arXiv:1505.00171 ."},{"key":"1222_CR30","unstructured":"Handa, A., Patraucean, V., Badrinarayanan, V., Stent, S., & Cipolla, R. (2016). Understanding real world indoor scenes with synthetic data. In CVPR."},{"key":"1222_CR31","doi-asserted-by":"crossref","unstructured":"Hao, Z., Huang, X., & Belongie, S. (2018). Controllable video generation with sparse trajectories. In CVPR.","DOI":"10.1109\/CVPR.2018.00819"},{"key":"1222_CR32","doi-asserted-by":"crossref","unstructured":"Hattori, H., Boddeti, V. N., Kitani, K. M., & Kanade, T. (2015) Learning scene-specific pedestrian detectors without real data. In CVPR.","DOI":"10.1109\/CVPR.2015.7299006"},{"key":"1222_CR33","unstructured":"Ioffe, S., & Szegedy, C. (2015). Batch normalization: Accelerating deep network training by reducing internal covariate shift. In ICML (Vol. 37)."},{"key":"1222_CR34","doi-asserted-by":"crossref","unstructured":"Jhuang, H., Gall, J., Zuffi, S., Schmid, C., & Black, M. J. (2013). Towards understanding action recognition. In ICCV.","DOI":"10.1109\/ICCV.2013.396"},{"key":"1222_CR35","unstructured":"Jiang, Y. G., Liu, J., Roshan Zamir, A., Laptev, I., Piccardi, M., Shah, M., & Sukthankar, R. (2013). THUMOS challenge: Action recognition with a large number of classes."},{"key":"1222_CR36","doi-asserted-by":"crossref","unstructured":"Kaneva, B., Torralba, A., & Freeman, W. (2011). Evaluation of image features using a photorealistic virtual world. In ICCV.","DOI":"10.1109\/ICCV.2011.6126508"},{"key":"1222_CR37","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Toderici, G., Shetty, S., Leung, T., Sukthankar, R., & Fei-Fei, L. (2014). Large-scale video classification with convolutional neural networks. In CVPR.","DOI":"10.1109\/CVPR.2014.223"},{"key":"1222_CR38","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H. H., Garrote-Contreras, E., Poggio, T., & Serre, T. (2011). HMDB: A large video database for human motion recognition. In ICCV.","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"1222_CR39","unstructured":"Lan, Z., Lin, M., Li, X., Hauptmann, A. G., & Raj, B. (2015). Beyond Gaussian pyramid: Multi-skip feature stacking for action recognition. In CVPR."},{"issue":"6","key":"1222_CR40","doi-asserted-by":"publisher","first-page":"649","DOI":"10.1068\/p3060","volume":"29","author":"MS Langer","year":"2000","unstructured":"Langer, M. S., & B\u00fclthoff, H. H. (2000). Depth discrimination from shading under diffuse lighting. Perception, 29(6), 649\u2013660.","journal-title":"Perception"},{"key":"1222_CR41","unstructured":"Lerer, A., Gross, S., & Fergus, R. (2016). Learning physical intuition of block towers by example. In Proceedings of machine learning research (Vol. 48)."},{"key":"1222_CR42","doi-asserted-by":"crossref","unstructured":"Li, Y., Min, M. R., Shen, D., Carlson, D. E., & Carin, L. (2018). Video generation from text. In AAAI.","DOI":"10.1609\/aaai.v32i1.12233"},{"key":"1222_CR43","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., et al. (2014). Microsoft COCO: Common objects in context. In ECCV.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1222_CR44","doi-asserted-by":"crossref","unstructured":"Mar\u00edn, J., V\u00e1zquez, D., Ger\u00f3nimo, D., & L\u00f3pez, A. M. (2010). Learning appearance in virtual scenarios for pedestrian detection. In CVPR.","DOI":"10.1109\/CVPR.2010.5540218"},{"key":"1222_CR45","doi-asserted-by":"crossref","unstructured":"Marwah, T., Mittal, G., & Balasubramanian, V. N. (2017). Attentive semantic video generation using captions. In ICCV.","DOI":"10.1109\/ICCV.2017.159"},{"key":"1222_CR46","doi-asserted-by":"crossref","unstructured":"Massa, F., Russell, B., & Aubry, M. (2016). Deep exemplar 2D\u20133D detection by adapting from real to rendered views. In CVPR.","DOI":"10.1109\/CVPR.2016.648"},{"key":"1222_CR47","doi-asserted-by":"crossref","unstructured":"Matikainen, P., Sukthankar, R., & Hebert, M. (2011). Feature seeding for action recognition. In ICCV.","DOI":"10.1109\/ICCV.2011.6126435"},{"key":"1222_CR48","doi-asserted-by":"crossref","unstructured":"Mayer, N., Ilg, E., Hausser, P., Fischer, P., Cremers, D., Dosovitskiy, A., & Brox, T. (2016). A large dataset to train convolutional networks for disparity, optical flow, and scene flow estimation. In CVPR.","DOI":"10.1109\/CVPR.2016.438"},{"key":"1222_CR49","unstructured":"Meister, S., & Kondermann, D. (2011). Real versus realistically rendered scenes for optical flow evaluation. In CEMT."},{"key":"1222_CR50","doi-asserted-by":"crossref","unstructured":"Miller, G. (1994). Efficient algorithms for local and global accessibility shading. In SIGGRAPH.","DOI":"10.1145\/192161.192244"},{"key":"1222_CR51","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Graves, A., Antonoglou, I., Wierstra, D., et al. (2013). Playing Atari with deep reinforcement learning. In NIPS workshops."},{"key":"1222_CR52","unstructured":"Molnar, S. (1991). Efficient supersampling antialiasing for high-performance architectures. Technical report, North Carolina University at Chapel Hill."},{"key":"1222_CR53","doi-asserted-by":"publisher","first-page":"126","DOI":"10.1016\/j.cviu.2017.06.012","volume":"163","author":"F Nian","year":"2017","unstructured":"Nian, F., Li, T., Wang, Y., Wu, X., Ni, B., & Xu, C. (2017). Learning explicit video attributes from mid-level representation for video captioning. Computer Vision and Image Understanding, 163, 126\u2013138.","journal-title":"Computer Vision and Image Understanding"},{"issue":"9","key":"1222_CR54","doi-asserted-by":"publisher","first-page":"3121","DOI":"10.1007\/s11042-013-1771-7","volume":"74","author":"N Onkarappa","year":"2015","unstructured":"Onkarappa, N., & Sappa, A. (2015). Synthetic sequences and ground-truth flow field generation for algorithm validation. Multimedia Tools and Applications, 74(9), 3121\u20133135.","journal-title":"Multimedia Tools and Applications"},{"key":"1222_CR55","doi-asserted-by":"crossref","unstructured":"Papon, J., & Schoeler, M. (2015). Semantic pose using deep networks trained on synthetic RGB-D. In ICCV.","DOI":"10.1109\/ICCV.2015.95"},{"key":"1222_CR56","doi-asserted-by":"crossref","unstructured":"Peng, X., Zou, C., Qiao, Y., & Peng, Q. (2014). Action recognition with stacked fisher vectors. In ECCV.","DOI":"10.1007\/978-3-319-10602-1_38"},{"key":"1222_CR57","doi-asserted-by":"crossref","unstructured":"Peng, X., Sun, B., Ali, K., & Saenko, K. (2015). Learning deep object detectors from 3D models. In ICCV.","DOI":"10.1109\/ICCV.2015.151"},{"issue":"1","key":"1222_CR58","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1109\/2945.468392","volume":"1","author":"K Perlin","year":"1995","unstructured":"Perlin, K. (1995). Real time responsive animation with personality. IEEE Transactions on Visualization and Computer Graphics, 1(1), 5\u201315.","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"key":"1222_CR59","doi-asserted-by":"crossref","unstructured":"Perlin, K., & Seidman, G. (2008). Autonomous digital actors. In Motion in games.","DOI":"10.1007\/978-3-540-89220-5_24"},{"key":"1222_CR60","doi-asserted-by":"crossref","unstructured":"Richter, S., Vineet, V., Roth, S., & Vladlen, K. (2016). Playing for data: Ground truth from computer games. In ECCV.","DOI":"10.1007\/978-3-319-46475-6_7"},{"key":"1222_CR61","doi-asserted-by":"crossref","unstructured":"Ritschel, T., Grosch, T., & Seidel, H. P. (2009). Approximating dynamic global illumination in image space. In Proceedings of the 2009 symposium on interactive 3D graphics and games\u2014I3D \u201909.","DOI":"10.1145\/1507149.1507161"},{"key":"1222_CR62","doi-asserted-by":"crossref","unstructured":"Ros, G., Sellart, L., Materzyska, J., V\u00e1zquez, D., & L\u00f3pez, A. (2016). The SYNTHIA dataset: A large collection of synthetic images for semantic segmentation of urban scenes. In CVPR.","DOI":"10.1109\/CVPR.2016.352"},{"key":"1222_CR63","doi-asserted-by":"crossref","unstructured":"Saito, M., Matsumoto, E., & Saito, S. (2017). Temporal generative adversarial nets with singular value clipping. In ICCV.","DOI":"10.1109\/ICCV.2017.308"},{"key":"1222_CR64","doi-asserted-by":"crossref","unstructured":"Selan, J. (2012). Cinematic color. In SIGGRAPH.","DOI":"10.1145\/2343483.2343492"},{"key":"1222_CR65","doi-asserted-by":"crossref","unstructured":"Shafaei, A., Little, J., & Schmidt, M. (2016). Play and learn: Using video games to train computer vision models. In BMVC.","DOI":"10.5244\/C.30.26"},{"key":"1222_CR66","doi-asserted-by":"crossref","unstructured":"Shotton, J., Fitzgibbon, A., Cook, M., Sharp, T., Finocchio, M., Moore, R., et al. (2011). Real-time human pose recognition in parts from a single depth image. In CVPR.","DOI":"10.1109\/CVPR.2011.5995316"},{"key":"1222_CR67","unstructured":"Simonyan, K., & Zisserman, A. (2014). Two-stream convolutional networks for action recognition in videos. In NIPS."},{"key":"1222_CR68","doi-asserted-by":"crossref","unstructured":"Sizikova1, E., Singh, V. K., Georgescu, B., Halber, M., Ma, K., & Chen, T. (2016). Enhancing place recognition using joint intensity\u2014depth analysis and synthetic data. In ECCV workshops.","DOI":"10.1007\/978-3-319-49409-8_74"},{"key":"1222_CR69","unstructured":"Soomro, K., Zamir, A. R., & Shah, M. (2012). UCF101: A dataset of 101 human actions classes from videos in the wild. CoRR. arXiv:1212.0402 ."},{"key":"1222_CR70","unstructured":"Sousa, T., Kasyan, N., & Schulz, N. (2011). Secrets of cryengine 3 graphics technology. In SIGGRAPH."},{"key":"1222_CR71","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: A simple way to prevent neural networks from overfitting. Journal on Machine Learning Research, 15, 1929\u20131958.","journal-title":"Journal on Machine Learning Research"},{"key":"1222_CR72","unstructured":"Steiner, B. (2011). Post processing effects. Institute of Graphics and Algorithms, Vienna University of Technology, Bachelour\u2019s thesis."},{"key":"1222_CR73","doi-asserted-by":"crossref","unstructured":"Su, H., Qi, C., Yi, Y., & Guibas, L. (2015a). Render for CNN: Viewpoint estimation in images using CNNs trained with rendered 3D model views. In ICCV.","DOI":"10.1109\/ICCV.2015.308"},{"key":"1222_CR74","doi-asserted-by":"crossref","unstructured":"Su, H., Wang, F., Yi, Y., & Guibas, L. (2015b). 3D-assisted feature synthesis for novel views of an object. In ICCV.","DOI":"10.1109\/ICCV.2015.307"},{"key":"1222_CR75","doi-asserted-by":"crossref","unstructured":"Sun, S., Kuang, Z., Sheng, L., Ouyang, W., & Zhang, W. (2018). Optical flow guided feature: A fast and robust motion representation for video action recognition. In CVPR.","DOI":"10.1109\/CVPR.2018.00151"},{"key":"1222_CR76","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., et al. (2015). Going deeper with convolutions. In CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1222_CR77","doi-asserted-by":"crossref","unstructured":"Taylor, G., Chosak, A., & Brewer, P. (2007). OVVV: Using virtual worlds to design and evaluate surveillance systems. In CVPR.","DOI":"10.1109\/CVPR.2007.383518"},{"key":"1222_CR78","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., & Paluri, M. (2015). Learning spatiotemporal features with 3D convolutional networks. In CVPR.","DOI":"10.1109\/ICCV.2015.510"},{"key":"1222_CR79","doi-asserted-by":"crossref","unstructured":"Tulyakov, S., Liu, M. Y., Yang, X., & Kautz, J. (2018). MoCoGAN: Decomposing motion and content for video generation. In CVPR.","DOI":"10.1109\/CVPR.2018.00165"},{"key":"1222_CR80","unstructured":"V\u00e1zquez, D., L\u00f3pez, A., Ponsa, D., & Mar\u00edn, J. (2011). Cool world: Domain adaptation of virtual and real worlds for human detection using active learning. In NIPS workshops."},{"issue":"4","key":"1222_CR81","doi-asserted-by":"publisher","first-page":"797","DOI":"10.1109\/TPAMI.2013.163","volume":"36","author":"D Vazquez","year":"2014","unstructured":"Vazquez, D., L\u00f3pez, A. M., Mar\u00edn, J., Ponsa, D., & Ger\u00f3nimo, D. (2014). Virtual and real world adaptation for pedestrian detection. T-PAMI, 36(4), 797\u2013809.","journal-title":"T-PAMI"},{"key":"1222_CR82","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lin, X., Batra, T., Zitnick, C., & Parikh, D. (2015). Learning common sense through visual abstraction. In ICCV.","DOI":"10.1109\/ICCV.2015.292"},{"key":"1222_CR83","unstructured":"Veeravasarapu, V., Hota, R., Rothkopf, C., & Visvanathan, R. (2015). Simulations for validation of vision systems. CoRR. arXiv:1512.01030 ."},{"key":"1222_CR84","unstructured":"Veeravasarapu, V., Rothkopf, C., & Visvanathan, R. (2016). Model-driven simulations for deep convolutional neural networks. CoRR. arXiv:1605.09582 ."},{"key":"1222_CR85","unstructured":"Vondrick, C., Pirsiavash, H., & Torralba, A. (2016). Generating videos with scene dynamics. In NIPS."},{"key":"1222_CR86","doi-asserted-by":"crossref","unstructured":"Wang, H., & Schmid, C. (2013). Action recognition with improved trajectories. In ICCV.","DOI":"10.1109\/ICCV.2013.441"},{"key":"1222_CR87","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1007\/s11263-012-0594-8","volume":"103","author":"H Wang","year":"2013","unstructured":"Wang, H., Kl\u00e4ser, A., Schmid, C., & Liu, C. L. (2013). Dense trajectories and motion boundary descriptors for action recognition. IJCV, 103, 60\u201379.","journal-title":"IJCV"},{"issue":"3","key":"1222_CR88","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1007\/s11263-015-0846-5","volume":"119","author":"H Wang","year":"2016","unstructured":"Wang, H., Oneata, D., Verbeek, J., & Schmid, C. (2016a). A robust and efficient video representation for action recognition. IJCV, 119(3), 219\u2013238.","journal-title":"IJCV"},{"key":"1222_CR89","doi-asserted-by":"crossref","unstructured":"Wang, L., Qiao, Y., & Tang, X. (2015). Action recognition with trajectory-pooled deep-convolutional descriptors. In CVPR.","DOI":"10.1109\/CVPR.2015.7299059"},{"key":"1222_CR90","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., & van Gool, L. (2016b). Temporal segment networks: Towards good practices for deep action recognition. In ECCV.","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1222_CR91","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., et al. (2017). Temporal segment networks for action recognition in videos. CoRR. arXiv:1705.02953 ."},{"key":"1222_CR92","doi-asserted-by":"crossref","unstructured":"Wang, X., Farhadi, A., & Gupta, A. (2016c). Actions $$\\sim $$ Transformations. In CVPR.","DOI":"10.1109\/CVPR.2016.291"},{"key":"1222_CR93","unstructured":"van Welbergen, H., van Basten, B. J. H., Egges, A., Ruttkay, Z. M., & Overmars, M. H. (2009). Real time character animation: A trade-off between naturalness and control. In Proceedings of the Eurographics."},{"key":"1222_CR94","doi-asserted-by":"crossref","unstructured":"Wu, W., Zhang, Y., Li, C., Qian, C., & Loy, C. C. (2018). Reenactgan: Learning to reenact faces via boundary transfer. In ECCV.","DOI":"10.1007\/978-3-030-01246-5_37"},{"key":"1222_CR95","doi-asserted-by":"crossref","unstructured":"Xiong, W., Luo, W., Ma, L., Liu, W., & Luo, J. (2018). Learning to generate time-lapse videos using multi-stage dynamic generative adversarial networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00251"},{"issue":"5","key":"1222_CR96","first-page":"2121","volume":"15","author":"J Xu","year":"2014","unstructured":"Xu, J., V\u00e1zquez, D., L\u00f3pez, A., Mar\u00edn, J., & Ponsa, D. (2014). Learning a part-based pedestrian detector in a virtual world. T-ITS, 15(5), 2121\u20132131.","journal-title":"T-ITS"},{"key":"1222_CR97","doi-asserted-by":"crossref","unstructured":"Yan, X., Rastogi, A., Villegas, R., Sunkavalli, K., Shechtman, E., Hadap, S., et al. (2018). MT-VAE: Learning motion transformations to generate multimodal human dynamics. In ECCV (Vol. 11209).","DOI":"10.1007\/978-3-030-01228-1_17"},{"key":"1222_CR98","doi-asserted-by":"crossref","unstructured":"Yan, Y., Xu, J., Ni, B., Zhang, W., & Yang, X. (2017). Skeleton-aided articulated motion generation. In ACM-MM.","DOI":"10.1145\/3123266.3123277"},{"key":"1222_CR99","doi-asserted-by":"crossref","unstructured":"Yang, C., Wang, Z., Zhu, X., Huang, C., Shi, J., & Lin, D. (2018). Pose guided human video generation. In ECCV (Vol. 11214).","DOI":"10.1007\/978-3-030-01249-6_13"},{"key":"1222_CR100","unstructured":"Zach, C., Pock, T., & Bischof, H. (2007). A duality based approach for realtime TV-L1 optical flow. In Proceedings of the 29th DAGM conference on pattern recognition."},{"key":"1222_CR101","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Xiong, Y., & Lin, D. (2018). Recognize actions by disentangling components of dynamics. In CVPR.","DOI":"10.1109\/CVPR.2018.00687"},{"key":"1222_CR102","doi-asserted-by":"publisher","first-page":"2243","DOI":"10.1109\/TPAMI.2008.263","volume":"31","author":"Y Zheng","year":"2009","unstructured":"Zheng, Y., Lin, S., Kambhamettu, C., Yu, J., & Kang, S. B. (2009). Single-image vignetting correction. T-PAMI, 31, 2243\u20132256.","journal-title":"T-PAMI"},{"key":"1222_CR103","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., & Torralba, A. (2017). Scene parsing through ADE20K dataset. In CVPR.","DOI":"10.1109\/CVPR.2017.544"},{"key":"1222_CR104","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Mottaghi, R., Kolve, E., Lim, J. J., Gupta, A., Fei-Fei, L., & Farhadi, A. (2017). Target-driven visual navigation in indoor scenes using deep reinforcement learning. In ICRA.","DOI":"10.1109\/ICRA.2017.7989381"},{"issue":"4","key":"1222_CR105","doi-asserted-by":"publisher","first-page":"627","DOI":"10.1109\/TPAMI.2014.2366143","volume":"38","author":"C Zitnick","year":"2016","unstructured":"Zitnick, C., Vedantam, R., & Parikh, D. (2016). Adopting abstract images for semantic scene understanding. T-PAMI, 38(4), 627\u2013638.","journal-title":"T-PAMI"},{"key":"1222_CR106","doi-asserted-by":"crossref","unstructured":"Zolfaghari, M., Oliveira, G. L., Sedaghat, N., & Brox, T. (2017). Chained multi-stream networks exploiting pose, motion, and appearance for action classification and detection. In ICCV.","DOI":"10.1109\/ICCV.2017.316"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-019-01222-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-019-01222-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-019-01222-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,2]],"date-time":"2022-10-02T11:43:48Z","timestamp":1664711028000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-019-01222-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,23]]},"references-count":106,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2020,5]]}},"alternative-id":["1222"],"URL":"https:\/\/doi.org\/10.1007\/s11263-019-01222-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,10,23]]},"assertion":[{"value":"16 November 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 August 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 October 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}