{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:24:15Z","timestamp":1779380655919,"version":"3.53.1"},"reference-count":90,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2020,6,15]],"date-time":"2020-06-15T00:00:00Z","timestamp":1592179200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,6,15]],"date-time":"2020-06-15T00:00:00Z","timestamp":1592179200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100010198","name":"Ministerio de Econom\u00eda, Industria y Competitividad, Gobierno de Espa\u00f1a","doi-asserted-by":"publisher","award":["TIN2016-79717-R"],"award-info":[{"award-number":["TIN2016-79717-R"]}],"id":[{"id":"10.13039\/501100010198","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010198","name":"Ministerio de Econom\u00eda, Industria y Competitividad, Gobierno de Espa\u00f1a","doi-asserted-by":"publisher","award":["PCIN-2015-251"],"award-info":[{"award-number":["PCIN-2015-251"]}],"id":[{"id":"10.13039\/501100010198","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100011264","name":"FP7 People: Marie-Curie Actions","doi-asserted-by":"publisher","award":["665919"],"award-info":[{"award-number":["665919"]}],"id":[{"id":"10.13039\/100011264","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2020,12]]},"DOI":"10.1007\/s11263-020-01340-z","type":"journal-article","created":{"date-parts":[[2020,6,15]],"date-time":"2020-06-15T03:22:33Z","timestamp":1592191353000},"page":"2849-2872","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Mix and Match Networks: Cross-Modal Alignment for Zero-Pair Image-to-Image Translation"],"prefix":"10.1007","volume":"128","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6055-7164","authenticated-orcid":false,"given":"Yaxing","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7022-3395","authenticated-orcid":false,"given":"Luis","family":"Herranz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9656-9706","authenticated-orcid":false,"given":"Joost","family":"van de Weijer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,6,15]]},"reference":[{"issue":"7","key":"1340_CR1","doi-asserted-by":"publisher","first-page":"1425","DOI":"10.1109\/TPAMI.2015.2487986","volume":"38","author":"Z Akata","year":"2016","unstructured":"Akata, Z., Perronnin, F., Harchaoui, Z., & Schmid, C. (2016). Label-embedding for image classification. IEEE Transactions on Pattern Analysis and Machine Intelligence, 38(7), 1425\u20131438.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1340_CR2","doi-asserted-by":"crossref","unstructured":"Alharbi, Y., Smith, N., & Wonka, P. (2019). Latent filter scaling for multimodal unsupervised image-to-image translation. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 1458\u20131466).","DOI":"10.1109\/CVPR.2019.00155"},{"key":"1340_CR3","unstructured":"Almahairi, A., Rajeswar, S., Sordoni, A., Bachman, P., & Courville, A. (2018). Augmented cyclegan: Learning many-to-many mappings from unpaired data. International Conference on Machine Learning."},{"key":"1340_CR4","doi-asserted-by":"crossref","unstructured":"Amodio, M., & Krishnaswamy, S. (2019). Travelgan: Image-to-image translation by transformation vector learning. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.00919"},{"key":"1340_CR5","doi-asserted-by":"crossref","unstructured":"Anoosheh, A., Agustsson, E., Timofte, R., & Van\u00a0Gool, L. (2018). Combogan: Unrestrained scalability for image domain translation. In 2018 IEEE\/CVF conference on computer vision and pattern recognition workshops (CVPRW) , http:\/\/dx.doi.org\/10.1109\/CVPRW.2018.00122.","DOI":"10.1109\/CVPRW.2018.00122"},{"key":"1340_CR6","unstructured":"Badrinarayanan, V., Handa, A., & Cipolla, R. (2015). Segnet: A deep convolutional encoder-decoder architecture for robust semantic pixel-wise labelling. In Proceedings of the IEEE conference on computer vision and pattern recognition."},{"key":"1340_CR7","unstructured":"Cadena, C., Dick, A. R., & Reid, I. D. (2016). Multi-modal auto-encoders as joint estimators for robotics scene understanding. In Robotics: Science and systems."},{"key":"1340_CR8","doi-asserted-by":"crossref","unstructured":"Castrejon, L., Aytar, Y., Vondrick, C., Pirsiavash, H., & Torralba, A. (2016). Learning aligned cross-modal representations from weakly aligned data. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2940\u20132949).","DOI":"10.1109\/CVPR.2016.321"},{"issue":"4","key":"1340_CR9","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2018","unstructured":"Chen, L. C., Papandreou, G., Kokkinos, I., Murphy, K., & Yuille, A. L. (2018). Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Transactions on Pattern Analysis and Machine Intelligence, 40(4), 834\u2013848.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1340_CR10","doi-asserted-by":"crossref","unstructured":"Chen, Q., & Koltun, V. (2017). Photographic image synthesis with cascaded refinement networks.","DOI":"10.1109\/ICCV.2017.168"},{"key":"1340_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Y., Liu, Y., Cheng, Y., & Li, V. O. (2017). A teacher\u2013student framework for zero-resource neural machine translation. Preprint arXiv:170500753.","DOI":"10.18653\/v1\/P17-1176"},{"key":"1340_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Y. C., Xu, X., Tian, Z., & Jia, J. (2019). Homomorphic latent space interpolation for unpaired image-to-image translation. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2408\u20132416).","DOI":"10.1109\/CVPR.2019.00251"},{"key":"1340_CR13","unstructured":"Cheng, Y., Zhao, X., Cai, R., Li, Z., Huang, K., Rui, Y., et\u00a0al. (2016). Semi-supervised multimodal deep learning for RGB-D object recognition. In Proceedings of the international joint conference on artificial intelligence."},{"key":"1340_CR14","doi-asserted-by":"crossref","unstructured":"Cho, W., Choi, S., Park, D. K., Shin, I., & Choo, J. (2019). Image-to-image translation via group-wise deep whitening-and-coloring transformation. In The IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2019.01089"},{"key":"1340_CR15","doi-asserted-by":"crossref","unstructured":"Choi, Y., Choi, M., Kim, M., Ha, J. W., Kim, S., & Choo, J. (2018). Stargan: Unified generative adversarial networks for multi-domain image-to-image translation. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00916"},{"key":"1340_CR16","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L. J., Li, K., & Fei-Fei, L. (2009). ImageNet: A Large-Scale Hierarchical Image Database. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1340_CR17","doi-asserted-by":"crossref","unstructured":"Eigen, D., & Fergus, R. (2015). Predicting depth, surface normals and semantic labels with a common multi-scale convolutional architecture. In Proceedings of the international conference on computer vision (pp. 2650\u20132658).","DOI":"10.1109\/ICCV.2015.304"},{"key":"1340_CR18","doi-asserted-by":"crossref","unstructured":"Eitel, A., Springenberg, J. T., Spinello, L., Riedmiller, M., & Burgard, W. (2015). Multimodal deep learning for robust rgb-d object recognition. In Proceedings of the IEEE\/RSJ conference on intelligent robots and systems (pp. 681\u2013687), IEEE.","DOI":"10.1109\/IROS.2015.7353446"},{"key":"1340_CR19","doi-asserted-by":"crossref","unstructured":"Fergus, R., Bernal, H., Weiss, Y., & Torralba, A. (2010). Semantic label sharing for learning with many categories. In Proceedings of the European conference on computer vision (pp. 762\u2013775).","DOI":"10.1007\/978-3-642-15549-9_55"},{"key":"1340_CR20","doi-asserted-by":"crossref","unstructured":"Firat, O., Cho, K., & Bengio, Y. (2016). Multi-way, multilingual neural machine translation with a shared attention mechanism. Preprint arXiv:160101073.","DOI":"10.18653\/v1\/N16-1101"},{"key":"1340_CR21","unstructured":"Fu, Y., Xiang, T., Jiang, Y. G., Xue, X., Sigal, L., & Gong, S. (2017). Recent advances in zero-shot recognition. Preprint arXiv:171004837."},{"key":"1340_CR22","unstructured":"Ganin, Y., & Lempitsky, V. (2015). Unsupervised domain adaptation by backpropagation. In International conference on machine learning (pp. 1180\u20131189)."},{"key":"1340_CR23","doi-asserted-by":"crossref","unstructured":"Gatys, L. A., Ecker, A. S., & Bethge, M. (2016). Image style transfer using convolutional neural networks. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2414\u20132423).","DOI":"10.1109\/CVPR.2016.265"},{"issue":"12","key":"1340_CR24","doi-asserted-by":"publisher","first-page":"1338","DOI":"10.1109\/34.977559","volume":"23","author":"JM Geusebroek","year":"2001","unstructured":"Geusebroek, J. M., Van den Boomgaard, R., Smeulders, A. W. M., & Geerts, H. (2001). Color invariance. IEEE Transactions on Pattern Analysis and Machine Intelligence, 23(12), 1338\u20131350.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1340_CR25","unstructured":"Gong, B., Shi, Y., Sha, F., & Grauman, K. (2012). Geodesic flow kernel for unsupervised domain adaptation. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2066\u20132073), IEEE."},{"key":"1340_CR26","unstructured":"Gonzalez-Garcia, A., van\u00a0de Weijer, J., & Bengio, Y. (2018). Image-to-image translation for cross-domain disentanglement. In Advances in neural information processing systems (pp. 1294\u20131305)."},{"key":"1340_CR27","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2014). Generative adversarial nets. In Advances in neural information processing systems (pp. 2672\u20132680)."},{"key":"1340_CR28","doi-asserted-by":"crossref","unstructured":"Gupta, S., Hoffman, J., & Malik, J. (2016). Cross modal distillation for supervision transfer. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2016.309"},{"key":"1340_CR29","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 770\u2013778).","DOI":"10.1109\/CVPR.2016.90"},{"key":"1340_CR30","doi-asserted-by":"crossref","unstructured":"Hoffman, J., Gupta, S., & Darrell, T. (2016a). Learning with side information through modality hallucination. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 826\u2013834).","DOI":"10.1109\/CVPR.2016.96"},{"key":"1340_CR31","doi-asserted-by":"crossref","unstructured":"Hoffman, J., Gupta, S., Leong, J., Guadarrama, S., & Darrell, T. (2016b). Cross-modal adaptation for rgb-d detection. In 2016 IEEE international conference on robotics and automation (ICRA) (pp. 5032\u20135039), IEEE.","DOI":"10.1109\/ICRA.2016.7487708"},{"key":"1340_CR32","doi-asserted-by":"crossref","unstructured":"Huang, X., Liu, M. Y., Belongie, S., & Kautz, J. (2018). Multimodal unsupervised image-to-image translation. In Proceedings of the European conference on computer vision (pp. 172\u2013189).","DOI":"10.1007\/978-3-030-01219-9_11"},{"key":"1340_CR33","doi-asserted-by":"crossref","unstructured":"Isola, P., Zhu, J. Y., Zhou, T., & Efros, A. A. (2017). Image-to-image translation with conditional adversarial networks. In Proceedings of the IEEE conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2017.632"},{"key":"1340_CR34","unstructured":"Jayaraman, D., & Grauman, K. (2014). Zero-shot recognition with unreliable attributes. In Advances in neural information processing systems (pp. 3464\u20133472)."},{"key":"1340_CR35","doi-asserted-by":"crossref","unstructured":"Johnson, M., Schuster, M., Le, Q. V., Krikun, M., Wu, Y., Chen, Z., Thorat, N., Vi\u00e9gas, F., Wattenberg, M., Corrado, G., et\u00a0al. (2016). Google\u2019s multilingual neural machine translation system: Enabling zero-shot translation. Preprint arXiv:161104558.","DOI":"10.1162\/tacl_a_00065"},{"key":"1340_CR36","unstructured":"Kendall, A., Gal, Y., & Cipolla, R. (2018). Multi-task learning using uncertainty to weigh losses for scene geometry and semantics. In Proceedings of the IEEE conference on computer vision and pattern recognition."},{"key":"1340_CR37","doi-asserted-by":"crossref","unstructured":"Kim, S., Park, K., Sohn, K., & Lin, S. (2016). Unified depth prediction and intrinsic image decomposition from a single image via joint convolutional neural fields. In Proceedings of the European conference on computer vision (pp. 143\u2013159), Springer.","DOI":"10.1007\/978-3-319-46484-8_9"},{"key":"1340_CR38","unstructured":"Kim, T., Cha, M., Kim, H., Lee, J., & Kim, J. (2017). Learning to discover cross-domain relations with generative adversarial networks."},{"key":"1340_CR39","unstructured":"Kingma, D., & Ba, J. (2014). Adam: A method for stochastic optimization. In International conference on learning representations."},{"key":"1340_CR40","doi-asserted-by":"crossref","unstructured":"Kuga, R., Kanezaki, A., Samejima, M., Sugano, Y., & Matsushita, Y. (2017). Multi-task learning using multi-modal encoder\u2013decoder networks with shared skip connections. In Proceedings of the international conference on computer vision.","DOI":"10.1109\/ICCVW.2017.54"},{"key":"1340_CR41","doi-asserted-by":"crossref","unstructured":"Kuznietsov, Y., St\u00fcckler, J., Leibe, B. (2017). Semi-supervised deep learning for monocular depth map prediction. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6647\u20136655).","DOI":"10.1109\/CVPR.2017.238"},{"key":"1340_CR42","doi-asserted-by":"crossref","unstructured":"Lai, K., Bo, L., Ren, X., & Fox, D. (2011). A large-scale hierarchical multi-view rgb-d object dataset. In Proceedings of IEEE international conference on robotics and automation (pp. 1817\u20131824), IEEE.","DOI":"10.1109\/ICRA.2011.5980382"},{"key":"1340_CR43","doi-asserted-by":"crossref","unstructured":"Laina, I., Rupprecht, C., Belagiannis, V., Tombari, F., & Navab, N. (2016). Deeper depth prediction with fully convolutional residual networks. In 2016 fourth international conference on 3D vision (3DV) (pp. 239\u2013248), IEEE.","DOI":"10.1109\/3DV.2016.32"},{"issue":"3","key":"1340_CR44","doi-asserted-by":"publisher","first-page":"453","DOI":"10.1109\/TPAMI.2013.140","volume":"36","author":"CH Lampert","year":"2014","unstructured":"Lampert, C. H., Nickisch, H., & Harmeling, S. (2014). Attribute-based classification for zero-shot visual object categorization. IEEE Transactions on Pattern Analysis and Machine Intelligence, 36(3), 453\u2013465.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1340_CR45","doi-asserted-by":"crossref","unstructured":"Lee, H. Y., Tseng, H. Y., Huang, J. B., Singh, M., & Yang, M. H. (2018). Diverse image-to-image translation via disentangled representations. In Proceedings of the European conference on computer vision (pp. 35\u201351).","DOI":"10.1007\/978-3-030-01246-5_3"},{"key":"1340_CR46","doi-asserted-by":"crossref","unstructured":"Li, Y., Liu, M. Y., Li, X., Yang, M. H., & Kautz, J. (2018). A closed-form solution to photorealistic image stylization. In Proceedings of the European conference on computer vision (pp. 453\u2013468).","DOI":"10.1007\/978-3-030-01219-9_28"},{"key":"1340_CR47","doi-asserted-by":"crossref","unstructured":"Lin, J., Xia, Y., Qin, T., Chen, Z., & Liu, T. Y. (2018). Conditional image-to-image translation. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 5524\u20135532).","DOI":"10.1109\/CVPR.2018.00579"},{"issue":"10","key":"1340_CR48","doi-asserted-by":"publisher","first-page":"2024","DOI":"10.1109\/TPAMI.2015.2505283","volume":"38","author":"F Liu","year":"2016","unstructured":"Liu, F., Shen, C., Lin, G., & Reid, I. (2016). Learning depth from single monocular images using deep convolutional neural fields. IEEE Transactions on Pattern Analysis and Machine Intelligence, 38(10), 2024\u20132039.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1340_CR49","doi-asserted-by":"crossref","unstructured":"Liu, M. Y., Breuel, T., & Kautz, J. (2017). Unsupervised image-to-image translation networks. In Advances in neural information processing systems.","DOI":"10.1007\/978-3-319-70139-4"},{"key":"1340_CR50","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., & Darrell, T. (2015). Fully convolutional networks for semantic segmentation. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 3431\u20133440).","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"1340_CR51","unstructured":"Mao, X., Li, Q., Xie, H., Lau, R. Y., & Wang, Z. (2016). Multi-class generative adversarial networks with the l2 loss function. Preprint arXiv:161104076."},{"key":"1340_CR52","unstructured":"Mathieu, M. F., Zhao, J. J., Zhao, J., Ramesh, A., Sprechmann, P., & LeCun, Y. (2016). Disentangling factors of variation in deep representation using adversarial training. In Advances in neural information processing systems (pp. 5040\u20135048)."},{"key":"1340_CR53","doi-asserted-by":"crossref","unstructured":"McCormac, J., Handa, A., Leutenegger, S., & JDavison, A. (2017). Scenenet rgb-d: Can 5m synthetic images beat generic imagenet pre-training on indoor segmentation? In Proceedings of the international conference on computer vision.","DOI":"10.1109\/ICCV.2017.292"},{"key":"1340_CR54","unstructured":"Mejjati, Y. A., Richardt, C., Tompkin, J., Cosker, D., & Kim, K. I. (2018). Unsupervised attention-guided image-to-image translation. In Advances in neural information processing systems (pp. 3697\u20133707)."},{"key":"1340_CR55","unstructured":"Mirza, M., & Osindero, S. (2014). Conditional generative adversarial nets. Preprint arXiv:14111784."},{"key":"1340_CR56","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., & Ng, A. Y. (2011). Multimodal deep learning. In International conference on machine learning (pp. 689\u2013696)."},{"key":"1340_CR57","doi-asserted-by":"crossref","unstructured":"Nilsback, M. E., & Zisserman, A. (2008). Automated flower classification over a large number of classes. In Sixth Indian conference on computer vision, graphics & image processing, 2008. ICVGIP\u201908 (pp. 722\u2013729), IEEE.","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"1340_CR58","unstructured":"Perarnau, G., Van De\u00a0Weijer, J., Raducanu, B., & \u00c1lvarez, J. M. (2016). Invertible conditional gans for image editing. Preprint arXiv:161106355."},{"key":"1340_CR59","doi-asserted-by":"crossref","unstructured":"Reed, S., Akata, Z., Lee, H., & Schiele, B. (2016). Learning deep representations of fine-grained visual descriptions. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 49\u201358).","DOI":"10.1109\/CVPR.2016.13"},{"key":"1340_CR60","doi-asserted-by":"crossref","unstructured":"Rohrbach, M., Stark, M., & Schiele, B. (2011). Evaluating knowledge transfer and zero-shot learning in a large-scale setting. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 1641\u20131648), IEEE.","DOI":"10.1109\/CVPR.2011.5995627"},{"key":"1340_CR61","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In International conference on medical image computing and computer-assisted intervention (pp. 234\u2013241), Springer.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"1340_CR62","doi-asserted-by":"crossref","unstructured":"Roy, A., & Todorovic, S. (2016). Monocular depth estimation using neural regression forest. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 5506\u20135514).","DOI":"10.1109\/CVPR.2016.594"},{"key":"1340_CR63","unstructured":"Saito, K., Ushiku, Y., & Harada, T. (2017). Asymmetric tri-training for unsupervised domain adaptation."},{"key":"1340_CR64","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., & Fergus, R. (2012). Indoor segmentation and support inference from RGBD images. In Proceedings of the European conference on computer vision (pp. 746\u2013760), Springer.","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"1340_CR65","unstructured":"Simonyan, K., & Zisserman, A. (2015). Very deep convolutional networks for large-scale image recognition."},{"key":"1340_CR66","doi-asserted-by":"crossref","unstructured":"Song, S., Lichtenberg, S. P., & Xiao, J. (2015). Sun rgb-d: A rgb-d scene understanding benchmark suite. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 567\u2013576).","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"1340_CR67","doi-asserted-by":"crossref","unstructured":"Song, X., Herranz, L., & Jiang, S. (2017). Depth CNNs for RGB-D scene recognition: Learning from scratch better than transferring from rgb-cnns. In Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v31i1.11226"},{"key":"1340_CR68","unstructured":"Taigman, Y., Polyak, A., & Wolf, L. (2017). Unsupervised cross-domain image generation."},{"key":"1340_CR69","doi-asserted-by":"crossref","unstructured":"Tsai, Y. H., Hung, W. C., Schulter, S., Sohn, K., Yang, M. H., & Chandraker, M. (2018). Learning to adapt structured output space for semantic segmentation. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00780"},{"key":"1340_CR70","doi-asserted-by":"crossref","unstructured":"Valada, A., Oliveira, G. L., Brox, T., & Burgard, W. (2016). Deep multispectral semantic scene understanding of forested environments using multimodal fusion. In International symposium on experimental robotics (pp. 465\u2013477), Springer.","DOI":"10.1007\/978-3-319-50115-4_41"},{"key":"1340_CR71","doi-asserted-by":"crossref","unstructured":"Wang, P., Shen, X., Lin, Z., Cohen, S., Price, B., & Yuille, A. L. (2015). Towards unified depth and semantic prediction from a single image. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2800\u20132809).","DOI":"10.1109\/CVPR.2015.7298897"},{"key":"1340_CR72","doi-asserted-by":"crossref","unstructured":"Wang, T. C., Liu, M. Y., Zhu, J. Y., Tao, A., Kautz, J., & Catanzaro, B. (2018a). High-resolution image synthesis and semantic manipulation with conditional gans. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 8798\u20138807).","DOI":"10.1109\/CVPR.2018.00917"},{"key":"1340_CR73","doi-asserted-by":"crossref","unstructured":"Wang, W., & Neumann, U. (2018). Depth-aware CNN for RGB-D segmentation. In Proceedings of the European conference on computer vision (pp. 135\u2013150).","DOI":"10.1007\/978-3-030-01252-6_9"},{"key":"1340_CR74","doi-asserted-by":"crossref","unstructured":"Wang, Y., Gonzalez-Garcia, A., van\u00a0de Weijer, J., & Herranz, L. (2019). Sdit: Scalable and diverse cross-domain image translation. Preprint arXiv:190806881.","DOI":"10.1145\/3343031.3351004"},{"key":"1340_CR75","doi-asserted-by":"crossref","unstructured":"Wang, Y., van\u00a0de Weijer, J., & Herranz, L. (2018b). Mix and match networks: Encoder\u2013decoder alignment for zero-pair image translation. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 5467\u20135476).","DOI":"10.1109\/CVPR.2018.00573"},{"key":"1340_CR76","doi-asserted-by":"crossref","unstructured":"Wu, W., Cao, K., Li, C., Qian, C., & Loy, C. C. (2019). Transgaga: Geometry-aware unsupervised image-to-image translation. In The IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2019.00820"},{"key":"1340_CR77","doi-asserted-by":"crossref","unstructured":"Wu, Z., Han, X., Lin, Y. L., Uzunbas, M. G., Goldstein, T., Lim, S. N., & Davis, L. S. (2018). Dcan: Dual channel-wise alignment networks for unsupervised scene adaptation. In Proceedings of the European conference on computer vision.","DOI":"10.1007\/978-3-030-01228-1_32"},{"key":"1340_CR78","doi-asserted-by":"crossref","unstructured":"Xian, Y., Lampert, C. H., Schiele, B., & Akata, Z. (2018a). Zero-shot learning-a comprehensive evaluation of the good, the bad and the ugly. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/CVPR.2017.328"},{"key":"1340_CR79","doi-asserted-by":"crossref","unstructured":"Xian, Y., Lorenz, T., Schiele, B., & Akata, Z. (2018b). Feature generating networks for zero-shot learning. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 5542\u20135551).","DOI":"10.1109\/CVPR.2018.00581"},{"key":"1340_CR80","doi-asserted-by":"crossref","unstructured":"Xu, D., Ouyang, W., Ricci, E., Wang, X., & Sebe, N. (2017). Learning cross-modal deep representations for robust pedestrian detection. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 5363\u20135371).","DOI":"10.1109\/CVPR.2017.451"},{"key":"1340_CR81","doi-asserted-by":"crossref","unstructured":"Yi, Z., Zhang, H., Gong, P. T., et\u00a0al. (2017). Dualgan: Unsupervised dual learning for image-to-image translation. In Proceedings of the international conference on computer vision.","DOI":"10.1109\/ICCV.2017.310"},{"key":"1340_CR82","unstructured":"Yu, F., & Koltun, V. (2016). Multi-scale context aggregation by dilated convolutions."},{"issue":"2","key":"1340_CR83","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1007\/s00138-017-0902-y","volume":"29","author":"L Yu","year":"2018","unstructured":"Yu, L., Zhang, L., van de Weijer, J., Khan, F. S., Cheng, Y., & Parraga, C. A. (2018). Beyond eleven color names for image understanding. Machine Vision and Applications, 29(2), 361\u2013373.","journal-title":"Machine Vision and Applications"},{"issue":"4","key":"1340_CR84","doi-asserted-by":"publisher","first-page":"1837","DOI":"10.1109\/TIP.2018.2879249","volume":"28","author":"L Zhang","year":"2019","unstructured":"Zhang, L., Gonzalez-Garcia, A., van de Weijer, J., Danelljan, M., & Khan, F. S. (2019). Synthetic data generation for end-to-end thermal infrared tracking. IEEE Transactions on Image Processing, 28(4), 1837\u20131850.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1340_CR85","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., & Efros, A. A. (2016). Colorful image colorization. In Proceedings of the European conference on computer vision","DOI":"10.1007\/978-3-319-46487-9_40"},{"key":"1340_CR86","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., & Jia, J. (2017). Pyramid scene parsing network. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2881\u20132890).","DOI":"10.1109\/CVPR.2017.660"},{"key":"1340_CR87","doi-asserted-by":"crossref","unstructured":"Zheng, H., Cheng, Y., & Liu, Y. (2017). Maximum expected likelihood estimation for zero-resource neural machine translation. In Proceedings of the international joint conference on artificial intelligence.","DOI":"10.24963\/ijcai.2017\/594"},{"key":"1340_CR88","doi-asserted-by":"crossref","unstructured":"Zhu, J. Y., Park, T., Isola, P., & Efros, A. A. (2017a). Unpaired image-to-image translation using cycle-consistent adversarial networks. In Proceedings of the international conference on computer vision.","DOI":"10.1109\/ICCV.2017.244"},{"key":"1340_CR89","unstructured":"Zhu, J. Y., Zhang, R., Pathak, D., Darrell, T., Efros, A. A., Wang, O., & Shechtman, E. (2017b). Toward multimodal image-to-image translation. In Advances in neural information processing systems (pp. 465\u2013476)."},{"key":"1340_CR90","doi-asserted-by":"crossref","unstructured":"Zou, Y., Yu, Z., Vijaya\u00a0Kumar, B., & Wang, J. (2018). Unsupervised domain adaptation for semantic segmentation via class-balanced self-training. In Proceedings of the European conference on computer vision.","DOI":"10.1007\/978-3-030-01219-9_18"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01340-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-020-01340-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01340-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,28]],"date-time":"2022-10-28T13:14:54Z","timestamp":1666962894000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-020-01340-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,6,15]]},"references-count":90,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2020,12]]}},"alternative-id":["1340"],"URL":"https:\/\/doi.org\/10.1007\/s11263-020-01340-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,6,15]]},"assertion":[{"value":"1 March 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 May 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 June 2020","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}