{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,13]],"date-time":"2025-11-13T12:43:02Z","timestamp":1763037782684,"version":"3.37.3"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2022,7,21]],"date-time":"2022-07-21T00:00:00Z","timestamp":1658361600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,7,21]],"date-time":"2022-07-21T00:00:00Z","timestamp":1658361600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,4]]},"DOI":"10.1007\/s10489-022-03936-z","type":"journal-article","created":{"date-parts":[[2022,7,21]],"date-time":"2022-07-21T15:02:58Z","timestamp":1658415778000},"page":"8467-8481","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["SSDMM-VAE: variational multi-modal disentangled representation learning"],"prefix":"10.1007","volume":"53","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7297-374X","authenticated-orcid":false,"given":"Arnab Kumar","family":"Mondal","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ajay","family":"Sailopal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Parag","family":"Singla","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Prathosh","family":"AP","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,7,21]]},"reference":[{"issue":"11","key":"3936_CR1","doi-asserted-by":"publisher","first-page":"8088","DOI":"10.1007\/s10489-021-02268-8","volume":"51","author":"D Koch","year":"2021","unstructured":"Koch D, Despotovic M, Thaler S, Zeppelzauer M (2021) Where do university graduates live?\u2013a computer vision approach using satellite images. Appl Intell 51(11):8088\u20138105","journal-title":"Appl Intell"},{"key":"3936_CR2","doi-asserted-by":"crossref","unstructured":"Hassan H, Mishra P, Ahmad M, Bashir AK, Huang B, Luo B (2022) Effects of haze and dehazing on deep learning-based vision models. Appl Intell:1\u201319","DOI":"10.1007\/s10489-022-03245-5"},{"issue":"7","key":"3936_CR3","doi-asserted-by":"publisher","first-page":"2105","DOI":"10.1007\/s10489-020-01641-3","volume":"50","author":"X Lin","year":"2020","unstructured":"Lin X, Wang X, Li L (2020) Intelligent detection of edge inconsistency for mechanical workpiece by machine vision with deep learning and variable geometry model. Appl Intell 50(7):2105\u20132119","journal-title":"Appl Intell"},{"issue":"2","key":"3936_CR4","doi-asserted-by":"publisher","first-page":"1878","DOI":"10.1007\/s10489-021-02306-5","volume":"52","author":"X Lu","year":"2022","unstructured":"Lu X, Deng Y, Sun T, Gao Y, Feng J, Sun X, Sutcliffe R (2022) Mkpm: multi keyword-pair matching for natural language sentences. Appl Intell 52(2):1878\u20131892","journal-title":"Appl Intell"},{"key":"3936_CR5","doi-asserted-by":"crossref","unstructured":"Zhao S, Zhang T, Hu M, Chang W, You F (2022) Ap-bert: enhanced pre-trained model through average pooling. Appl Intell:1\u20139","DOI":"10.1007\/s10489-022-03190-3"},{"key":"3936_CR6","doi-asserted-by":"publisher","first-page":"228450","DOI":"10.1016\/j.jpowsour.2020.228450","volume":"471","author":"S Wang","year":"2020","unstructured":"Wang S, Fernandez C, Yu C, Fan Y, Cao W, Stroe D-I (2020) A novel charged state prediction method of the lithium ion battery packs based on the composite equivalent modeling and improved splice kalman filtering algorithm. J Power Sources 471:228450","journal-title":"J Power Sources"},{"issue":"15","key":"3936_CR7","doi-asserted-by":"publisher","first-page":"1308","DOI":"10.1016\/j.cub.2009.06.060","volume":"19","author":"R Quian Quiroga","year":"2009","unstructured":"Quian Quiroga R, Kraskov A, Koch C, Fried I (2009) Explicit encoding of multimodal percepts by single neurons in the human brain. Curr Biol CB 19(15):1308\u20131313","journal-title":"Curr Biol CB"},{"issue":"1-2","key":"3936_CR8","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1016\/j.heares.2009.03.012","volume":"258","author":"BE Stein","year":"2009","unstructured":"Stein BE, Stanford TR, Rowland BA (2009) The neural basis of multisensory integration in the midbrain: its organization and maturation. Hear Res 258(1-2):4\u201315","journal-title":"Hear Res"},{"key":"3936_CR9","unstructured":"Suzuki M, Nakayama K, Matsuo Y (2017) Joint multimodal learning with deep generative models. In: ICLR Wrokshop"},{"key":"3936_CR10","unstructured":"Vedantam R, Fischer I, Huang J, Murphy K (2018) Generative models of visually grounded imagination. Proc of ICLR"},{"key":"3936_CR11","unstructured":"Wu M, Goodman N (2018) Multimodal generative models for scalable weakly-supervised learning. In: Proc. of neruIPS"},{"key":"3936_CR12","doi-asserted-by":"crossref","unstructured":"Yadav R, Sardana A, Namboodiri VP, Hegde RM (2020) Bridged variational autoencoders for joint modeling of images and attributes. In: Proc. of WACV","DOI":"10.1109\/WACV45572.2020.9093565"},{"key":"3936_CR13","unstructured":"Shi Y, Siddharth N, Paige B, Torr PHS (2019) Variational mixture-of-experts autoencoders for multi-modal deep generative models. In: Proc. of neruIPS"},{"key":"3936_CR14","unstructured":"Kingma DP, Welling M (2014) Auto-encoding variational bayes. In: Proc. of ICLR"},{"key":"3936_CR15","unstructured":"Do K, Tran T (2020) Theory and evaluation metrics for learning disentangled representations. In: Proc. of ICLR"},{"key":"3936_CR16","unstructured":"Parascandolo G, Kilbertus N, Rojas-Carulla M, Sch\u00f6lkopf B (2018) Learning independent causal mechanisms. In: Proc. of ICML"},{"key":"3936_CR17","unstructured":"Besserve M, Mehrjou A, Sun R, Sch\u00f6lkopf B (2020) Counterfactuals uncover the modular structure of deep generative models. In: Proc. of ICLR"},{"issue":"5","key":"3936_CR18","doi-asserted-by":"publisher","first-page":"612","DOI":"10.1109\/JPROC.2021.3058954","volume":"109","author":"B Sch\u00f6lkopf","year":"2021","unstructured":"Sch\u00f6lkopf B, Locatello F, Bauer S, Ke NR, Kalchbrenner N, Goyal A, Bengio Y (2021) Toward causal representation learning. Proc IEEE 109(5):612\u2013634. https:\/\/doi.org\/10.1109\/JPROC.2021.3058954https:\/\/doi.org\/10.1109\/JPROC.2021.3058954","journal-title":"Proc IEEE"},{"key":"3936_CR19","unstructured":"Louizos C, Swersky K, Li Y, Welling M, Zemel R (2016) The variational fair autoencoder. In: Proc. of ICLR"},{"key":"3936_CR20","unstructured":"Creager E, Madras D, Jacobsen J-H, Weis M, Swersky K, Pitassi T, Zemel R (2019) Flexibly fair representation learning by disentanglement. In: Proc. of ICML"},{"key":"3936_CR21","unstructured":"Locatello F, Abbati G, Rainforth T, Bauer S, Sch\u00f6lkopf B, Bachem O (2019) On the fairness of disentangled representations. In: Proc. of neurIPS"},{"key":"3936_CR22","unstructured":"Achille A, Eccles T, Matthey L, Burgess CP, Watters N, Lerchner A, Higgins I (2018) Life-long disentangled representation learning with cross-domain latent homologies. In: Proc. of neurIPS"},{"key":"3936_CR23","doi-asserted-by":"publisher","first-page":"216","DOI":"10.1016\/j.neucom.2020.09.065","volume":"426","author":"B Li","year":"2021","unstructured":"Li B, Han C, Guo T, Zhao T (2021) Disentangled features with direct sum decomposition for zero shot learning. Neurocomputing 426:216\u2013226. https:\/\/doi.org\/10.1016\/j.neucom.2020.09.065","journal-title":"Neurocomputing"},{"issue":"12","key":"3936_CR24","doi-asserted-by":"publisher","first-page":"4261","DOI":"10.1007\/s10489-020-01750-z","volume":"50","author":"P Sun","year":"2020","unstructured":"Sun P, Su X, Guo S, Chen F (2020) Cycle representation-disentangling network: learning to completely disentangle spatial-temporal features in video. Appl Intell 50(12):4261\u20134280. https:\/\/doi.org\/10.1007\/s10489-020-01750-z","journal-title":"Appl Intell"},{"key":"3936_CR25","doi-asserted-by":"publisher","first-page":"270","DOI":"10.1016\/j.neucom.2022.01.066","volume":"481","author":"W Hou","year":"2022","unstructured":"Hou W, Qin Z, Xi X, Lu X, Yin Y (2022) Learning disentangled representation for self-supervised video object segmentation. Neurocomputing 481:270\u2013280. https:\/\/doi.org\/10.1016\/j.neucom.2022.01.066https:\/\/doi.org\/10.1016\/j.neucom.2022.01.066","journal-title":"Neurocomputing"},{"issue":"10","key":"3936_CR26","doi-asserted-by":"publisher","first-page":"2402","DOI":"10.1007\/s11263-019-01284-z","volume":"128","author":"H-Y Lee","year":"2020","unstructured":"Lee H-Y, Tseng H-Y, Mao Q, Huang J-B, Lu Y-D, Singh M, Yang M-H (2020) Drit++: Diverse image-to-image translation via disentangled representations. Int J Comput Vis 128(10):2402\u20132417. https:\/\/doi.org\/10.1007\/s11263-019-01284-z","journal-title":"Int J Comput Vis"},{"key":"3936_CR27","unstructured":"Higgins I, Matthey L, Pal A, Burgess C, Glorot X, Botvinick M, Mohamed S, Lerchner A (2017) \u03b2-VAE: Learning basic visual concepts with a constrained variational framework. In: Proc. of ICLR"},{"key":"3936_CR28","unstructured":"Chen TQ, Li X, Grosse RB, Duvenaud DK (2018) Isolating sources of disentanglement in variational autoencoders. In: Proc. of neuRIPS"},{"key":"3936_CR29","unstructured":"Kim H, Mnih A (2018) Disentangling by factorising. In: Proc. of ICML"},{"key":"3936_CR30","unstructured":"Jeong Y, Song HO (2019) Learning discrete and continuous factors of data via alternating disentanglement"},{"key":"3936_CR31","doi-asserted-by":"crossref","unstructured":"Locatello F, Bauer S, Lucic M, R\u00e4tsch G, Gelly S, Sch\u00f6lkopf B, Bachem O (2019) Challenging common assumptions in the unsupervised learning of disentangled representations. In: Proc. of ICML","DOI":"10.1609\/aaai.v34i09.7120"},{"key":"3936_CR32","first-page":"209","volume":"21","author":"F Locatello","year":"2020","unstructured":"Locatello F, Bauer S, Lucic M, R\u00e4tsch G, Gelly S, Sch\u00f6lkopf B, Bachem O (2020) A sober look at the unsupervised learning of disentangled representations and their evaluation. J Mach Learn Res 21:209\u2013120962","journal-title":"J Mach Learn Res"},{"key":"3936_CR33","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1016\/j.ins.2018.12.057","volume":"482","author":"Y Li","year":"2019","unstructured":"Li Y, Pan Q, Wang S, Peng H, Yang T, Cambria E (2019) Disentangled variational auto-encoder for semi-supervised learning. Inf Sci 482:73\u201385","journal-title":"Inf Sci"},{"key":"3936_CR34","doi-asserted-by":"crossref","unstructured":"Bouchacourt D, Tomioka R, Nowozin S (2018) Multi-level variational autoencoder: learning disentangled representations from grouped observations","DOI":"10.1609\/aaai.v32i1.11867"},{"key":"3936_CR35","doi-asserted-by":"crossref","unstructured":"Hosoya H (2019) Group-based learning of disentangled representations with generalizability for novel contents. In: Proc. of IJCAI","DOI":"10.24963\/ijcai.2019\/348"},{"key":"3936_CR36","unstructured":"Shu R, Chen Y, Kumar A, Ermon S, Poole B (2020) Weakly supervised disentanglement with guarantees. In: Proc. of ICLR"},{"key":"3936_CR37","unstructured":"Locatello F, Poole B, Raetsch G, Sch\u00f6lkopf B, Bachem O, Tschannen M (2020) Weakly-supervised disentanglement without compromises. In: Proc. of ICML"},{"key":"3936_CR38","unstructured":"Locatello F, Tschannen M, Bauer S, R\u00e4tsch G, Sch\u00f6lkopf B, Bachem O (2020) Disentangling factors of variation using few labels. In: Proc. of ICLR"},{"issue":"8","key":"3936_CR39","doi-asserted-by":"publisher","first-page":"1771","DOI":"10.1162\/089976602760128018","volume":"14","author":"GE Hinton","year":"2002","unstructured":"Hinton GE (2002) Training products of experts by minimizing contrastive divergence. Neural Comput 14(8):1771\u20131800","journal-title":"Neural Comput"},{"key":"3936_CR40","unstructured":"Burgess CP, Higgins I, Pal A, Matthey L, Watters N, Desjardins G, Lerchner A (2017) Understanding disentangling in \u03b2 -VAE. In: NeuRIPS workshop"},{"key":"3936_CR41","unstructured":"Dupont E. (2018) Learning disentangled joint continuous and discrete representations. In: Proc. of neurIPS"},{"key":"3936_CR42","unstructured":"Lample G, Zeghidour N, Usunier N, Bordes A, Denoyer L, Ranzato M (2017) Fader networks: manipulating images by sliding attributes"},{"key":"3936_CR43","unstructured":"Reed S, Sohn K, Zhang Y, Lee H (2014) Learning to disentangle factors of variation with manifold interaction. In: Proc. of ICML"},{"key":"3936_CR44","unstructured":"Cheung B, Livezey JA, Bansal AK, Olshausen BA (2015) Discovering hidden factors of variation in deep networks. In: Proc. of ICLR workshop"},{"key":"3936_CR45","unstructured":"Mathieu MF, Zhao JJ, Zhao J, Ramesh A, Sprechmann P, LeCun Y (2016) Disentangling factors of variation in deep representation using adversarial training. In: Proc. of neurIPS"},{"key":"3936_CR46","unstructured":"Siddharth N, Paige B, van de Meent J-W, Desmaison A, Goodman ND, Kohli P, Wood F, Torr PHS (2017) Learning disentangled representations with semi-supervised deep generative models. In: Proc of neurIPS"},{"key":"3936_CR47","doi-asserted-by":"crossref","unstructured":"Lee M, Pavlovic V (2021) Private-shared disentangled multimodal vae for learning of latent representations. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshop, pp 1692\u20131700","DOI":"10.1109\/CVPRW53098.2021.00185"},{"key":"3936_CR48","unstructured":"Cao Y, Fleet DJ (2014) Generalized product of experts for automatic and principled fusion of gaussian process predictions. In: Proc. of modern nonparametrics 3: automating the learning pipeline workshop at neurIPS"},{"key":"3936_CR49","unstructured":"Hoffman MD, Johnson MJ (2016) Elbo surgery: yet another way to carve up the variational evidence lower bound. In: NeurIPS workshop"},{"key":"3936_CR50","unstructured":"Matthey L, Higgins I, Hassabis D, Lerchner A (2017) dSprites: disentanglement testing sprites dataset. https:\/\/github.com\/deepmind\/dsprites-dataset\/. Accessed 16 Feb 2022"},{"key":"3936_CR51","unstructured":"Burgess C, Kim H (2018) 3D shapes dataset. https:\/\/github.com\/deepmind\/3dshapes-dataset\/. Accessed 16 Feb 2022"},{"key":"3936_CR52","unstructured":"Lecun Y (2010) The mnist database of handwritten digits. http:\/\/yann.lecun.com\/exdb\/mnist\/. Accessed 16 Feb 2022"},{"key":"3936_CR53","doi-asserted-by":"crossref","unstructured":"El-Sawy A, EL-Bakry H, Loey M (2016) Cnn for handwritten arabic digits recognition based on lenet-5. In: Proc. of the international conference on advanced intelligent systems and informatics","DOI":"10.1007\/978-3-319-48308-5_54"},{"key":"3936_CR54","unstructured":"Theis L, Oord Avd, Bethge M (2016) A note on the evaluation of generative models. In: Proc. of ICLR"},{"key":"3936_CR55","unstructured":"Lucic M, Kurach K, Michalski M, Bousquet O, Gelly S (2018) Are gans created equal? A large-scale study. In: Proc. of neuRIPS"},{"key":"3936_CR56","unstructured":"Sajjadi MSM, Bachem O, Lucic M, Bousquet O, Gelly S (2018) Assessing generative models via precision and recall. In: Proc. of neuRIPS"},{"key":"3936_CR57","doi-asserted-by":"crossref","unstructured":"Grover A, Dhar M, Ermon S (2018) Flow-gan: combining maximum likelihood and adversarial learning in generative models. In: Proc. of AAAI","DOI":"10.1609\/aaai.v32i1.11829"},{"key":"3936_CR58","unstructured":"Heusel M, Ramsauer H, Unterthiner T, Nessler B, Hochreiter S (2017) Gans trained by a two time-scale update rule converge to a local nash equilibrium. In: Proc. of neuRIPS"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-03936-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-022-03936-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-03936-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,3,16]],"date-time":"2023-03-16T02:20:25Z","timestamp":1678933225000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-022-03936-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,21]]},"references-count":58,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2023,4]]}},"alternative-id":["3936"],"URL":"https:\/\/doi.org\/10.1007\/s10489-022-03936-z","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2022,7,21]]},"assertion":[{"value":"24 June 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 July 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Competing interests"}}]}}