{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T19:42:35Z","timestamp":1771702955241,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,15]],"date-time":"2019-10-15T00:00:00Z","timestamp":1571097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["1637312"],"award-info":[{"award-number":["1637312"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012659","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61672267 and U1836220"],"award-info":[{"award-number":["61672267 and U1836220"]}],"id":[{"id":"10.13039\/501100012659","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,15]]},"DOI":"10.1145\/3343031.3351086","type":"proceedings-article","created":{"date-parts":[[2019,10,21]],"date-time":"2019-10-21T16:32:26Z","timestamp":1571675546000},"page":"2006-2014","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["On Learning Disentangled Representation for Acoustic Event Detection"],"prefix":"10.1145","author":[{"given":"Lijian","family":"Gao","sequence":"first","affiliation":[{"name":"Jiangsu University, Zhenjiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qirong","family":"Mao","sequence":"additional","affiliation":[{"name":"Jiangsu University, Zhenjiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Dong","sequence":"additional","affiliation":[{"name":"Wayne State University, Detroit, MI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Jing","sequence":"additional","affiliation":[{"name":"Wayne State University, Detroit, MI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ratna","family":"Chinnam","sequence":"additional","affiliation":[{"name":"Wayne State University, Detroit, MI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Technical Report. DCASE2017 Challenge.","author":"Adavanne Sharath","year":"2017","unstructured":"Sharath Adavanne and Tuomas Virtanen . 2017 . A Report on Sound Event Detection with Different Binaural Features . Technical Report. DCASE2017 Challenge. Sharath Adavanne and Tuomas Virtanen. 2017. A Report on Sound Event Detection with Different Binaural Features . Technical Report. DCASE2017 Challenge."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.50"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2015.7280624"},{"key":"e_1_3_2_1_4_1","volume-title":"Infogan: Interpretable representation learning by information maximizing generative adversarial nets. In Advances in Neural Information Processing Systems. 2172--2180.","author":"Chen Xi","year":"2016","unstructured":"Xi Chen , Yan Duan , Rein Houthooft , John Schulman , Ilya Sutskever , and Pieter Abbeel . 2016 . Infogan: Interpretable representation learning by information maximizing generative adversarial nets. In Advances in Neural Information Processing Systems. 2172--2180. Xi Chen, Yan Duan, Rein Houthooft, John Schulman, Ilya Sutskever, and Pieter Abbeel. 2016. Infogan: Interpretable representation learning by information maximizing generative adversarial nets. In Advances in Neural Information Processing Systems. 2172--2180."},{"key":"e_1_3_2_1_5_1","volume-title":"Discovering hidden factors of variation in deep networks. arXiv preprint arXiv:1412.6583","author":"Cheung Brian","year":"2014","unstructured":"Brian Cheung , Jesse A Livezey , Arjun K Bansal , and Bruno A Olshausen . 2014. Discovering hidden factors of variation in deep networks. arXiv preprint arXiv:1412.6583 ( 2014 ). Brian Cheung, Jesse A Livezey, Arjun K Bansal, and Bruno A Olshausen. 2014. Discovering hidden factors of variation in deep networks. arXiv preprint arXiv:1412.6583 (2014)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2006.262661"},{"key":"e_1_3_2_1_7_1","volume-title":"International Conference on Representation Learning","author":"Cohen Taco S","year":"2015","unstructured":"Taco S Cohen and Max Welling . 2015 . Transformation properties of learned visual representations . International Conference on Representation Learning (2015). Taco S Cohen and Max Welling. 2015. Transformation properties of learned visual representations. International Conference on Representation Learning (2015)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/B978-0-08-051584-7.50010-3"},{"key":"e_1_3_2_1_9_1","volume-title":"Matrix Information Geometry","author":"Dessein Arnaud","unstructured":"Arnaud Dessein , Arshia Cont , and Guillaume Lemaitre . 2013. Real-time detection of overlapping sound events with non-negative matrix factorization . In Matrix Information Geometry . Springer , 341--371. Arnaud Dessein, Arshia Cont, and Guillaume Lemaitre. 2013. Real-time detection of overlapping sound events with non-negative matrix factorization. In Matrix Information Geometry . Springer, 341--371."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502245"},{"key":"e_1_3_2_1_11_1","volume-title":"Signal Processing Conference . 506--510","author":"Gencoglu Oguzhan","year":"2010","unstructured":"Oguzhan Gencoglu , Tuomas Virtanen , and Heikki Huttunen . 2010 . Recognition of acoustic events using deep neural networks . In Signal Processing Conference . 506--510 . Oguzhan Gencoglu, Tuomas Virtanen, and Heikki Huttunen. 2010. Recognition of acoustic events using deep neural networks. In Signal Processing Conference . 506--510."},{"key":"e_1_3_2_1_12_1","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative adversarial nets. In Advances in Neural Information Processing Systems. 2672--2680.  Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative adversarial nets. In Advances in Neural Information Processing Systems. 2672--2680."},{"key":"e_1_3_2_1_13_1","volume-title":"Studies in Computational Intelligence","volume":"385","author":"Graves Alex","year":"2008","unstructured":"Alex Graves . 2008 . Supervised Sequence Labelling with Recurrent Neural Networks . Studies in Computational Intelligence , Vol. 385 (2008). Alex Graves. 2008. Supervised Sequence Labelling with Recurrent Neural Networks. Studies in Computational Intelligence , Vol. 385 (2008)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2005.1521503"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639360"},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Learning Representations .","author":"Higgins Irina","year":"2017","unstructured":"Irina Higgins , Loic Matthey , Arka Pal , Christopher Burgess , Xavier Glorot , Matthew Botvinick , Shakir Mohamed , and Alexander Lerchner . 2017 . Beta-vae: Learning basic visual concepts with a constrained variational framework . In International Conference on Learning Representations . Irina Higgins, Loic Matthey, Arka Pal, Christopher Burgess, Xavier Glorot, Matthew Botvinick, Shakir Mohamed, and Alexander Lerchner. 2017. Beta-vae: Learning basic visual concepts with a constrained variational framework. In International Conference on Learning Representations ."},{"key":"e_1_3_2_1_17_1","first-page":"2579","article-title":"Visualizing High-Dimensional Data Using t-SNE","volume":"9","author":"Hinton G","year":"2008","unstructured":"G Hinton . 2008 . Visualizing High-Dimensional Data Using t-SNE . Journal of Machnine Learning Research , Vol. 9 (2008), 2579 -- 2605 . G Hinton. 2008. Visualizing High-Dimensional Data Using t-SNE . Journal of Machnine Learning Research , Vol. 9 (2008), 2579--2605.","journal-title":"Journal of Machnine Learning Research"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.camwa.2012.03.077"},{"key":"e_1_3_2_1_19_1","unstructured":"Diederik P Kingma Tim Salimans Rafal Jozefowicz Xi Chen Ilya Sutskever and Max Welling. 2016. Improved variational inference with inverse autoregressive flow. In Advances in Neural Information Processing Systems. 4743--4751.  Diederik P Kingma Tim Salimans Rafal Jozefowicz Xi Chen Ilya Sutskever and Max Welling. 2016. Improved variational inference with inverse autoregressive flow. In Advances in Neural Information Processing Systems. 4743--4751."},{"key":"e_1_3_2_1_20_1","volume-title":"International Conference on Representation Learning","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Max Welling . 2014 . Auto-encoding variational bayes . International Conference on Representation Learning (2014). Diederik P Kingma and Max Welling. 2014. Auto-encoding variational bayes. International Conference on Representation Learning (2014)."},{"key":"e_1_3_2_1_21_1","volume-title":"Neuroevolution for Sound Event Detection in Real Life Audio: A Pilot Study. In Detection & Classification of Acoustic Scenes & Events Workshop .","author":"Kroos Christian","unstructured":"Christian Kroos and Mark D. Plumbley . 2017 . Neuroevolution for Sound Event Detection in Real Life Audio: A Pilot Study. In Detection & Classification of Acoustic Scenes & Events Workshop . Christian Kroos and Mark D. Plumbley. 2017. Neuroevolution for Sound Event Detection in Real Life Audio: A Pilot Study. In Detection & Classification of Acoustic Scenes & Events Workshop ."},{"key":"e_1_3_2_1_22_1","volume-title":"Deep Convolutional Inverse Graphics Network. In Conference and Workshop on Neural Information Processing Systems. 2539--2547","author":"Kulkarni Tejas D","year":"2015","unstructured":"Tejas D Kulkarni , William F. Whitney , Pushmeet Kohli , and Josh Tenenbaum . 2015 . Deep Convolutional Inverse Graphics Network. In Conference and Workshop on Neural Information Processing Systems. 2539--2547 . Tejas D Kulkarni, William F. Whitney, Pushmeet Kohli, and Josh Tenenbaum. 2015. Deep Convolutional Inverse Graphics Network. In Conference and Workshop on Neural Information Processing Systems. 2539--2547."},{"key":"e_1_3_2_1_23_1","volume-title":"Nature","volume":"401","author":"Lee Daniel D","year":"1999","unstructured":"Daniel D Lee and H Sebastian Seung . 1999 . Learning the parts of objects by non-negative matrix factorization . Nature , Vol. 401 , 6755 (1999), 788. Daniel D Lee and H Sebastian Seung. 1999. Learning the parts of objects by non-negative matrix factorization. Nature , Vol. 401, 6755 (1999), 788."},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning. 5670--5679","author":"Li Yingzhen","year":"2018","unstructured":"Yingzhen Li and Stephan Mandt . 2018 . Disentangled Sequential Autoencoder . In Proceedings of the 35th International Conference on Machine Learning. 5670--5679 . Yingzhen Li and Stephan Mandt. 2018. Disentangled Sequential Autoencoder. In Proceedings of the 35th International Conference on Machine Learning. 5670--5679."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning. 2995--3004","author":"Li Zhuohan","year":"2018","unstructured":"Zhuohan Li , Di He , Fei Tian , Wei Chen , Tao Qin , Liwei Wang , and Tieyan Liu . 2018 . Towards Binary-Valued Gates for Robust LS\u2122 Training . In Proceedings of the 35th International Conference on Machine Learning. 2995--3004 . Zhuohan Li, Di He, Fei Tian, Wei Chen, Tao Qin, Liwei Wang, and Tieyan Liu. 2018. Towards Binary-Valued Gates for Robust LS\u2122 Training. In Proceedings of the 35th International Conference on Machine Learning. 2995--3004."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2389618"},{"key":"e_1_3_2_1_27_1","volume-title":"DCASE 2017 challenge setup: Tasks, datasets and baseline system. In DCASE 2017-Workshop on Detection and Classification of Acoustic Scenes and Events .","author":"Mesaros Annamaria","year":"2017","unstructured":"Annamaria Mesaros , Toni Heittola , Aleksandr Diment , Benjamin Elizalde , Ankit Shah , Emmanuel Vincent , Bhiksha Raj , and Tuomas Virtanen . 2017 . DCASE 2017 challenge setup: Tasks, datasets and baseline system. In DCASE 2017-Workshop on Detection and Classification of Acoustic Scenes and Events . Annamaria Mesaros, Toni Heittola, Aleksandr Diment, Benjamin Elizalde, Ankit Shah, Emmanuel Vincent, Bhiksha Raj, and Tuomas Virtanen. 2017. DCASE 2017 challenge setup: Tasks, datasets and baseline system. In DCASE 2017-Workshop on Detection and Classification of Acoustic Scenes and Events ."},{"key":"e_1_3_2_1_28_1","volume-title":"Signal Processing Conference. IEEE, 1267--1271","author":"Mesaros Annamaria","year":"2010","unstructured":"Annamaria Mesaros , Toni Heittola , Antti Eronen , and Tuomas Virtanen . 2010 . Acoustic event detection in real life recordings . In Signal Processing Conference. IEEE, 1267--1271 . Annamaria Mesaros, Toni Heittola, Antti Eronen, and Tuomas Virtanen. 2010. Acoustic event detection in real life recordings. In Signal Processing Conference. IEEE, 1267--1271."},{"key":"e_1_3_2_1_29_1","unstructured":"Seongkyu Mun Suwon Shon Wooil Kim and Hanseok Ko. 2016. Deep Neural Network Bottleneck Features for Acoustic Event Recognition.. In INTERSPEECH . 2954--2957.  Seongkyu Mun Suwon Shon Wooil Kim and Hanseok Ko. 2016. Deep Neural Network Bottleneck Features for Acoustic Event Recognition.. In INTERSPEECH . 2954--2957."},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Acoustics, Speech, and Signal Processing","author":"Parascandolo Giambattista","year":"2016","unstructured":"Giambattista Parascandolo , Heikki Huttunen , and Tuomas Virtanen . 2016 . Recurrent neural networks for polyphonic sound event detection in real life recordings . International Conference on Acoustics, Speech, and Signal Processing (2016), 6440--6444. Giambattista Parascandolo, Heikki Huttunen, and Tuomas Virtanen. 2016. Recurrent neural networks for polyphonic sound event detection in real life recordings. International Conference on Acoustics, Speech, and Signal Processing (2016), 6440--6444."},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Machine Learning. 1431--1439","author":"Reed Scott","year":"2014","unstructured":"Scott Reed , Kihyuk Sohn , Yuting Zhang , and Honglak Lee . 2014 . Learning to disentangle factors of variation with manifold interaction . In International Conference on Machine Learning. 1431--1439 . Scott Reed, Kihyuk Sohn, Yuting Zhang, and Honglak Lee. 2014. Learning to disentangle factors of variation with manifold interaction. In International Conference on Machine Learning. 1431--1439."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10618-008-0091-4"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 30th International Conference on Machine Learning. 1058--1066","author":"Wan Li","year":"2013","unstructured":"Li Wan , Matthew Zeiler , Sixin Zhang , Yann Le Cun , and Rob Fergus . 2013 . Regularization of Neural Networks using DropConnect . In Proceedings of the 30th International Conference on Machine Learning. 1058--1066 . Li Wan, Matthew Zeiler, Sixin Zhang, Yann Le Cun, and Rob Fergus. 2013. Regularization of Neural Networks using DropConnect. In Proceedings of the 30th International Conference on Machine Learning. 1058--1066."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Yun Wang and Florian Metze. 2017. A Transfer Learning Based Feature Extractor for Polyphonic Sound Event Detection Using Connectionist Temporal Classification. In INTERSPEECH . 3097--3101.  Yun Wang and Florian Metze. 2017. A Transfer Learning Based Feature Extractor for Polyphonic Sound Event Detection Using Connectionist Temporal Classification. In INTERSPEECH . 3097--3101.","DOI":"10.21437\/Interspeech.2017-1469"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Xianjun Xia Roberto Togneri Ferdous Sohel and David Huang. 2017. Frame-Wise Dynamic Threshold Based Polyphonic Acoustic Event Detection. In INTERSPEECH .  Xianjun Xia Roberto Togneri Ferdous Sohel and David Huang. 2017. Frame-Wise Dynamic Threshold Based Polyphonic Acoustic Event Detection. In INTERSPEECH .","DOI":"10.21437\/Interspeech.2017-746"},{"key":"e_1_3_2_1_36_1","unstructured":"Jimei Yang Scott E Reed Minghsuan Yang and Honglak Lee. 2015. Weakly-supervised disentangling with recurrent transformations for 3d view synthesis. In Advances in Neural Information Processing Systems. 1099--1107.  Jimei Yang Scott E Reed Minghsuan Yang and Honglak Lee. 2015. Weakly-supervised disentangling with recurrent transformations for 3d view synthesis. In Advances in Neural Information Processing Systems. 1099--1107."},{"key":"e_1_3_2_1_37_1","volume-title":"Detecting sound events in basketball video archive. Department of Electrical Engineering","author":"Zhang Dongqing","year":"2001","unstructured":"Dongqing Zhang and Dan Ellis . 2001. Detecting sound events in basketball video archive. Department of Electrical Engineering , Columbia University , New York ( 2001 ). Dongqing Zhang and Dan Ellis. 2001. Detecting sound events in basketball video archive. Department of Electrical Engineering, Columbia University, New York (2001)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Matthias Zhrer and Franz Pernkopf. 2017. Virtual Adversarial Training and Data Augmentation for Acoustic Event Detection with Gated Recurrent Neural Networks. In INTERSPEECH. 493--497.  Matthias Zhrer and Franz Pernkopf. 2017. Virtual Adversarial Training and Data Augmentation for Acoustic Event Detection with Gated Recurrent Neural Networks. In INTERSPEECH. 493--497.","DOI":"10.21437\/Interspeech.2017-1238"}],"event":{"name":"MM '19: The 27th ACM International Conference on Multimedia","location":"Nice France","acronym":"MM '19","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 27th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3351086","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3351086","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3351086","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:12Z","timestamp":1750201992000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3351086"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,15]]},"references-count":38,"alternative-id":["10.1145\/3343031.3351086","10.1145\/3343031"],"URL":"https:\/\/doi.org\/10.1145\/3343031.3351086","relation":{},"subject":[],"published":{"date-parts":[[2019,10,15]]},"assertion":[{"value":"2019-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}