{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T19:09:38Z","timestamp":1783537778467,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,2,14]],"date-time":"2022-02-14T00:00:00Z","timestamp":1644796800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,2,14]]},"DOI":"10.1145\/3511616.3513104","type":"proceedings-article","created":{"date-parts":[[2022,3,22]],"date-time":"2022-03-22T00:21:13Z","timestamp":1647908473000},"page":"96-105","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["A Novel Policy for Pre-trained Deep Reinforcement Learning for Speech Emotion Recognition"],"prefix":"10.1145","author":[{"given":"Thejan","family":"Rajapakshe","sequence":"first","affiliation":[{"name":"University of Southern Queensland, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rajib","family":"Rana","sequence":"additional","affiliation":[{"name":"University of Southern Queensland, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sara","family":"Khalifa","sequence":"additional","affiliation":[{"name":"Distributed Sensing Systems Group, Data61, CSIRO, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiajun","family":"Liu","sequence":"additional","affiliation":[{"name":"Distributed Sensing Systems Group, Data61, CSIRO, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bjorn","family":"Schuller","sequence":"additional","affiliation":[{"name":"GLAM Group on Language, Audio &amp; Music, Imperial College London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,3,21]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg\u00a0S Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Shlens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/  Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg\u00a0S Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Shlens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"e_1_3_2_1_3_1","volume-title":"Learning from demonstration (programming by demonstration). Encyclopedia of Robotics(2018), 1\u20138","author":"Calinon Sylvain","unstructured":"Sylvain Calinon . 2018. Learning from demonstration (programming by demonstration). Encyclopedia of Robotics(2018), 1\u20138 . Sylvain Calinon. 2018. Learning from demonstration (programming by demonstration). Encyclopedia of Robotics(2018), 1\u20138."},{"key":"e_1_3_2_1_4_1","volume-title":"Boltzmann exploration done right(NIPS\u201917)","author":"Cesa-Bianchi Nicol\u00f2","unstructured":"Nicol\u00f2 Cesa-Bianchi , Claudio Gentile , G\u00e1bor Lugosi , and Gergely Neu . 2017. Boltzmann exploration done right(NIPS\u201917) . Curran Associates Inc ., 6287\u20136296. Nicol\u00f2 Cesa-Bianchi, Claudio Gentile, G\u00e1bor Lugosi, and Gergely Neu. 2017. Boltzmann exploration done right(NIPS\u201917). Curran Associates Inc., 6287\u20136296."},{"key":"e_1_3_2_1_5_1","unstructured":"Fran\u00e7ois Chollet and others. 2015. Keras. https:\/\/keras.io  Fran\u00e7ois Chollet and others. 2015. Keras. https:\/\/keras.io"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN48605.2020.9207023"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Gabriel V de\u00a0la Cruz\u00a0Jr Yunshu Du and Matthew\u00a0E Taylor. 2019. Pre-training Neural Networks with Human Demonstrations for Deep Reinforcement Learning. In Adaptive Learning Agents (ALA). http:\/\/arxiv.org\/abs\/1709.04083  Gabriel V de\u00a0la Cruz\u00a0Jr Yunshu Du and Matthew\u00a0E Taylor. 2019. Pre-training Neural Networks with Human Demonstrations for Deep Reinforcement Learning. In Adaptive Learning Agents (ALA). http:\/\/arxiv.org\/abs\/1709.04083","DOI":"10.1017\/S0269888919000055"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1980.1163420"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.21437\/SMM.2018-5"},{"key":"e_1_3_2_1_10_1","unstructured":"Rasool Fakoor Xiaodong He Ivan Tashev and Shuayb Zarar. 2018. Reinforcement Learning To Adapt Speech Enhancement to Instantaneous Input Signal Quality. arXiv:1711.10791 [cs](2018). http:\/\/arxiv.org\/abs\/1711.10791  Rasool Fakoor Xiaodong He Ivan Tashev and Shuayb Zarar. 2018. Reinforcement Learning To Adapt Speech Enhancement to Instantaneous Input Signal Quality. arXiv:1711.10791 [cs](2018). http:\/\/arxiv.org\/abs\/1711.10791"},{"key":"e_1_3_2_1_11_1","volume-title":"A Theoretical Analysis of Deep Q-Learning. arXiv (1","author":"Fan Jianqing","year":"2019","unstructured":"Jianqing Fan , Zhaoran Wang , Yuchen Xie , and Zhuoran Yang . 2019. A Theoretical Analysis of Deep Q-Learning. arXiv (1 2019 ). http:\/\/arxiv.org\/abs\/1901.00137 Jianqing Fan, Zhaoran Wang, Yuchen Xie, and Zhuoran Yang. 2019. A Theoretical Analysis of Deep Q-Learning. arXiv (1 2019). http:\/\/arxiv.org\/abs\/1901.00137"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3264869.3264875"},{"key":"e_1_3_2_1_13_1","volume-title":"Interspeech","author":"Han Kun","year":"2014","unstructured":"Kun Han , Dong Yu , and Ivan Tashev . 2014. Speech Emotion Recognition Using Deep Neural Network and Extreme Learning Machine . In Interspeech 2014 . Kun Han, Dong Yu, and Ivan Tashev. 2014. Speech Emotion Recognition Using Deep Neural Network and Extreme Learning Machine. In Interspeech 2014."},{"key":"e_1_3_2_1_14_1","unstructured":"S Haq P\u00a0J\u00a0B Jackson and J Edge. 2008. Audio-visual feature selection and reduction for emotion classification.  S Haq P\u00a0J\u00a0B Jackson and J Edge. 2008. Audio-visual feature selection and reduction for emotion classification."},{"key":"e_1_3_2_1_15_1","unstructured":"Todd Hester Matej Vecerik Olivier Pietquin Marc Lanctot Tom Schaul Bilal Piot Dan Horgan John Quan Andrew Sendonaris Gabriel Dulac-Arnold Ian Osband John Agapiou Joel\u00a0Z Leibo and Audrunas Gruslys. 2017. Deep Q-learning from Demonstrations. arXiv:1704.03732 [cs](2017). http:\/\/arxiv.org\/abs\/1704.03732  Todd Hester Matej Vecerik Olivier Pietquin Marc Lanctot Tom Schaul Bilal Piot Dan Horgan John Quan Andrew Sendonaris Gabriel Dulac-Arnold Ian Osband John Agapiou Joel\u00a0Z Leibo and Audrunas Gruslys. 2017. Deep Q-learning from Demonstrations. arXiv:1704.03732 [cs](2017). http:\/\/arxiv.org\/abs\/1704.03732"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings AAAI.","author":"Hester Todd","year":"2018","unstructured":"Todd Hester , Matej Vecerik , Olivier Pietquin , Marc Lanctot , Tom Schaul , Bilal Piot , Dan Horgan , John Quan , Andrew Sendonaris , Ian Osband , 2018 . Deep q-learning from demonstrations . In Proceedings AAAI. Todd Hester, Matej Vecerik, Olivier Pietquin, Marc Lanctot, Tom Schaul, Bilal Piot, Dan Horgan, John Quan, Andrew Sendonaris, Ian Osband, 2018. Deep q-learning from demonstrations. In Proceedings AAAI."},{"key":"e_1_3_2_1_17_1","volume-title":"IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings, Vol.\u00a02019-May. Institute of Electrical and Electronics Engineers Inc., 5866\u20135870","author":"Huang Kun\u00a0Yi","year":"2019","unstructured":"Kun\u00a0Yi Huang , Chung\u00a0Hsien Wu , Qian\u00a0Bei Hong , Ming\u00a0Hsiang Su , and Yi\u00a0Hsuan Chen . 2019 . Speech Emotion Recognition Using Deep Neural Network Considering Verbal and Nonverbal Speech Sounds. In ICASSP , IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings, Vol.\u00a02019-May. Institute of Electrical and Electronics Engineers Inc., 5866\u20135870 . https:\/\/doi.org\/10.1109\/ICASSP.2019.8682283 Kun\u00a0Yi Huang, Chung\u00a0Hsien Wu, Qian\u00a0Bei Hong, Ming\u00a0Hsiang Su, and Yi\u00a0Hsuan Chen. 2019. Speech Emotion Recognition Using Deep Neural Network Considering Verbal and Nonverbal Speech Sounds. In ICASSP, IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings, Vol.\u00a02019-May. Institute of Electrical and Electronics Engineers Inc., 5866\u20135870. https:\/\/doi.org\/10.1109\/ICASSP.2019.8682283"},{"key":"e_1_3_2_1_18_1","volume-title":"Speech emotion recognition with deep convolutional neural networks. Biomedical Signal Processing and Control 59 (5","author":"Issa Dias","year":"2020","unstructured":"Dias Issa , M. Fatih\u00a0Demirci , and Adnan Yazici . 2020. Speech emotion recognition with deep convolutional neural networks. Biomedical Signal Processing and Control 59 (5 2020 ), 101894. https:\/\/doi.org\/10.1016\/j.bspc.2020.101894 Dias Issa, M. Fatih\u00a0Demirci, and Adnan Yazici. 2020. Speech emotion recognition with deep convolutional neural networks. Biomedical Signal Processing and Control 59 (5 2020), 101894. https:\/\/doi.org\/10.1016\/j.bspc.2020.101894"},{"key":"e_1_3_2_1_19_1","unstructured":"Vitaly Kurin Sebastian Nowozin Katja Hofmann Lucas Beyer and Bastian Leibe. 2017. The atari grand challenge dataset. arXiv1705.10998(2017).  Vitaly Kurin Sebastian Nowozin Katja Hofmann Lucas Beyer and Bastian Leibe. 2017. The atari grand challenge dataset. arXiv1705.10998(2017)."},{"key":"e_1_3_2_1_20_1","volume-title":"Reinforcement learning as classification: leveraging modern classifiers(ICML\u201903)","author":"Lagoudakis G","unstructured":"Michail\u00a0 G Lagoudakis and Ronald Parr . 2003. Reinforcement learning as classification: leveraging modern classifiers(ICML\u201903) . AAAI Press , 424\u2013431. Michail\u00a0G Lagoudakis and Ronald Parr. 2003. Reinforcement learning as classification: leveraging modern classifiers(ICML\u201903). AAAI Press, 424\u2013431."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8461058"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Siddique Latif Rajib Rana Sara Khalifa Raja Jurdak and Julien Epps. 2019. Direct Modelling of Speech Emotion from Raw Speech. http:\/\/arxiv.org\/abs\/1904.03833  Siddique Latif Rajib Rana Sara Khalifa Raja Jurdak and Julien Epps. 2019. Direct Modelling of Speech Emotion from Raw Speech. http:\/\/arxiv.org\/abs\/1904.03833","DOI":"10.21437\/Interspeech.2019-3252"},{"key":"e_1_3_2_1_23_1","unstructured":"Siddique Latif Rajib Rana Sara Khalifa Raja Jurdak Junaid Qadir and Bj\u00f6rn\u00a0W Schuller. 2020. Deep Representation Learning in Speech Processing: Challenges Recent Advances and Future Trends. arXiv:2001.00378 [cs eess](2020). http:\/\/arxiv.org\/abs\/2001.00378  Siddique Latif Rajib Rana Sara Khalifa Raja Jurdak Junaid Qadir and Bj\u00f6rn\u00a0W Schuller. 2020. Deep Representation Learning in Speech Processing: Challenges Recent Advances and Future Trends. arXiv:2001.00378 [cs eess](2020). http:\/\/arxiv.org\/abs\/2001.00378"},{"key":"e_1_3_2_1_24_1","volume-title":"Adversarial Machine Learning And Speech Emotion Recognition: Utilizing Generative Adversarial Networks For Robustness. arXiv (11","author":"Latif Siddique","year":"2018","unstructured":"Siddique Latif , Rajib Rana , and Junaid Qadir . 2018. Adversarial Machine Learning And Speech Emotion Recognition: Utilizing Generative Adversarial Networks For Robustness. arXiv (11 2018 ). http:\/\/arxiv.org\/abs\/1811.11402 Siddique Latif, Rajib Rana, and Junaid Qadir. 2018. Adversarial Machine Learning And Speech Emotion Recognition: Utilizing Generative Adversarial Networks For Robustness. arXiv (11 2018). http:\/\/arxiv.org\/abs\/1811.11402"},{"key":"e_1_3_2_1_25_1","unstructured":"Siddique Latif Rajib Rana Shahzad Younis Junaid Qadir and Julien Epps. 2018. Cross Corpus Speech Emotion Classification - An Effective Transfer Learning Technique. (2018).  Siddique Latif Rajib Rana Shahzad Younis Junaid Qadir and Julien Epps. 2018. Cross Corpus Speech Emotion Classification - An Effective Transfer Learning Technique. (2018)."},{"key":"e_1_3_2_1_26_1","volume-title":"Model-Based Regularization for Deep Reinforcement Learning with Transcoder Networks. arXiv (9","author":"Leibfried Felix","year":"2018","unstructured":"Felix Leibfried and Peter Vrancx . 2018. Model-Based Regularization for Deep Reinforcement Learning with Transcoder Networks. arXiv (9 2018 ). http:\/\/arxiv.org\/abs\/1809.01906 Felix Leibfried and Peter Vrancx. 2018. Model-Based Regularization for Deep Reinforcement Learning with Transcoder Networks. arXiv (9 2018). http:\/\/arxiv.org\/abs\/1809.01906"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2014.7078570"},{"key":"e_1_3_2_1_28_1","volume-title":"Matt McVicar, Eric Battenberg, and Oriol Nieto.","author":"McFee Brian","year":"2015","unstructured":"Brian McFee , Colin Raffel , Dawen Liang , Daniel P\u00a0W Ellis , Matt McVicar, Eric Battenberg, and Oriol Nieto. 2015 . librosa: Audio and music signal analysis in python, Vol .\u00a08. Brian McFee, Colin Raffel, Dawen Liang, Daniel P\u00a0W Ellis, Matt McVicar, Eric Battenberg, and Oriol Nieto. 2015. librosa: Audio and music signal analysis in python, Vol.\u00a08."},{"key":"e_1_3_2_1_29_1","volume-title":"A breakthrough in Speech emotion recognition using Deep Retinal Convolution Neural Networks. arXiv (7","author":"Niu Yafeng","year":"2017","unstructured":"Yafeng Niu , Dongsheng Zou , Yadong Niu , Zhongshi He , and Hua Tan . 2017. A breakthrough in Speech emotion recognition using Deep Retinal Convolution Neural Networks. arXiv (7 2017 ). http:\/\/arxiv.org\/abs\/1707.09917 Yafeng Niu, Dongsheng Zou, Yadong Niu, Zhongshi He, and Hua Tan. 2017. A breakthrough in Speech emotion recognition using Deep Retinal Convolution Neural Networks. arXiv (7 2017). http:\/\/arxiv.org\/abs\/1707.09917"},{"key":"e_1_3_2_1_30_1","volume-title":"Proc. Dialog-on-Dialog Workshop, Interspeech.","author":"Paek Tim","year":"2006","unstructured":"Tim Paek . 2006 . Reinforcement learning for spoken dialogue systems: Comparing strengths and weaknesses for practical deployment . In Proc. Dialog-on-Dialog Workshop, Interspeech. Tim Paek. 2006. Reinforcement learning for spoken dialogue systems: Comparing strengths and weaknesses for practical deployment. In Proc. Dialog-on-Dialog Workshop, Interspeech."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/276"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/5326.983933"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1111\/ecc.13033"},{"key":"e_1_3_2_1_34_1","volume-title":"Reinforcement Learning Based Speech Enhancement for Robust Speech Recognition. In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 6750\u20136754","author":"Shen Y","year":"2019","unstructured":"Y Shen , C Huang , S Wang , Y Tsao , H Wang , and T Chi . 2019 . Reinforcement Learning Based Speech Enhancement for Robust Speech Recognition. In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 6750\u20136754 . https:\/\/doi.org\/10.1109\/ICASSP.2019.8683648 Y Shen, C Huang, S Wang, Y Tsao, H Wang, and T Chi. 2019. Reinforcement Learning Based Speech Enhancement for Robust Speech Recognition. In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 6750\u20136754. https:\/\/doi.org\/10.1109\/ICASSP.2019.8683648"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"David Silver Aja Huang Chris\u00a0J Maddison Arthur Guez Laurent Sifre George van\u00a0den Driessche Julian Schrittwieser Ioannis Antonoglou Veda Panneershelvam Marc Lanctot Sander Dieleman Dominik Grewe John Nham Nal Kalchbrenner Ilya Sutskever Timothy Lillicrap Madeleine Leach Koray Kavukcuoglu Thore Graepel and Demis Hassabis. 2016. Mastering the game of Go with deep neural networks and tree search. Nature 529 7587 (2016) 484\u2013489. https:\/\/doi.org\/10.1038\/nature16961  David Silver Aja Huang Chris\u00a0J Maddison Arthur Guez Laurent Sifre George van\u00a0den Driessche Julian Schrittwieser Ioannis Antonoglou Veda Panneershelvam Marc Lanctot Sander Dieleman Dominik Grewe John Nham Nal Kalchbrenner Ilya Sutskever Timothy Lillicrap Madeleine Leach Koray Kavukcuoglu Thore Graepel and Demis Hassabis. 2016. Mastering the game of Go with deep neural networks and tree search. Nature 529 7587 (2016) 484\u2013489. https:\/\/doi.org\/10.1038\/nature16961","DOI":"10.1038\/nature16961"},{"key":"e_1_3_2_1_36_1","unstructured":"Satinder\u00a0P Singh Michael\u00a0J Kearns Diane\u00a0J Litman and Marilyn\u00a0A Walker. 1999. Reinforcement learning for spoken dialogue systems.. In Nips. 956\u2013962.  Satinder\u00a0P Singh Michael\u00a0J Kearns Diane\u00a0J Litman and Marilyn\u00a0A Walker. 1999. Reinforcement learning for spoken dialogue systems.. In Nips. 956\u2013962."},{"key":"e_1_3_2_1_37_1","volume-title":"Reinforcement Learning of Listener Response for Mood Classification of Audio. In 2009 International Conference on Computational Science and Engineering, Vol.\u00a04. 849\u2013853","author":"Stockholm J","year":"2009","unstructured":"J Stockholm and P Pasquier . 2009 . Reinforcement Learning of Listener Response for Mood Classification of Audio. In 2009 International Conference on Computational Science and Engineering, Vol.\u00a04. 849\u2013853 . https:\/\/doi.org\/10.1109\/CSE.2009.184 J Stockholm and P Pasquier. 2009. Reinforcement Learning of Listener Response for Mood Classification of Audio. In 2009 International Conference on Computational Science and Engineering, Vol.\u00a04. 849\u2013853. https:\/\/doi.org\/10.1109\/CSE.2009.184"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638959"},{"key":"e_1_3_2_1_39_1","volume-title":"Multi-Modal Emotion recognition on IEMOCAP Dataset using Deep Learning. (4","author":"Tripathi Samarth","year":"2018","unstructured":"Samarth Tripathi , Sarthak Tripathi , and Homayoon Beigi . 2018. Multi-Modal Emotion recognition on IEMOCAP Dataset using Deep Learning. (4 2018 ). http:\/\/arxiv.org\/abs\/1804.05788 Samarth Tripathi, Sarthak Tripathi, and Homayoon Beigi. 2018. Multi-Modal Emotion recognition on IEMOCAP Dataset using Deep Learning. (4 2018). http:\/\/arxiv.org\/abs\/1804.05788"},{"key":"e_1_3_2_1_40_1","volume-title":"Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575, 7782","author":"Vinyals Oriol","year":"2019","unstructured":"Oriol Vinyals , Igor Babuschkin , Wojciech\u00a0 M Czarnecki , Micha\u00ebl Mathieu , Andrew Dudzik , Junyoung Chung , David\u00a0 H Choi , Richard Powell , Timo Ewalds , Petko Georgiev , Junhyuk Oh , Dan Horgan , Manuel Kroiss , Ivo Danihelka , Aja Huang , Laurent Sifre , Trevor Cai , John\u00a0 P Agapiou , Max Jaderberg , Alexander\u00a0 S Vezhnevets , R\u00e9mi Leblond , Tobias Pohlen , Valentin Dalibard , David Budden , Yury Sulsky , James Molloy , Tom\u00a0 L Paine , Caglar Gulcehre , Ziyu Wang , Tobias Pfaff , Yuhuai Wu , Roman Ring , Dani Yogatama , Dario W\u00fcnsch , Katrina McKinney , Oliver Smith , Tom Schaul , Timothy Lillicrap , Koray Kavukcuoglu , Demis Hassabis , Chris Apps , and David Silver . 2019. Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575, 7782 ( 2019 ), 350\u2013354. https:\/\/doi.org\/10.1038\/s41586-019-1724-z Oriol Vinyals, Igor Babuschkin, Wojciech\u00a0M Czarnecki, Micha\u00ebl Mathieu, Andrew Dudzik, Junyoung Chung, David\u00a0H Choi, Richard Powell, Timo Ewalds, Petko Georgiev, Junhyuk Oh, Dan Horgan, Manuel Kroiss, Ivo Danihelka, Aja Huang, Laurent Sifre, Trevor Cai, John\u00a0P Agapiou, Max Jaderberg, Alexander\u00a0S Vezhnevets, R\u00e9mi Leblond, Tobias Pohlen, Valentin Dalibard, David Budden, Yury Sulsky, James Molloy, Tom\u00a0L Paine, Caglar Gulcehre, Ziyu Wang, Tobias Pfaff, Yuhuai Wu, Roman Ring, Dani Yogatama, Dario W\u00fcnsch, Katrina McKinney, Oliver Smith, Tom Schaul, Timothy Lillicrap, Koray Kavukcuoglu, Demis Hassabis, Chris Apps, and David Silver. 2019. Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575, 7782 (2019), 350\u2013354. https:\/\/doi.org\/10.1038\/s41586-019-1724-z"},{"key":"e_1_3_2_1_41_1","volume-title":"Starcraft ii: A new challenge for reinforcement learning. arXiv","author":"Vinyals Oriol","year":"2017","unstructured":"Oriol Vinyals , Timo Ewalds , Sergey Bartunov , Petko Georgiev , Alexander\u00a0Sasha Vezhnevets , Michelle Yeo , Alireza Makhzani , Heinrich K\u00fcttler , John Agapiou , Julian Schrittwieser , 2017. Starcraft ii: A new challenge for reinforcement learning. arXiv 2017 , 1708.04782 (2017). Oriol Vinyals, Timo Ewalds, Sergey Bartunov, Petko Georgiev, Alexander\u00a0Sasha Vezhnevets, Michelle Yeo, Alireza Makhzani, Heinrich K\u00fcttler, John Agapiou, Julian Schrittwieser, 2017. Starcraft ii: A new challenge for reinforcement learning. arXiv 2017, 1708.04782 (2017)."},{"key":"e_1_3_2_1_42_1","volume-title":"Learning from Delayed Rewards. Ph.\u00a0D","author":"Cornish\u00a0Hellaby Watkins Christopher John","unstructured":"Christopher John Cornish\u00a0Hellaby Watkins . 1989. Learning from Delayed Rewards. Ph.\u00a0D . Dissertation . Cambridge, UK . Christopher John Cornish\u00a0Hellaby Watkins. 1989. Learning from Delayed Rewards. Ph.\u00a0D. Dissertation. Cambridge, UK."},{"key":"e_1_3_2_1_43_1","unstructured":"Marco Wiering. 1999. Explorations in Efficient Reinforcement Learning. Ph.\u00a0D. Dissertation. https:\/\/dare.uva.nl\/search?identifier=6ac07651-85ee-4c7b-9cab-86eea5b818f4  Marco Wiering. 1999. Explorations in Efficient Reinforcement Learning. Ph.\u00a0D. Dissertation. https:\/\/dare.uva.nl\/search?identifier=6ac07651-85ee-4c7b-9cab-86eea5b818f4"},{"key":"e_1_3_2_1_44_1","volume-title":"Proc. NIPS Workshop on Deep Learning and Unsupervised Feature Learning.","author":"Yu Dong","year":"2010","unstructured":"Dong Yu , Li Deng , and George Dahl . 2010 . Roles of pre-training and fine-tuning in context-dependent DBN-HMMs for real-world speech recognition . In Proc. NIPS Workshop on Deep Learning and Unsupervised Feature Learning. Dong Yu, Li Deng, and George Dahl. 2010. Roles of pre-training and fine-tuning in context-dependent DBN-HMMs for real-world speech recognition. In Proc. NIPS Workshop on Deep Learning and Unsupervised Feature Learning."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICNSC.2019.8743211"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-17313-4_1"}],"event":{"name":"ACSW 2022: Australasian Computer Science Week 2022","location":"Brisbane Australia","acronym":"ACSW 2022"},"container-title":["Australasian Computer Science Week 2022"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3511616.3513104","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3511616.3513104","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:11:47Z","timestamp":1750191107000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3511616.3513104"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,2,14]]},"references-count":46,"alternative-id":["10.1145\/3511616.3513104","10.1145\/3511616"],"URL":"https:\/\/doi.org\/10.1145\/3511616.3513104","relation":{},"subject":[],"published":{"date-parts":[[2022,2,14]]},"assertion":[{"value":"2022-03-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}