{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T06:44:48Z","timestamp":1764225888761,"version":"3.28.0"},"reference-count":43,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,4]]},"DOI":"10.1109\/icassp49357.2023.10095055","type":"proceedings-article","created":{"date-parts":[[2023,5,5]],"date-time":"2023-05-05T13:28:30Z","timestamp":1683293310000},"page":"1-5","source":"Crossref","is-referenced-by-count":2,"title":["Joint Modelling of Spoken Language Understanding Tasks with Integrated Dialog History"],"prefix":"10.1109","author":[{"given":"Siddhant","family":"Arora","sequence":"first","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hayato","family":"Futami","sequence":"additional","affiliation":[{"name":"Sony Group Corporation,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Emiru","family":"Tsunoo","sequence":"additional","affiliation":[{"name":"Sony Group Corporation,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Brian","family":"Yan","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-312"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2928"},{"key":"ref34","first-page":"8024","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume":"32","author":"paszke","year":"2019","journal-title":"Proc NeurIPS"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-5514"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414858"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1004"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3173"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2012-324"},{"key":"ref10","first-page":"1131","article-title":"Gated embeddings in e2e speech recognition for conversational-context fusion","author":"kim","year":"2019","journal-title":"Proc ACL"},{"key":"ref32","first-page":"6256","article-title":"End-to-end monaural multi-speaker asr system without pretraining","author":"chang","year":"2019","journal-title":"Proc ICASSP"},{"key":"ref2","first-page":"27","article-title":"Toward conversational human-computer interaction","volume":"22","author":"allen","year":"2001","journal-title":"AI Magazine"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2008.918413"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1460"},{"key":"ref39","first-page":"5998","article-title":"Attention is all you need","volume":"30","author":"vaswani","year":"2017","journal-title":"Proc NeurIPS"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1301"},{"article-title":"WavLM: Large-scale self-supervised pre-training for full stack speech processing","year":"2021","author":"chen","key":"ref38"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053247"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747871"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414594"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683109"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747674"},{"article-title":"Harpervalley-bank: A domain-specific spoken dialog corpus","year":"2020","author":"wu","key":"ref25"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10890"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2014.80"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414464"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9687942"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/2792745.2792775"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref27","first-page":"4171","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"Proc NAACL-HLT"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2444659"},{"article-title":"A context-based approach for dialogue act recognition using simple recurrent neural networks","year":"2018","author":"bothe","key":"ref8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6259"},{"article-title":"Dialogue act classification with context-aware self-attention","year":"2019","author":"raheja","key":"ref9"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-788"},{"article-title":"Tie your embeddings down: Cross-modal latent spaces for end-to-end spoken language understanding","year":"2020","author":"agrawal","key":"ref3"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853573"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639291"},{"key":"ref40","first-page":"449","article-title":"A comparative study on transformer vs RNN in speech applications","author":"karita","year":"2019","journal-title":"Proc ASRU"}],"event":{"name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2023,6,4]]},"location":"Rhodes Island, Greece","end":{"date-parts":[[2023,6,10]]}},"container-title":["ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10094559\/10094560\/10095055.pdf?arnumber=10095055","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,20]],"date-time":"2023-11-20T13:57:01Z","timestamp":1700488621000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10095055\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,4]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/icassp49357.2023.10095055","relation":{},"subject":[],"published":{"date-parts":[[2023,6,4]]}}}