{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:22:43Z","timestamp":1750220563058,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":20,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,3,5]],"date-time":"2021-03-05T00:00:00Z","timestamp":1614902400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,3,5]]},"DOI":"10.1145\/3461353.3461377","type":"proceedings-article","created":{"date-parts":[[2021,9,6]],"date-time":"2021-09-06T17:21:15Z","timestamp":1630948875000},"page":"196-201","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Soft-Gated Self-Supervision Network for Action Reasoning"],"prefix":"10.1145","author":[{"given":"Shengli","family":"Wang","sequence":"first","affiliation":[{"name":"Northwestern Polytechnical University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lifang","family":"Wang","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Gu","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shichang","family":"He","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huijuan","family":"Hao","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meiling","family":"Yao","sequence":"additional","affiliation":[{"name":"Beijing University of Chemical Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,9,4]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"64","article-title":"ConvLab","volume":"3","author":"Lee Sungjin","year":"2019","journal-title":"Multi-Domain End-to-End Dialog System Platform. ACL"},{"volume-title":"Steve J. Young:Sample-efficient Actor-Critic Reinforcement Learning with Supervised Data for Dialogue Management. SIGDIAL Conference 2017:  147-157","author":"Su Pei-Hao","key":"e_1_3_2_1_2_1"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2998772"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2996946"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3017164"},{"key":"e_1_3_2_1_6_1","first-page":"120","article-title":"Multi-doma","volume":"2016","author":"Wen Tsung-Hsien","journal-title":"Neural Network Language Generation for Spoken Dialogue Systems. HLT-NAACL"},{"volume-title":"Steve J. Young:Semantically Conditioned LSTM-based Natural Language Generation for Spoken Dialogue Systems. EMNLP 2015:  1711-1721","author":"Wen Tsung-Hsien","key":"e_1_3_2_1_7_1"},{"key":"e_1_3_2_1_8_1","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Alex Graves Ioannis Antonoglou Daan Wierstra Martin A. Riedmiller:Playing Atari with Deep Reinforcement Learning. CoRR abs\/1312.5602 (2013)  Volodymyr Mnih Koray Kavukcuoglu David Silver Alex Graves Ioannis Antonoglou Daan Wierstra Martin A. Riedmiller:Playing Atari with Deep Reinforcement Learning. CoRR abs\/1312.5602 (2013)"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_1_10_1","first-page":"2182","article-title":"Deep Dyna-Q","volume":"1","author":"Peng Baolin","year":"2018","journal-title":"Integrating Planning for Task-Completion Dialogue Policy Learning. ACL"},{"key":"e_1_3_2_1_11_1","unstructured":"Pei-Hao Su Milica Gasic Nikola Mrksic Lina Maria Rojas-Barahona Stefan Ultes David Vandyke Tsung-Hsien Wen Steve J. Young:On-line Active Reward Learning for Policy Optimisation in Spoken Dialogue Systems. ACL ( (1) 2016  Pei-Hao Su Milica Gasic Nikola Mrksic Lina Maria Rojas-Barahona Stefan Ultes David Vandyke Tsung-Hsien Wen Steve J. Young:On-line Active Reward Learning for Policy Optimisation in Spoken Dialogue Systems. ACL (1) 2016"},{"volume-title":"Tony Jebara:Subgoal Discovery for Hierarchical Dialogue Policy Learning. EMNLP 2018:  2298-2309","author":"Tang Da","key":"e_1_3_2_1_12_1"},{"key":"e_1_3_2_1_13_1","unstructured":"Vladimir Vlasov Johannes E. M. Mosig Alan Nichol:Dialogue Transformers. CoRR abs\/1910.00486 (2019)  Vladimir Vlasov Johannes E. M. Mosig Alan Nichol:Dialogue Transformers. CoRR abs\/1910.00486 (2019)"},{"key":"e_1_3_2_1_14_1","unstructured":"Vladimir Vlasov Akela Drissner-Schmid Alan Nichol:Few-Shot Generalization Across Dialogue Tasks. CoRR abs\/1811.11707 (2018)  Vladimir Vlasov Akela Drissner-Schmid Alan Nichol:Few-Shot Generalization Across Dialogue Tasks. CoRR abs\/1811.11707 (2018)"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_16_1","first-page":"1468","article-title":"Mem2Seq: Effectively Incorporating Knowledge Bases into End-to-End Task-Oriented Dialog Systems","volume":"1","author":"Madotto Andrea","year":"2018","journal-title":"ACL"},{"key":"e_1_3_2_1_17_1","unstructured":"Stephen Merity Caiming Xiong James Bradbury Richard Socher:Pointer Sentinel Mixture Models. ICLR (Poster) 2017  Stephen Merity Caiming Xiong James Bradbury Richard Socher:Pointer Sentinel Mixture Models. ICLR (Poster) 2017"},{"key":"e_1_3_2_1_18_1","first-page":"4171","article-title":"BERT","volume":"1","author":"Devlin Jacob","year":"2019","journal-title":"Pre-training of Deep Bidirectional Transformers for Language Understanding. NAACL-HLT"},{"key":"e_1_3_2_1_19_1","first-page":"7289","article-title":"Switch-Based Active Deep Dyna-Q","volume":"2019","author":"Wu Yuexin","journal-title":"Efficient Adaptive Planning for Task-Completion Dialogue Policy Learning. AAAI"},{"key":"e_1_3_2_1_20_1","first-page":"100","article-title":"Guided Dialog Policy Learning","volume":"1","author":"Takanobu Ryuichi","year":"2019","journal-title":"Reward Estimation for Multi-Domain Task-Oriented Dialog. EMNLP\/IJCNLP"}],"event":{"name":"ICIAI 2021: 2021 the 5th International Conference on Innovation in Artificial Intelligence","acronym":"ICIAI 2021","location":"Xia men China"},"container-title":["2021 the 5th International Conference on Innovation in Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3461353.3461377","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3461353.3461377","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:28:35Z","timestamp":1750195715000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3461353.3461377"}},"subtitle":["Soft-Gated Self-Supervision Network with Attention Mechanism and Joint Multi-Task Training Strategy for Action Reasoning"],"short-title":[],"issued":{"date-parts":[[2021,3,5]]},"references-count":20,"alternative-id":["10.1145\/3461353.3461377","10.1145\/3461353"],"URL":"https:\/\/doi.org\/10.1145\/3461353.3461377","relation":{},"subject":[],"published":{"date-parts":[[2021,3,5]]},"assertion":[{"value":"2021-09-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}