{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T23:16:03Z","timestamp":1784675763298,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T00:00:00Z","timestamp":1720396800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,8]]},"DOI":"10.1145\/3640794.3665573","type":"proceedings-article","created":{"date-parts":[[2024,7,7]],"date-time":"2024-07-07T06:24:56Z","timestamp":1720333496000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Advancing Faithfulness of Large Language Models in Goal-Oriented Dialogue Question Answering"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-9687-6763","authenticated-orcid":false,"given":"Abigail","family":"Sticha","sequence":"first","affiliation":[{"name":"Engineering Department, University of Cambridge, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1388-440X","authenticated-orcid":false,"given":"Norbert","family":"Braunschweiler","sequence":"additional","affiliation":[{"name":"Cambridge Research Laboratory, Toshiba Europe Limited, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1061-9512","authenticated-orcid":false,"given":"Rama Sanand","family":"Doddipatla","sequence":"additional","affiliation":[{"name":"Speech Technology Group, Toshiba Europe Limited, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1292-2769","authenticated-orcid":false,"given":"Kate M","family":"Knill","sequence":"additional","affiliation":[{"name":"Engineering Department, University of Cambridge, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,8]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","unstructured":"K. Adnan and R. Akbar. 2019. An analytical study of information extraction from unstructured and multidimensional big data.Journal of Big Data 6 91 (2019). https:\/\/doi.org\/10.1186\/s40537-019-0254-8","DOI":"10.1186\/s40537-019-0254-8"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 1st Workshop on Taming Large Language Models: Controllability in the era of Interactive Assistants!Association for Computational Linguistics, Prague, Czechia.","author":"Braunschweiler Norbert","year":"2023","unstructured":"Norbert Braunschweiler, Rama Doddipatla, Simon Keizer, and Svetlana Stoyanchev. 2023. Evaluating Large Language Models for Document-grounded Response Generation in Information-Seeking Dialogues. In Proceedings of the 1st Workshop on Taming Large Language Models: Controllability in the era of Interactive Assistants!Association for Computational Linguistics, Prague, Czechia."},{"key":"e_1_3_2_1_3_1","volume-title":"Language models are few-shot learners. Advances in neural information processing systems 33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared\u00a0D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020), 1877\u20131901."},{"key":"e_1_3_2_1_4_1","volume-title":"UPRISE: Universal Prompt Retrieval for Improving Zero-Shot Evaluation. arXiv preprint arXiv:2303.08518","author":"Cheng Daixuan","year":"2023","unstructured":"Daixuan Cheng, Shaohan Huang, Junyu Bi, Yuefeng Zhan, Jianfeng Liu, Yujing Wang, Hao Sun, Furu Wei, Denvy Deng, and Qi Zhang. 2023. UPRISE: Universal Prompt Retrieval for Improving Zero-Shot Evaluation. arXiv preprint arXiv:2303.08518 (2023). https:\/\/arxiv.org\/abs\/2303.08518"},{"key":"e_1_3_2_1_5_1","volume-title":"QuAC: Question answering in context. arXiv preprint arXiv:1808.07036","author":"Choi Eunsol","year":"2018","unstructured":"Eunsol Choi, He He, Mohit Iyyer, Mark Yatskar, Wen-tau Yih, Yejin Choi, Percy Liang, and Luke Zettlemoyer. 2018. QuAC: Question answering in context. arXiv preprint arXiv:1808.07036 (2018). https:\/\/arxiv.org\/abs\/1808.07036"},{"key":"e_1_3_2_1_6_1","volume-title":"PaLM: Scaling Language Modeling with Pathways. arXiv preprint arXiv:2204.02311","author":"Chowdhery Aakanksha","year":"2022","unstructured":"Aakanksha Chowdhery, Sharan Narang, Jacob Devlin, Maarten Bosma, Gaurav Mishra, Adam Roberts, Paul Barham, Hyung\u00a0Won Chung, Charles Sutton, Sebastian Gehrmann, 2022. PaLM: Scaling Language Modeling with Pathways. arXiv preprint arXiv:2204.02311 (2022). https:\/\/arxiv.org\/abs\/2204.02311"},{"key":"e_1_3_2_1_7_1","volume-title":"Wizard of Wikipedia: Knowledge-powered conversational agents. arXiv preprint arXiv:1811.01241","author":"Dinan Emily","year":"2018","unstructured":"Emily Dinan, Stephen Roller, Kurt Shuster, Angela Fan, Michael Auli, and Jason Weston. 2018. Wizard of Wikipedia: Knowledge-powered conversational agents. arXiv preprint arXiv:1811.01241 (2018). https:\/\/arxiv.org\/abs\/1811.01241"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00529"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.dialdoc-1.18"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.498"},{"key":"e_1_3_2_1_11_1","volume-title":"doc2dial: A goal-oriented document-grounded dialogue dataset. arXiv preprint arXiv:2011.06623","author":"Feng Song","year":"2020","unstructured":"Song Feng, Hui Wan, Chulaka Gunasekara, Siva\u00a0Sankalp Patel, Sachindra Joshi, and Luis\u00a0A Lastras. 2020. doc2dial: A goal-oriented document-grounded dialogue dataset. arXiv preprint arXiv:2011.06623 (2020). https:\/\/arxiv.org\/abs\/2011.06623"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3079"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10506-023-09374-7"},{"key":"e_1_3_2_1_14_1","volume-title":"Rethinking with retrieval: Faithful large language model inference. arXiv preprint arXiv:2301.00303","author":"He Hangfeng","year":"2022","unstructured":"Hangfeng He, Hongming Zhang, and Dan Roth. 2022. Rethinking with retrieval: Faithful large language model inference. arXiv preprint arXiv:2301.00303 (2022). https:\/\/arxiv.org\/abs\/2301.00303"},{"key":"e_1_3_2_1_15_1","volume-title":"Lisa\u00a0Anne Hendricks","author":"Hoffmann Jordan","year":"2022","unstructured":"Jordan Hoffmann, Sebastian Borgeaud, Arthur Mensch, Elena Buchatskaya, Trevor Cai, Eliza Rutherford, Diego de\u00a0Las Casas, Lisa\u00a0Anne Hendricks, Johannes Welbl, Aidan Clark, 2022. Training compute-optimal large language models. arXiv preprint arXiv:2203.15556 (2022). https:\/\/arxiv.org\/abs\/2203.15556"},{"key":"e_1_3_2_1_16_1","volume-title":"Q2: Evaluating Factual Consistency in Knowledge-Grounded Dialogues via Question Generation and Question Answering. arXiv preprint arXiv:2104.08202","author":"Honovich Or","year":"2021","unstructured":"Or Honovich, Leshem Choshen, Roee Aharoni, Ella Neeman, Idan Szpektor, and Omri Abend. 2021. Q2: Evaluating Factual Consistency in Knowledge-Grounded Dialogues via Question Generation and Question Answering. arXiv preprint arXiv:2104.08202 (2021). https:\/\/arxiv.org\/abs\/2104.08202"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.sdp-1.2"},{"key":"e_1_3_2_1_18_1","volume-title":"Triviaqa: A large scale distantly supervised challenge dataset for reading comprehension. arXiv preprint arXiv:1705.03551","author":"Joshi Mandar","year":"2017","unstructured":"Mandar Joshi, Eunsol Choi, Daniel\u00a0S Weld, and Luke Zettlemoyer. 2017. Triviaqa: A large scale distantly supervised challenge dataset for reading comprehension. arXiv preprint arXiv:1705.03551 (2017). https:\/\/arxiv.org\/abs\/1705.03551"},{"key":"e_1_3_2_1_19_1","volume-title":"Dense passage retrieval for open-domain question answering. arXiv preprint arXiv:2004.04906","author":"Karpukhin Vladimir","year":"2020","unstructured":"Vladimir Karpukhin, Barlas O\u011fuz, Sewon Min, Patrick Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih. 2020. Dense passage retrieval for open-domain question answering. arXiv preprint arXiv:2004.04906 (2020). https:\/\/arxiv.org\/abs\/2004.04906"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00276"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3496517"},{"key":"e_1_3_2_1_23_1","unstructured":"Junyi Li Xiaoxue Cheng Wayne\u00a0Xin Zhao Jian-Yun Nie and Ji-Rong Wen. 2023. HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models. arxiv:2305.11747\u00a0[cs.CL]"},{"key":"e_1_3_2_1_24_1","unstructured":"Jerry Liu. 2022. LlamaIndex. https:\/\/github.com\/jerryjliu\/llama_index"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Nick McKenna Tianyi Li Liang Cheng Mohammad\u00a0Javad Hosseini Mark Johnson and Mark Steedman. 2023. Sources of Hallucination by Large Language Models on Inference Tasks. arxiv:2305.14552\u00a0[cs.CL]","DOI":"10.18653\/v1\/2023.findings-emnlp.182"},{"key":"e_1_3_2_1_26_1","volume-title":"USR: An unsupervised and reference free evaluation metric for dialog generation. arXiv preprint arXiv:2005.00456","author":"Mehri Shikib","year":"2020","unstructured":"Shikib Mehri and Maxine Eskenazi. 2020. USR: An unsupervised and reference free evaluation metric for dialog generation. arXiv preprint arXiv:2005.00456 (2020). https:\/\/arxiv.org\/abs\/2005.00456"},{"key":"e_1_3_2_1_27_1","unstructured":"OpenAI. 2023. GPT-3 (Jul 05 version) [Large language model]. https:\/\/platform.openai.com\/docs\/models\/gpt-3-5"},{"key":"e_1_3_2_1_28_1","volume-title":"100,000+ questions for machine comprehension of text. arXiv preprint arXiv:1606.05250","author":"Rajpurkar Pranav","year":"2016","unstructured":"Pranav Rajpurkar, Jian Zhang, Konstantin Lopyrev, and Percy Liang. 2016. Squad: 100,000+ questions for machine comprehension of text. arXiv preprint arXiv:1606.05250 (2016). https:\/\/arxiv.org\/abs\/1606.05250"},{"key":"e_1_3_2_1_29_1","volume-title":"Toolformer: Language models can teach themselves to use tools. arXiv preprint arXiv:2302.04761","author":"Schick Timo","year":"2023","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess\u00ec, Roberta Raileanu, Maria Lomeli, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. 2023. Toolformer: Language models can teach themselves to use tools. arXiv preprint arXiv:2302.04761 (2023). https:\/\/arxiv.org\/abs\/2302.04761"},{"key":"e_1_3_2_1_30_1","volume-title":"Retrieval augmentation reduces hallucination in conversation. arXiv preprint arXiv:2104.07567","author":"Shuster Kurt","year":"2021","unstructured":"Kurt Shuster, Spencer Poff, Moya Chen, Douwe Kiela, and Jason Weston. 2021. Retrieval augmentation reduces hallucination in conversation. arXiv preprint arXiv:2104.07567 (2021). https:\/\/arxiv.org\/abs\/2104.07567"},{"key":"e_1_3_2_1_31_1","volume-title":"Nenad Tomasev, Yun Liu, Renee Wong, Christopher Semturs, S.\u00a0Sara Mahdavi, Joelle Barral, Dale Webster, Greg\u00a0S. Corrado, Yossi Matias, Shekoofeh Azizi, Alan Karthikesalingam, and Vivek Natarajan.","author":"Singhal Karan","year":"2023","unstructured":"Karan Singhal, Tao Tu, Juraj Gottweis, Rory Sayres, Ellery Wulczyn, Le Hou, Kevin Clark, Stephen Pfohl, Heather Cole-Lewis, Darlene Neal, Mike Schaekermann, Amy Wang, Mohamed Amin, Sami Lachgar, Philip Mansfield, Sushant Prakash, Bradley Green, Ewa Dominowska, Blaise\u00a0Aguera y Arcas, Nenad Tomasev, Yun Liu, Renee Wong, Christopher Semturs, S.\u00a0Sara Mahdavi, Joelle Barral, Dale Webster, Greg\u00a0S. Corrado, Yossi Matias, Shekoofeh Azizi, Alan Karthikesalingam, and Vivek Natarajan. 2023. Towards Expert-Level Medical Question Answering with Large Language Models. arxiv:2305.09617\u00a0[cs.CL]"},{"key":"e_1_3_2_1_32_1","volume-title":"Interleaving Retrieval with Chain-of-Thought Reasoning for Knowledge-Intensive Multi-Step Questions. arXiv preprint arXiv:2212.10509","author":"Trivedi Harsh","year":"2022","unstructured":"Harsh Trivedi, Niranjan Balasubramanian, Tushar Khot, and Ashish Sabharwal. 2022. Interleaving Retrieval with Chain-of-Thought Reasoning for Knowledge-Intensive Multi-Step Questions. arXiv preprint arXiv:2212.10509 (2022). https:\/\/arxiv.org\/abs\/2212.10509"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10439-023-03327-6"},{"key":"e_1_3_2_1_34_1","volume-title":"ReAct: Synergizing Reasoning and Acting in Language Models. arXiv preprint arXiv:2210.03629","author":"Yao Shunyu","year":"2022","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao. 2022. ReAct: Synergizing Reasoning and Acting in Language Models. arXiv preprint arXiv:2210.03629 (2022). https:\/\/arxiv.org\/abs\/2210.03629"},{"key":"e_1_3_2_1_35_1","unstructured":"Muru Zhang Ofir Press William Merrill Alisa Liu and Noah\u00a0A. Smith. 2023. How Language Model Hallucinations Can Snowball. arxiv:2305.13534\u00a0[cs.CL]"},{"key":"e_1_3_2_1_36_1","volume-title":"Verify-and-Edit: A Knowledge-Enhanced Chain-of-Thought Framework. arXiv preprint arXiv:2305.03268","author":"Zhao Ruochen","year":"2023","unstructured":"Ruochen Zhao, Xingxuan Li, Shafiq Joty, Chengwei Qin, and Lidong Bing. 2023. Verify-and-Edit: A Knowledge-Enhanced Chain-of-Thought Framework. arXiv preprint arXiv:2305.03268 (2023). https:\/\/arxiv.org\/abs\/2305.03268"},{"key":"e_1_3_2_1_37_1","volume-title":"Towards a Unified Multi-Dimensional Evaluator for Text Generation. arXiv preprint arXiv:2210.07197","author":"Zhong Ming","year":"2022","unstructured":"Ming Zhong, Yang Liu, Da Yin, Yuning Mao, Yizhu Jiao, Pengfei Liu, Chenguang Zhu, Heng Ji, and Jiawei Han. 2022. Towards a Unified Multi-Dimensional Evaluator for Text Generation. arXiv preprint arXiv:2210.07197 (2022). https:\/\/arxiv.org\/abs\/2210.07197"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1076"}],"event":{"name":"CUI '24: ACM Conversational User Interfaces 2024","location":"Luxembourg Luxembourg","acronym":"CUI '24","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["ACM Conversational User Interfaces 2024"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640794.3665573","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3640794.3665573","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T18:04:41Z","timestamp":1755885881000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640794.3665573"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,8]]},"references-count":38,"alternative-id":["10.1145\/3640794.3665573","10.1145\/3640794"],"URL":"https:\/\/doi.org\/10.1145\/3640794.3665573","relation":{},"subject":[],"published":{"date-parts":[[2024,7,8]]},"assertion":[{"value":"2024-07-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}