{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T12:29:24Z","timestamp":1743769764235,"version":"3.40.3"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030997359"},{"type":"electronic","value":"9783030997366"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-030-99736-6_13","type":"book-chapter","created":{"date-parts":[[2022,4,4]],"date-time":"2022-04-04T23:02:47Z","timestamp":1649113367000},"page":"184-198","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A Dependency-Aware Utterances Permutation Strategy to\u00a0Improve Conversational Evaluation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5070-2049","authenticated-orcid":false,"given":"Guglielmo","family":"Faggioli","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0894-4175","authenticated-orcid":false,"given":"Marco","family":"Ferrante","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9219-6239","authenticated-orcid":false,"given":"Nicola","family":"Ferro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7189-4724","authenticated-orcid":false,"given":"Raffaele","family":"Perego","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7427-1001","authenticated-orcid":false,"given":"Nicola","family":"Tonellotto","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,4,5]]},"reference":[{"key":"13_CR1","unstructured":"Anand, A., Cavedon, L., Joho, H., Sanderson, M., Stein, B.: Conversational search (Dagstuhl Seminar 19461). In: Dagstuhl Reports, vol. 9 (2020)"},{"key":"13_CR2","unstructured":"Banerjee, S., Lavie, A.: METEOR: an automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp. 65\u201372 (2005)"},{"issue":"7","key":"13_CR3","doi-asserted-by":"publisher","first-page":"1249","DOI":"10.1109\/TASL.2008.2001102","volume":"16","author":"S Bangalore","year":"2008","unstructured":"Bangalore, S., Di Fabbrizio, G., Stent, A.: Learning the structure of task-driven human-human dialogs. IEEE Trans. Audio Speech Lang. Process. 16(7), 1249\u20131259 (2008)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"1\u20132","key":"13_CR4","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1023\/A:1009984519381","volume":"1","author":"D Banks","year":"1999","unstructured":"Banks, D., Over, P., Zhang, N.F.: Blind men and elephants: six approaches to TREC data. Inf. Retriev. J. 1(1\u20132), 7\u201334 (1999)","journal-title":"Inf. Retriev. J."},{"issue":"1","key":"13_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3470563","volume":"40","author":"JS Culpepper","year":"2021","unstructured":"Culpepper, J.S., Faggioli, G., Ferro, N., Kurland, O.: Topic difficulty: collection and query formulation effects. ACM Trans. Inf. Syst. 40(1), 1\u201336 (2021)","journal-title":"ACM Trans. Inf. Syst."},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Dalton, J., Xiong, C., Callan, J.: TREC CAsT 2019: the conversational assistance track overview. In: Proceedings of TREC (2020)","DOI":"10.6028\/NIST.SP.1266.cast-overview"},{"key":"13_CR7","doi-asserted-by":"crossref","unstructured":"Dalton, J., Xiong, C., Callan, J.: TREC CAsT 2020: the conversational assistance track overview. In: Proceedings of TREC (2021)","DOI":"10.6028\/NIST.SP.500-335.cast-overview"},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Dietz, L., Verma, M., Radlinski, F., Craswell, N.: TREC complex answer retrieval overview. In: Proceedings of TREC (2017)","DOI":"10.6028\/NIST.SP.500-324.car-overview"},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Faggioli, G., Ferrante, M., Ferro, N., Perego, R., Tonellotto, N.: Hierarchical dependence-aware evaluation measures for conversational search. In: Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 1935\u20131939 (2021)","DOI":"10.1145\/3404835.3463090"},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Faggioli, G., Zendel, O., Culpepper, J.S., Ferro, N., Scholer, F.: An enhanced evaluation framework for query performance prediction. In: Proceedings of the 43rd European Conference on Information Retrieval, pp. 115\u2013129 (2021)","DOI":"10.1007\/978-3-030-72113-8_8"},{"key":"13_CR11","doi-asserted-by":"publisher","unstructured":"Ferro, N., Harman, D.: CLEF 2009: Grid@CLEF pilot track overview. In: Peters, C., et al. (eds.) CLEF 2009. LNCS, vol. 6241, pp. 552\u2013565. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15754-7_68","DOI":"10.1007\/978-3-642-15754-7_68"},{"key":"13_CR12","doi-asserted-by":"crossref","unstructured":"Ferro, N., Sanderson, M.: Improving the accuracy of system performance estimation by using shards. In: Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 805\u2013814 (2019)","DOI":"10.1145\/3331184.3338062"},{"key":"13_CR13","doi-asserted-by":"crossref","unstructured":"Ferro, N., Silvello, G.: A general linear mixed models approach to study system component effects. In: Proceedings of the 39th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 25\u201334 (2016)","DOI":"10.1145\/2911451.2911530"},{"key":"13_CR14","doi-asserted-by":"publisher","first-page":"369","DOI":"10.1109\/TASLP.2019.2955290","volume":"28","author":"JC Gu","year":"2020","unstructured":"Gu, J.C., Ling, Z.H., Liu, Q.: Utterance-to-utterance interactive matching network for multi-turn response selection in retrieval-based chatbots. IEEE\/ACM Trans. Audio Speech Lang. Proc. 28, 369\u2013379 (2020)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Proc."},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"J\u00e4rvelin, K., Kek\u00e4 l\u00e4inen, J.: Cumulated gain-based evaluation of IR techniques. ACM Trans. Inf. Syst. 20(4), 422\u2013446 (2002)","DOI":"10.1145\/582415.582418"},{"key":"13_CR16","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1162\/tacl_a_00300","volume":"8","author":"M Joshi","year":"2020","unstructured":"Joshi, M., Chen, D., Liu, Y., Weld, D.S., Zettlemoyer, L., Levy, O.: Spanbert: improving pre-training by representing and predicting spans. Trans. Assoc. Comput. Linguist. 8, 64\u201377 (2020)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Lee, K., He, L., Zettlemoyer, L.: Higher-order coreference resolution with coarse-to-fine inference. In: Proceedings of the 2018 Conference of the NAACL-HLT, pp. 687\u2013692 (2018)","DOI":"10.18653\/v1\/N18-2108"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Li, J., et al.: Dialogue history matters! personalized response selection in multi-turn retrieval-based chatbots. ACM Trans. Inf. Syst. 39(4), 1\u201325 (2021)","DOI":"10.1145\/3453183"},{"key":"13_CR19","doi-asserted-by":"crossref","unstructured":"Lipani, A., Carterette, B., Yilmaz, E.: How am i doing?: evaluating conversational search systems offline. ACM Trans. Inf. Syst. 39(4), 1\u201322 (2021)","DOI":"10.1145\/3451160"},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Liu, C.W., Lowe, R., Serban, I.V., Noseworthy, M., Charlin, L., Pineau, J.: How NOT to Evaluate Your Dialogue System: An Empirical Study of Unsupervised Evaluation Metrics for Dialogue Response Generation (2017)","DOI":"10.18653\/v1\/D16-1230"},{"key":"13_CR21","doi-asserted-by":"crossref","unstructured":"Liu, Z., Zhou, K., Wilson, M.L.: Meta-evaluation of conversational search evaluation metrics. ACM Trans. Inf. Syst. 39(4), 1\u201342 (2021)","DOI":"10.1145\/3445029"},{"key":"13_CR22","doi-asserted-by":"crossref","unstructured":"Lv, Y., Zhai, C.: Positional relevance model for pseudo-relevance feedback. In: Proceedings of the 33rd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 579\u2013586 (2010)","DOI":"10.1145\/1835449.1835546"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Mele, I., Muntean, C.I., Nardini, F.M., Perego, R., Tonellotto, N., Frieder, O.: Topic propagation in conversational search. In: Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 2057\u20132060 (2020)","DOI":"10.1145\/3397271.3401268"},{"issue":"6","key":"13_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2021.102682","volume":"58","author":"I Mele","year":"2021","unstructured":"Mele, I., Muntean, C.I., Nardini, F.M., Perego, R., Tonellotto, N., Frieder, O.: Adaptive utterance rewriting for conversational search. Inf. Process. Manag. 58(6), 102682 (2021)","journal-title":"Inf. Process. Manag."},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"13_CR26","unstructured":"Penha, G., Hauff, C.: Challenges in the evaluation of conversational search systems. In: Workshop on Conversational Systems Towards Mainstream Adoption, KDD-Converse (2020)"},{"key":"13_CR27","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(140), 1\u201367 (2020)"},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Rutherford, A.: ANOVA and ANCOVA, 2nd edn. A GLM Approach. Wiley, New York (2011)","DOI":"10.1002\/9781118491683"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Sakai, T.: Evaluating evaluation metrics based on the bootstrap. In: Proceedings of the 29th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 525\u2013532 (2006)","DOI":"10.1145\/1148170.1148261"},{"key":"13_CR30","doi-asserted-by":"crossref","unstructured":"Smucker, M.D., Allan, J., Carterette, B.A.: A comparison of statistical significance tests for information retrieval evaluation. In: Proceedings of the Sixteenth ACM Conference on Conference on Information and Knowledge Management, pp. 623\u2013632 (2007)","DOI":"10.1145\/1321440.1321528"},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Tao, C., Wu, W., Xu, C., Hu, W., Zhao, D., Yan, R.: Multi-representation fusion network for multi-turn response selection in retrieval-based chatbots. In: Proceedings of the 12th ACM International Conference on Web Search and Data Mining, pp. 267\u2013275 (2019)","DOI":"10.1145\/3289600.3290985"},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Urbano, J., Lima, H., Hanjalic, A.: Statistical significance testing in information retrieval: an empirical analysis of type I, type II and type III errors. In: Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 505\u2013514 (2019)","DOI":"10.1145\/3331184.3331259"},{"key":"13_CR33","doi-asserted-by":"crossref","unstructured":"Vakulenko, S., Longpre, S., Tu, Z., Anantha, R.: Question rewriting for conversational question answering. In: Proceedings of the Fourth ACM International Conference on Web Search and Data Mining (WSDM), pp. 355\u2013363 (2021)","DOI":"10.1145\/3437963.3441748"},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Voorhees, E.M., Samarov, D., Soboroff, I.: Using replicates in information retrieval evaluation. ACM Trans. Inf. Syst. 36(2), 1\u201321, 102682 (2017)","DOI":"10.1145\/3086701"},{"key":"13_CR35","doi-asserted-by":"crossref","unstructured":"Wu, Y., Wu, W., Xing, C., Zhou, M., Li, Z.: Sequential matching network: a new architecture for multi-turn response selection in retrieval-based chatbots. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics, pp. 496\u2013505 (2017)","DOI":"10.18653\/v1\/P17-1046"},{"key":"13_CR36","doi-asserted-by":"crossref","unstructured":"Yan, R.: \u201cChitty-chitty-chat bot\": deep learning for conversational AI. In: Proceedings of the 27th International Joint Conference on Artificial Intelligence (IJCAI), vol. 18, pp. 5520\u20135526 (2018)","DOI":"10.24963\/ijcai.2018\/778"},{"key":"13_CR37","doi-asserted-by":"crossref","unstructured":"Yu, Z., Xu, Z., Black, A.W., Rudnicky, A.: Strategy and policy learning for non-task-oriented conversational systems. In: Proceedings of the 17th Annual Meeting of the Special Interest Group on Discourse and Dialogue, pp. 404\u2013412 (2016)","DOI":"10.18653\/v1\/W16-3649"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"Zhang, S., Balog, K.: Evaluating conversational recommender systems via user simulation. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 1512\u20131520 (2020)","DOI":"10.1145\/3394486.3403202"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-99736-6_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,21]],"date-time":"2024-09-21T15:41:25Z","timestamp":1726933285000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-99736-6_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783030997359","9783030997366"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-99736-6_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"5 April 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Stavanger","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Norway","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 April 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 April 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"44","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2022.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"395","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"35","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"29","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"9% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4-6","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Additionally, there are other papers: 11 reproducibility, 12 doctoral, 13 CLEF Labs, 5 workshops and 4 tutorials.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}