{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T08:01:27Z","timestamp":1784966487593,"version":"3.55.0"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T00:00:00Z","timestamp":1784937600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T00:00:00Z","timestamp":1784937600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"NSERC of Canada"},{"DOI":"10.13039\/501100004489","name":"Mitacs","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004489","id-type":"DOI","asserted-by":"publisher"}]},{"name":"NSERC of Canada"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Empir Software Eng"],"published-print":{"date-parts":[[2027,2]]},"DOI":"10.1007\/s10664-026-10923-2","type":"journal-article","created":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T07:49:37Z","timestamp":1784965777000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["The impact of critique on LLM-based model generation from natural language: the case of activity diagrams"],"prefix":"10.1007","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-9187-3845","authenticated-orcid":false,"given":"Parham","family":"Khamsepour","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mark","family":"Cole","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ish","family":"Ashraf","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"DaYuan","family":"Tan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sandeep","family":"Puri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4711-8319","authenticated-orcid":false,"given":"Mehrdad","family":"Sabetzadeh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0281-8231","authenticated-orcid":false,"given":"Shiva","family":"Nejati","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,25]]},"reference":[{"issue":"6","key":"10923_CR1","doi-asserted-by":"publisher","first-page":"5454","DOI":"10.1007\/S10664-020-09864-1","volume":"25","author":"S Abualhaija","year":"2020","unstructured":"Abualhaija S, Arora C, Sabetzadeh M, Briand LC, Traynor M (2020) Automated demarcation of requirements in textual specifications: a machine learning-based approach. Empir Softw Eng 25(6):5454\u20135497. https:\/\/doi.org\/10.1007\/S10664-020-09864-1","journal-title":"Empir Softw Eng"},{"key":"10923_CR2","doi-asserted-by":"crossref","unstructured":"Arora C, Sabetzadeh M, Briand LC, Zimmer F (2016) Extracting domain models from natural-language requirements: approach and industrial evaluation. In: Baudry B, Combemale B (eds) Proceedings of the ACM\/IEEE 19th International Conference on Model Driven Engineering Languages and Systems, Saint-Malo, France, October 2-7, 2016, ACM, pp 250\u2013260. http:\/\/dl.acm.org\/citation.cfm?id=2976769","DOI":"10.1145\/2976767.2976769"},{"issue":"1","key":"10923_CR3","doi-asserted-by":"publisher","first-page":"4:1","DOI":"10.1145\/3293454","volume":"28","author":"C Arora","year":"2019","unstructured":"Arora C, Sabetzadeh M, Nejati S, Briand LC (2019) An active learning approach for improving the accuracy of automated domain model extraction. ACM Trans Softw Eng Methodol 28(1):4:1-4:34. https:\/\/doi.org\/10.1145\/3293454","journal-title":"ACM Trans Softw Eng Methodol"},{"key":"10923_CR4","doi-asserted-by":"publisher","unstructured":"Bashir S, Abbas M, Saadatmand M, Enoiu EP, Bohlin M, Lindberg P (2023) Requirement or not, that is the question: A case from the railway industry. In: Ferrari A, Penzenstadler B (eds) Requirements engineering: Foundation for software quality - 29th international working conference, REFSQ 2023, Barcelona, Spain, April 17-20, 2023, Proceedings, Springer, Lecture Notes in Computer Science, vol 13975, pp 105\u2013121. https:\/\/doi.org\/10.1007\/978-3-031-29786-1_8","DOI":"10.1007\/978-3-031-29786-1_8"},{"issue":"1","key":"10923_CR5","doi-asserted-by":"publisher","first-page":"289","DOI":"10.1111\/j.2517-6161.1995.tb02031.x","volume":"57","author":"Y Benjamini","year":"1995","unstructured":"Benjamini Y, Hochberg Y (1995) Controlling the false discovery rate: a practical and powerful approach to multiple testing. J Roy Stat Soc: Ser B (Methodol) 57(1):289\u2013300","journal-title":"J Roy Stat Soc: Ser B (Methodol)"},{"key":"10923_CR6","doi-asserted-by":"publisher","unstructured":"Chen B, Wei O, Zheng B, Mussbacher G (2025) Accurate and consistent graph model generation from text with large language models. In: 28th ACM\/IEEE International conference on Model Driven Engineering Languages and Systems, MODELS 2025, Grand Rapids, MI, USA, October 5-10, 2025, IEEE, pp 130\u2013141. https:\/\/doi.org\/10.1109\/MODELS67397.2025.00018","DOI":"10.1109\/MODELS67397.2025.00018"},{"key":"10923_CR7","unstructured":"Chiang W, Zheng L, Sheng Y, Angelopoulos AN, Li T, Li D, Zhu B, Zhang H, Jordan MI, Gonzalez JE, Stoica I (2024) Chatbot arena: An open platform for evaluating llms by human preference. https:\/\/openreview.net\/forum?id=3MW8GKNyzI"},{"issue":"1","key":"10923_CR8","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1177\/001316446002000104","volume":"20","author":"J Cohen","year":"1960","unstructured":"Cohen J (1960) A coefficient of agreement for nominal scales. Educ Psychol Measur 20(1):37\u201346. https:\/\/doi.org\/10.1177\/001316446002000104","journal-title":"Educ Psychol Measur"},{"key":"10923_CR9","unstructured":"Cook S, Bock C, Rivett P, Rutt T, Seidewitz E, Selic B, Tolbert D (2017) Unified modeling language (UML) version 2.5.1. Standard, Object Management Group (OMG). https:\/\/www.omg.org\/spec\/UML\/2.5.1"},{"key":"10923_CR10","doi-asserted-by":"publisher","unstructured":"DeepSeek-AI, Guo D, Yang D, Zhang H, Song J, Zhang R, Xu R, Zhu Q, Ma S, Wang P, Bi X, Others (2025) Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. https:\/\/doi.org\/10.48550\/ARXIV.2501.12948. arxiv:2501.12948","DOI":"10.48550\/ARXIV.2501.12948"},{"key":"10923_CR11","doi-asserted-by":"publisher","unstructured":"Du W, Liao W, Liang H, Lei W (2024) PAGED: A benchmark for procedural graphs extraction from documents. In: Ku L, Martins A, Srikumar V (eds) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2024, Bangkok, Thailand, August 11-16, 2024, Association for Computational Linguistics, pp 10829\u201310846. https:\/\/doi.org\/10.18653\/V1\/2024.ACL-LONG.583","DOI":"10.18653\/V1\/2024.ACL-LONG.583"},{"key":"10923_CR12","doi-asserted-by":"publisher","unstructured":"Ferrari A, Abualhaija S, Arora C (2024) Model generation with llms: From requirements to UML sequence diagrams. In: 32nd IEEE International Requirements Engineering Conference, RE 2024 - Workshops, Reykjavik, Iceland, June 24-25, 2024, IEEE, pp 291\u2013300. https:\/\/doi.org\/10.1109\/REW61692.2024.00044","DOI":"10.1109\/REW61692.2024.00044"},{"key":"10923_CR13","doi-asserted-by":"publisher","unstructured":"G\u00e9rard S, Dumoulin C, Tessier P, Selic B (2010) 19 papyrus: A uml2 tool for domain-specific language modeling. In: Giese H, Karsai G, Lee E, Rumpe B, Sch\u00e4tz B (eds) Model-Based Engineering of Embedded Real-Time Systems: International Dagstuhl Workshop, Dagstuhl Castle, Germany, November 4-9, 2007. Revised Selected Papers, Springer Berlin Heidelberg, Berlin, Heidelberg, pp 361\u2013368. https:\/\/doi.org\/10.1007\/978-3-642-16277-0_19","DOI":"10.1007\/978-3-642-16277-0_19"},{"key":"10923_CR14","doi-asserted-by":"publisher","unstructured":"Herbold S, Knieke C, Rausch A, Schindler C (2025) Neurosymbolic architectural reasoning: Towards formal analysis through neural software architecture inference. In: 1st IEEE\/ACM International Workshop on Neuro-Symbolic Software Engineering, NSE@ICSE 2025, Ottawa, ON, Canada, May 3, 2025, IEEE, pp 5\u201310. https:\/\/doi.org\/10.1109\/NSE66660.2025.00008","DOI":"10.1109\/NSE66660.2025.00008"},{"key":"#cr-split#-10923_CR15.1","unstructured":"Herwanto GB (2024) Automating data flow diagram generation from user stories using large language models. In: DM et al"},{"key":"#cr-split#-10923_CR15.2","unstructured":"(ed) Joint proceedings of REFSQ-2024 workshops, doctoral symposium, posters & tools track, and education and training track co-located with the 30th International Conference on Requirements Engineering: Foundation for Software Quality (REFSQ 2024), Winterthur, Switzerland, April 8-11, 2024, CEUR-WS.org, CEUR Workshop Proceedings, vol 3672. https:\/\/ceur-ws.org\/Vol-3672\/NLP4RE-paper3.pdf"},{"key":"10923_CR16","doi-asserted-by":"publisher","unstructured":"Hurst A, Lerer A, Goucher AP, Perelman A, Ramesh A, Clark A, Ostrow A, Welihinda A, Hayes A, Radford A, Madry A, Baker-Whitcomb A, Beutel A, Borzunov A, Carney A et al (2024) Gpt-4o system card. https:\/\/doi.org\/10.48550\/ARXIV.2410.21276. arxiv:2410.21276","DOI":"10.48550\/ARXIV.2410.21276"},{"key":"10923_CR17","doi-asserted-by":"publisher","unstructured":"Ibrahimzada AR, Ke K, Pawagi M, Abid MS, Pan R, Sinha S, Jabbarvand R (2025) Alphatrans: A neuro-symbolic compositional approach for repository-level code translation and validation. Proc ACM Softw Eng 2(FSE):2454\u20132476. https:\/\/doi.org\/10.1145\/3729379","DOI":"10.1145\/3729379"},{"key":"10923_CR18","doi-asserted-by":"publisher","unstructured":"Jaech A, Kalai A, Lerer A, Richardson A, El-Kishky A, Low A, Helyar A, Madry A, Beutel A, Carney A et al (2024) Openai o1 system card. https:\/\/doi.org\/10.48550\/ARXIV.2412.16720. arxiv:2412.16720","DOI":"10.48550\/ARXIV.2412.16720"},{"key":"10923_CR19","doi-asserted-by":"publisher","unstructured":"Jahan M, Hassan MM, Golpayegani R, Ranjbaran G, Roy C, Roy B, Schneider KA (2024) Automated derivation of UML sequence diagrams from user stories: Unleashing the power of generative AI vs. a rule-based approach. In: Egyed A, Wimmer M, Chechik M, Combemale B (eds) Proceedings of the ACM\/IEEE 27th International Conference on Model Driven Engineering Languages and Systems, MODELS 2024, Linz, Austria, September 22-27, 2024, ACM, pp 138\u2013148. https:\/\/doi.org\/10.1145\/3640310.3674081","DOI":"10.1145\/3640310.3674081"},{"key":"10923_CR20","unstructured":"JGraph (2021) draw.io. https:\/\/www.draw.io\/. Accessed 21 Aug 2025"},{"key":"10923_CR21","unstructured":"Lehmann EL, D\u2019Abrera HJ (2006) Nonparametrics: statistical methods based on ranks, vol 464. Springer, New York"},{"issue":"11","key":"10923_CR22","doi-asserted-by":"publisher","first-page":"405","DOI":"10.3390\/IJGI13110405","volume":"13","author":"D Li","year":"2024","unstructured":"Li D, Zhao Y, Wang Z, Jung C, Zhang Z (2024) Large language model-driven structured output: A comprehensive benchmark and spatial data generation framework. ISPRS Int J Geo Inf 13(11):405. https:\/\/doi.org\/10.3390\/IJGI13110405","journal-title":"ISPRS Int J Geo Inf"},{"key":"10923_CR23","doi-asserted-by":"publisher","unstructured":"Li Z, Zhang X, Zhang Y, Long D, Xie P, Zhang M (2023) Towards general text embeddings with multi-stage contrastive learning. https:\/\/doi.org\/10.48550\/ARXIV.2308.03281, arxiv:2308.03281","DOI":"10.48550\/ARXIV.2308.03281"},{"key":"10923_CR24","doi-asserted-by":"publisher","unstructured":"Maoz S, Ringert JO, Rumpe B (2011) Addiff: semantic differencing for activity diagrams. In: Gyim\u00f3thy T, Zeller A (eds) SIGSOFT\/FSE\u201911 19th ACM SIGSOFT Symposium on the Foundations of Software Engineering (FSE-19) and ESEC\u201911: 13th European Software Engineering Conference (ESEC-13), Szeged, Hungary, September 5-9, 2011, ACM, pp 179\u2013189. https:\/\/doi.org\/10.1145\/2025113.2025140","DOI":"10.1145\/2025113.2025140"},{"key":"10923_CR25","unstructured":"Microsoft (2025) Azure openai reasoning models - gpt-5 series, o3-mini, o1, o1-mini. https:\/\/learn.microsoft.com\/en-us\/azure\/ai-foundry\/openai\/how-to\/reasoning. Accessed 21 Aug 2025"},{"key":"10923_CR26","doi-asserted-by":"publisher","unstructured":"Nejati S, Sabetzadeh M, Chechik M, Easterbrook SM, Zave P (2007) Matching and merging of statecharts specifications. In: 29th International Conference on Software Engineering (ICSE 2007), Minneapolis, MN, USA, May 20-26, 2007, IEEE Computer Society, pp 54\u201364. https:\/\/doi.org\/10.1109\/ICSE.2007.50","DOI":"10.1109\/ICSE.2007.50"},{"issue":"6","key":"10923_CR27","doi-asserted-by":"publisher","first-page":"1355","DOI":"10.1109\/TSE.2011.112","volume":"38","author":"S Nejati","year":"2012","unstructured":"Nejati S, Sabetzadeh M, Chechik M, Easterbrook SM, Zave P (2012) Matching and merging of variant feature specifications. IEEE Trans Softw Eng 38(6):1355\u20131375. https:\/\/doi.org\/10.1109\/TSE.2011.112","journal-title":"IEEE Trans Softw Eng"},{"key":"10923_CR28","doi-asserted-by":"publisher","unstructured":"Nejati S, Sabetzadeh M, Arora C, Briand LC, Mandoux F (2016) Automated change impact analysis between sysml models of requirements and design. In: Zimmermann T, Cleland-Huang J, Su Z (eds) Proceedings of the 24th ACM SIGSOFT international symposium on foundations of software engineering, FSE 2016, Seattle, WA, USA, November 13-18, 2016, ACM, pp 242\u2013253. https:\/\/doi.org\/10.1145\/2950290.2950293","DOI":"10.1145\/2950290.2950293"},{"key":"10923_CR29","unstructured":"Object Management Group (2021) Semantics of a foundational subset for executable UML models (fUML) version 1.5. Standard, Object Management Group (OMG). https:\/\/www.omg.org\/spec\/FUML\/1.5"},{"key":"10923_CR30","unstructured":"Ollama (2023) Ollama. https:\/\/github.com\/ollama\/ollama. Accessed 21 Aug 2025"},{"key":"10923_CR31","unstructured":"OMG (2011) Business Process Model and Notation (BPMN), Version 2.0. Object Management Group. http:\/\/www.omg.org\/spec\/BPMN\/2.0. Accessed 21 Aug 2025"},{"key":"10923_CR32","doi-asserted-by":"publisher","unstructured":"Pezeshkpour P, Hruschka E (2024) Large language models sensitivity to the order of options in multiple-choice questions. In: Duh K, G\u00f3mez-Adorno H, Bethard S (eds) Findings of the Association for Computational Linguistics: NAACL 2024, Mexico City, Mexico, June 16-21, 2024, Association for Computational Linguistics, pp 2006\u20132017. https:\/\/doi.org\/10.18653\/V1\/2024.FINDINGS-NAACL.130","DOI":"10.18653\/V1\/2024.FINDINGS-NAACL.130"},{"key":"10923_CR33","doi-asserted-by":"publisher","unstructured":"Reimers N, Gurevych I (2019) Sentence-bert: Sentence embeddings using siamese bert-networks. In: Inui K, Jiang J, Ng V, Wan X (eds) Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing, EMNLP-IJCNLP 2019, Hong Kong, China, November 3-7, 2019, Association for Computational Linguistics, pp 3980\u20133990. https:\/\/doi.org\/10.18653\/V1\/D19-1410","DOI":"10.18653\/V1\/D19-1410"},{"key":"10923_CR34","unstructured":"ReplicationPackage (2025) LADEX replication package. https:\/\/github.com\/parham-box\/EMSE-LADEX. Accessed 24 Nov 2025"},{"key":"10923_CR35","unstructured":"Roques A, Contributors P (2009) Plantuml software. https:\/\/github.com\/plantuml\/plantuml. Accessed 21 Aug 2025"},{"key":"10923_CR36","doi-asserted-by":"publisher","unstructured":"Sokolsky O, Kannan S, Lee I (2006) Simulation-based graph similarity. In: Hermanns H, Palsberg J (eds) Tools and algorithms for the construction and analysis of systems, 12th international conference, TACAS 2006 held as part of the joint European conferences on theory and practice of software, ETAPS 2006, Vienna, Austria, March 25 - April 2, 2006, Proceedings, Springer, Lecture Notes in Computer Science, vol 3920, pp 426\u2013440. https:\/\/doi.org\/10.1007\/11691372_28","DOI":"10.1007\/11691372_28"},{"issue":"3\/4","key":"10923_CR37","doi-asserted-by":"publisher","first-page":"441","DOI":"10.2307\/1422689","volume":"100","author":"C Spearman","year":"1987","unstructured":"Spearman C (1987) The proof and measurement of association between two things. Am J Psychol 100(3\/4):441\u2013471. https:\/\/doi.org\/10.2307\/1422689","journal-title":"Am J Psychol"},{"issue":"2","key":"10923_CR38","first-page":"101","volume":"25","author":"A Vargha","year":"2000","unstructured":"Vargha A, Delaney HD (2000) A critique and improvement of the cl common language effect size statistics of mcgraw and wong. J Educ Behav Stat 25(2):101\u2013132","journal-title":"J Educ Behav Stat"},{"issue":"3","key":"10923_CR39","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1109\/MS.2013.65","volume":"31","author":"J Whittle","year":"2014","unstructured":"Whittle J, Hutchinson J, Rouncefield M (2014) The state of practice in model-driven engineering. IEEE Softw 31(3):79\u201385. https:\/\/doi.org\/10.1109\/MS.2013.65","journal-title":"IEEE Softw"},{"key":"10923_CR40","doi-asserted-by":"crossref","unstructured":"Wilcoxon F (1992) Individual comparisons by ranking methods. Breakthroughs in statistics: Methodology and distribution. Springer, pp 196\u2013202","DOI":"10.1007\/978-1-4612-4380-9_16"},{"key":"10923_CR41","doi-asserted-by":"publisher","unstructured":"Yang Y, Chen B, Chen K, Mussbacher G, Varr\u00f3 D (2024) Multi-step iterative automated domain modeling with large language models. In: Wimmer M, Egyed A, Combemale B, Chechik M (eds) Proceedings of the ACM\/IEEE 27th international conference on Model Driven Engineering Languages and Systems, MODELS Companion 2024, Linz, Austria, September 22-27, 2024, ACM, pp 587\u2013595. https:\/\/doi.org\/10.1145\/3652620.3687807","DOI":"10.1145\/3652620.3687807"},{"key":"10923_CR42","doi-asserted-by":"publisher","unstructured":"Zhang X, Zhang Y, Long D, Xie W, Dai Z, Tang J, Lin H, Yang B, Xie P, Huang F, Zhang M, Li W, Zhang M (2024) mgte: Generalized long-context text representation and reranking models for multilingual text retrieval. https:\/\/doi.org\/10.18653\/V1\/2024.EMNLP-INDUSTRY.103","DOI":"10.18653\/V1\/2024.EMNLP-INDUSTRY.103"}],"container-title":["Empirical Software Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10923-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10664-026-10923-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10923-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T07:49:39Z","timestamp":1784965779000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10664-026-10923-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,25]]},"references-count":43,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2027,2]]}},"alternative-id":["10923"],"URL":"https:\/\/doi.org\/10.1007\/s10664-026-10923-2","relation":{},"ISSN":["1382-3256","1573-7616"],"issn-type":[{"value":"1382-3256","type":"print"},{"value":"1573-7616","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,25]]},"assertion":[{"value":"27 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 July 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"This research did not involve human participants or animals; therefore, ethical approval was not required.","order":1,"name":"Ethics","label":"Ethical Approval","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"No personal data or identifiable information is reported in this research; informed consent is not applicable.","order":2,"name":"Ethics","label":"Informed Consent","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflicts of interest.","order":3,"name":"Ethics","label":"Conflicts of Interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"12"}}