{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T21:17:48Z","timestamp":1783113468521,"version":"3.54.6"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T00:00:00Z","timestamp":1783036800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T00:00:00Z","timestamp":1783036800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["The VLDB Journal"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1007\/s00778-026-00993-5","type":"journal-article","created":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T20:27:15Z","timestamp":1783110435000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Distilled documentation for dialect-specific SQL generation"],"prefix":"10.1007","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7483-0099","authenticated-orcid":false,"given":"Till","family":"D\u00f6hmen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Adithya","family":"Krishnan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hamilton","family":"Ulmer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peter","family":"Boncz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sebastian","family":"Schelter","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,3]]},"reference":[{"key":"993_CR1","unstructured":"Albrecht, A.: python-sqlparse \u2013 a non-validating sql parser for python. https:\/\/github.com\/andialbrecht\/sqlparse (2023)"},{"key":"993_CR2","unstructured":"Atwal, R., Boyd, R., Courtney, A., D\u00f6hmen, T., Gerlinghoff, F., Huang, J., Hwang, J., Hyde, R., Felder, E., et\u00a0al.: Motherduck: Duckdb in the cloud and in the client. (2024)"},{"issue":"1","key":"993_CR3","first-page":"1","volume":"3","author":"S Chen","year":"2025","unstructured":"Chen, S., Fan, J., Wu, B., Tang, N., Deng, C., Wang, P., Li, Y., Tan, J., Li, F., Zhou, J., et al.: Automatic database configuration debugging using retrieval-augmented language models. Proceedings of the ACM on Management of Data 3(1), 1\u201327 (2025)","journal-title":"Proceedings of the ACM on Management of Data"},{"issue":"3","key":"993_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3654975","volume":"2","author":"T D\u00f6hmen","year":"2024","unstructured":"D\u00f6hmen, T., Geacu, R., Hulsebos, M., Schelter, S.: Schemapile: a large collection of relational database schemas. Proceedings of the ACM on Management of Data 2(3), 1\u201325 (2024)","journal-title":"Proceedings of the ACM on Management of Data"},{"key":"993_CR5","unstructured":"D\u00f6hmen, T., Wu, S., Orr, L.: Duckdb nsql 7b model (huggingface) (2024). https:\/\/huggingface.co\/motherduckdb\/DuckDB-NSQL-7B-v0.1"},{"key":"993_CR6","unstructured":"D\u00f6hmen, T., Wu, S., Orr, L., Fahlgreen, C.: Duckdb nsql hub (2024). https:\/\/huggingface.co\/duckdb-nsql-hub"},{"key":"993_CR7","unstructured":"Duan, Y., Yu, Y., Zhao, X., Wu, Y., Liu, W.: Pdc & dm-sft: A road for llm sql bug-fix enhancing. In: Proceedings of the 31st International Conference on Computational Linguistics: Industry Track, pp. 76\u201390 (2025)"},{"key":"993_CR8","unstructured":"Floratou, A., Psallidas, F., Zhao, F., Deep, S., Hagleither, G., Tan, W., Cahoon, J., Alotaibi, R., Henkel, J., Singla, A., et\u00a0al.: Nl2sql is a solved problem... not! In: CIDR (2024)"},{"issue":"6","key":"993_CR9","doi-asserted-by":"publisher","first-page":"1534","DOI":"10.14778\/3583140.3583165","volume":"16","author":"H Fu","year":"2023","unstructured":"Fu, H., Liu, C., Wu, B., Li, F., Tan, J., Sun, J.: Catsql: Towards real world natural language to sql applications. Proceedings of the VLDB Endowment 16(6), 1534\u20131547 (2023)","journal-title":"Proceedings of the VLDB Endowment"},{"key":"993_CR10","unstructured":"Ganesh, S., Purwar, A., et\u00a0al.: Context-augmented retrieval: A novel framework for fast information retrieval based response generation using large language model. arXiv preprint arXiv:2406.16383 (2024)"},{"issue":"5","key":"993_CR11","doi-asserted-by":"publisher","first-page":"1132","DOI":"10.14778\/3641204.3641221","volume":"17","author":"D Gao","year":"2024","unstructured":"Gao, D., Wang, H., Li, Y., Sun, X., Qian, Y., Ding, B., Zhou, J.: Text-to-sql empowered by large language models: A benchmark evaluation. Proceedings of the VLDB Endowment 17(5), 1132\u20131145 (2024)","journal-title":"Proceedings of the VLDB Endowment"},{"issue":"8","key":"993_CR12","doi-asserted-by":"publisher","first-page":"1939","DOI":"10.14778\/3659437.3659449","volume":"17","author":"J Lao","year":"2024","unstructured":"Lao, J., Wang, Y., Li, Y., Wang, J., Zhang, Y., Cheng, Z., Chen, W., Tang, M., Wang, J.: Gptuner: A manual-reading database tuning system via gpt-guided bayesian optimization. Proceedings of the VLDB Endowment 17(8), 1939\u20131952 (2024)","journal-title":"Proceedings of the VLDB Endowment"},{"key":"993_CR13","unstructured":"Lei, F., Chen, J., Ye, Y., Cao, R., Shin, D., Hongjin, S., SUO, Z., Gao, H., Hu, W., Yin, P., et\u00a0al.: Spider 2.0: Evaluating language models on real-world enterprise text-to-sql workflows. In: The Thirteenth International Conference on Learning Representations"},{"key":"993_CR14","unstructured":"Leviathan, Y., Kalman, M., Matias, Y.: Fast inference from transformers via speculative decoding. In: International Conference on Machine Learning, pp. 19,274\u201319,286. PMLR (2023)"},{"issue":"11","key":"993_CR15","doi-asserted-by":"publisher","first-page":"3318","DOI":"10.14778\/3681954.3682003","volume":"17","author":"B Li","year":"2024","unstructured":"Li, B., Luo, Y., Chai, C., Li, G., Tang, N.: The dawn of natural language to sql: Are we fully ready? Proceedings of the VLDB Endowment 17(11), 3318\u20133331 (2024)","journal-title":"Proceedings of the VLDB Endowment"},{"issue":"3","key":"993_CR16","first-page":"1","volume":"2","author":"H Li","year":"2024","unstructured":"Li, H., Zhang, J., Liu, H., Fan, J., Zhang, X., Zhu, J., Wei, R., Pan, H., Li, C., Chen, H.: Codes: Towards building open-source language models for text-to-sql. Proceedings of the ACM on Management of Data 2(3), 1\u201328 (2024)","journal-title":"Proceedings of the ACM on Management of Data"},{"key":"993_CR17","doi-asserted-by":"publisher","first-page":"42330","DOI":"10.52202\/075280-1835","volume":"36","author":"J Li","year":"2023","unstructured":"Li, J., Hui, B., Qu, G., Yang, J., Li, B., Li, B., Wang, B., Qin, B., Geng, R., Huo, N., et al.: Can llm already serve as a database interface? a big bench for large-scale database grounded text-to-sqls. Adv. Neural. Inf. Process. Syst. 36, 42330\u201342357 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"993_CR18","unstructured":"Li, J., et.al.: Bird-critic (2025). https:\/\/bird-critic.github.io\/"},{"key":"993_CR19","unstructured":"Liu, A., Feng, B., Xue, B., Wang, B., Wu, B., Lu, C., Zhao, C., Deng, C., Zhang, C., Ruan, C., et\u00a0al.: Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)"},{"key":"993_CR20","unstructured":"Modarressi, A., Deilamsalehy, H., Dernoncourt, F., Bui, T., Rossi, R.A., Yoon, S., Sch\u00fctze, H.: Nolima: Long-context evaluation beyond literal matching. arXiv preprint arXiv:2502.05167 (2025)"},{"key":"993_CR21","doi-asserted-by":"publisher","first-page":"19327","DOI":"10.52202\/075280-0848","volume":"36","author":"J Mu","year":"2023","unstructured":"Mu, J., Li, X., Goodman, N.: Learning to compress prompts with gist tokens. Adv. Neural. Inf. Process. Syst. 36, 19327\u201319352 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"993_CR22","unstructured":"M\u00fchleisen, H., Raasveldt, M.: Runtime-extensible parsers. CIDR (2025)"},{"key":"993_CR23","doi-asserted-by":"crossref","unstructured":"Ngom, A.L., Kraska, T.: Mallet: Sql dialect translation with llm rule generation. In: Proceedings of the Seventh International Workshop on Exploiting Artificial Intelligence Techniques for Data Management, pp. 1\u20135 (2024)","DOI":"10.1145\/3663742.3663973"},{"key":"993_CR24","doi-asserted-by":"publisher","first-page":"1591","DOI":"10.1007\/s00778-024-00864-x","volume":"33","author":"JJ Pan","year":"2024","unstructured":"Pan, J.J., Wang, J., Li, G.: Survey of vector database management systems. VLDB J. 33, 1591\u20131615 (2024)","journal-title":"VLDB J."},{"key":"993_CR25","unstructured":"Pourreza, M., Sun, R., Li, H., Miculicich, L., Pfister, T., Arik, S.\u00d6.: Sql-gen: Bridging the dialect gap for text-to-sql via synthetic data and model merging. CoRR (2024)"},{"key":"993_CR26","doi-asserted-by":"crossref","unstructured":"Raasveldt, M., M\u00fchleisen, H.: Duckdb: an embeddable analytical database. In: Proceedings of the 2019 international conference on management of data, pp. 1981\u20131984 (2019)","DOI":"10.1145\/3299869.3320212"},{"key":"993_CR27","unstructured":"Renggli, C., Ilyas, I.F., Rekatsinas, T.: Fundamental challenges in evaluating text2sql solutions and detecting their limitations. arXiv preprint arXiv:2501.18197 (2025)"},{"issue":"4","key":"993_CR28","first-page":"333","volume":"3","author":"S Robertson","year":"2009","unstructured":"Robertson, S., Zaragoza, H.: The probabilistic relevance framework: BM25 and beyond. Found. Trends Inf. Retr. 3(4), 333\u2013389 (2009)","journal-title":"Found. Trends Inf. Retr."},{"key":"993_CR29","unstructured":"Snell, C., Klein, D., Zhong, R.: Learning by distilling context. arXiv preprint arXiv:2209.15189 (2022)"},{"key":"993_CR30","unstructured":"Thakur, N., Reimers, N., Daxenberger, J., Gurevych, I.: BEIR: A heterogeneous benchmark for zero-shot evaluation of information retrieval models. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 5500\u20135522 (2021)"},{"issue":"11","key":"993_CR31","doi-asserted-by":"publisher","first-page":"3511","DOI":"10.14778\/3681954.3682017","volume":"17","author":"I Trummer","year":"2024","unstructured":"Trummer, I.: Generating succinct descriptions of database schemata for cost-efficient prompting of large language models. Proceedings of the VLDB Endowment 17(11), 3511\u20133523 (2024)","journal-title":"Proceedings of the VLDB Endowment"},{"key":"993_CR32","unstructured":"Verma, S.: Contextual compression in retrieval-augmented generation for large language models: A survey. arXiv preprint arXiv:2409.13385 (2024)"},{"key":"993_CR33","doi-asserted-by":"crossref","unstructured":"Yu, T., Zhang, R., Yang, K., Yasunaga, M., Wang, D., Li, Z., Ma, J., Li, I., Yao, Q., Roman, S., et\u00a0al.: Spider: A large-scale human-labeled dataset for complex and cross-domain semantic parsing and text-to-sql task. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing (2018)","DOI":"10.18653\/v1\/D18-1425"},{"key":"993_CR34","unstructured":"Zhong, S., Rigger, M.: Testing database systems with large language model synthesized fragments. CoRR arXiv:abs\/2505.02012 (2025)"},{"key":"993_CR35","doi-asserted-by":"crossref","unstructured":"Zhou, X., Li, G., Sun, Z., Liu, Z., Chen, W., Wu, J., Liu, J., Feng, R., Zeng, G.: D-bot: Database diagnosis system using large language models. Proceedings of the VLDB Endowment 17(10), 2514\u20132527 (2024)","DOI":"10.14778\/3675034.3675043"}],"container-title":["The VLDB Journal"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00778-026-00993-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00778-026-00993-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00778-026-00993-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T20:27:31Z","timestamp":1783110451000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00778-026-00993-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,3]]},"references-count":35,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,9]]}},"alternative-id":["993"],"URL":"https:\/\/doi.org\/10.1007\/s00778-026-00993-5","relation":{},"ISSN":["1066-8888","0949-877X"],"issn-type":[{"value":"1066-8888","type":"print"},{"value":"0949-877X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,3]]},"assertion":[{"value":"16 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 May 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 July 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"37"}}