{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T18:31:17Z","timestamp":1782930677341,"version":"3.54.5"},"reference-count":77,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100000140","name":"U.S. Department of Transportation","doi-asserted-by":"publisher","award":["69A3552348303"],"award-info":[{"award-number":["69A3552348303"]}],"id":[{"id":"10.13039\/100000140","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Scientometrics"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11192-026-05643-9","type":"journal-article","created":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T04:05:26Z","timestamp":1780632326000},"page":"3853-3895","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mapping scientific literature with large language models and topic modeling"],"prefix":"10.1007","volume":"131","author":[{"given":"Mason","family":"Smetana","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lev","family":"Khazanovich","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,5]]},"reference":[{"issue":"49","key":"5643_CR1","doi-asserted-by":"publisher","first-page":"20899","DOI":"10.1073\/pnas.1013452107","volume":"107","author":"E Airoldi","year":"2010","unstructured":"Airoldi, E., Erosheva, E., Fienberg, S., et al. (2010). Reconceptualizing the classification of PNAS articles. Proceedings of the National Academy of Sciences, 107(49), 20899\u201320904. https:\/\/doi.org\/10.1073\/pnas.1013452107","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"5","key":"5643_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3589955","volume":"30","author":"T August","year":"2023","unstructured":"August, T., Wang, L., Bragg, J., et al. (2023). Paper plain: Making medical research papers approachable to healthcare consumers with natural language processing. ACM Transactions on Computer-Human Interaction, 30(5), 1\u201338. https:\/\/doi.org\/10.1145\/3589955","journal-title":"ACM Transactions on Computer-Human Interaction"},{"issue":"7","key":"5643_CR3","doi-asserted-by":"publisher","first-page":"3839","DOI":"10.1007\/s11192-025-05339-6","volume":"130","author":"P Benz","year":"2025","unstructured":"Benz, P., Pradier, C., Kozlowski, D., Shokida, N. S., & Larivi\u00e9re, V. (2025). Mapping the unseen in practice: Comparing latent Dirichlet allocation and BERTopic for navigating topic spaces. Scientometrics, 130(7), 3839\u20133870. https:\/\/doi.org\/10.1007\/s11192-025-05339-6","journal-title":"Scientometrics"},{"key":"5643_CR4","doi-asserted-by":"publisher","first-page":"993","DOI":"10.5555\/944919.944937","volume":"3","author":"DM Blei","year":"2003","unstructured":"Blei, D. M., Ng, A. Y., & Jordan, M. I. (2003). Latent Dirichlet allocation. Journal of Machine Learning Research, 3, 993\u20131022. https:\/\/doi.org\/10.5555\/944919.944937","journal-title":"Journal of Machine Learning Research"},{"key":"5643_CR5","doi-asserted-by":"publisher","first-page":"5192","DOI":"10.1073\/pnas.0307509100","volume":"101","author":"K Boyack","year":"2004","unstructured":"Boyack, K. (2004). Mapping knowledge domains: Characterizing PNAS. Proceedings of the National Academy of Sciences, 101, 5192\u20135199. https:\/\/doi.org\/10.1073\/pnas.0307509100","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"5643_CR6","unstructured":"Brown, T., Mann, B., Ryder, N., et al. (2020). Language Models are Few-Shot Learners. Proceedings of the 34th international conference on neural information processing systems, nips (Vol. 33, pp. 1877\u20131907)"},{"key":"5643_CR7","doi-asserted-by":"publisher","first-page":"14062","DOI":"10.1073\/pnas.1212729110","volume":"110","author":"W Bruine de Bruin","year":"2013","unstructured":"Bruine de Bruin, W., & Bostrom, A. (2013). Assessing what to address in science communication. Proceedings of the National Academy of Sciences, 110, 14062\u201314068. https:\/\/doi.org\/10.1073\/pnas.1212729110","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"6","key":"5643_CR8","doi-asserted-by":"publisher","first-page":"514","DOI":"10.1038\/nbt0609-514","volume":"27","author":"T Bubela","year":"2009","unstructured":"Bubela, T., Nisbet, M., Borchelt, R., et al. (2009). Science communication reconsidered. Nature Biotechnology, 27(6), 514\u2013518. https:\/\/doi.org\/10.1038\/nbt0609-514","journal-title":"Nature Biotechnology"},{"issue":"1","key":"5643_CR9","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-024-45563-x","volume":"15","author":"J Dagdelen","year":"2024","unstructured":"Dagdelen, J., Dunn, A., Lee, S., et al. (2024). Structured information extraction from scientific text with large language models. Nature Communications, 15(1), Article 1418. https:\/\/doi.org\/10.1038\/s41467-024-45563-x","journal-title":"Nature Communications"},{"issue":"3","key":"5643_CR10","doi-asserted-by":"publisher","first-page":"1817","DOI":"10.1007\/s11192-018-2812-9","volume":"116","author":"J Ding","year":"2018","unstructured":"Ding, J., Ahlgren, P., Yang, L., & Yue, T. (2018). Disciplinary structures in Nature, Science and PNAS: Journal and country levels. Scientometrics, 116(3), 1817\u20131852. https:\/\/doi.org\/10.1007\/s11192-018-2812-9","journal-title":"Scientometrics"},{"key":"5643_CR11","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1016\/j.jbusres.2021.04.070","volume":"133","author":"A Donthu","year":"2021","unstructured":"Donthu, A., Kumar, S., Debmalya, M., Pandey, N., & Lim, W. M. (2021). How to conduct a bibliometric analysis: An overview and guidelines. Journal of Business Research, 133, 285\u2013296. https:\/\/doi.org\/10.1016\/j.jbusres.2021.04.070","journal-title":"Journal of Business Research"},{"key":"5643_CR12","doi-asserted-by":"crossref","unstructured":"Doogan, C., & Buntine, W. (2021). Topic Model or Topic Twaddle? Re-evaluating Semantic Interpretability Measures. Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (pp. 3824\u20133848). Online: Association for Computational Linguistics. Retrieved from DOI: https:\/\/doi.org\/10.18653\/v1\/2021.naaclmain.300","DOI":"10.18653\/v1\/2021.naacl-main.300"},{"key":"5643_CR13","doi-asserted-by":"publisher","first-page":"675","DOI":"10.1007\/s10115-020-01530-8","volume":"63","author":"L Duan","year":"2021","unstructured":"Duan, L., Ma, S., Aggarwal, C., & Saket, S. (2021). Improving spectral clustering with deep embedding, cluster estimation and metric learning. Knowledge and Information Systems, 63, 675\u2013694. https:\/\/doi.org\/10.1007\/s10115-020-01530-8","journal-title":"Knowledge and Information Systems"},{"key":"5643_CR14","doi-asserted-by":"publisher","DOI":"10.3389\/fsoc.2022.886498","volume":"7","author":"R Egger","year":"2022","unstructured":"Egger, R., & Yu, J. (2022). A topic modeling comparison between LDA, NMF, Top2Vec, and BERTopic to demystify Twitter posts. Frontiers in Sociology, 7, Article Article 886498. https:\/\/doi.org\/10.3389\/fsoc.2022.886498","journal-title":"Frontiers in Sociology"},{"key":"5643_CR15","doi-asserted-by":"publisher","first-page":"625","DOI":"10.1038\/s41586-024-07421-0","volume":"630","author":"S Farquhar","year":"2024","unstructured":"Farquhar, S., Kossen, J., Kuhn, L., & Gal, Y. (2024). Detecting hallucinations in large language models using semantic entropy. Nature, 630, 625\u2013630. https:\/\/doi.org\/10.1038\/s41586-024-07421-0","journal-title":"Nature"},{"key":"5643_CR16","doi-asserted-by":"crossref","unstructured":"Ferrando, J., G\u00e1llego, G., Tsiamas, I., & Costa-juss\u2018a, M. (2023). Explaining How Transformers Use Context to Build Predictions. Proceedings of the 61st annual meeting of the association for computational linguistics, acl (Vol. 1, pp. 5486\u20135513). Toronto, Canada: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.301","DOI":"10.18653\/v1\/2023.acl-long.301"},{"issue":"4","key":"5643_CR17","doi-asserted-by":"publisher","first-page":"421","DOI":"10.3109\/a036879","volume":"10","author":"H Field","year":"2001","unstructured":"Field, H., & Powell, P. (2001). Public understanding of science versus public understanding of research. Public Understanding of Science, 10(4), 421\u2013426. https:\/\/doi.org\/10.3109\/a036879","journal-title":"Public Understanding of Science"},{"key":"5643_CR18","unstructured":"Grootendorst, M. (2022). Bertopic: Neural topic modeling with a class-based tf-idf procedure. arXiv preprint arXiv:2203.05794"},{"key":"5643_CR19","doi-asserted-by":"crossref","unstructured":"Guo, Y., Chang, J., Antoniak, M., et al. (2024). Personalized Jargon Identification for Enhanced Interdisciplinary Communication. arXiv preprint arXiv:2311.09481","DOI":"10.18653\/v1\/2024.naacl-long.255"},{"key":"5643_CR20","unstructured":"Guo, Y., Qiu, W., Wang, Y., & Cohen, T. (2020). Automated Lay Language Summarization of Biomedical Scientific Reviews. arXiv preprint arXiv:2012.12573"},{"key":"5643_CR21","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1007\/s11573-016-0822-8","volume":"87","author":"M Hahsler","year":"2017","unstructured":"Hahsler, M., & Karpienko, R. (2017). Visualizing association rules in hierarchical groups. Journal of Business Economics, 87, 317\u2013335. https:\/\/doi.org\/10.1007\/s11573-016-0822-8","journal-title":"Journal of Business Economics"},{"issue":"12","key":"5643_CR22","doi-asserted-by":"publisher","first-page":"12661","DOI":"10.1007\/s10115-025-02605-0","volume":"67","author":"I Hernandez-Camero","year":"2025","unstructured":"Hernandez-Camero, I., Garcia-Cabot, A., Garcia-Lopez, E., Caro-Alvaro, S., & MorenoCediel, A. (2025). Leveraging large language models to enhance clustering-based topic modeling. Knowledge and Information Systems, 67(12), 12661\u201312697. https:\/\/doi.org\/10.1007\/s10115-025-02605-0","journal-title":"Knowledge and Information Systems"},{"key":"5643_CR23","doi-asserted-by":"publisher","unstructured":"Hoyle, A., Goel, P., Hian-Cheong, et al. (2021). Is Automated Topic Model Evaluation Broken? The Incoherence of Coherence. Proceedings of the 35th Conference on Neural Information Processing Systems, NuerIPS (Vol. 34, pp. 2018\u20132033). Online: Curran Associates, Inc. Retrieved from https:\/\/doi.org\/10.5555\/3540261.3540416","DOI":"10.5555\/3540261.3540416"},{"key":"5643_CR24","doi-asserted-by":"crossref","unstructured":"Huang, J., & Chang, K. (2023). Towards Reasoning in Large Language Models: A Survey. arXiv preprint arXiv:2212.10403","DOI":"10.18653\/v1\/2023.findings-acl.67"},{"key":"5643_CR25","doi-asserted-by":"publisher","first-page":"777","DOI":"10.1016\/j.procs.2022.03.106","volume":"201","author":"S Ibrihich","year":"2022","unstructured":"Ibrihich, S., Oussous, A., Ibrihich, O., & Esghir, M. (2022). A review on recent research in information retrieval. Procedia Computer Science, 201, 777\u2013782. https:\/\/doi.org\/10.1016\/j.procs.2022.03.106","journal-title":"Procedia Computer Science"},{"issue":"3","key":"5643_CR26","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pbio.2005468","volume":"16","author":"J Ioannidis","year":"2018","unstructured":"Ioannidis, J. (2018). Meta-research: Why research on research matters. PLoS Biology, 16(3), Article e2005468. https:\/\/doi.org\/10.1371\/journal.pbio.2005468","journal-title":"PLoS Biology"},{"issue":"22","key":"5643_CR27","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2401409121","volume":"121","author":"K Kamani","year":"2024","unstructured":"Kamani, K., & Rogers, S. (2024). Brittle and ductile yielding in soft materials. Proceedings of the National Academy of Sciences, 121(22), Article e2401409121. https:\/\/doi.org\/10.1073\/pnas.2401409121","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"5643_CR28","doi-asserted-by":"publisher","unstructured":"Kaur, A., & Wallace, J.R. (2024). Moving Beyond LDA: A Comparison of Unsupervised Topic Modelling Techniques for Qualitative Data Analysis of Online Communities. https:\/\/doi.org\/10.48550\/arXiv.2412.14486","DOI":"10.48550\/arXiv.2412.14486"},{"issue":"14","key":"5643_CR29","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2320442121","volume":"121","author":"D Koo","year":"2024","unstructured":"Koo, D., Mao, Z., Noguchi, M., et al. (2024). Defining T cell receptor repertoires using nanovial-based binding and functional screening. Proceedings of the National Academy of Sciences, 121(14), Article e2320442121. https:\/\/doi.org\/10.1073\/pnas.2320442121","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"5643_CR30","doi-asserted-by":"publisher","unstructured":"K\u00f6tter, T., G\u00fcnnemann, S., Berthold, M., Faloutsos, C. (2015). Automatic Taxonomy Extraction from Bipartite Graphs. Proceedings of the ieee international conference on data mining (pp. 221\u2013230). Atlantic City, NJ, USA: IEEE. (https:\/\/doi.org\/10.1109\/ICDM.2015.24)","DOI":"10.1109\/ICDM.2015.24)"},{"key":"5643_CR31","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1038\/44565","volume":"401","author":"DD Lee","year":"1999","unstructured":"Lee, D. D., & Seung, H. S. (1999). Learning the parts of objects by non-negative matrix factorization. Nature, 401, 788\u2013791. https:\/\/doi.org\/10.1038\/44565","journal-title":"Nature"},{"key":"5643_CR32","doi-asserted-by":"publisher","unstructured":"Li, P., Wang, Z., Zhang, X., et al. (2025). SciTopic: Enhancing Topic Discovery in Scientific Literature through Advanced LLM. https:\/\/doi.org\/10.48550\/arXiv.2508.20514","DOI":"10.48550\/arXiv.2508.20514"},{"key":"5643_CR33","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1162\/tacla00638","volume":"12","author":"N Liu","year":"2024","unstructured":"Liu, N., Lin, K., Hewitt, J., et al. (2024). Lost in the middle: How language models use long contexts. Transactions of the Association for Computational Linguistics, 12, 157\u2013173. https:\/\/doi.org\/10.1162\/tacla00638","journal-title":"Transactions of the Association for Computational Linguistics"},{"issue":"11","key":"5643_CR34","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2319488121","volume":"121","author":"A Lupia","year":"2024","unstructured":"Lupia, A., Allison, D., Jamieson, K., et al. (2024). Trends in US public confidence in science and opportunities for progress. Proceedings of the National Academy of Sciences, 121(11), Article e2319488121. https:\/\/doi.org\/10.1073\/pnas.2319488121","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"4","key":"5643_CR35","doi-asserted-by":"publisher","first-page":"428","DOI":"10.1177\/0741088313493610","volume":"30","author":"M Luz\u00f3n","year":"2013","unstructured":"Luz\u00f3n, M. (2013). Public communication of science in blogs: Recontextualizing scientific discourse for a diversified audience. Written Communication, 30(4), 428\u2013457. https:\/\/doi.org\/10.1177\/0741088313493610","journal-title":"Written Communication"},{"key":"5643_CR36","doi-asserted-by":"crossref","unstructured":"Ma, L., Chen, R., Ge, W., et al. (2025). AI-powered topic modeling: comparing LDA and BERTopic in analyzing opioid-related cardiovascular risks in women.","DOI":"10.3389\/ebm.2025.10389"},{"key":"5643_CR37","doi-asserted-by":"publisher","first-page":"5287","DOI":"10.1073\/pnas.0307626100","volume":"101","author":"K Mane","year":"2004","unstructured":"Mane, K., & B\u00f6rner, K. (2004). Mapping topics and topic bursts in PNAS. Proceedings of the National Academy of Sciences, 101, 5287\u20135290. https:\/\/doi.org\/10.1073\/pnas.0307626100","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"9","key":"5643_CR38","doi-asserted-by":"publisher","DOI":"10.1093\/pnasnexus\/pgae387","volume":"3","author":"D Markowitz","year":"2024","unstructured":"Markowitz, D. (2024). From complexity to clarity: How AI enhances perceptions of scientists and the public\u2019s understanding of science. PNAS Nexus, 3(9), Article pgae387. https:\/\/doi.org\/10.1093\/pnasnexus\/pgae387","journal-title":"PNAS Nexus"},{"issue":"3","key":"5643_CR39","doi-asserted-by":"publisher","first-page":"1301","DOI":"10.1007\/s11192-020-03441-5","volume":"123","author":"S Milojevi\u0107","year":"2020","unstructured":"Milojevi\u0107, S. (2020). Nature, science, and PNAS: Disciplinary profiles and impact. Scientometrics, 123(3), 1301\u20131315. https:\/\/doi.org\/10.1007\/s11192-020-03441-5","journal-title":"Scientometrics"},{"key":"5643_CR40","unstructured":"Mimno, D., Wallach, H., Talley, E., Leenders, M., & McCallum, A. (2011). Optimizing Semantic Coherence in Topic Models. Proceedings of the 2011 conference on empirical methods in natural language processing (pp. 262\u2013272). Edinburgh, Scotland, UK.: Association for Computational Linguistics. Retrieved from https:\/\/aclanthology.org\/D11-1024\/"},{"key":"5643_CR41","doi-asserted-by":"publisher","unstructured":"Mu, Y., Dong, C., Bontcheva, K., & Song, X. (2024). Large Language Models Offer an Alternative to the Traditional Approach of Topic Modeling. https:\/\/doi.org\/10.48550\/arXiv.2403.16248","DOI":"10.48550\/arXiv.2403.16248"},{"key":"5643_CR42","doi-asserted-by":"crossref","unstructured":"Muennighoff, N., Tazi, N., Magne, L., & Reimers, N. (2023). MTEB: Massive Text Embedding Benchmark. arXiv preprint arXiv:2210.07316","DOI":"10.18653\/v1\/2023.eacl-main.148"},{"key":"5643_CR43","unstructured":"National Academy of Sciences (2025). PNAS home. Proceedings of the National Academy of Sciences Website. Available at https:\/\/www.pnas.org\/. Accessed 3 January 2025."},{"key":"5643_CR44","unstructured":"Neelakantan, A., Xu, T., Puri, R., et al. (2022). Text and Code Embeddings by Contrastive Pre-Training. arXiv preprint arXiv:2201.10005"},{"issue":"5","key":"5643_CR45","doi-asserted-by":"publisher","first-page":"323","DOI":"10.1080\/00107510500052444","volume":"46","author":"M Newman","year":"2005","unstructured":"Newman, M. (2005). Power laws, Pareto distributions, and Zipf\u2019s law. Contemporary Physics, 46(5), 323\u2013351. https:\/\/doi.org\/10.1080\/00107510500052444","journal-title":"Contemporary Physics"},{"key":"5643_CR46","unstructured":"OpenAI (2024a). GPT-4o mini. (OpenAI Platform Website. Available at https:\/\/ platform.openai.com\/docs\/models\/gpt-4o-mini. Accessed 9 February 2025.)"},{"key":"5643_CR47","unstructured":"OpenAI (2024b). text-embedding-3-small. OpenAI Platform Website. Available at https:\/\/platform.openai.com\/docs\/models\/text-embedding-3-small\/. Accessed 9 Feb 2025."},{"key":"5643_CR48","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/s41746-023-00958-w","volume":"6","author":"C Peng","year":"2023","unstructured":"Peng, C., Yang, X., Chen, A., et al. (2023). A study of generative large language model for medical research and healthcare. Npj Digital Medicine, 6, 1\u201317. https:\/\/doi.org\/10.1038\/s41746-023-00958-w","journal-title":"Npj Digital Medicine"},{"issue":"24","key":"5643_CR49","doi-asserted-by":"publisher","DOI":"10.3390\/electronics12244957","volume":"12","author":"D-M Petrosanu","year":"2023","unstructured":"Petrosanu, D.-M., P\u00eerjan, A., & T\u0103busc\u0103, A. (2023). Tracing the influence of large language models across the most impactful scientific works. Electronics, 12(24), Article 4957. https:\/\/doi.org\/10.3390\/electronics12244957","journal-title":"Electronics"},{"issue":"3","key":"5643_CR50","doi-asserted-by":"publisher","first-page":"549","DOI":"10.1111\/cgf.13441","volume":"37","author":"N Pezzoti","year":"2018","unstructured":"Pezzoti, N., Fekete, J.-D., H\u00f3llt, T., et al. (2018). Multiscale visualization and exploration of large bipartite graphs. Computer Graphics Forum, 37(3), 549\u2013560. https:\/\/doi.org\/10.1111\/cgf.13441","journal-title":"Computer Graphics Forum"},{"key":"5643_CR51","doi-asserted-by":"crossref","unstructured":"Pham, C.M., Hoyle, A., Sun, S., Resnik, P., & Iyyer, M. (2024). TopicGPT: A Promptbased Topic Modeling Framework. Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) (pp. 2956\u20132984). Mexico City, Mexico: Association for Computational Linguistics. Retrieved from https:\/\/aclanthology.org\/2024.naacl-long.164\/","DOI":"10.18653\/v1\/2024.naacl-long.164"},{"key":"5643_CR52","doi-asserted-by":"publisher","first-page":"1112","DOI":"10.3758\/s13423-014-0585-6","volume":"21","author":"S Piantadosi","year":"2014","unstructured":"Piantadosi, S. (2014). Zipf\u2019s word frequency law in natural language: A critical review and future directions. Psychonomic Bulletin & Review, 21, 1112\u20131130. https:\/\/doi.org\/10.3758\/s13423-014-0585-6","journal-title":"Psychonomic Bulletin & Review"},{"key":"5643_CR53","doi-asserted-by":"publisher","unstructured":"Qader, W., Ameen, M., & Ahmed, B. (2019). An Overview of Bag of Words;Importance, Implementation, Applications, and Challenges. Proceedings of the international engineering conference, iec (pp. 200\u2013204). Erbil, Iraq: IEEE. (https:\/\/doi.org\/10.1109\/IEC47844.2019.8950616","DOI":"10.1109\/IEC47844.2019.8950616"},{"key":"5643_CR54","unstructured":"Radford, A., Wu, J., Child, R., et al. (2019). Language Models are Unsupervised Multitask Learners. Retrieved from https:\/\/cdn.openai.com\/better-languagemodels\/language models are unsupervised multitask learners.pdf (OpenAI preprint. Accesssed 14 Nov 2024)"},{"key":"5643_CR55","doi-asserted-by":"publisher","unstructured":"R\u00f6der, M., Both, A., & Hinneburg, A. (2015). Exploring the Space of Topic Coherence Measures. Proceedings of the Eighth ACM International Conference on Web Search and Data Mining (pp. 399\u2013408). New York, NY, USA: Association for Computing Machinery. Retrieved from https:\/\/doi.org\/10.1145\/2684822.2685324","DOI":"10.1145\/2684822.2685324"},{"issue":"11","key":"5643_CR56","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2212270120","volume":"120","author":"P Saha","year":"2023","unstructured":"Saha, P., Garimella, K., Kalyan, N., et al. (2023). On the rise of fear speech in online social media. Proceedings of the National Academy of Sciences, 120(11), Article Article e2212270120. https:\/\/doi.org\/10.1073\/pnas.2212270120","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"8","key":"5643_CR57","doi-asserted-by":"publisher","first-page":"2755","DOI":"10.1073\/pnas.0800528105","volume":"105","author":"R Schekman","year":"2008","unstructured":"Schekman, R. (2008). Charting the course for PNAS. Proceedings of the National Academy of Sciences, 105(8), 2755\u20132756. https:\/\/doi.org\/10.1073\/pnas.0800528105","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"5","key":"5643_CR58","doi-asserted-by":"publisher","first-page":"528","DOI":"10.1177\/0963662512469916","volume":"23","author":"A Sharon","year":"2014","unstructured":"Sharon, A., & Baram-Tsabari, A. (2014). Measuring mumbo jumbo: A preliminary quantification of the use of jargon in science communication. Public Understanding of Science, 23(5), 528\u2013546. https:\/\/doi.org\/10.1177\/0963662512469916","journal-title":"Public Understanding of Science"},{"key":"5643_CR59","doi-asserted-by":"publisher","first-page":"5183","DOI":"10.1073\/pnas.0307852100","volume":"101","author":"R Shiffrin","year":"2004","unstructured":"Shiffrin, R., & B\u00f3rner, K. (2004). Mapping knowledge domains. Proceedings of the National Academy of Sciences, 101, 5183\u20135185.","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"5643_CR60","doi-asserted-by":"publisher","DOI":"10.1186\/s13638-021-01910-w","author":"C Shi","year":"2021","unstructured":"Shi, C., Wei, B., Wei, S., Wang, W., & Liu, J. (2021). A quantitative discriminant method of elbow point for the optimal number of clusters in clustering algorithm. EURASIP Journal on Wireless Communications and Networking. https:\/\/doi.org\/10.1186\/s13638-021-01910-w","journal-title":"EURASIP Journal on Wireless Communications and Networking"},{"issue":"10","key":"5643_CR61","doi-asserted-by":"publisher","first-page":"2294","DOI":"10.1093\/jamia\/ocae186","volume":"31","author":"C Shyr","year":"2024","unstructured":"Shyr, C., Grout, R., Kennedy, N., et al. (2024). Leveraging artificial intelligence to summarize abstracts in lay language for increasing research accessibility and transparency. Journal of the American Medical Informatics Association, 31(10), 2294\u20132303. https:\/\/doi.org\/10.1093\/jamia\/ocae186","journal-title":"Journal of the American Medical Informatics Association"},{"issue":"4","key":"5643_CR62","doi-asserted-by":"publisher","DOI":"10.3390\/app14041352","volume":"14","author":"M Smetana","year":"2024","unstructured":"Smetana, M., Salles de Salles, L., Sukharev, I., & Khazanovich, L. (2024). Highway construction safety analysis using large language models. Applied Sciences, 14(4), Article 1352. https:\/\/doi.org\/10.3390\/app14041352","journal-title":"Applied Sciences"},{"key":"5643_CR63","doi-asserted-by":"publisher","first-page":"2647","DOI":"10.1007\/s10439-023-03284-0","volume":"51","author":"S Thapa","year":"2023","unstructured":"Thapa, S., & Adhikari, S. (2023). ChatGPT, bard, and large language models for biomedical research: Opportunities and pitfalls. Annals of Biomedical Engineering, 51, 2647\u20132651. https:\/\/doi.org\/10.1007\/s10439-023-03284-0","journal-title":"Annals of Biomedical Engineering"},{"issue":"29","key":"5643_CR64","doi-asserted-by":"publisher","first-page":"24421","DOI":"10.1007\/s00521-025-11593-9","volume":"37","author":"PC Theocharopoulos","year":"2025","unstructured":"Theocharopoulos, P. C., Anagnostou, P., Georgakopoulos, S. V., Tasoulis, S. K., & Plagianakos, V. P. (2025). Large language models for efficient topic modeling. Neural Computing and Applications, 37(29), 24421\u201324439. https:\/\/doi.org\/10.1007\/s00521-025-11593-9","journal-title":"Neural Computing and Applications"},{"key":"5643_CR65","doi-asserted-by":"publisher","unstructured":"Tran, N.K., Zerr, S., Bischoff, K., Nieder\u00e9e, C., & Krestel, R. (2013). Topic Cropping: Leveraging Latent Topics for the Analysis of Small Corpora. Research and Advanced Technology for Digital Libraries (Vol. 8092, pp. 297\u2013308). Berlin, Heidelberg: Springer. Retrieved from https:\/\/doi.org\/10.1007\/978-3-642-40501-3_30","DOI":"10.1007\/978-3-642-40501-3_30"},{"issue":"86","key":"5643_CR66","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten, L., & Hinton, G. (2008). Visualizing data using t-SNE. Journal of Machine Learning Research, 9(86), 2579\u20132605.","journal-title":"Journal of Machine Learning Research"},{"key":"5643_CR67","doi-asserted-by":"publisher","first-page":"523","DOI":"10.1007\/s11192-009-0146-3","volume":"84","author":"NJ Van Eck","year":"2010","unstructured":"Van Eck, N. J., & Waltman, L. (2010). Software survey: VOSviewer, a computer program for bibliometric mapping. Scientometrics, 84, 523\u2013538. https:\/\/doi.org\/10.1007\/s11192-009-0146-3","journal-title":"Scientometrics"},{"key":"5643_CR68","doi-asserted-by":"publisher","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al. (2017). Attention Is All You Need. Proceedings of the 31st conference on neural information processing systems, nips (Vol. 30). Long Beach, California: Curran Associates, Inc. Retrieved from https:\/\/doi.org\/10.5555\/3295222.3295349","DOI":"10.5555\/3295222.3295349"},{"issue":"26","key":"5643_CR69","doi-asserted-by":"publisher","first-page":"7875","DOI":"10.1073\/pnas.1509912112","volume":"112","author":"I Verma","year":"2015","unstructured":"Verma, I. (2015). Impact, not impact factor. Proceedings of the National Academy of Sciences, 112(26), 7875\u20137876. https:\/\/doi.org\/10.1073\/pnas.1509912112","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"11","key":"5643_CR70","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pbio.2006930","volume":"16","author":"J Wallach","year":"2018","unstructured":"Wallach, J., Boyack, K., & Ioannidis, J. (2018). Reproducible research practices, transparency, and open access data in the biomedical literature 2015\u20132017. PLoS Biology, 16(11), Article e2006930. https:\/\/doi.org\/10.1371\/journal.pbio.2006930","journal-title":"PLoS Biology"},{"issue":"21","key":"5643_CR71","doi-asserted-by":"publisher","DOI":"10.3390\/app122111220","volume":"12","author":"M-H Weng","year":"2022","unstructured":"Weng, M.-H., Wu, S., & Dyer, M. (2022). Identification and visualization of key topics in scientific publications with transformer-based language models and document clustering methods. Applied Sciences, 12(21), Article 11220. https:\/\/doi.org\/10.3390\/app122111220","journal-title":"Applied Sciences"},{"issue":"2","key":"5643_CR72","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1007\/s10462-023-10661-7","volume":"57","author":"X Wu","year":"2024","unstructured":"Wu, X., Nguyen, T., & Luu, A. T. (2024). A survey on neural topic models: Methods, applications, and challenges. Artificial Intelligence Review, 57(2), 18. https:\/\/doi.org\/10.1007\/s10462-023-10661-7","journal-title":"Artificial Intelligence Review"},{"key":"5643_CR73","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1140\/epjds\/s13688-018-0134-z","volume":"7","author":"Z Xie","year":"2018","unstructured":"Xie, Z., Li, M., Li, J., Duan, X., & Ouyang, Z. (2018). Feature analysis of multidisciplinary scientific collaboration patterns based on PNAS. EPJ Data Science, 7, 1\u201317. https:\/\/doi.org\/10.1140\/epjds\/s13688-018-0134-z","journal-title":"EPJ Data Science"},{"issue":"3","key":"5643_CR74","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1080\/01587919.2016.1185079","volume":"37","author":"O Zawacki-Richter","year":"2016","unstructured":"Zawacki-Richter, O., & Naidu, S. (2016). Mapping research trends from 35 years of publications in distance education. Distance Education, 37(3), 245\u2013269. https:\/\/doi.org\/10.1080\/01587919.2016.1185079","journal-title":"Distance Education"},{"issue":"6","key":"5643_CR75","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3715318","volume":"57","author":"Q Zhang","year":"2025","unstructured":"Zhang, Q., Ding, K., Lv, T., et al. (2025). Scientific large language models: A survey on biological & chemical domains. ACM Computing Surveys, 57(6), 1\u201338. https:\/\/doi.org\/10.1145\/3715318","journal-title":"ACM Computing Surveys"},{"key":"5643_CR76","doi-asserted-by":"publisher","DOI":"10.1186\/1471-2105-16-S13-S8","volume":"16","author":"W Zhao","year":"2015","unstructured":"Zhao, W., Chen, J. J., Perkins, R., et al. (2015). A heuristic approach to determine an appropriate number of topics in topic modeling. BMC Bioinformatics, 16, Article S8. https:\/\/doi.org\/10.1186\/1471-2105-16-S13-S8","journal-title":"BMC Bioinformatics"},{"key":"5643_CR77","unstructured":"Zipf, G. (1935). The Psycho-Biology of Language: an Introduction to Dynamic Philology. Boston, MA: Houghton Mifflin. (Reprinted by Routledge, 2014)"}],"container-title":["Scientometrics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11192-026-05643-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11192-026-05643-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11192-026-05643-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T17:42:55Z","timestamp":1782927775000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11192-026-05643-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":77,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["5643"],"URL":"https:\/\/doi.org\/10.1007\/s11192-026-05643-9","relation":{},"ISSN":["0138-9130","1588-2861"],"issn-type":[{"value":"0138-9130","type":"print"},{"value":"1588-2861","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"5 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}