{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T07:16:22Z","timestamp":1760426182548,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":34,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819981779"},{"type":"electronic","value":"9789819981786"}],"license":[{"start":{"date-parts":[[2023,11,30]],"date-time":"2023-11-30T00:00:00Z","timestamp":1701302400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,30]],"date-time":"2023-11-30T00:00:00Z","timestamp":1701302400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-99-8178-6_38","type":"book-chapter","created":{"date-parts":[[2023,11,29]],"date-time":"2023-11-29T10:02:54Z","timestamp":1701252174000},"page":"504-517","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Assessing and\u00a0Enhancing LLMs: A Physics and\u00a0History Dataset and\u00a0One-More-Check Pipeline Method"],"prefix":"10.1007","author":[{"given":"Chaofan","family":"He","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunhui","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianyuan","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liping","family":"Shen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,30]]},"reference":[{"key":"38_CR1","doi-asserted-by":"crossref","unstructured":"Borji, A.: A categorical archive of ChatGPT failures (2023)","DOI":"10.21203\/rs.3.rs-2895792\/v1"},{"key":"38_CR2","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"38_CR3","doi-asserted-by":"publisher","unstructured":"Creswell, A., White, T., Dumoulin, V., Arulkumaran, K., Sengupta, B., Bharath, A.A.: Generative adversarial networks: an overview. IEEE Signal Process. Mag 35(1), 53\u201365 (2018). https:\/\/doi.org\/10.1109\/msp.2017.2765202 , https:\/\/doi.org\/10.1109\/2Fmsp.2017.2765202","DOI":"10.1109\/msp.2017.2765202 10.1109\/2Fmsp.2017.2765202"},{"key":"38_CR4","doi-asserted-by":"crossref","unstructured":"Dhingra, S., Singh, M., SB, V., Malviya, N., Gill, S.S.: Mind meets machine: unravelling GPT-4\u2019s cognitive psychology (2023)","DOI":"10.1016\/j.tbench.2023.100139"},{"key":"38_CR5","unstructured":"Dong, Q., et al: A survey on in-context learning (2023)"},{"key":"38_CR6","doi-asserted-by":"crossref","unstructured":"Du, Z., Qian, Y., Liu, X., Ding, M., Qiu, J., Yang, Z., Tang, J.: GLM: general language model pretraining with autoregressive blank infilling. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 320\u2013335 (2022)","DOI":"10.18653\/v1\/2022.acl-long.26"},{"key":"38_CR7","unstructured":"Frieder, S., et al.: Mathematical capabilities of ChatGPT (2023)"},{"key":"38_CR8","unstructured":"Huang, Y., et al.: C-Eval: a multi-level multi-discipline Chinese evaluation suite for foundation models. arXiv preprint arXiv:2305.08322 (2023)"},{"key":"38_CR9","doi-asserted-by":"crossref","unstructured":"Inaba, T., Kiyomaru, H., Cheng, F., Kurohashi, S.: Multitool-cot: GPT-3 can use multiple external tools with chain of thought prompting (2023)","DOI":"10.18653\/v1\/2023.acl-short.130"},{"key":"38_CR10","unstructured":"Kasai, J., Kasai, Y., Sakaguchi, K., Yamada, Y., Radev, D.: Evaluating GPT-4 and ChatGPT on Japanese medical licensing examinations (2023)"},{"key":"38_CR11","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners (2023)"},{"key":"38_CR12","unstructured":"Li, X., et al.: Chain of knowledge: a framework for grounding large language models with structured knowledge bases (2023)"},{"key":"38_CR13","doi-asserted-by":"crossref","unstructured":"Min, S., et al.: Rethinking the role of demonstrations: what makes in-context learning work? In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 11048\u201311064. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates (2022). https:\/\/aclanthology.org\/2022.emnlp-main.759","DOI":"10.18653\/v1\/2022.emnlp-main.759"},{"key":"38_CR14","unstructured":"M\u00fcndler, N., He, J., Jenko, S., Vechev, M.: Self-contradictory hallucinations of large language models: evaluation, detection and mitigation (2023)"},{"key":"38_CR15","unstructured":"Nori, H., King, N., McKinney, S.M., Carignan, D., Horvitz, E.: Capabilities of GPT-4 on medical challenge problems (2023)"},{"key":"38_CR16","unstructured":"Nunes, D., Primi, R., Pires, R., Lotufo, R., Nogueira, R.: Evaluating GPT-3.5 and GPT-4 models on Brazilian university admission exams (2023)"},{"key":"38_CR17","unstructured":"OpenAI: GPT-4 technical report (2023)"},{"key":"38_CR18","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback (2022)"},{"key":"38_CR19","unstructured":"Rae, J.W., et al.: Scaling language models: methods, analysis & insights from training gopher (2022)"},{"key":"38_CR20","doi-asserted-by":"crossref","unstructured":"Savelka, J., Agarwal, A., Bogart, C., Song, Y., Sakr, M.: Can generative pre-trained transformers (GPT) pass assessments in higher education programming courses? (2023)","DOI":"10.1145\/3587102.3588792"},{"key":"38_CR21","unstructured":"Turpin, M., Michael, J., Perez, E., Bowman, S.R.: Language models don\u2019t always say what they think: unfaithful explanations in chain-of-thought prompting (2023)"},{"key":"38_CR22","unstructured":"Wang, B., Yue, X., Sun, H.: Can ChatGPT defend the truth? automatic dialectical evaluation elicits LLMs\u2019 deficiencies in reasoning (2023)"},{"key":"38_CR23","unstructured":"Wang, X., et al.: Self-consistency improves chain of thought reasoning in language models (2023)"},{"key":"38_CR24","unstructured":"Wei, J., et al.: Emergent abilities of large language models (2022)"},{"key":"38_CR25","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models (2023)"},{"key":"38_CR26","unstructured":"Yao, S., et al.: Tree of thoughts: deliberate problem solving with large language models (2023)"},{"key":"38_CR27","doi-asserted-by":"crossref","unstructured":"Yao, Y., Li, Z., Zhao, H.: Beyond chain-of-thought, effective graph-of-thought reasoning in large language models (2023)","DOI":"10.18653\/v1\/2024.findings-naacl.183"},{"key":"38_CR28","unstructured":"Ye, X., Durrett, G.: The unreliability of explanations in few-shot prompting for textual reasoning (2022)"},{"key":"38_CR29","unstructured":"Yuan, Z., Yuan, H., Tan, C., Wang, W., Huang, S.: How well do large language models perform in arithmetic tasks? (2023)"},{"key":"38_CR30","unstructured":"Zeng, A., et al.: GLM-130B: an open bilingual pre-trained model. arXiv preprint arXiv:2210.02414 (2022)"},{"key":"38_CR31","unstructured":"Zhang, X., Li, C., Zong, Y., Ying, Z., He, L., Qiu, X.: Evaluating the performance of large language models on GAOKAO benchmark (2023)"},{"key":"38_CR32","unstructured":"Zhao, W.X., et al.: A survey of large language models (2023)"},{"key":"38_CR33","doi-asserted-by":"crossref","unstructured":"Zhu, W., Thomason, J., Jia, R.: Chain-of-questions training with latent answers for robust multistep question answering (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.547"},{"key":"38_CR34","unstructured":"Ziegler, D.M., et al.: Fine-tuning language models from human preferences (2020)"}],"container-title":["Communications in Computer and Information Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-99-8178-6_38","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T07:38:37Z","timestamp":1730705917000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-99-8178-6_38"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,30]]},"ISBN":["9789819981779","9789819981786"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-981-99-8178-6_38","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2023,11,30]]},"assertion":[{"value":"30 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Changsha","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 November 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2023.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1274","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"650","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"51% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.14","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.46","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}