<?xml version="1.0" encoding="UTF-8"?><doi_batch version="4.3.7" xmlns="http://www.crossref.org/schema/4.3.7" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.crossref.org/schema/4.3.7 http://www.crossref.org/schema/deposit/crossref4.3.7.xsd">
		<head>
		<doi_batch_id>cirpublications.com-WeFv-1790880674-f858630908</doi_batch_id>
		<timestamp>1790880674</timestamp>
		<depositor>
			<depositor_name>Clinical Intelligence Research Press</depositor_name>
			<email_address>info@cirpublications.com</email_address>
		</depositor>
		<registrant>Clinical Intelligence Research Press</registrant>
	</head>
	<body>
		<journal>
			<journal_metadata>
				<full_title>Journal of Artificial Intelligence for Healthcare Systems</full_title>
				<abbrev_title>J. Artif. Intell. Healthc. Syst.</abbrev_title>
				<issn>3149-8981</issn>
			</journal_metadata>
			<journal_issue>
				<publication_date>
					<year>2026</year>
				</publication_date>
				<journal_volume>
					<volume>5</volume>
				</journal_volume>
				<issue>1</issue>
			</journal_issue>
			<journal_article publication_type="full_text">
				<titles>
					<title>Large Language Models in Clinical Medicine from 2017 to 2025: A Systematic Review of Performance on Medical Licensing Examinations, Clinical Documentation, Decision Support, and Safety Concerns</title>
				</titles>
								<contributors>
          					<person_name sequence="first" contributor_role="author">
            <given_name>Gabriel</given_name>
            <surname>Costa</surname>
					</person_name>
          					<person_name sequence="additional" contributor_role="author">
            <given_name>Rafael</given_name>
            <surname>Mendes</surname>
					</person_name>
          					<person_name sequence="additional" contributor_role="author">
            <given_name>Bruno</given_name>
            <surname>Teixeira</surname>
					</person_name>
          					<person_name sequence="additional" contributor_role="author">
            <given_name>Lucas</given_name>
            <surname>Ribeiro</surname>
					</person_name>
          				</contributors>
								<publication_date>
					<year>2026</year>
				</publication_date>
				<doi_data>
					<doi>10.68159/f858630908</doi>
					<resource>https://cirpublications.com/pub/journal/1/article/f858630908</resource>
				</doi_data>
				<citation_list>
          					<citation key="rk-10.68159/f858630908-7f98bd7f-7ed8-4f97-9fd7-282c55c0e42d">
					  <unstructured_citation>Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large language models encode clinical knowledge. Nat. 2023;620(7972):172-180.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-76775071-2d25-461b-a12e-3fa16bac8a7f">
					  <unstructured_citation>Singhal K, Tu T, Gottweis J, Sayres R, Wulczyn E, Amin M, et al. Toward expert-level medical question answering with large language models. Nat Med. 2025;31(3):943-950.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-bfcb9c94-ddb0-4b72-bf72-7ce33abbd189">
					  <unstructured_citation>Kung TH, Cheatham M, Medenilla A, Sillos C, De Leon L, Elepaño C, et al. Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models. PLOS Digit Health. 2023;2(2):e0000198.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-1fa361e3-2640-4079-93bd-fff2e7c03152">
					  <unstructured_citation>Gilson A, Safranek CW, Huang T, Socrates V, Chi L, Taylor RA, et al. How does ChatGPT perform on the United States Medical Licensing Examination (USMLE)? The implications of large language models for medical education and knowledge assessment. JMIR Med Educ. 2023;9:e45312.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-6347fb4c-c47f-4a0c-ba58-cda587afa4dc">
					  <unstructured_citation>Lee P, Bubeck S, Petro J. Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine. N Engl J Med. 2023;388(13):1233-9.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-7503c016-92aa-4392-bbca-5934ea4f56a9">
					  <unstructured_citation>Goh E, Gallo R, Hom J, Strong E, Weng Y, Kerman H, et al. Large language model influence on diagnostic reasoning: a randomized clinical trial. JAMA Netw Open. 2024;7(10):e2440969.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-2a73b479-bf9e-41a9-87b7-7b93316a5b13">
					  <unstructured_citation>Gaber F, Shaik M, Allega F, Bilecz AJ, Busch F, Goon K, et al. Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis. NPJ Digit Med. 2025;8(1):263.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-4716aad8-f3d0-463d-adeb-78c5e1d99598">
					  <unstructured_citation>Weissman GE, Mankowitz T, Kanter GP. Unregulated large language models produce medical device-like output. NPJ Digit Med. 2025;8(1):148.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-bcbb83a2-bc9a-4318-a40f-cdad1a308472">
					  <unstructured_citation>Brin D, Sorin V, Konen E, Nadkarni G, Glicksberg BS, Klang E. How GPT models perform on the United States medical licensing examination: a systematic review. Discov Appl Sci. 2024;6(10):500.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-353a3679-1eb8-4110-aa34-252d0bb02364">
					  <unstructured_citation>Mihalache A, Huang RS, Popovic MM, Muni RH. ChatGPT-4: an assessment of an upgraded artificial intelligence chatbot in the United States Medical Licensing Examination. Med Teach. 2024;46(3):366-72.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-14664772-57f8-4f63-9034-91ce2cea1e44">
					  <unstructured_citation>Song JW, Park J, Kim JH, You SC. Large language model assistant for emergency department discharge documentation. JAMA Netw Open. 2025;8(10):e2538427.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-faf94dcb-94f2-4d9f-8fad-38e31f98f37f">
					  <unstructured_citation>Ganzinger M, Kunz N, Fuchs P, Lyu CK, Loos M, Dugas M, et al. Automated generation of discharge summaries: leveraging large language models with clinical data. Sci Rep. 2025;15(1):16466.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-fb65eaca-44a0-41c2-822a-7e9a2a559fac">
					  <unstructured_citation>Rust P, Frings J, Meister S, Fehring L. Evaluation of a large language model to simplify discharge summaries and provide cardiological lifestyle recommendations. Commun Med. 2025;5(1):208.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-8a083682-10a2-4f65-a2bb-dd7503b55ee1">
					  <unstructured_citation>Strong E, DiGiammarino A, Weng Y, Kumar A, Hosamani P, Hom J, et al. Chatbot vs medical student performance on free-response clinical reasoning examinations. JAMA Intern Med. 2023;183(9):1028-30.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-4d25f836-c8fa-47ae-873f-ecf0a640a64b">
					  <unstructured_citation>Hains L, Kleinig O, Murugappa A, Gluck S, Marks J, Gilbert T, et al. Large language model discharge summary preparation using real world electronic medical record data shows promise. Intern Med J. 2025;55(7):1188-92.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-4f4e6de5-b2ae-4d1e-9493-528790a6690f">
					  <unstructured_citation>Williams CY, Subramanian CR, Ali SS, Apolinario M, Askin E, Barish P, et al. Physician- and large language model–generated hospital discharge summaries. JAMA Intern Med. 2025;185(7):818-25.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-1c489b2b-cdf7-44d4-9f67-98cef9c6fb95">
					  <unstructured_citation>Croxford E, Gao Y, First E, Pellegrino N, Schnier M, Caskey J, et al. Evaluating clinical AI summaries with large language models as judges. NPJ Digit Med. 2025;8(1):640.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-06d82056-edbb-492c-9af5-08157e6790d8">
					  <unstructured_citation>Woo BF, Cato K, Cho H, You SB, Song J. The use of large language models in clinical documentation: a scoping review. Int J Nurs Stud. 2025:105322.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-41e3d1bf-2d25-4560-8709-89e5cc072c5c">
					  <unstructured_citation>Mbakwe AB, Lourentzou I, Celi LA, Mechanic OJ, Dagan A. ChatGPT passing USMLE shines a spotlight on the flaws of medical education. PLOS Digit Health. 2023;2(2):e0000205.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-763b4987-231b-48ae-875d-50315c6b273d">
					  <unstructured_citation>Wu S, Koo M, Blum L, Black A, Kao L, Fei Z, et al. Benchmarking open-source large language models, GPT-4 and Claude 2 on multiple-choice questions in nephrology. NEJM AI. 2024;1(2):AIdbp2300092.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-56536e9d-fa8d-4784-ae87-b89b17faa63f">
					  <unstructured_citation>Peng C, Yang X, Chen A, Smith KE, PourNejatian N, Costa AB, et al. A study of generative large language model for medical research and healthcare. NPJ Digit Med. 2023;6(1):210.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-187bc822-dd83-4758-bca8-fb0825f530a5">
					  <unstructured_citation>Yang X, Chen A, PourNejatian N, Shin HC, Smith KE, Parisien C, et al. A large language model for electronic health records. NPJ Digit Med. 2022;5(1):194.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-82f07dde-23c7-4f0e-ab13-032ba9aa1d81">
					  <unstructured_citation>Asgari E, Montaña-Brown N, Dubois M, Khalil S, Balloch J, Yeung JA, et al. A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation. NPJ Digit Med. 2025;8(1):274.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-d760b649-efae-4da3-a41b-f4d32ae24839">
					  <unstructured_citation>Rutledge GW. Diagnostic accuracy of GPT-4 on common clinical scenarios and challenging cases. Learn Health Syst. 2024;8(3):e10438.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-906222e0-730c-419f-a01f-c8a2eee7fc35">
					  <unstructured_citation>Ong JC, Jin L, Elangovan K, San Lim GY, Lim DY, Sng GG, et al. Large language model as clinical decision support system augments medication safety in 16 clinical specialties. Cell Rep Med. 2025;6(10):101869.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-a147b1c5-e3be-4970-a85e-b11a9b85ff62">
					  <unstructured_citation>Katz U, Cohen E, Shachar E, Somer J, Fink A, Morse E, et al. GPT versus resident physicians—a benchmark based on official board scores. NEJM AI. 2024;1(5):AIdbp2300192.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-3c187bd8-0066-49e7-9de2-6a49714ca792">
					  <unstructured_citation>Shan G, Chen X, Wang C, Liu L, Gu Y, Jiang H, et al. Comparing diagnostic accuracy of clinical professionals and large language models: systematic review and meta-analysis. JMIR Med Inform. 2025;13(1):e64963.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-9e81a965-0a53-4444-a388-4dd0c597dc46">
					  <unstructured_citation>Jin Q, Chen F, Zhou Y, Xu Z, Cheung JM, Chen R, et al. Hidden flaws behind expert-level accuracy of multimodal GPT-4 vision in medicine. NPJ Digit Med. 2024;7(1):190.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-19140d58-e085-4ec6-a045-fb49784152e5">
					  <unstructured_citation>Siam MK, Varela A, Faruk MJ, Cheng JQ, Gu H, Maruf AA, et al. Benchmarking large language models on the United States medical licensing examination for clinical reasoning and medical licensing scenarios. Sci Rep. 2025;15(1):38421.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-15d6f548-e7b7-4d28-8a25-d93f83e1cf5e">
					  <unstructured_citation>Chen Y, Huang X, Yang F, Lin H, Lin H, Zheng Z, et al. Performance of ChatGPT and Bard on medical licensing examinations varies across different cultures: comparison study. BMC Med Educ. 2024;24(1):1372.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-ce87c978-d13b-46b0-b26c-dd17bac6e522">
					  <unstructured_citation>Wang L, Chen X, Deng X, Wen H, You M, Liu W, et al. Prompt engineering in consistency and reliability with evidence-based guideline for LLMs. NPJ Digit Med. 2024;7(1):41.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-d6d1bc12-c4c5-4dc3-b4f6-3827e23eb0ce">
					  <unstructured_citation>Tanaka Y, Nakata T, Aiga K, Etani T, Muramatsu R, Katagiri S, et al. Performance of generative pretrained transformer on the national medical licensing examination in Japan. PLOS Digit Health. 2024;3(1):e0000433.</unstructured_citation>
											</citation>
          					<citation key="rk-10.68159/f858630908-0df23560-e925-4106-87df-c7570d66bd91">
					  <unstructured_citation>Kresevic S, Giuffrè M, Ajcevic M, Accardo A, Crocè LS, Shung DL. Optimization of hepatological clinical guidelines interpretation by large language models: a retrieval augmented generation-based framework. NPJ Digit Med. 2024;7(1):102</unstructured_citation>
											</citation>
          				</citation_list>
			</journal_article>
		</journal>
	</body>
</doi_batch>
